From 4fb47535762e0bc0670b7bfce4f533f408c5f762 Mon Sep 17 00:00:00 2001 From: alexsoyes Date: Thu, 1 Oct 2026 06:28:48 +0800 Subject: [PATCH 1/6] feat(aidd-dev): use the smallest worker model with maximum reasoning --- plugins/aidd-dev/skills/09-for-sure/SKILL.md | 3 ++- .../aidd-dev/skills/09-for-sure/actions/03-autonomous-loop.md | 3 ++- .../skills/09-for-sure/assets/autonomous-loop-worker-prompt.md | 3 +++ 3 files changed, 7 insertions(+), 2 deletions(-) diff --git a/plugins/aidd-dev/skills/09-for-sure/SKILL.md b/plugins/aidd-dev/skills/09-for-sure/SKILL.md index 31d5be3e9..b58508fb3 100644 --- a/plugins/aidd-dev/skills/09-for-sure/SKILL.md +++ b/plugins/aidd-dev/skills/09-for-sure/SKILL.md @@ -13,7 +13,7 @@ Run an autonomous loop until a success condition is verified. An interactive pre | # | Action | Phase | Role | | --- | ----------------- | ---------------------- | -------------------------------------------------------------------------- | | 01 | `init-tracking` | interactive pre-flight | validate the goal, build the journey map, create the tracking file, spawn the loop | -| 02 | `auto-accept` | autonomous | decide and act as the user, stopping only on money or destructive actions | +| 02 | `auto-accept` | autonomous | decide and act as the user under the auto-accept rules | | 03 | `autonomous-loop` | autonomous | spawn one worker per step, verify, retry, evaluate the success condition | Run `01` interactively; it spawns `03`, which runs unattended under the `02` auto-accept rules until the success condition passes. @@ -26,6 +26,7 @@ Before running an action, read its file in `actions/`, not only the table or ass - Honesty over escape: never set `status: implemented` until the success condition genuinely passes. - Auto-accept: when a decision or approval is needed, act as the user (create accounts, generate keys, approve prompts, install tools), never asking. Stop only on a payment or a destructive action. - The loop spawns one worker agent per step and never does the work itself. +- Worker dispatch: at every launch or relaunch, use the smallest available model with its highest supported reasoning effort. The orchestrator retains reflection, framing, and replanning; workers execute their assigned step and return evidence. ## Assets diff --git a/plugins/aidd-dev/skills/09-for-sure/actions/03-autonomous-loop.md b/plugins/aidd-dev/skills/09-for-sure/actions/03-autonomous-loop.md index 853ed248d..aa26f3b14 100644 --- a/plugins/aidd-dev/skills/09-for-sure/actions/03-autonomous-loop.md +++ b/plugins/aidd-dev/skills/09-for-sure/actions/03-autonomous-loop.md @@ -16,7 +16,7 @@ The success condition verified and the plan's `status` set to `implemented`, wit 2. **Mark.** Increment `iteration` in the frontmatter, setting `status: in-progress` when still `pending`. 3. **Learn.** Read the Log to learn from prior attempts. 4. **Next.** Find the next unchecked step. -5. **Spawn.** Spawn a worker for that step with [autonomous-loop-worker-prompt.md](../assets/autonomous-loop-worker-prompt.md), passing the step description and the relevant context (objective, rules, prior Log entries for the step). +5. **Spawn.** Read the worker dispatch rule in [../SKILL.md](../SKILL.md), then spawn the worker with [autonomous-loop-worker-prompt.md](../assets/autonomous-loop-worker-prompt.md), passing the step description and relevant context (objective, rules, prior Log entries for the step). 6. **Verify.** Read the worker's result, then verify concretely by running a check command, reading a file, or testing the output. Never trust the worker's claim alone. 7. **Record.** Read the worker's outcome. When it stopped at a money or destructive gate, surface the reason to the user and stop the loop; never retry, which would re-trigger the action. On verified success, tick the step `[x]`. On a plain failure, spawn another worker with the error context. Append a Log entry per [autonomous-loop-log-format.md](../references/autonomous-loop-log-format.md). 8. **Loop.** Move to the next unchecked step and repeat from Read. @@ -26,4 +26,5 @@ The success condition verified and the plan's `status` set to `implemented`, wit - Each step attempt has exactly one Log entry. - Every checked step has a `= ✓` entry whose verification cites a concrete command or file. +- Each worker launch or relaunch uses the smallest available model and its highest supported reasoning effort. - `status: implemented` is set only after the `success_condition` command has been re-run and exits zero. diff --git a/plugins/aidd-dev/skills/09-for-sure/assets/autonomous-loop-worker-prompt.md b/plugins/aidd-dev/skills/09-for-sure/assets/autonomous-loop-worker-prompt.md index 3e4a7e57b..f8c8ea30d 100644 --- a/plugins/aidd-dev/skills/09-for-sure/assets/autonomous-loop-worker-prompt.md +++ b/plugins/aidd-dev/skills/09-for-sure/assets/autonomous-loop-worker-prompt.md @@ -4,6 +4,9 @@ Execute this step. Auto-accept everything, act as the user, and make every decision yourself (approve prompts, generate keys, install tools, click buttons). Do not ask for permission. Just do it. +Worker policy: execute only the assigned step and return concrete evidence. The +orchestrator retains reflection, framing, and replanning. + Signing in via an existing account (Google Sign-in, GitHub OAuth, SSO) is NOT account creation; it uses the user's active browser session. Do it. From 3601a88eab1838323b49b545861c893d88e4d31c Mon Sep 17 00:00:00 2001 From: alexsoyes Date: Thu, 1 Oct 2026 11:32:35 +0800 Subject: [PATCH 2/6] feat(aidd-dev)!: turn for-sure into goalify Keep goal framing and verification on a powerful model while delegating execution to the smallest available model. Analyze failures before changing the plan and retrying, without changing the tracking contract or safety boundaries. BREAKING CHANGE: Invoke aidd-dev:09-goalify instead of aidd-dev:09-for-sure. The old skill has no alias; existing tracking files remain compatible. --- .claude-plugin/marketplace.json | 2 +- aidd_docs/README.md | 4 +- docs/CATALOG.md | 4 +- plugins/aidd-dev/.claude-plugin/plugin.json | 4 +- plugins/aidd-dev/CATALOG.md | 18 +++--- plugins/aidd-dev/README.md | 4 +- plugins/aidd-dev/skills/09-for-sure/SKILL.md | 38 ------------- .../09-for-sure/actions/03-autonomous-loop.md | 30 ---------- plugins/aidd-dev/skills/09-goalify/SKILL.md | 55 +++++++++++++++++++ .../actions/01-init-tracking.md | 14 +++-- .../actions/02-auto-accept.md | 0 .../09-goalify/actions/03-autonomous-loop.md | 36 ++++++++++++ .../assets/autonomous-loop-worker-prompt.md | 0 .../assets/plan-template.md | 2 +- .../references/autonomous-loop-log-format.md | 0 15 files changed, 119 insertions(+), 92 deletions(-) delete mode 100644 plugins/aidd-dev/skills/09-for-sure/SKILL.md delete mode 100644 plugins/aidd-dev/skills/09-for-sure/actions/03-autonomous-loop.md create mode 100644 plugins/aidd-dev/skills/09-goalify/SKILL.md rename plugins/aidd-dev/skills/{09-for-sure => 09-goalify}/actions/01-init-tracking.md (70%) rename plugins/aidd-dev/skills/{09-for-sure => 09-goalify}/actions/02-auto-accept.md (100%) create mode 100644 plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md rename plugins/aidd-dev/skills/{09-for-sure => 09-goalify}/assets/autonomous-loop-worker-prompt.md (100%) rename plugins/aidd-dev/skills/{09-for-sure => 09-goalify}/assets/plan-template.md (95%) rename plugins/aidd-dev/skills/{09-for-sure => 09-goalify}/references/autonomous-loop-log-format.md (100%) diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 1e92608d2..c67764ff3 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -19,7 +19,7 @@ { "name": "aidd-dev", "source": "./plugins/aidd-dev", - "description": "Code transformation: plan, implement, assert, audit, review, test, refactor, debug, for-sure. Hosts engineering agents.", + "description": "Code transformation: plan, implement, assert, audit, review, test, refactor, debug, goalify. Hosts engineering agents.", "strict": true, "metadata": { "recommended": true diff --git a/aidd_docs/README.md b/aidd_docs/README.md index 0d6bb246a..1c7313c46 100644 --- a/aidd_docs/README.md +++ b/aidd_docs/README.md @@ -40,7 +40,7 @@ Skills are grouped into plugins by domain. Install only the plugins you need. | aidd-context | Bootstrap, project init, generation of context artifacts (skills, agents, rules, commands, hooks, plugins, marketplaces), mermaid diagrams, learn, discovery | `02-project-memory`, `03-context-generate`, `09-mermaid` | | aidd-refine | Meta-cognition: brainstorm, challenge prior work, blind-spot scan, fact-check | `01-brainstorm`, `02-challenge`, `03-shadow-areas` | | aidd-pm | Product management: backlog artifacts, refinement, Product Briefs, PRD, spec | `02-user-stories`, `05-spike`, `07-epic`, `09-defect`, `10-task` | -| aidd-dev | Code transformation: plan, implement, assert, audit, review, test, refactor, debug, for-sure | `01-plan`, `02-implement`, `05-review`, `06-test` | +| aidd-dev | Code transformation: plan, implement, assert, audit, review, test, refactor, debug, goalify | `01-plan`, `02-implement`, `05-review`, `06-test` | | aidd-vcs | VCS workflows: commit, pull/merge request, release tag, issue creation | `01-commit`, `02-pull-request`, `04-issue-create` | | aidd-orchestrator | Synchronous SDLC, async issue-to-PR automation, and product backlog | `00-async-dev`, `01-sdlc`, `02-backlog` | @@ -104,7 +104,7 @@ AIDD is delivered as a plugin marketplace. Pick what you need; do not install ev | ------------ | ------------------------------------------------------------------------------------------------------------------- | | aidd-context | 00-onboard, 01-bootstrap, 02-project-memory, 03-context-generate, 09-mermaid, 10-learn, 11-explore | | aidd-refine | 01-brainstorm, 02-challenge, 03-shadow-areas, 04-fact-check | -| aidd-dev | 01-plan, 02-implement, 03-assert, 04-audit, 05-review, 06-test, 07-refactor, 08-debug, 09-for-sure, 10-todo | +| aidd-dev | 01-plan, 02-implement, 03-assert, 04-audit, 05-review, 06-test, 07-refactor, 08-debug, 09-goalify, 10-todo | | aidd-orchestrator | 00-async-dev, 01-sdlc | | aidd-vcs | 01-commit, 02-pull-request, 03-release-tag, 04-issue-create | | aidd-pm | 01-ticket-info, 02-user-stories, 03-prd, 04-spec, 05-spike, 06-product-brief, 07-epic, 08-three-amigos, 09-defect, 10-task | diff --git a/docs/CATALOG.md b/docs/CATALOG.md index 66890a81c..660281985 100644 --- a/docs/CATALOG.md +++ b/docs/CATALOG.md @@ -35,7 +35,7 @@ Bootstrap, project init, context-artifact generation, diagrams, learning, and ex ## 💻 aidd-dev -Code transformation: plan, implement, assert, audit, review, test, refactor, debug, for-sure, todo. Standalone Browser QA records short web evidence. +Code transformation: plan, implement, assert, audit, review, test, refactor, debug, goalify, todo. Standalone Browser QA records short web evidence. | Skill | Role | Actions | | --------------- | -------------------------------------------------------------------------- | ------------------------------------------------------------------------------- | @@ -47,7 +47,7 @@ Code transformation: plan, implement, assert, audit, review, test, refactor, deb | `06-test` | Write and iterate tests, validate user journeys in the browser | `01-test`, `02-test-journey` | | `07-refactor` | Improve code without changing behavior across four axes | `01-performance`, `02-security`, `03-cleanup`, `04-architecture` | | `08-debug` | Reproduce and fix bugs with a test-driven workflow | `01-reproduce`, `02-debug`, `03-reflect-issue` | -| `09-for-sure` | Iterative loop that retries until a success condition is met | `01-init-tracking`, `02-auto-accept`, `03-autonomous-loop` | +| `09-goalify` | Autonomous loop that replans and retries until a runnable success condition passes | `01-init-tracking`, `02-auto-accept`, `03-autonomous-loop` | | `10-todo` | Split the prompt into independent todos, run one implementer agent per todo in parallel | `01-todo` | | `11-browser-qa` | Record short reviewer videos for browser-scoped happy and edge cases | `00-prerequisites`, `01-load-scope`, `02-prepare-run`, `03-run-scenarios` | diff --git a/plugins/aidd-dev/.claude-plugin/plugin.json b/plugins/aidd-dev/.claude-plugin/plugin.json index 917597521..8b0afd25e 100644 --- a/plugins/aidd-dev/.claude-plugin/plugin.json +++ b/plugins/aidd-dev/.claude-plugin/plugin.json @@ -2,7 +2,7 @@ "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "aidd-dev", "version": "2.5.0", - "description": "Code transformation: plan, implement, assert, audit, review, test, refactor, debug, for-sure, plus short standalone Browser QA evidence. Hosts engineering agents.", + "description": "Code transformation: plan, implement, assert, audit, review, test, refactor, debug, goalify, plus short standalone Browser QA evidence. Hosts engineering agents.", "author": { "name": "AI-Driven Dev", "url": "https://github.com/ai-driven-dev" @@ -16,7 +16,7 @@ "./skills/06-test", "./skills/07-refactor", "./skills/08-debug", - "./skills/09-for-sure", + "./skills/09-goalify", "./skills/10-todo", "./skills/11-browser-qa" ], diff --git a/plugins/aidd-dev/CATALOG.md b/plugins/aidd-dev/CATALOG.md index 671f54194..1147d152d 100644 --- a/plugins/aidd-dev/CATALOG.md +++ b/plugins/aidd-dev/CATALOG.md @@ -17,7 +17,7 @@ Auto-generated index of skills, agents, references and assets shipped by the `ai - [`skills/06-test`](#skills06-test) - [`skills/07-refactor`](#skills07-refactor) - [`skills/08-debug`](#skills08-debug) - - [`skills/09-for-sure`](#skills09-for-sure) + - [`skills/09-goalify`](#skills09-goalify) - [`skills/10-todo`](#skills10-todo) - [`skills/11-browser-qa`](#skills11-browser-qa) @@ -126,17 +126,17 @@ Auto-generated index of skills, agents, references and assets shipped by the `ai | `references` | [mermaid-conventions.md](skills/08-debug/references/mermaid-conventions.md) | `Rules for generating valid, high-quality Mermaid diagrams. Apply when creating or reviewing any Mermaid diagram (flowchart, state, ER, sequence, gantt).` | | `-` | [SKILL.md](skills/08-debug/SKILL.md) | `Reproduce and fix a known bug, or find an unknown root cause by hypothesis validation. Use when the user wants to fix a bug, find why something breaks, or reopen a stuck investigation. Not for building a feature or reviewing a diff.` | -#### `skills/09-for-sure` +#### `skills/09-goalify` | Group | File | Description | |-------|------|---| -| `actions` | [01-init-tracking.md](skills/09-for-sure/actions/01-init-tracking.md) | - | -| `actions` | [02-auto-accept.md](skills/09-for-sure/actions/02-auto-accept.md) | - | -| `actions` | [03-autonomous-loop.md](skills/09-for-sure/actions/03-autonomous-loop.md) | - | -| `assets` | [autonomous-loop-worker-prompt.md](skills/09-for-sure/assets/autonomous-loop-worker-prompt.md) | - | -| `assets` | [plan-template.md](skills/09-for-sure/assets/plan-template.md) | - | -| `references` | [autonomous-loop-log-format.md](skills/09-for-sure/references/autonomous-loop-log-format.md) | - | -| `-` | [SKILL.md](skills/09-for-sure/SKILL.md) | `Run an iterative agent loop that retries until a runnable success condition passes. Use when the user says "for sure", "keep trying until", or wants guaranteed completion against a success command. Not for one-shot tasks or uncheckable goals.` | +| `actions` | [01-init-tracking.md](skills/09-goalify/actions/01-init-tracking.md) | - | +| `actions` | [02-auto-accept.md](skills/09-goalify/actions/02-auto-accept.md) | - | +| `actions` | [03-autonomous-loop.md](skills/09-goalify/actions/03-autonomous-loop.md) | - | +| `assets` | [autonomous-loop-worker-prompt.md](skills/09-goalify/assets/autonomous-loop-worker-prompt.md) | - | +| `assets` | [plan-template.md](skills/09-goalify/assets/plan-template.md) | - | +| `references` | [autonomous-loop-log-format.md](skills/09-goalify/references/autonomous-loop-log-format.md) | - | +| `-` | [SKILL.md](skills/09-goalify/SKILL.md) | `Turn a goal into an autonomous loop that replans and retries until a runnable success condition passes. Use when the user says "goalify", "keep trying until", or wants a goal verified by a command. Not for one-shot tasks or uncheckable goals.` | #### `skills/10-todo` diff --git a/plugins/aidd-dev/README.md b/plugins/aidd-dev/README.md index a23bc830b..1150be1a5 100644 --- a/plugins/aidd-dev/README.md +++ b/plugins/aidd-dev/README.md @@ -8,7 +8,7 @@ Code transformation plugin for the AI-Driven Development framework. First time? Install with `/plugin install aidd-dev@aidd-framework`, then run `aidd-dev:01-plan`. -Covers code transformation: planning, implementation, assertions, audits, code review, testing, refactoring, debugging, for-sure, and parallel todo fan-out. Standalone Browser QA records short web evidence. Also hosts AI agents. +Covers code transformation: planning, implementation, assertions, audits, code review, testing, refactoring, debugging, goalify, and parallel todo fan-out. Standalone Browser QA records short web evidence. Also hosts AI agents. ## Skills @@ -22,7 +22,7 @@ Covers code transformation: planning, implementation, assertions, audits, code r | [2.6] | [test](skills/06-test/SKILL.md) | Write and iterate on tests until they pass, and validate user journeys end-to-end in the browser. | | [2.7] | [refactor](skills/07-refactor/SKILL.md) | Optimize code for performance and fix security vulnerabilities following OWASP guidelines. | | [2.8] | [debug](skills/08-debug/SKILL.md) | Reproduce and fix bugs systematically using test-driven workflow, root cause analysis, and hypothesis validation. | -| [2.9] | [for-sure](skills/09-for-sure/SKILL.md) | Iterative agent loop that tracks attempts and retries until a success condition is met. | +| [2.9] | [goalify](skills/09-goalify/SKILL.md) | Autonomous loop that replans and retries until a runnable success condition passes. | | [2.10] | [todo](skills/10-todo/SKILL.md) | Split the prompt into independent todos, run one executor agent per todo in parallel, then report a minimal table. | | [2.11] | [browser-qa](skills/11-browser-qa/SKILL.md) | Record one short named video for a locked browser happy path and each sourced browser edge case. | diff --git a/plugins/aidd-dev/skills/09-for-sure/SKILL.md b/plugins/aidd-dev/skills/09-for-sure/SKILL.md deleted file mode 100644 index b58508fb3..000000000 --- a/plugins/aidd-dev/skills/09-for-sure/SKILL.md +++ /dev/null @@ -1,38 +0,0 @@ ---- -name: 09-for-sure -description: Run an iterative agent loop that retries until a runnable success condition passes. Use when the user says "for sure", "keep trying until", or wants guaranteed completion against a success command. Not for one-shot tasks or uncheckable goals. -argument-hint: task | command ---- - -# Skill: for-sure - -Run an autonomous loop until a success condition is verified. An interactive pre-flight (human present) sets it up, then autonomous execution (human gone) retries until success, acting as the user and never stopping early. - -## Actions - -| # | Action | Phase | Role | -| --- | ----------------- | ---------------------- | -------------------------------------------------------------------------- | -| 01 | `init-tracking` | interactive pre-flight | validate the goal, build the journey map, create the tracking file, spawn the loop | -| 02 | `auto-accept` | autonomous | decide and act as the user under the auto-accept rules | -| 03 | `autonomous-loop` | autonomous | spawn one worker per step, verify, retry, evaluate the success condition | - -Run `01` interactively; it spawns `03`, which runs unattended under the `02` auto-accept rules until the success condition passes. -Before running an action, read its file in `actions/`, not only the table or assets. - -## Transversal rules - -- Single source of truth: all task state lives in `aidd_docs/tasks/.md` and nowhere else. -- No repeated failures: never retry a failed approach without a meaningful change. -- Honesty over escape: never set `status: implemented` until the success condition genuinely passes. -- Auto-accept: when a decision or approval is needed, act as the user (create accounts, generate keys, approve prompts, install tools), never asking. Stop only on a payment or a destructive action. -- The loop spawns one worker agent per step and never does the work itself. -- Worker dispatch: at every launch or relaunch, use the smallest available model with its highest supported reasoning effort. The orchestrator retains reflection, framing, and replanning; workers execute their assigned step and return evidence. - -## Assets - -- `assets/plan-template.md`: the tracking file format (frontmatter, phases, acceptance criteria, Log). -- `assets/autonomous-loop-worker-prompt.md`: the prompt the loop spawns each per-step worker with. - -## References - -- `references/autonomous-loop-log-format.md`: the Log entry format the loop appends per attempt. diff --git a/plugins/aidd-dev/skills/09-for-sure/actions/03-autonomous-loop.md b/plugins/aidd-dev/skills/09-for-sure/actions/03-autonomous-loop.md deleted file mode 100644 index aa26f3b14..000000000 --- a/plugins/aidd-dev/skills/09-for-sure/actions/03-autonomous-loop.md +++ /dev/null @@ -1,30 +0,0 @@ -# 03 - Autonomous loop - -Orchestrate the loop: for each unchecked step spawn a worker, verify the result, then check the box or retry. One step is one agent is one log entry, and the orchestrator never does the work itself. - -## Input - -The tracking file `aidd_docs/tasks/.md` produced by `01-init-tracking`. The loop runs with no human interaction, reading and writing through that file. - -## Output - -The success condition verified and the plan's `status` set to `implemented`, with every step checked and one Log entry per attempt. - -## Process - -1. **Read.** Read the entire file: frontmatter, journey map, steps, and full Log. -2. **Mark.** Increment `iteration` in the frontmatter, setting `status: in-progress` when still `pending`. -3. **Learn.** Read the Log to learn from prior attempts. -4. **Next.** Find the next unchecked step. -5. **Spawn.** Read the worker dispatch rule in [../SKILL.md](../SKILL.md), then spawn the worker with [autonomous-loop-worker-prompt.md](../assets/autonomous-loop-worker-prompt.md), passing the step description and relevant context (objective, rules, prior Log entries for the step). -6. **Verify.** Read the worker's result, then verify concretely by running a check command, reading a file, or testing the output. Never trust the worker's claim alone. -7. **Record.** Read the worker's outcome. When it stopped at a money or destructive gate, surface the reason to the user and stop the loop; never retry, which would re-trigger the action. On verified success, tick the step `[x]`. On a plain failure, spawn another worker with the error context. Append a Log entry per [autonomous-loop-log-format.md](../references/autonomous-loop-log-format.md). -8. **Loop.** Move to the next unchecked step and repeat from Read. -9. **Evaluate.** Once every step is checked, run the `success_condition` command and verify the result yourself. On success, set `status: implemented` and stop. On failure, add new steps addressing the root cause and continue the loop. - -## Test - -- Each step attempt has exactly one Log entry. -- Every checked step has a `= ✓` entry whose verification cites a concrete command or file. -- Each worker launch or relaunch uses the smallest available model and its highest supported reasoning effort. -- `status: implemented` is set only after the `success_condition` command has been re-run and exits zero. diff --git a/plugins/aidd-dev/skills/09-goalify/SKILL.md b/plugins/aidd-dev/skills/09-goalify/SKILL.md new file mode 100644 index 000000000..b789034bf --- /dev/null +++ b/plugins/aidd-dev/skills/09-goalify/SKILL.md @@ -0,0 +1,55 @@ +--- +name: 09-goalify +description: Turn a goal into an autonomous loop that replans and retries until a runnable success condition passes. Use when the user says "goalify", "keep trying until", or wants a goal verified by a command. Not for one-shot tasks or uncheckable goals. +argument-hint: task | command +--- + +# Skill: goalify + +Frame a checkable goal interactively, then run unattended until its success condition passes or a safety stop is reached. + +```mermaid +flowchart TD + Start[Setup or resume] --> Init[init-tracking] + Init -->|ready| Loop[autonomous-loop under auto-accept] + Init -->|already implemented| Done[Complete] + Init -->|unresolved prerequisite| Stop[Stop and report] + Loop --> Worker[Execute next step] + Worker -->|safety stop| Stop + Worker --> Verify[Verify evidence] + Verify -->|failure| Replan[Analyze and replan] + Replan -->|retry| Loop + Verify -->|more steps| Loop + Verify -->|all steps checked| Success{success_condition passes?} + Success -->|no| Replan + Success -->|yes| Done +``` + +## Actions + +| Action | Does | +| --- | --- | +| [init-tracking](actions/01-init-tracking.md) | frame the goal, create or resume tracking, launch the loop | +| [auto-accept](actions/02-auto-accept.md) | decide and act within the task's safety limits | +| [autonomous-loop](actions/03-autonomous-loop.md) | dispatch workers, verify evidence, replan failures, check completion | + +Run `init-tracking` interactively; it launches or resumes `autonomous-loop` under `auto-accept`. +Before running an action, read its file in `actions/`, not only the table or assets. + +## Transversal rules + +- Single source of truth: all task state lives in `aidd_docs/tasks/.md` and nowhere else. +- No repeated failures: never retry a failed approach without a meaningful change. +- Honesty over escape: never set `status: implemented` until the success condition genuinely passes. +- Auto-accept: follow the action's rules within the original task; stop on payment or destructive actions. +- The loop spawns one worker agent per step and never does the work itself. +- Model policy: use a powerful available model for framing, planning, verification, and replanning, including every orchestrator launch or resume. At every worker launch or relaunch, use the smallest available model with its highest supported reasoning effort. Workers execute only their assigned step and return evidence. + +## Assets + +- `assets/plan-template.md`: the tracking file format (frontmatter, phases, acceptance criteria, Log). +- `assets/autonomous-loop-worker-prompt.md`: the prompt the loop spawns each per-step worker with. + +## References + +- `references/autonomous-loop-log-format.md`: the Log entry format the loop appends per attempt. diff --git a/plugins/aidd-dev/skills/09-for-sure/actions/01-init-tracking.md b/plugins/aidd-dev/skills/09-goalify/actions/01-init-tracking.md similarity index 70% rename from plugins/aidd-dev/skills/09-for-sure/actions/01-init-tracking.md rename to plugins/aidd-dev/skills/09-goalify/actions/01-init-tracking.md index a96224abe..bb74b72f8 100644 --- a/plugins/aidd-dev/skills/09-for-sure/actions/01-init-tracking.md +++ b/plugins/aidd-dev/skills/09-goalify/actions/01-init-tracking.md @@ -12,7 +12,7 @@ The tracking file at `aidd_docs/tasks/.md`, marked created or resumed ## Process -1. **Resume.** Check `aidd_docs/tasks/` for a file matching the task name and read its frontmatter `status`. +1. **Resume.** Apply the [model policy](../SKILL.md#transversal-rules), then check `aidd_docs/tasks/` for a file matching the task name and read its frontmatter `status`. - `pending` or `in-progress`: report the status (iteration, steps remaining), then skip to Spawn to resume. - `implemented`: report "Task already completed" and stop. - No file: continue to Collect. @@ -24,10 +24,14 @@ The tracking file at `aidd_docs/tasks/.md`, marked created or resumed 7. **Map.** Project the whole path as an ASCII map of steps, dependencies, tools, and blockers. Ask the user to confirm and iterate until they do. 8. **Scaffold.** Load [plan-template.md](../assets/plan-template.md), creating `aidd_docs/tasks/` when missing. 9. **Create.** Write `aidd_docs/tasks/.md` from the template. Fill the frontmatter (`objective`, `success_condition`, `iteration: 0`, `status: pending`), the phases with their tasks and acceptance criteria, and the journey map. -10. **Spawn.** Read the orchestrator recipe from [03-autonomous-loop.md](./03-autonomous-loop.md) and hand it to the Agent tool with `` filled in. +10. **Spawn.** Apply the router's model policy to launch or resume the orchestrator with [03-autonomous-loop.md](./03-autonomous-loop.md) and `` filled in. ## Test -- The tracking file exists with frontmatter `status: pending` at creation. -- Its `success_condition` is a runnable command and the journey map is present. -- Every `[!]` blocker was resolved before the spawn, and an autonomous agent was launched. +| Case | Pass | +| --- | --- | +| New task | The tracking file exists at `aidd_docs/tasks/.md` with `status: pending`, a runnable `success_condition`, and a journey map. | +| Pending or in-progress task | The existing file is retained and the orchestrator resumes from its recorded state. | +| Implemented task | "Task already completed" is reported and no agent launches. | +| Unresolved hard prerequisite | No orchestrator launches until every `[!]` is resolved. | +| Setup or resume | Framing and orchestrator model selections follow the router's model policy. | diff --git a/plugins/aidd-dev/skills/09-for-sure/actions/02-auto-accept.md b/plugins/aidd-dev/skills/09-goalify/actions/02-auto-accept.md similarity index 100% rename from plugins/aidd-dev/skills/09-for-sure/actions/02-auto-accept.md rename to plugins/aidd-dev/skills/09-goalify/actions/02-auto-accept.md diff --git a/plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md b/plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md new file mode 100644 index 000000000..6ecb1d072 --- /dev/null +++ b/plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md @@ -0,0 +1,36 @@ +# 03 - Autonomous loop + +Orchestrate the loop: dispatch each unchecked step, verify the result, and replan failures before retrying. One attempt is one worker and one log entry; the orchestrator never does the work itself. + +## Input + +The tracking file `aidd_docs/tasks/.md` produced by `01-init-tracking`. The loop runs with no human interaction, reading and writing through that file. + +## Output + +The success condition verified and the plan's `status` set to `implemented`, with every step checked and one Log entry per attempt. Or a safety stop with its reason reported. + +## Process + +1. **Read.** Apply the [model policy](../SKILL.md#transversal-rules), then read the entire file: frontmatter, journey map, steps, and full Log. +2. **Mark.** Increment `iteration` in the frontmatter, setting `status: in-progress` when still `pending`. +3. **Learn.** Read the Log to learn from prior attempts. +4. **Next.** Find the next unchecked step. +5. **Spawn.** Apply the router's model policy to every worker launch or relaunch, using [autonomous-loop-worker-prompt.md](../assets/autonomous-loop-worker-prompt.md) with the step description and relevant context (objective, rules, prior Log entries for the step). +6. **Verify.** Read the worker's result, then verify concretely by running a check command, reading a file, or testing the output. Never trust the worker's claim alone. +7. **Record.** On verified success, tick the step `[x]`; on failure, leave it unchecked. Append a Log entry per [autonomous-loop-log-format.md](../references/autonomous-loop-log-format.md). If the worker stopped for payment, a destructive action, or an out-of-scope request, report the reason and stop; never retry the stopped action. +8. **Replan.** After a failure, analyze the worker and verification evidence and amend the tracking plan with a changed approach before retrying. Prefix each amendment with 🤖 and a brief rationale. Keep the original objective, success condition, and rules; pass the updated step and failure context to the next worker. +9. **Loop.** Move to the next unchecked step and repeat from Read. +10. **Evaluate.** Once every step is checked, run the `success_condition` command yourself. On exit 0, set `status: implemented` and stop. On failure, analyze the evidence and amend the plan as in Replan, adding unchecked steps for the root cause, then continue the loop. + +## Test + +| Case | Pass | +| --- | --- | +| Step attempt | Exactly one Log entry records the worker's attempt and the orchestrator's verification. | +| Verified step | The checked step has a `= ✓` entry citing a concrete command, file, or output. | +| Model selection | Orchestrator verification/replanning and every worker launch/relaunch follow the router's model policy. | +| Failed step | Before another worker launches, the tracking plan contains a changed approach and a 🤖 rationale based on the failure evidence. | +| Safety stop | The reason is reported, the stopped action is not retried, and `status` stays `in-progress`. | +| Final condition fails | `status` stays `in-progress`; the plan gains unchecked steps addressing the failure. | +| Final condition passes | `status: implemented` is set only after the orchestrator re-runs `success_condition` and observes exit 0. | diff --git a/plugins/aidd-dev/skills/09-for-sure/assets/autonomous-loop-worker-prompt.md b/plugins/aidd-dev/skills/09-goalify/assets/autonomous-loop-worker-prompt.md similarity index 100% rename from plugins/aidd-dev/skills/09-for-sure/assets/autonomous-loop-worker-prompt.md rename to plugins/aidd-dev/skills/09-goalify/assets/autonomous-loop-worker-prompt.md diff --git a/plugins/aidd-dev/skills/09-for-sure/assets/plan-template.md b/plugins/aidd-dev/skills/09-goalify/assets/plan-template.md similarity index 95% rename from plugins/aidd-dev/skills/09-for-sure/assets/plan-template.md rename to plugins/aidd-dev/skills/09-goalify/assets/plan-template.md index e5f274b24..dcf8aa3ee 100644 --- a/plugins/aidd-dev/skills/09-for-sure/assets/plan-template.md +++ b/plugins/aidd-dev/skills/09-goalify/assets/plan-template.md @@ -14,7 +14,7 @@ status: pending - Interpret comments on this file to help you fill it. - Each phase MUST have acceptance criteria. - During implementation, the AI may amend this plan. Every AI change MUST be prefixed with 🤖 and include a brief rationale. -- This file IS the live tracking file for For Sure. State lives in the `status` frontmatter field (`pending → in-progress → implemented`). +- This file IS the live tracking file for Goalify. State lives in the `status` frontmatter field (`pending → in-progress → implemented`). - `success_condition` MUST be a runnable command. The loop sets `status: implemented` only when it passes. - Log is APPEND-ONLY. One entry per step attempt. Never rewrite history. --> diff --git a/plugins/aidd-dev/skills/09-for-sure/references/autonomous-loop-log-format.md b/plugins/aidd-dev/skills/09-goalify/references/autonomous-loop-log-format.md similarity index 100% rename from plugins/aidd-dev/skills/09-for-sure/references/autonomous-loop-log-format.md rename to plugins/aidd-dev/skills/09-goalify/references/autonomous-loop-log-format.md From e4ecb1fc9f82dbe87541456b0bf7adda372ccfa5 Mon Sep 17 00:00:00 2001 From: alexsoyes Date: Mon, 5 Oct 2026 09:40:36 +0700 Subject: [PATCH 3/6] fix(aidd-context): forbid skill links back to loaded routers Skills already load their router; linking to it repeats context. Guard this invariant without restricting documentation links. --- .../references/skill-authoring.md | 2 +- .../09-goalify/actions/01-init-tracking.md | 2 +- .../09-goalify/actions/03-autonomous-loop.md | 2 +- .../a-skill-links-only-inside-itself.test.js | 127 ++++++++++++++---- 4 files changed, 103 insertions(+), 30 deletions(-) diff --git a/plugins/aidd-context/skills/04-skill-generate/references/skill-authoring.md b/plugins/aidd-context/skills/04-skill-generate/references/skill-authoring.md index 9170aefa1..c9fd9528a 100644 --- a/plugins/aidd-context/skills/04-skill-generate/references/skill-authoring.md +++ b/plugins/aidd-context/skills/04-skill-generate/references/skill-authoring.md @@ -16,7 +16,7 @@ The contract every generated skill satisfies. `skill-generate` obeys it too. - **R7.** The mermaid flow shows every path a run can take: the nominal chain, one entry node per case, a back-edge per loop, a terminal node per outcome. A branch stated in prose is a branch missing from the flow. - **R8.** The action table is `| Action | Does |`, one row per action file, in run order. `Action` is the bare slug, no backticks and no number. `Does` is a lowercase imperative half-line with no final period. Above the table, one sentence: what to read next, nothing more. - **R9.** `## Transversal rules` holds the rules no single action or reference owns. A rule stated there is stated nowhere else. -- **R10.** The router is loaded on every call, an action only when its turn comes: the router carries nothing an action or a reference could carry. +- **R10.** The router is loaded on every call, an action only when its turn comes: the router carries nothing an action or a reference could carry. Never link to `SKILL.md` from a skill file. ## An action diff --git a/plugins/aidd-dev/skills/09-goalify/actions/01-init-tracking.md b/plugins/aidd-dev/skills/09-goalify/actions/01-init-tracking.md index bb74b72f8..517ce2cb7 100644 --- a/plugins/aidd-dev/skills/09-goalify/actions/01-init-tracking.md +++ b/plugins/aidd-dev/skills/09-goalify/actions/01-init-tracking.md @@ -12,7 +12,7 @@ The tracking file at `aidd_docs/tasks/.md`, marked created or resumed ## Process -1. **Resume.** Apply the [model policy](../SKILL.md#transversal-rules), then check `aidd_docs/tasks/` for a file matching the task name and read its frontmatter `status`. +1. **Resume.** Apply the router's model policy, then check `aidd_docs/tasks/` for a file matching the task name and read its frontmatter `status`. - `pending` or `in-progress`: report the status (iteration, steps remaining), then skip to Spawn to resume. - `implemented`: report "Task already completed" and stop. - No file: continue to Collect. diff --git a/plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md b/plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md index 6ecb1d072..5ae4bce62 100644 --- a/plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md +++ b/plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md @@ -12,7 +12,7 @@ The success condition verified and the plan's `status` set to `implemented`, wit ## Process -1. **Read.** Apply the [model policy](../SKILL.md#transversal-rules), then read the entire file: frontmatter, journey map, steps, and full Log. +1. **Read.** Apply the router's model policy, then read the entire file: frontmatter, journey map, steps, and full Log. 2. **Mark.** Increment `iteration` in the frontmatter, setting `status: in-progress` when still `pending`. 3. **Learn.** Read the Log to learn from prior attempts. 4. **Next.** Find the next unchecked step. diff --git a/scripts/__tests__/a-skill-links-only-inside-itself.test.js b/scripts/__tests__/a-skill-links-only-inside-itself.test.js index 78ce3209f..37ef7e655 100644 --- a/scripts/__tests__/a-skill-links-only-inside-itself.test.js +++ b/scripts/__tests__/a-skill-links-only-inside-itself.test.js @@ -1,9 +1,11 @@ const assert = require("node:assert/strict"); const fs = require("node:fs"); +const os = require("node:os"); const path = require("node:path"); const { describe, it } = require("node:test"); const ROOT = path.resolve(__dirname, "../.."); +const PROBE_SOURCE = path.posix.join("plugins", "probe", "skills", "01-probe", "actions", "probe.md"); /** * A skill ships two ways and a relative path survives only one of them: the tree ships flat @@ -14,12 +16,20 @@ const ROOT = path.resolve(__dirname, "../.."); * The skill's own directory is the boundary, not the plugin's. A link to a sibling skill, to * the plugin's README, or to anything in the repository is equally unreachable once a tool * has installed the skill somewhere of its own choosing; name the file in prose instead. + * The router is already loaded, so links back to SKILL.md are redundant even when they + * stay inside the skill. */ const SKILL_ROOT = /^plugins\/[^/]+\/skills\/[^/]+$/u; +const LINK_DESTINATION = /\]\(\s*(<[^>\n]+>|[^\s)]+)[^)\n]*\)/gu; -/** Markdown link targets that are relative paths — not anchors, not URLs, not mail. */ -const RELATIVE_LINK = /\]\((\.[^)\s]*)\)/gu; +function decodedPath(target) { + try { + return decodeURI(target); + } catch { + return target; + } +} function everyMarkdownUnderPlugins(directory, into) { for (const entry of fs.readdirSync(directory, { withFileTypes: true })) { @@ -38,49 +48,112 @@ function skillRootOf(relativePath) { return SKILL_ROOT.test(root) ? root : null; } -function linksLeavingTheirSkill() { - const escaping = []; - for (const file of everyMarkdownUnderPlugins(path.join(ROOT, "plugins"), [])) { - const relative = path.relative(ROOT, file).split(path.sep).join("/"); +function skillLinkViolations(repository = ROOT) { + const violations = []; + for (const file of everyMarkdownUnderPlugins(path.join(repository, "plugins"), [])) { + const relative = path.relative(repository, file).split(path.sep).join("/"); const root = skillRootOf(relative); if (root === null) continue; const text = fs.readFileSync(file, "utf8"); - for (const [, target] of text.matchAll(RELATIVE_LINK)) { + for (const [, raw] of text.matchAll(LINK_DESTINATION)) { + const target = raw.startsWith("<") && raw.endsWith(">") ? raw.slice(1, -1) : raw; + if (/^(?:[a-z][a-z\d+.-]*:|\/\/|#)/iu.test(target)) continue; const resolved = path - .normalize(path.join(path.dirname(relative), target.split("#")[0])) + .normalize(path.join(path.dirname(relative), decodedPath(target.split("#")[0]))) .split(path.sep) .join("/"); - if (resolved !== root && !resolved.startsWith(`${root}/`)) { - escaping.push(`${relative} -> ${target}`); + if ( + path.posix.basename(resolved) === "SKILL.md" + || (resolved !== root && !resolved.startsWith(`${root}/`)) + ) { + violations.push(`${relative} -> ${raw}`); } } } - return escaping; + return violations; } -describe("a skill links only inside itself", () => { - it("has no markdown link reaching out of the skill that ships it", () => { +function probeSkillLinks(markdown) { + const repository = fs.mkdtempSync(path.join(os.tmpdir(), "aidd-skill-links-")); + try { + const skill = path.join(repository, "plugins", "probe", "skills", "01-probe"); + fs.mkdirSync(path.join(skill, "actions"), { recursive: true }); + fs.writeFileSync(path.join(skill, "actions", "probe.md"), markdown, "utf8"); + for (const file of ["README.md", "CATALOG.md"]) { + fs.writeFileSync( + path.join(repository, "plugins", "probe", file), + "[skill](skills/01-probe/SKILL.md#rules)\n", + "utf8", + ); + } + return skillLinkViolations(repository); + } finally { + fs.rmSync(repository, { recursive: true, force: true }); + } +} + +describe("a skill links only inside itself, never back to its loaded router", () => { + it("has no markdown link leaving its skill or returning to SKILL.md", () => { + assert.doesNotMatch( + fs.readFileSync(__filename, "utf8"), + /(plugins|scripts)\//u, + "synthetic fixture paths must not hide this repository-wide guard from changed-test selection", + ); assert.deepEqual( - linksLeavingTheirSkill(), + skillLinkViolations(), + [], + "links must stay inside their skill and never return to SKILL.md; the router is already loaded", + ); + }); + + it("rejects inline links back to a router, with or without a fragment", () => { + const cases = [ + ["[router](../SKILL.md)", "../SKILL.md"], + ["[rules](../SKILL.md#transversal-rules)", "../SKILL.md#transversal-rules"], + ["[router](SKILL.md)", "SKILL.md"], + ["[router](./SKILL.md)", "./SKILL.md"], + ["[router](../SKILL.md \"already loaded\")", "../SKILL.md"], + ["[rules](../SKILL.md#transversal-rules \"already loaded\")", "../SKILL.md#transversal-rules"], + ["[router](<../SKILL.md>)", "<../SKILL.md>"], + ["[rules](<../SKILL.md#transversal-rules>)", "<../SKILL.md#transversal-rules>"], + ["[router](../SKILL%2Emd)", "../SKILL%2Emd"], + ["[rules](../SKILL%2Emd#transversal-rules)", "../SKILL%2Emd#transversal-rules"], + ]; + assert.deepEqual( + probeSkillLinks(cases.map(([link]) => link).join("\n")), + cases.map(([, target]) => `${PROBE_SOURCE} -> ${target}`), + ); + }); + + it("accepts actions, references, bare router mentions and README/catalog skill links", () => { + assert.deepEqual( + probeSkillLinks([ + "[action](./next.md)", + "[reference](../references/rules.md#rule)", + "[reference](<../references/rules.md> \"rules\")", + "[reference](../references/rules%2Emd#rule)", + "[reference](../references/100%coverage.md)", + "The router SKILL.md is already loaded; apply `../SKILL.md` rules.", + "[docs](https://example.com/guide)", + "[section](#process)", + ].join("\n")), [], - "a relative link out of a skill is dead in every installed copy — name the file in prose instead", ); }); // The guard has to be able to see one. A checker that resolves every path against this // repository, the way `check-markdown-links.js` does, reports nothing here at all. it("sees an escaping link when one is put in front of it", () => { - const root = path.join(ROOT, "plugins", "aidd-dev", "skills", "01-plan"); - const probe = path.join(root, "escaping-link-probe.md"); - fs.writeFileSync(probe, "[out](../../../../README.md)\n", "utf8"); - try { - const found = linksLeavingTheirSkill(); - assert.ok( - found.some((entry) => entry.includes("escaping-link-probe.md")), - `the probe was not detected; found: ${JSON.stringify(found)}`, - ); - } finally { - fs.rmSync(probe); - } + const cases = [ + ["[out](../../../../README.md)", "../../../../README.md"], + ["[out\ncontinued](../../../../README.md)", "../../../../README.md"], + ["[out [nested]](../../../../README.md)", "../../../../README.md"], + ["[out](%2E%2E/%2E%2E/%2E%2E/%2E%2E/README.md)", "%2E%2E/%2E%2E/%2E%2E/%2E%2E/README.md"], + ["[out](../../../../README%ZZ.md)", "../../../../README%ZZ.md"], + ]; + assert.deepEqual( + probeSkillLinks(cases.map(([link]) => link).join("\n")), + cases.map(([, target]) => `${PROBE_SOURCE} -> ${target}`), + ); }); }); From 18122ca2fbffcc651501068a6d5efec686779c07 Mon Sep 17 00:00:00 2001 From: alexsoyes Date: Fri, 9 Oct 2026 13:35:21 +0700 Subject: [PATCH 4/6] feat(aidd-dev)!: delegate independent goal steps to batch Keep Goalify framing, model selection, verification and retries with the owner while Batch dispatches scoped executor leaves in parallel. Return per-item evidence and announce real batch launches without adding an agent tier. BREAKING CHANGE: Rename /aidd-dev:10-todo to /aidd-dev:10-batch without a compatibility alias. --- .claude-plugin/marketplace.json | 2 +- aidd_docs/README.md | 2 +- docs/ARCHITECTURE.md | 4 +- docs/CATALOG.md | 4 +- plugins/aidd-dev/.claude-plugin/plugin.json | 4 +- plugins/aidd-dev/CATALOG.md | 8 ++-- plugins/aidd-dev/README.md | 4 +- plugins/aidd-dev/skills/09-goalify/SKILL.md | 11 +++-- .../09-goalify/actions/03-autonomous-loop.md | 15 ++++--- .../assets/autonomous-loop-worker-prompt.md | 8 ++-- plugins/aidd-dev/skills/10-batch/SKILL.md | 23 ++++++++++ .../skills/10-batch/actions/01-batch.md | 37 +++++++++++++++ plugins/aidd-dev/skills/10-todo/SKILL.md | 17 ------- .../skills/10-todo/actions/01-todo.md | 31 ------------- .../skills/01-sdlc/references/03-check.md | 10 ++--- .../goalify-batch-publication.test.js | 45 +++++++++++++++++++ 16 files changed, 146 insertions(+), 79 deletions(-) create mode 100644 plugins/aidd-dev/skills/10-batch/SKILL.md create mode 100644 plugins/aidd-dev/skills/10-batch/actions/01-batch.md delete mode 100644 plugins/aidd-dev/skills/10-todo/SKILL.md delete mode 100644 plugins/aidd-dev/skills/10-todo/actions/01-todo.md create mode 100644 scripts/__tests__/goalify-batch-publication.test.js diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index c67764ff3..db0361b91 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -19,7 +19,7 @@ { "name": "aidd-dev", "source": "./plugins/aidd-dev", - "description": "Code transformation: plan, implement, assert, audit, review, test, refactor, debug, goalify. Hosts engineering agents.", + "description": "Code transformation: plan, implement, assert, audit, review, test, refactor, debug, goalify, batch. Hosts engineering agents.", "strict": true, "metadata": { "recommended": true diff --git a/aidd_docs/README.md b/aidd_docs/README.md index 1c7313c46..cb1866e77 100644 --- a/aidd_docs/README.md +++ b/aidd_docs/README.md @@ -104,7 +104,7 @@ AIDD is delivered as a plugin marketplace. Pick what you need; do not install ev | ------------ | ------------------------------------------------------------------------------------------------------------------- | | aidd-context | 00-onboard, 01-bootstrap, 02-project-memory, 03-context-generate, 09-mermaid, 10-learn, 11-explore | | aidd-refine | 01-brainstorm, 02-challenge, 03-shadow-areas, 04-fact-check | -| aidd-dev | 01-plan, 02-implement, 03-assert, 04-audit, 05-review, 06-test, 07-refactor, 08-debug, 09-goalify, 10-todo | +| aidd-dev | 01-plan, 02-implement, 03-assert, 04-audit, 05-review, 06-test, 07-refactor, 08-debug, 09-goalify, 10-batch | | aidd-orchestrator | 00-async-dev, 01-sdlc | | aidd-vcs | 01-commit, 02-pull-request, 03-release-tag, 04-issue-create | | aidd-pm | 01-ticket-info, 02-user-stories, 03-prd, 04-spec, 05-spike, 06-product-brief, 07-epic, 08-three-amigos, 09-defect, 10-task | diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 1b0f28551..5761e423c 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -144,8 +144,8 @@ A skill never links outside itself. The same tree ships flat, where the skill fo Choose by context, not complexity: keep the work visible to the caller → skill; isolate it and take only the result → agent. -- **Spawning is authorized by the high-level orchestrator, never invented by a recipe skill.** A recipe skill normally runs in the caller's context. A bounded fan-out capability may mechanically spawn leaf agents only when the orchestrator explicitly delegates that responsibility and retains routing ownership. -- An orchestrator spawns each isolated step as a leaf agent that runs a recipe, or runs the recipe itself when isolation is unnecessary. The SDLC owns planning, delegates delivery to `executor`, and delegates independent judgments to a fresh `checker`. For independent repair findings, it may explicitly delegate bounded fan-out to `10-todo`; Todo's leaf executors return their results to the SDLC. A recipe invoked inside an agent never spawns again. +- **Spawning is authorized by the high-level orchestrator or goal owner, never invented by a recipe skill.** A recipe skill normally runs in the caller's context. A bounded fan-out capability may mechanically spawn leaf agents only when the caller explicitly delegates that responsibility and retains routing ownership. +- An orchestrator spawns each isolated step as a leaf agent that runs a recipe, or runs the recipe itself when isolation is unnecessary. The SDLC owns planning, delegates delivery to `executor`, and delegates independent judgments to a fresh `checker`. For independent repair findings, it may explicitly delegate bounded fan-out to `10-batch`; Batch's leaf executors return their results to the SDLC. Goalify may likewise delegate ready independent tasks to Batch while retaining its plan, verification, and safety responsibilities. Batch runs in the caller's context, adding no agent tier; its leaf executors never spawn. A recipe invoked inside an agent never spawns again. - An agent invokes only the recipe skills it declares under `# Skills you may invoke`, never an orchestrator skill, and never reads a skill's files. It names every skill by its canonical `/plugin:folder` address so its permissions are explicit and auditable. - An agent never delegates flow work to another agent and never invokes an orchestrator skill. It may spawn a read-only recon helper (for example `Explore`) that mutates nothing and spawns nothing. So the write path stays two layers deep and delegation can never cycle. diff --git a/docs/CATALOG.md b/docs/CATALOG.md index 660281985..3a967540c 100644 --- a/docs/CATALOG.md +++ b/docs/CATALOG.md @@ -35,7 +35,7 @@ Bootstrap, project init, context-artifact generation, diagrams, learning, and ex ## 💻 aidd-dev -Code transformation: plan, implement, assert, audit, review, test, refactor, debug, goalify, todo. Standalone Browser QA records short web evidence. +Code transformation: plan, implement, assert, audit, review, test, refactor, debug, goalify, batch. Standalone Browser QA records short web evidence. | Skill | Role | Actions | | --------------- | -------------------------------------------------------------------------- | ------------------------------------------------------------------------------- | @@ -48,7 +48,7 @@ Code transformation: plan, implement, assert, audit, review, test, refactor, deb | `07-refactor` | Improve code without changing behavior across four axes | `01-performance`, `02-security`, `03-cleanup`, `04-architecture` | | `08-debug` | Reproduce and fix bugs with a test-driven workflow | `01-reproduce`, `02-debug`, `03-reflect-issue` | | `09-goalify` | Autonomous loop that replans and retries until a runnable success condition passes | `01-init-tracking`, `02-auto-accept`, `03-autonomous-loop` | -| `10-todo` | Split the prompt into independent todos, run one implementer agent per todo in parallel | `01-todo` | +| `10-batch` | Run ready independent tasks through bounded parallel executor agents in the caller's context | `01-batch` | | `11-browser-qa` | Record short reviewer videos for browser-scoped happy and edge cases | `00-prerequisites`, `01-load-scope`, `02-prepare-run`, `03-run-scenarios` | ## 📋 aidd-pm diff --git a/plugins/aidd-dev/.claude-plugin/plugin.json b/plugins/aidd-dev/.claude-plugin/plugin.json index 8b0afd25e..defa6d84f 100644 --- a/plugins/aidd-dev/.claude-plugin/plugin.json +++ b/plugins/aidd-dev/.claude-plugin/plugin.json @@ -2,7 +2,7 @@ "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "aidd-dev", "version": "2.5.0", - "description": "Code transformation: plan, implement, assert, audit, review, test, refactor, debug, goalify, plus short standalone Browser QA evidence. Hosts engineering agents.", + "description": "Code transformation: plan, implement, assert, audit, review, test, refactor, debug, goalify, batch, plus short standalone Browser QA evidence. Hosts engineering agents.", "author": { "name": "AI-Driven Dev", "url": "https://github.com/ai-driven-dev" @@ -17,7 +17,7 @@ "./skills/07-refactor", "./skills/08-debug", "./skills/09-goalify", - "./skills/10-todo", + "./skills/10-batch", "./skills/11-browser-qa" ], "agents": [ diff --git a/plugins/aidd-dev/CATALOG.md b/plugins/aidd-dev/CATALOG.md index 1147d152d..e12593094 100644 --- a/plugins/aidd-dev/CATALOG.md +++ b/plugins/aidd-dev/CATALOG.md @@ -18,7 +18,7 @@ Auto-generated index of skills, agents, references and assets shipped by the `ai - [`skills/07-refactor`](#skills07-refactor) - [`skills/08-debug`](#skills08-debug) - [`skills/09-goalify`](#skills09-goalify) - - [`skills/10-todo`](#skills10-todo) + - [`skills/10-batch`](#skills10-batch) - [`skills/11-browser-qa`](#skills11-browser-qa) --- @@ -138,12 +138,12 @@ Auto-generated index of skills, agents, references and assets shipped by the `ai | `references` | [autonomous-loop-log-format.md](skills/09-goalify/references/autonomous-loop-log-format.md) | - | | `-` | [SKILL.md](skills/09-goalify/SKILL.md) | `Turn a goal into an autonomous loop that replans and retries until a runnable success condition passes. Use when the user says "goalify", "keep trying until", or wants a goal verified by a command. Not for one-shot tasks or uncheckable goals.` | -#### `skills/10-todo` +#### `skills/10-batch` | Group | File | Description | |-------|------|---| -| `actions` | [01-todo.md](skills/10-todo/actions/01-todo.md) | - | -| `-` | [SKILL.md](skills/10-todo/SKILL.md) | `Split the user prompt into independent todos and run one executor agent per todo in parallel, then report a minimal table. Use when the user says "todo" or asks to fan out a multi-part request into parallel implementations.` | +| `actions` | [01-batch.md](skills/10-batch/actions/01-batch.md) | - | +| `-` | [SKILL.md](skills/10-batch/SKILL.md) | `Run independent tasks through bounded parallel executor agents and return per-item evidence. Use when the user says "batch", asks for parallel implementation, or a caller delegates ready independent tasks. Not for dependent tasks, goal planning, verification, or retries.` | #### `skills/11-browser-qa` diff --git a/plugins/aidd-dev/README.md b/plugins/aidd-dev/README.md index 1150be1a5..d0546d60f 100644 --- a/plugins/aidd-dev/README.md +++ b/plugins/aidd-dev/README.md @@ -8,7 +8,7 @@ Code transformation plugin for the AI-Driven Development framework. First time? Install with `/plugin install aidd-dev@aidd-framework`, then run `aidd-dev:01-plan`. -Covers code transformation: planning, implementation, assertions, audits, code review, testing, refactoring, debugging, goalify, and parallel todo fan-out. Standalone Browser QA records short web evidence. Also hosts AI agents. +Covers code transformation: planning, implementation, assertions, audits, code review, testing, refactoring, debugging, goalify, and bounded parallel execution through Batch. Standalone Browser QA records short web evidence. Also hosts AI agents. ## Skills @@ -23,7 +23,7 @@ Covers code transformation: planning, implementation, assertions, audits, code r | [2.7] | [refactor](skills/07-refactor/SKILL.md) | Optimize code for performance and fix security vulnerabilities following OWASP guidelines. | | [2.8] | [debug](skills/08-debug/SKILL.md) | Reproduce and fix bugs systematically using test-driven workflow, root cause analysis, and hypothesis validation. | | [2.9] | [goalify](skills/09-goalify/SKILL.md) | Autonomous loop that replans and retries until a runnable success condition passes. | -| [2.10] | [todo](skills/10-todo/SKILL.md) | Split the prompt into independent todos, run one executor agent per todo in parallel, then report a minimal table. | +| [2.10] | [Batch](skills/10-batch/SKILL.md) | Run ready independent tasks through bounded parallel executor agents in the caller's context, then report their results. | | [2.11] | [browser-qa](skills/11-browser-qa/SKILL.md) | Record one short named video for a locked browser happy path and each sourced browser edge case. | ## Agents diff --git a/plugins/aidd-dev/skills/09-goalify/SKILL.md b/plugins/aidd-dev/skills/09-goalify/SKILL.md index b789034bf..c6d13ff23 100644 --- a/plugins/aidd-dev/skills/09-goalify/SKILL.md +++ b/plugins/aidd-dev/skills/09-goalify/SKILL.md @@ -14,9 +14,12 @@ flowchart TD Init -->|ready| Loop[autonomous-loop under auto-accept] Init -->|already implemented| Done[Complete] Init -->|unresolved prerequisite| Stop[Stop and report] - Loop --> Worker[Execute next step] - Worker -->|safety stop| Stop - Worker --> Verify[Verify evidence] + Loop --> Select{Ready independent steps?} + Select -->|parallel capability available| Batch[Delegate parallel execution] + Select -->|otherwise| Worker[Execute next step] + Batch --> Verify[Verify each result] + Worker --> Verify + Verify -->|safety stop| Stop Verify -->|failure| Replan[Analyze and replan] Replan -->|retry| Loop Verify -->|more steps| Loop @@ -42,7 +45,7 @@ Before running an action, read its file in `actions/`, not only the table or ass - No repeated failures: never retry a failed approach without a meaningful change. - Honesty over escape: never set `status: implemented` until the success condition genuinely passes. - Auto-accept: follow the action's rules within the original task; stop on payment or destructive actions. -- The loop spawns one worker agent per step and never does the work itself. +- The loop dispatches one leaf worker per step, directly or through a discovered parallel-execution capability in its own context. It retains the plan, safety decisions, per-item verification, and retries; it never does the work itself or adds a controller agent. - Model policy: use a powerful available model for framing, planning, verification, and replanning, including every orchestrator launch or resume. At every worker launch or relaunch, use the smallest available model with its highest supported reasoning effort. Workers execute only their assigned step and return evidence. ## Assets diff --git a/plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md b/plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md index 5ae4bce62..30f3bd5bf 100644 --- a/plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md +++ b/plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md @@ -15,11 +15,11 @@ The success condition verified and the plan's `status` set to `implemented`, wit 1. **Read.** Apply the router's model policy, then read the entire file: frontmatter, journey map, steps, and full Log. 2. **Mark.** Increment `iteration` in the frontmatter, setting `status: in-progress` when still `pending`. 3. **Learn.** Read the Log to learn from prior attempts. -4. **Next.** Find the next unchecked step. -5. **Spawn.** Apply the router's model policy to every worker launch or relaunch, using [autonomous-loop-worker-prompt.md](../assets/autonomous-loop-worker-prompt.md) with the step description and relevant context (objective, rules, prior Log entries for the step). -6. **Verify.** Read the worker's result, then verify concretely by running a check command, reading a file, or testing the output. Never trust the worker's claim alone. -7. **Record.** On verified success, tick the step `[x]`; on failure, leave it unchecked. Append a Log entry per [autonomous-loop-log-format.md](../references/autonomous-loop-log-format.md). If the worker stopped for payment, a destructive action, or an out-of-scope request, report the reason and stop; never retry the stopped action. -8. **Replan.** After a failure, analyze the worker and verification evidence and amend the tracking plan with a changed approach before retrying. Prefix each amendment with 🤖 and a brief rationale. Keep the original objective, success condition, and rules; pass the updated step and failure context to the next worker. +4. **Select.** Find the next unchecked step, or a set of ready independent steps whose prerequisites are already verified. Respect the recorded phases and journey-map dependencies. Frame each item with its existing step identifier (or checkbox position), description, acceptance criteria, dependencies, allowed write scope, and relevant context. Unknown dependencies, overlapping write scopes, or shared mutable resources require sequential execution. If every step is checked, proceed to Evaluate. +5. **Dispatch.** Apply the router's model policy to every worker launch or relaunch, using [autonomous-loop-worker-prompt.md](../assets/autonomous-loop-worker-prompt.md) with the item and context (objective, rules, prior Log entries). For two or more ready independent items, discover a parallel-execution capability by description. Use it only if it accepts pre-framed items, concrete worker model/reasoning settings and safety instructions, spawns leaf executors in the caller's context, and returns per-item evidence without owning tracking or retries. Invoke it in this orchestrator's context, never as another agent. Supply the frames, resolved worker settings, and worker prompt. Immediately before the actual parallel launch, announce: `Batch: executing A, B and C in parallel. Goalify will verify each result.` Substitute the selected item identifiers; do not announce a batch for a rejected or sequential dispatch. If no compatible capability is available, spawn a single worker for the next step instead. +6. **Verify.** Collect results by item identifier. Treat missing or failed results as failures. Verify each returned result concretely by running a check command, reading a file, or testing the output; never trust a worker or Batch claim alone. Dependent steps remain blocked until their prerequisites pass this verification. +7. **Record.** For each attempted item, tick `[x]` only on verified success and append one Log entry per [autonomous-loop-log-format.md](../references/autonomous-loop-log-format.md); leave failed or missing results unchecked. Only the orchestrator writes the tracking file. If any worker stopped for payment, a destructive action, or an out-of-scope request, preserve the batch's partial evidence, stop further dispatch, request running workers to stop where supported, report the reason, and stop; never retry the stopped action. +8. **Replan.** After a failure, analyze that item's worker and verification evidence and amend its tracking step with a changed approach before retrying. Prefix each amendment with 🤖 and a brief rationale. Keep the original objective, success condition, and rules; pass the updated step and failure context to the next worker. Retain verified successes and never relaunch them merely because another batch item failed. 9. **Loop.** Move to the next unchecked step and repeat from Read. 10. **Evaluate.** Once every step is checked, run the `success_condition` command yourself. On exit 0, set `status: implemented` and stop. On failure, analyze the evidence and amend the plan as in Replan, adding unchecked steps for the root cause, then continue the loop. @@ -30,6 +30,11 @@ The success condition verified and the plan's `status` set to `implemented`, wit | Step attempt | Exactly one Log entry records the worker's attempt and the orchestrator's verification. | | Verified step | The checked step has a `= ✓` entry citing a concrete command, file, or output. | | Model selection | Orchestrator verification/replanning and every worker launch/relaunch follow the router's model policy. | +| Independent steps | Ready disjoint items are delegated in the orchestrator's context with its resolved worker settings and safety rules; the launch announcement names those items. | +| Unavailable capability or uncertain independence | One worker launches sequentially, without a batch announcement. | +| Dependency | A downstream step cannot launch before each prerequisite is independently verified. | +| Partial batch failure | Successful items stay checked; failed or missing items stay unchecked and are replanned before selective relaunch. | +| Batch tracking | Each attempted item has one Log entry in the existing tracking file; Batch and its executors do not write that file. | | Failed step | Before another worker launches, the tracking plan contains a changed approach and a 🤖 rationale based on the failure evidence. | | Safety stop | The reason is reported, the stopped action is not retried, and `status` stays `in-progress`. | | Final condition fails | `status` stays `in-progress`; the plan gains unchecked steps addressing the failure. | diff --git a/plugins/aidd-dev/skills/09-goalify/assets/autonomous-loop-worker-prompt.md b/plugins/aidd-dev/skills/09-goalify/assets/autonomous-loop-worker-prompt.md index f8c8ea30d..f2edd5545 100644 --- a/plugins/aidd-dev/skills/09-goalify/assets/autonomous-loop-worker-prompt.md +++ b/plugins/aidd-dev/skills/09-goalify/assets/autonomous-loop-worker-prompt.md @@ -4,8 +4,10 @@ Execute this step. Auto-accept everything, act as the user, and make every decision yourself (approve prompts, generate keys, install tools, click buttons). Do not ask for permission. Just do it. -Worker policy: execute only the assigned step and return concrete evidence. The -orchestrator retains reflection, framing, and replanning. +Worker policy: execute only the assigned step within its allowed write scope and +return concrete evidence. Never spawn agents or edit the orchestrator's tracking +file. Other workers may be active; preserve their changes. The orchestrator +retains reflection, framing, and replanning. Signing in via an existing account (Google Sign-in, GitHub OAuth, SSO) is NOT account creation; it uses the user's active browser session. Do it. @@ -15,7 +17,7 @@ payment, subscription, or paid upgrade) or is destructive (deletes data, drops a database, force-pushes, resets git history, removes files recursively, or overwrites uncommitted work). Stay inside the task; skip unrelated signups. -STEP: +STEP: CONTEXT: Report: diff --git a/plugins/aidd-dev/skills/10-batch/SKILL.md b/plugins/aidd-dev/skills/10-batch/SKILL.md new file mode 100644 index 000000000..67c09a8d4 --- /dev/null +++ b/plugins/aidd-dev/skills/10-batch/SKILL.md @@ -0,0 +1,23 @@ +--- +name: 10-batch +description: Run independent tasks through bounded parallel executor agents and return per-item evidence. Use when the user says "batch", asks for parallel implementation, or a caller delegates ready independent tasks. Not for dependent tasks, goal planning, verification, or retries. +argument-hint: requirement | framed items +--- + +# Batch + +Execute independent work in parallel, in the caller's context, without adding an agent tier. Accept a standalone request or already-framed items from a goal owner. + +## Actions + +| Action | Does | +| --- | --- | +| batch | Dispatch independent items to leaf executors and return their evidence | + +Before running an action, read its file in `actions/`, not only the table or assets. + +## Transversal rules + +- Run only when the user or owning caller authorizes parallel execution. A leaf executor must never invoke this spawning capability. +- Preserve delegated item boundaries, acceptance criteria, worker settings, and safety rules; do not reframe or replan them. +- Return evidence, not a verification verdict. The owner retains routing, tracking, verification, retries, and the final success decision. diff --git a/plugins/aidd-dev/skills/10-batch/actions/01-batch.md b/plugins/aidd-dev/skills/10-batch/actions/01-batch.md new file mode 100644 index 000000000..9d4fe4b50 --- /dev/null +++ b/plugins/aidd-dev/skills/10-batch/actions/01-batch.md @@ -0,0 +1,37 @@ +# 01 - Batch + +Dispatch ready independent items to leaf executors and return one evidence row per item. + +## Input + +A standalone requirement, or delegated items with stable identifiers, step descriptions, acceptance criteria, dependencies, allowed write scopes, relevant context, worker instructions, and caller-selected execution settings (model and reasoning effort when supplied). + +## Output + +One result per item, returned to the caller. `returned` means the executor supplied a result, not that an independent check passed. Include commands, outputs, changed files or other concrete evidence, and any failure or safety-stop reason. Missing results are failures. + +```markdown +| Item | Status | Evidence | +| ---- | ------ | -------- | +``` + +## Process + +1. **Frame.** For delegated items, retain the supplied frames unchanged. For a standalone request, split and refine items inline in the caller's context, using a discovered non-interactive refinement capability when available; otherwise clarify obvious ambiguity inline. Assign identifiers, acceptance criteria, dependencies, and allowed write scopes. Ask for the requirement only when standalone input is empty; return incomplete delegated frames to their owner. +2. **Check readiness.** Require satisfied dependencies and disjoint write scopes and mutable resources. Do not launch dependent, conflicting, or insufficiently scoped items together; return the reason to the caller without guessing or widening scope. Require the host to support the supplied worker settings and inherited safety rules. +3. **Launch.** In the caller's context, spawn one leaf `executor` per item concurrently within the host's capacity. Apply the supplied model and reasoning effort exactly, or inherit the caller's settings when none were supplied. Pass each item's frame, worker instructions, and safety rules unchanged. Mandate execution only within that item and its write scope, concrete evidence on return, and no agent spawning. Explicitly override the executor's per-unit commit policy for this dispatch: no Git mutations (staging, commits, branch changes, or pushes); the owner serializes any authorized Git writes after the batch. Never spawn a Batch controller or modify the owner's tracking file. +4. **Collect.** Wait for every launched item and associate its result with its identifier. Preserve partial results. On a safety stop, stop further dispatch and request any running executors to stop where supported; report the gate without retrying it. Do not retry, replan, certify completion, or launch dependent follow-up work. +5. **Report.** Return one minimal table with `returned`, `failed`, or `stopped` and concrete evidence per item. When invoked by a goal owner, return it to that owner for verification; do not override its progress announcement or final report. Standalone output is the table only. + +## Test + +| Case | Pass | +| --- | --- | +| Standalone request | Items are refined in the caller's context before executor launches; the user receives one result table. | +| Delegated frames | Identifiers, criteria, scopes, worker settings, and safety rules are preserved without a second refinement. | +| Independent items | One leaf executor per item runs concurrently; no extra agent tier or nested spawning is added. | +| Shared checkout | Leaf executors perform no Git mutations; the owner handles authorized Git writes after all workers finish. | +| Dependency or shared writes | Unsafe items are not launched together; their owner receives the reason. | +| Unsupported worker settings | No silently substituted model or reasoning effort is used. | +| Partial failure or missing result | Each item has its own evidence or failure row; no retry or success certification occurs. | +| Safety stop | Further dispatch stops, partial evidence is returned, and the gate is not retried. | diff --git a/plugins/aidd-dev/skills/10-todo/SKILL.md b/plugins/aidd-dev/skills/10-todo/SKILL.md deleted file mode 100644 index a2083c643..000000000 --- a/plugins/aidd-dev/skills/10-todo/SKILL.md +++ /dev/null @@ -1,17 +0,0 @@ ---- -name: 10-todo -description: Split the user prompt into independent todos and run one executor agent per todo in parallel, then report a minimal table. Use when the user says "todo" or asks to fan out a multi-part request into parallel implementations. -argument-hint: requirement ---- - -# Todo - -Turn one prompt into N independent todos, implement them in parallel, report a table. - -## Actions - -```markdown -actions/01-todo.md -``` - -Before running an action, read its file in `actions/`, not only the table or assets. diff --git a/plugins/aidd-dev/skills/10-todo/actions/01-todo.md b/plugins/aidd-dev/skills/10-todo/actions/01-todo.md deleted file mode 100644 index 222401dc4..000000000 --- a/plugins/aidd-dev/skills/10-todo/actions/01-todo.md +++ /dev/null @@ -1,31 +0,0 @@ -# 01 - Todo - -Categorize the user prompt into independent todos, implement each in parallel, report. - -## Input -User's requirement. - -## Output -```markdown -| Category | Launched | Output | -| -------- | -------- | ------ | -``` - -## Process - -1. **Read.** Take `prompt` from the arguments; if empty, ask the user. -2. **Categorize.** Split the prompt into distinct, independent todos (category + task). Inline, no agent. -3. **Launch.** Spawn one `executor` agent per todo, all in parallel (one message, multiple Task calls). Each agent prompt mandates, in order: - ```markdown - 1. Refine the todo first: run a non-interactive refine capability if one is available (discovered at runtime, never a hardcoded plugin name); otherwise restate the todo clearly and resolve obvious ambiguity inline. Never block on the user. - 2. Implement the refined todo. - 3. Return a one-line output summary. - ``` -4. **Report.** Print exactly one table, nothing else. - -## Test - -- Every todo is one row in the table. -- Agents were spawned in a single parallel batch. -- Each agent ran a refine step before implementing. -- No output besides the table. diff --git a/plugins/aidd-orchestrator/skills/01-sdlc/references/03-check.md b/plugins/aidd-orchestrator/skills/01-sdlc/references/03-check.md index c871e419f..ada870226 100644 --- a/plugins/aidd-orchestrator/skills/01-sdlc/references/03-check.md +++ b/plugins/aidd-orchestrator/skills/01-sdlc/references/03-check.md @@ -4,7 +4,7 @@ Use one fresh checker, independent from implementation, to review the candidate against the contract, plan, and validation evidence. After the review clears, the same checker challenges whether the real outcome is trustworthy and serves the user. -Use product or contract findings as the next Frame source. Dispatch independent implementation findings through Todo and keep dependent repairs together in Deliver. Re-enter Check after every new candidate. Open the draft pull request when the checker returns no actionable finding. +Use product or contract findings as the next Frame source. Dispatch independent implementation findings through Batch and keep dependent repairs together in Deliver. Re-enter Check after every new candidate. Open the draft pull request when the checker returns no actionable finding. ```mermaid --- @@ -30,7 +30,7 @@ flowchart TD direction TB Findings["$findings"] Frame["01 Frame"] - Todo["/aidd-dev:10-todo"] + Batch["/aidd-dev:10-batch"] Deliver["02 Deliver"] end @@ -50,9 +50,9 @@ flowchart TD Challenge -- "Return actionable challenge findings." --> Findings Challenge -- "When the outcome is trustworthy, open the draft pull request." --> PullRequest Findings -- "Use product findings as the next Frame source." --> Frame - Findings -- "Repair independent implementation findings in parallel." --> Todo + Findings -- "Repair independent implementation findings in parallel." --> Batch Findings -- "Repair dependent implementation findings together." --> Deliver - Todo --> Deliver + Batch --> Deliver PullRequest --> PullRequestUrl classDef skill fill:#DBEAFE,stroke:#2563EB,color:#1E3A8A,stroke-width:2px @@ -60,7 +60,7 @@ flowchart TD classDef artifact fill:#DCFCE7,stroke:#16A34A,color:#14532D,stroke-width:2px classDef zone fill:#F1F5F9,stroke:#64748B,color:#0F172A,stroke-width:2px - class Review,Challenge,Todo,PullRequest skill + class Review,Challenge,Batch,PullRequest skill class Checker agent class Contract,Plan,CommittedCandidate,ValidationReports,Findings,PullRequestUrl artifact class Frame,Deliver zone diff --git a/scripts/__tests__/goalify-batch-publication.test.js b/scripts/__tests__/goalify-batch-publication.test.js new file mode 100644 index 000000000..90dcccf02 --- /dev/null +++ b/scripts/__tests__/goalify-batch-publication.test.js @@ -0,0 +1,45 @@ +const assert = require("node:assert/strict"); +const fs = require("node:fs"); +const path = require("node:path"); +const test = require("node:test"); +const yaml = require("js-yaml"); + +const ROOT = path.resolve(__dirname, "../.."); +const DEV = "plugins/aidd-dev"; +const BATCH = `${DEV}/skills/10-batch`; +const read = (file) => fs.readFileSync(path.join(ROOT, file), "utf8"); + +function obsoleteAddresses(text) { + return [...text.matchAll(/\b10-todo\b/gu)].map((match) => match[0]); +} + +test("the published batch capability resolves to its own router and action", () => { + const manifest = JSON.parse(read(`${DEV}/.claude-plugin/plugin.json`)); + assert.ok(manifest.skills.includes("./skills/10-batch"), "the batch capability is not exported"); + assert.ok(!manifest.skills.includes("./skills/10-todo"), "the obsolete invocation is still exported"); + + const frontmatter = yaml.load(read(`${BATCH}/SKILL.md`).split(/^---\s*$/mu)[1]); + assert.equal(frontmatter.name, "10-batch"); + assert.ok(frontmatter.description); + assert.ok(frontmatter["argument-hint"]); + assert.ok(fs.existsSync(path.join(ROOT, BATCH, "actions/01-batch.md"))); + assert.ok(!fs.existsSync(path.join(ROOT, DEV, "skills/10-todo/SKILL.md")), "no old skill alias ships"); +}); + +test("active documentation and orchestration addresses resolve to batch", () => { + const files = [ + `${DEV}/README.md`, + `${DEV}/CATALOG.md`, + "docs/CATALOG.md", + "docs/ARCHITECTURE.md", + "aidd_docs/README.md", + "plugins/aidd-orchestrator/skills/01-sdlc/references/03-check.md", + ]; + const stale = files.flatMap((file) => obsoleteAddresses(read(file)).map((address) => `${file}: ${address}`)); + assert.deepEqual(stale, [], "a public caller still points to a capability that no longer ships"); +}); + +test("the address guard detects a stale dispatch without rejecting ordinary todo terminology", () => { + assert.deepEqual(obsoleteAddresses("Dispatch /aidd-dev:10-todo."), ["10-todo"]); + assert.deepEqual(obsoleteAddresses("A board's Todo column and a tracked todo list."), []); +}); From 5eced3fbdb8f31f6b20d631eababf29c60c4cfd5 Mon Sep 17 00:00:00 2001 From: alexsoyes Date: Fri, 9 Oct 2026 13:52:19 +0700 Subject: [PATCH 5/6] fix(aidd-dev): leave batch rename to its dedicated pull request Remove the out-of-scope Batch implementation, rename and publication changes. Keep only Goalify delegation and its Batch announcement; the separate pull request owns the capability. --- .claude-plugin/marketplace.json | 2 +- aidd_docs/README.md | 2 +- docs/ARCHITECTURE.md | 4 +- docs/CATALOG.md | 4 +- plugins/aidd-dev/.claude-plugin/plugin.json | 4 +- plugins/aidd-dev/CATALOG.md | 8 ++-- plugins/aidd-dev/README.md | 4 +- .../09-goalify/actions/03-autonomous-loop.md | 3 +- plugins/aidd-dev/skills/10-batch/SKILL.md | 23 ---------- .../skills/10-batch/actions/01-batch.md | 37 --------------- plugins/aidd-dev/skills/10-todo/SKILL.md | 17 +++++++ .../skills/10-todo/actions/01-todo.md | 31 +++++++++++++ .../skills/01-sdlc/references/03-check.md | 10 ++--- .../goalify-batch-publication.test.js | 45 ------------------- 14 files changed, 69 insertions(+), 125 deletions(-) delete mode 100644 plugins/aidd-dev/skills/10-batch/SKILL.md delete mode 100644 plugins/aidd-dev/skills/10-batch/actions/01-batch.md create mode 100644 plugins/aidd-dev/skills/10-todo/SKILL.md create mode 100644 plugins/aidd-dev/skills/10-todo/actions/01-todo.md delete mode 100644 scripts/__tests__/goalify-batch-publication.test.js diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index db0361b91..c67764ff3 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -19,7 +19,7 @@ { "name": "aidd-dev", "source": "./plugins/aidd-dev", - "description": "Code transformation: plan, implement, assert, audit, review, test, refactor, debug, goalify, batch. Hosts engineering agents.", + "description": "Code transformation: plan, implement, assert, audit, review, test, refactor, debug, goalify. Hosts engineering agents.", "strict": true, "metadata": { "recommended": true diff --git a/aidd_docs/README.md b/aidd_docs/README.md index cb1866e77..1c7313c46 100644 --- a/aidd_docs/README.md +++ b/aidd_docs/README.md @@ -104,7 +104,7 @@ AIDD is delivered as a plugin marketplace. Pick what you need; do not install ev | ------------ | ------------------------------------------------------------------------------------------------------------------- | | aidd-context | 00-onboard, 01-bootstrap, 02-project-memory, 03-context-generate, 09-mermaid, 10-learn, 11-explore | | aidd-refine | 01-brainstorm, 02-challenge, 03-shadow-areas, 04-fact-check | -| aidd-dev | 01-plan, 02-implement, 03-assert, 04-audit, 05-review, 06-test, 07-refactor, 08-debug, 09-goalify, 10-batch | +| aidd-dev | 01-plan, 02-implement, 03-assert, 04-audit, 05-review, 06-test, 07-refactor, 08-debug, 09-goalify, 10-todo | | aidd-orchestrator | 00-async-dev, 01-sdlc | | aidd-vcs | 01-commit, 02-pull-request, 03-release-tag, 04-issue-create | | aidd-pm | 01-ticket-info, 02-user-stories, 03-prd, 04-spec, 05-spike, 06-product-brief, 07-epic, 08-three-amigos, 09-defect, 10-task | diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 5761e423c..1b0f28551 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -144,8 +144,8 @@ A skill never links outside itself. The same tree ships flat, where the skill fo Choose by context, not complexity: keep the work visible to the caller → skill; isolate it and take only the result → agent. -- **Spawning is authorized by the high-level orchestrator or goal owner, never invented by a recipe skill.** A recipe skill normally runs in the caller's context. A bounded fan-out capability may mechanically spawn leaf agents only when the caller explicitly delegates that responsibility and retains routing ownership. -- An orchestrator spawns each isolated step as a leaf agent that runs a recipe, or runs the recipe itself when isolation is unnecessary. The SDLC owns planning, delegates delivery to `executor`, and delegates independent judgments to a fresh `checker`. For independent repair findings, it may explicitly delegate bounded fan-out to `10-batch`; Batch's leaf executors return their results to the SDLC. Goalify may likewise delegate ready independent tasks to Batch while retaining its plan, verification, and safety responsibilities. Batch runs in the caller's context, adding no agent tier; its leaf executors never spawn. A recipe invoked inside an agent never spawns again. +- **Spawning is authorized by the high-level orchestrator, never invented by a recipe skill.** A recipe skill normally runs in the caller's context. A bounded fan-out capability may mechanically spawn leaf agents only when the orchestrator explicitly delegates that responsibility and retains routing ownership. +- An orchestrator spawns each isolated step as a leaf agent that runs a recipe, or runs the recipe itself when isolation is unnecessary. The SDLC owns planning, delegates delivery to `executor`, and delegates independent judgments to a fresh `checker`. For independent repair findings, it may explicitly delegate bounded fan-out to `10-todo`; Todo's leaf executors return their results to the SDLC. A recipe invoked inside an agent never spawns again. - An agent invokes only the recipe skills it declares under `# Skills you may invoke`, never an orchestrator skill, and never reads a skill's files. It names every skill by its canonical `/plugin:folder` address so its permissions are explicit and auditable. - An agent never delegates flow work to another agent and never invokes an orchestrator skill. It may spawn a read-only recon helper (for example `Explore`) that mutates nothing and spawns nothing. So the write path stays two layers deep and delegation can never cycle. diff --git a/docs/CATALOG.md b/docs/CATALOG.md index 3a967540c..660281985 100644 --- a/docs/CATALOG.md +++ b/docs/CATALOG.md @@ -35,7 +35,7 @@ Bootstrap, project init, context-artifact generation, diagrams, learning, and ex ## 💻 aidd-dev -Code transformation: plan, implement, assert, audit, review, test, refactor, debug, goalify, batch. Standalone Browser QA records short web evidence. +Code transformation: plan, implement, assert, audit, review, test, refactor, debug, goalify, todo. Standalone Browser QA records short web evidence. | Skill | Role | Actions | | --------------- | -------------------------------------------------------------------------- | ------------------------------------------------------------------------------- | @@ -48,7 +48,7 @@ Code transformation: plan, implement, assert, audit, review, test, refactor, deb | `07-refactor` | Improve code without changing behavior across four axes | `01-performance`, `02-security`, `03-cleanup`, `04-architecture` | | `08-debug` | Reproduce and fix bugs with a test-driven workflow | `01-reproduce`, `02-debug`, `03-reflect-issue` | | `09-goalify` | Autonomous loop that replans and retries until a runnable success condition passes | `01-init-tracking`, `02-auto-accept`, `03-autonomous-loop` | -| `10-batch` | Run ready independent tasks through bounded parallel executor agents in the caller's context | `01-batch` | +| `10-todo` | Split the prompt into independent todos, run one implementer agent per todo in parallel | `01-todo` | | `11-browser-qa` | Record short reviewer videos for browser-scoped happy and edge cases | `00-prerequisites`, `01-load-scope`, `02-prepare-run`, `03-run-scenarios` | ## 📋 aidd-pm diff --git a/plugins/aidd-dev/.claude-plugin/plugin.json b/plugins/aidd-dev/.claude-plugin/plugin.json index defa6d84f..8b0afd25e 100644 --- a/plugins/aidd-dev/.claude-plugin/plugin.json +++ b/plugins/aidd-dev/.claude-plugin/plugin.json @@ -2,7 +2,7 @@ "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "aidd-dev", "version": "2.5.0", - "description": "Code transformation: plan, implement, assert, audit, review, test, refactor, debug, goalify, batch, plus short standalone Browser QA evidence. Hosts engineering agents.", + "description": "Code transformation: plan, implement, assert, audit, review, test, refactor, debug, goalify, plus short standalone Browser QA evidence. Hosts engineering agents.", "author": { "name": "AI-Driven Dev", "url": "https://github.com/ai-driven-dev" @@ -17,7 +17,7 @@ "./skills/07-refactor", "./skills/08-debug", "./skills/09-goalify", - "./skills/10-batch", + "./skills/10-todo", "./skills/11-browser-qa" ], "agents": [ diff --git a/plugins/aidd-dev/CATALOG.md b/plugins/aidd-dev/CATALOG.md index e12593094..1147d152d 100644 --- a/plugins/aidd-dev/CATALOG.md +++ b/plugins/aidd-dev/CATALOG.md @@ -18,7 +18,7 @@ Auto-generated index of skills, agents, references and assets shipped by the `ai - [`skills/07-refactor`](#skills07-refactor) - [`skills/08-debug`](#skills08-debug) - [`skills/09-goalify`](#skills09-goalify) - - [`skills/10-batch`](#skills10-batch) + - [`skills/10-todo`](#skills10-todo) - [`skills/11-browser-qa`](#skills11-browser-qa) --- @@ -138,12 +138,12 @@ Auto-generated index of skills, agents, references and assets shipped by the `ai | `references` | [autonomous-loop-log-format.md](skills/09-goalify/references/autonomous-loop-log-format.md) | - | | `-` | [SKILL.md](skills/09-goalify/SKILL.md) | `Turn a goal into an autonomous loop that replans and retries until a runnable success condition passes. Use when the user says "goalify", "keep trying until", or wants a goal verified by a command. Not for one-shot tasks or uncheckable goals.` | -#### `skills/10-batch` +#### `skills/10-todo` | Group | File | Description | |-------|------|---| -| `actions` | [01-batch.md](skills/10-batch/actions/01-batch.md) | - | -| `-` | [SKILL.md](skills/10-batch/SKILL.md) | `Run independent tasks through bounded parallel executor agents and return per-item evidence. Use when the user says "batch", asks for parallel implementation, or a caller delegates ready independent tasks. Not for dependent tasks, goal planning, verification, or retries.` | +| `actions` | [01-todo.md](skills/10-todo/actions/01-todo.md) | - | +| `-` | [SKILL.md](skills/10-todo/SKILL.md) | `Split the user prompt into independent todos and run one executor agent per todo in parallel, then report a minimal table. Use when the user says "todo" or asks to fan out a multi-part request into parallel implementations.` | #### `skills/11-browser-qa` diff --git a/plugins/aidd-dev/README.md b/plugins/aidd-dev/README.md index d0546d60f..1150be1a5 100644 --- a/plugins/aidd-dev/README.md +++ b/plugins/aidd-dev/README.md @@ -8,7 +8,7 @@ Code transformation plugin for the AI-Driven Development framework. First time? Install with `/plugin install aidd-dev@aidd-framework`, then run `aidd-dev:01-plan`. -Covers code transformation: planning, implementation, assertions, audits, code review, testing, refactoring, debugging, goalify, and bounded parallel execution through Batch. Standalone Browser QA records short web evidence. Also hosts AI agents. +Covers code transformation: planning, implementation, assertions, audits, code review, testing, refactoring, debugging, goalify, and parallel todo fan-out. Standalone Browser QA records short web evidence. Also hosts AI agents. ## Skills @@ -23,7 +23,7 @@ Covers code transformation: planning, implementation, assertions, audits, code r | [2.7] | [refactor](skills/07-refactor/SKILL.md) | Optimize code for performance and fix security vulnerabilities following OWASP guidelines. | | [2.8] | [debug](skills/08-debug/SKILL.md) | Reproduce and fix bugs systematically using test-driven workflow, root cause analysis, and hypothesis validation. | | [2.9] | [goalify](skills/09-goalify/SKILL.md) | Autonomous loop that replans and retries until a runnable success condition passes. | -| [2.10] | [Batch](skills/10-batch/SKILL.md) | Run ready independent tasks through bounded parallel executor agents in the caller's context, then report their results. | +| [2.10] | [todo](skills/10-todo/SKILL.md) | Split the prompt into independent todos, run one executor agent per todo in parallel, then report a minimal table. | | [2.11] | [browser-qa](skills/11-browser-qa/SKILL.md) | Record one short named video for a locked browser happy path and each sourced browser edge case. | ## Agents diff --git a/plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md b/plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md index 30f3bd5bf..39296de35 100644 --- a/plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md +++ b/plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md @@ -16,7 +16,7 @@ The success condition verified and the plan's `status` set to `implemented`, wit 2. **Mark.** Increment `iteration` in the frontmatter, setting `status: in-progress` when still `pending`. 3. **Learn.** Read the Log to learn from prior attempts. 4. **Select.** Find the next unchecked step, or a set of ready independent steps whose prerequisites are already verified. Respect the recorded phases and journey-map dependencies. Frame each item with its existing step identifier (or checkbox position), description, acceptance criteria, dependencies, allowed write scope, and relevant context. Unknown dependencies, overlapping write scopes, or shared mutable resources require sequential execution. If every step is checked, proceed to Evaluate. -5. **Dispatch.** Apply the router's model policy to every worker launch or relaunch, using [autonomous-loop-worker-prompt.md](../assets/autonomous-loop-worker-prompt.md) with the item and context (objective, rules, prior Log entries). For two or more ready independent items, discover a parallel-execution capability by description. Use it only if it accepts pre-framed items, concrete worker model/reasoning settings and safety instructions, spawns leaf executors in the caller's context, and returns per-item evidence without owning tracking or retries. Invoke it in this orchestrator's context, never as another agent. Supply the frames, resolved worker settings, and worker prompt. Immediately before the actual parallel launch, announce: `Batch: executing A, B and C in parallel. Goalify will verify each result.` Substitute the selected item identifiers; do not announce a batch for a rejected or sequential dispatch. If no compatible capability is available, spawn a single worker for the next step instead. +5. **Dispatch.** Apply the router's model policy to every worker launch or relaunch, using [autonomous-loop-worker-prompt.md](../assets/autonomous-loop-worker-prompt.md) with the item and context (objective, rules, prior Log entries). For two or more ready independent items, discover the parallel-execution capability (Batch) by description. Use it only if it accepts pre-framed items, concrete worker model/reasoning settings and safety instructions, spawns leaf executors in the caller's context, and returns per-item evidence without owning tracking or retries. Invoke it in this orchestrator's context, never as another agent. Supply the frames, resolved worker settings, and worker prompt. For parallel workers, explicitly override any per-unit commit policy: no Git mutations during the batch; dispatch required Git-mutating steps separately, sequentially. Immediately before the actual parallel launch, announce: `Batch: executing A, B and C in parallel. Goalify will verify each result.` Substitute the selected item identifiers; do not announce a batch for a rejected or sequential dispatch. If no compatible capability is available, spawn a single worker for the next step instead. 6. **Verify.** Collect results by item identifier. Treat missing or failed results as failures. Verify each returned result concretely by running a check command, reading a file, or testing the output; never trust a worker or Batch claim alone. Dependent steps remain blocked until their prerequisites pass this verification. 7. **Record.** For each attempted item, tick `[x]` only on verified success and append one Log entry per [autonomous-loop-log-format.md](../references/autonomous-loop-log-format.md); leave failed or missing results unchecked. Only the orchestrator writes the tracking file. If any worker stopped for payment, a destructive action, or an out-of-scope request, preserve the batch's partial evidence, stop further dispatch, request running workers to stop where supported, report the reason, and stop; never retry the stopped action. 8. **Replan.** After a failure, analyze that item's worker and verification evidence and amend its tracking step with a changed approach before retrying. Prefix each amendment with 🤖 and a brief rationale. Keep the original objective, success condition, and rules; pass the updated step and failure context to the next worker. Retain verified successes and never relaunch them merely because another batch item failed. @@ -35,6 +35,7 @@ The success condition verified and the plan's `status` set to `implemented`, wit | Dependency | A downstream step cannot launch before each prerequisite is independently verified. | | Partial batch failure | Successful items stay checked; failed or missing items stay unchecked and are replanned before selective relaunch. | | Batch tracking | Each attempted item has one Log entry in the existing tracking file; Batch and its executors do not write that file. | +| Shared checkout | Parallel workers do not mutate Git state; required Git-mutating steps run sequentially. | | Failed step | Before another worker launches, the tracking plan contains a changed approach and a 🤖 rationale based on the failure evidence. | | Safety stop | The reason is reported, the stopped action is not retried, and `status` stays `in-progress`. | | Final condition fails | `status` stays `in-progress`; the plan gains unchecked steps addressing the failure. | diff --git a/plugins/aidd-dev/skills/10-batch/SKILL.md b/plugins/aidd-dev/skills/10-batch/SKILL.md deleted file mode 100644 index 67c09a8d4..000000000 --- a/plugins/aidd-dev/skills/10-batch/SKILL.md +++ /dev/null @@ -1,23 +0,0 @@ ---- -name: 10-batch -description: Run independent tasks through bounded parallel executor agents and return per-item evidence. Use when the user says "batch", asks for parallel implementation, or a caller delegates ready independent tasks. Not for dependent tasks, goal planning, verification, or retries. -argument-hint: requirement | framed items ---- - -# Batch - -Execute independent work in parallel, in the caller's context, without adding an agent tier. Accept a standalone request or already-framed items from a goal owner. - -## Actions - -| Action | Does | -| --- | --- | -| batch | Dispatch independent items to leaf executors and return their evidence | - -Before running an action, read its file in `actions/`, not only the table or assets. - -## Transversal rules - -- Run only when the user or owning caller authorizes parallel execution. A leaf executor must never invoke this spawning capability. -- Preserve delegated item boundaries, acceptance criteria, worker settings, and safety rules; do not reframe or replan them. -- Return evidence, not a verification verdict. The owner retains routing, tracking, verification, retries, and the final success decision. diff --git a/plugins/aidd-dev/skills/10-batch/actions/01-batch.md b/plugins/aidd-dev/skills/10-batch/actions/01-batch.md deleted file mode 100644 index 9d4fe4b50..000000000 --- a/plugins/aidd-dev/skills/10-batch/actions/01-batch.md +++ /dev/null @@ -1,37 +0,0 @@ -# 01 - Batch - -Dispatch ready independent items to leaf executors and return one evidence row per item. - -## Input - -A standalone requirement, or delegated items with stable identifiers, step descriptions, acceptance criteria, dependencies, allowed write scopes, relevant context, worker instructions, and caller-selected execution settings (model and reasoning effort when supplied). - -## Output - -One result per item, returned to the caller. `returned` means the executor supplied a result, not that an independent check passed. Include commands, outputs, changed files or other concrete evidence, and any failure or safety-stop reason. Missing results are failures. - -```markdown -| Item | Status | Evidence | -| ---- | ------ | -------- | -``` - -## Process - -1. **Frame.** For delegated items, retain the supplied frames unchanged. For a standalone request, split and refine items inline in the caller's context, using a discovered non-interactive refinement capability when available; otherwise clarify obvious ambiguity inline. Assign identifiers, acceptance criteria, dependencies, and allowed write scopes. Ask for the requirement only when standalone input is empty; return incomplete delegated frames to their owner. -2. **Check readiness.** Require satisfied dependencies and disjoint write scopes and mutable resources. Do not launch dependent, conflicting, or insufficiently scoped items together; return the reason to the caller without guessing or widening scope. Require the host to support the supplied worker settings and inherited safety rules. -3. **Launch.** In the caller's context, spawn one leaf `executor` per item concurrently within the host's capacity. Apply the supplied model and reasoning effort exactly, or inherit the caller's settings when none were supplied. Pass each item's frame, worker instructions, and safety rules unchanged. Mandate execution only within that item and its write scope, concrete evidence on return, and no agent spawning. Explicitly override the executor's per-unit commit policy for this dispatch: no Git mutations (staging, commits, branch changes, or pushes); the owner serializes any authorized Git writes after the batch. Never spawn a Batch controller or modify the owner's tracking file. -4. **Collect.** Wait for every launched item and associate its result with its identifier. Preserve partial results. On a safety stop, stop further dispatch and request any running executors to stop where supported; report the gate without retrying it. Do not retry, replan, certify completion, or launch dependent follow-up work. -5. **Report.** Return one minimal table with `returned`, `failed`, or `stopped` and concrete evidence per item. When invoked by a goal owner, return it to that owner for verification; do not override its progress announcement or final report. Standalone output is the table only. - -## Test - -| Case | Pass | -| --- | --- | -| Standalone request | Items are refined in the caller's context before executor launches; the user receives one result table. | -| Delegated frames | Identifiers, criteria, scopes, worker settings, and safety rules are preserved without a second refinement. | -| Independent items | One leaf executor per item runs concurrently; no extra agent tier or nested spawning is added. | -| Shared checkout | Leaf executors perform no Git mutations; the owner handles authorized Git writes after all workers finish. | -| Dependency or shared writes | Unsafe items are not launched together; their owner receives the reason. | -| Unsupported worker settings | No silently substituted model or reasoning effort is used. | -| Partial failure or missing result | Each item has its own evidence or failure row; no retry or success certification occurs. | -| Safety stop | Further dispatch stops, partial evidence is returned, and the gate is not retried. | diff --git a/plugins/aidd-dev/skills/10-todo/SKILL.md b/plugins/aidd-dev/skills/10-todo/SKILL.md new file mode 100644 index 000000000..a2083c643 --- /dev/null +++ b/plugins/aidd-dev/skills/10-todo/SKILL.md @@ -0,0 +1,17 @@ +--- +name: 10-todo +description: Split the user prompt into independent todos and run one executor agent per todo in parallel, then report a minimal table. Use when the user says "todo" or asks to fan out a multi-part request into parallel implementations. +argument-hint: requirement +--- + +# Todo + +Turn one prompt into N independent todos, implement them in parallel, report a table. + +## Actions + +```markdown +actions/01-todo.md +``` + +Before running an action, read its file in `actions/`, not only the table or assets. diff --git a/plugins/aidd-dev/skills/10-todo/actions/01-todo.md b/plugins/aidd-dev/skills/10-todo/actions/01-todo.md new file mode 100644 index 000000000..222401dc4 --- /dev/null +++ b/plugins/aidd-dev/skills/10-todo/actions/01-todo.md @@ -0,0 +1,31 @@ +# 01 - Todo + +Categorize the user prompt into independent todos, implement each in parallel, report. + +## Input +User's requirement. + +## Output +```markdown +| Category | Launched | Output | +| -------- | -------- | ------ | +``` + +## Process + +1. **Read.** Take `prompt` from the arguments; if empty, ask the user. +2. **Categorize.** Split the prompt into distinct, independent todos (category + task). Inline, no agent. +3. **Launch.** Spawn one `executor` agent per todo, all in parallel (one message, multiple Task calls). Each agent prompt mandates, in order: + ```markdown + 1. Refine the todo first: run a non-interactive refine capability if one is available (discovered at runtime, never a hardcoded plugin name); otherwise restate the todo clearly and resolve obvious ambiguity inline. Never block on the user. + 2. Implement the refined todo. + 3. Return a one-line output summary. + ``` +4. **Report.** Print exactly one table, nothing else. + +## Test + +- Every todo is one row in the table. +- Agents were spawned in a single parallel batch. +- Each agent ran a refine step before implementing. +- No output besides the table. diff --git a/plugins/aidd-orchestrator/skills/01-sdlc/references/03-check.md b/plugins/aidd-orchestrator/skills/01-sdlc/references/03-check.md index ada870226..c871e419f 100644 --- a/plugins/aidd-orchestrator/skills/01-sdlc/references/03-check.md +++ b/plugins/aidd-orchestrator/skills/01-sdlc/references/03-check.md @@ -4,7 +4,7 @@ Use one fresh checker, independent from implementation, to review the candidate against the contract, plan, and validation evidence. After the review clears, the same checker challenges whether the real outcome is trustworthy and serves the user. -Use product or contract findings as the next Frame source. Dispatch independent implementation findings through Batch and keep dependent repairs together in Deliver. Re-enter Check after every new candidate. Open the draft pull request when the checker returns no actionable finding. +Use product or contract findings as the next Frame source. Dispatch independent implementation findings through Todo and keep dependent repairs together in Deliver. Re-enter Check after every new candidate. Open the draft pull request when the checker returns no actionable finding. ```mermaid --- @@ -30,7 +30,7 @@ flowchart TD direction TB Findings["$findings"] Frame["01 Frame"] - Batch["/aidd-dev:10-batch"] + Todo["/aidd-dev:10-todo"] Deliver["02 Deliver"] end @@ -50,9 +50,9 @@ flowchart TD Challenge -- "Return actionable challenge findings." --> Findings Challenge -- "When the outcome is trustworthy, open the draft pull request." --> PullRequest Findings -- "Use product findings as the next Frame source." --> Frame - Findings -- "Repair independent implementation findings in parallel." --> Batch + Findings -- "Repair independent implementation findings in parallel." --> Todo Findings -- "Repair dependent implementation findings together." --> Deliver - Batch --> Deliver + Todo --> Deliver PullRequest --> PullRequestUrl classDef skill fill:#DBEAFE,stroke:#2563EB,color:#1E3A8A,stroke-width:2px @@ -60,7 +60,7 @@ flowchart TD classDef artifact fill:#DCFCE7,stroke:#16A34A,color:#14532D,stroke-width:2px classDef zone fill:#F1F5F9,stroke:#64748B,color:#0F172A,stroke-width:2px - class Review,Challenge,Batch,PullRequest skill + class Review,Challenge,Todo,PullRequest skill class Checker agent class Contract,Plan,CommittedCandidate,ValidationReports,Findings,PullRequestUrl artifact class Frame,Deliver zone diff --git a/scripts/__tests__/goalify-batch-publication.test.js b/scripts/__tests__/goalify-batch-publication.test.js deleted file mode 100644 index 90dcccf02..000000000 --- a/scripts/__tests__/goalify-batch-publication.test.js +++ /dev/null @@ -1,45 +0,0 @@ -const assert = require("node:assert/strict"); -const fs = require("node:fs"); -const path = require("node:path"); -const test = require("node:test"); -const yaml = require("js-yaml"); - -const ROOT = path.resolve(__dirname, "../.."); -const DEV = "plugins/aidd-dev"; -const BATCH = `${DEV}/skills/10-batch`; -const read = (file) => fs.readFileSync(path.join(ROOT, file), "utf8"); - -function obsoleteAddresses(text) { - return [...text.matchAll(/\b10-todo\b/gu)].map((match) => match[0]); -} - -test("the published batch capability resolves to its own router and action", () => { - const manifest = JSON.parse(read(`${DEV}/.claude-plugin/plugin.json`)); - assert.ok(manifest.skills.includes("./skills/10-batch"), "the batch capability is not exported"); - assert.ok(!manifest.skills.includes("./skills/10-todo"), "the obsolete invocation is still exported"); - - const frontmatter = yaml.load(read(`${BATCH}/SKILL.md`).split(/^---\s*$/mu)[1]); - assert.equal(frontmatter.name, "10-batch"); - assert.ok(frontmatter.description); - assert.ok(frontmatter["argument-hint"]); - assert.ok(fs.existsSync(path.join(ROOT, BATCH, "actions/01-batch.md"))); - assert.ok(!fs.existsSync(path.join(ROOT, DEV, "skills/10-todo/SKILL.md")), "no old skill alias ships"); -}); - -test("active documentation and orchestration addresses resolve to batch", () => { - const files = [ - `${DEV}/README.md`, - `${DEV}/CATALOG.md`, - "docs/CATALOG.md", - "docs/ARCHITECTURE.md", - "aidd_docs/README.md", - "plugins/aidd-orchestrator/skills/01-sdlc/references/03-check.md", - ]; - const stale = files.flatMap((file) => obsoleteAddresses(read(file)).map((address) => `${file}: ${address}`)); - assert.deepEqual(stale, [], "a public caller still points to a capability that no longer ships"); -}); - -test("the address guard detects a stale dispatch without rejecting ordinary todo terminology", () => { - assert.deepEqual(obsoleteAddresses("Dispatch /aidd-dev:10-todo."), ["10-todo"]); - assert.deepEqual(obsoleteAddresses("A board's Todo column and a tracked todo list."), []); -}); From 4d7f63d7f863dc5efbe12b07d1f214cec3d8ed1e Mon Sep 17 00:00:00 2001 From: alexsoyes Date: Sat, 10 Oct 2026 14:42:24 +0700 Subject: [PATCH 6/6] fix(aidd-dev): align goalify with skill authoring rules Keep the existing goal loop readable without losing its resume, retry,\nmodel-selection or safety contract. Preserve source checks during an\nunstaged action rename. --- .../2026_10_09_goalify-readability/review.md | 69 ++++++++++++ docs/CATALOG.md | 2 +- plugins/aidd-dev/CATALOG.md | 4 +- plugins/aidd-dev/skills/09-goalify/SKILL.md | 104 ++++++++++++------ .../09-goalify/actions/01-init-tracking.md | 50 ++++++--- .../09-goalify/actions/02-auto-accept.md | 31 +++--- .../09-goalify/actions/03-autonomous-loop.md | 42 ------- .../skills/09-goalify/actions/03-run-loop.md | 86 +++++++++++++++ .../assets/autonomous-loop-worker-prompt.md | 33 +++--- .../skills/09-goalify/assets/plan-template.md | 24 ++-- .../references/autonomous-loop-log-format.md | 5 +- scripts/__tests__/source-stays-text.test.js | 34 +++++- 12 files changed, 343 insertions(+), 141 deletions(-) create mode 100644 aidd_docs/tasks/2026_10/2026_10_09_goalify-readability/review.md delete mode 100644 plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md create mode 100644 plugins/aidd-dev/skills/09-goalify/actions/03-run-loop.md diff --git a/aidd_docs/tasks/2026_10/2026_10_09_goalify-readability/review.md b/aidd_docs/tasks/2026_10/2026_10_09_goalify-readability/review.md new file mode 100644 index 000000000..7a148cbf5 --- /dev/null +++ b/aidd_docs/tasks/2026_10/2026_10_09_goalify-readability/review.md @@ -0,0 +1,69 @@ +# Review: Goalify readability + +- **Verdict**: approve +- **Diff**: `HEAD -> working tree`, Goalify, its catalogue entries and the rename-safe source guard +- **Axes run**: code, functional, relevancy +- **Date**: 2026-10-10 +- **Findings**: 0 critical, 0 warning, 0 minor; six previous findings corrected + +## Phases + +### Phase 1: Preserve the existing contract + +- [x] Three actions retained; no replacement by a prompt generator or new execution mechanism. +- [x] Existing tracking path, frontmatter, section structure and resume paths retained; canonical Log shape is byte-identical to HEAD. +- [x] Powerful orchestration and smallest available workers with maximum supported reasoning retained, including launches and relaunches. +- [x] Failures require evidence analysis and a changed approach before retry; verified successes stay checked. +- [x] Final completion still requires the orchestrator's concrete success command and exit 0. +- [x] Compatible Batch dispatch, sequential fallback, per-item verification and launch announcement retained. +- [x] Payment, destructive-action and scope boundaries retained; checks now precede autonomous confirmations explicitly. +- [x] No changes to Todo, Batch, manifests or unrelated plugin files; both catalogue entries follow the loop action rename. + +### Phase 2: Correct the six reported authoring gaps + +- [x] R1: rename the noun action to `03-run-loop.md`, with one action table row and updated callers/catalogues; no compatibility alias. +- [x] R7: show framing, confirmation, self-fix and iteration back-edges, with distinct prerequisite, success and safety outcomes. +- [x] R12: place the conditional return under the replanning step, preserving final evaluation when no unchecked steps remain. +- [x] R15: present reference facts as individual list items and retain the exact emitted Log format. +- [x] R16: define filling sources, replace task examples and require removal of every scaffold comment and unresolved placeholder. +- [x] R17: keep Log shape in its reference; completion and amendment policy belong to the loop action, with the template citing their owner. + +### Phase 3: Keep validation usable during the rename + +- [x] The source guard scans present tracked and nonignored untracked files; a real temporary Git repository proves the old path disappears and the new path is checked. +- [x] Fixture Git commands and enumeration strip inherited `GIT_*`; a real caller repository remains byte-identical under poisoned Git environment variables. + +## Findings + +| Sev | Kind | Phase | Location | Issue | Fix | +| --- | ---- | ----- | -------- | ----- | --- | +| - | - | - | - | None in the corrected scope | - | + +## Verification + +| Metric | Value | +| --- | --- | +| Verified | 100% (16/16 scoped criteria); instruction comparison plus real guard execution, not a Goalify runtime success rate | +| Files checked | Seven Goalify files, both catalogue entries, source guard; HEAD diff and naming convention | +| Unchecked | None in the six-finding correction scope | +| Unplanned | Source-guard repair was required because the renamed action remains a deleted tracked path until staging | +| Changed tests | `pnpm test:changed`: exit 0, under the Git-hook protection wrapper | +| Full scripts suite | `node scripts/check-tests-leave-git-alone.js -- node --test 'scripts/__tests__/**/*.test.js'`: 444 passed, 0 failed, 0 cancelled, 0 skipped | +| Guard regression | Real unstaged-rename case and repository scan observed failing before the repair; both pass afterward | +| Git isolation | Guard suite passes with inherited `GIT_DIR`, `GIT_WORK_TREE`, `GIT_INDEX_FILE` and `GIT_COMMON_DIR`; all disposable caller `.git` file hashes unchanged | +| Static guards | `git diff --check`, skill argument-hint guard, Markdown links and referenced paths passed | +| Duplication guard limitation | Existing documentation guard does not scan skills; it is not evidence of R17 compliance | +| Live behavior | No Goalify goal loop executed; resume, failure/retry and final success are supported by instruction comparison only | +| Independent review | Fresh checker reviewed seven skill files and catalogue entries, then rechecked its four repair findings on disk: No remaining findings | +| Adjudication | Shared model/worker boundaries remain in the router; loop-only completion/amendment rules have one owner. Isolated workers retain their own safety payload | +| Resolved ambiguities | Confirmed ASCII journey explicitly converts to Mermaid without changing dependencies; nondecorative execution markers remain governed by the loop | +| Codex validator limitation | Installed `quick_validate.py` rejects required `argument-hint`; field retained, YAML/frontmatter and project hint guard validated instead | +| Router | Five distinct terminals, framing/confirmation/self-fix paths, recording before safety stops, and conditional returns represented; no router backlink introduced | +| Tracking setup | Updated action link, ASCII-to-Mermaid conversion and observable framing/confirmation cases | +| Autonomy action | Existing confirmation and safety rules retained; no change in this correction pass | +| Loop action | Verb-led name, subordinate loop return and reference-owned Log format; sequential fallback and Batch boundaries retained | +| Worker prompt | Existing isolated-worker scope, evidence and safety payload retained; no change in this correction pass | +| Tracking template | Filling/removal instructions and inline owner citations for amendments and Log format | +| Log reference | One fact per list item, append-only history and canonical UTC attempt format | +| Catalogues | Updated loop action name and resolvable link; no unrelated rows changed | +| Mutation scope | Scoped corrections and report only; no staging, commit or push | diff --git a/docs/CATALOG.md b/docs/CATALOG.md index 660281985..124d41147 100644 --- a/docs/CATALOG.md +++ b/docs/CATALOG.md @@ -47,7 +47,7 @@ Code transformation: plan, implement, assert, audit, review, test, refactor, deb | `06-test` | Write and iterate tests, validate user journeys in the browser | `01-test`, `02-test-journey` | | `07-refactor` | Improve code without changing behavior across four axes | `01-performance`, `02-security`, `03-cleanup`, `04-architecture` | | `08-debug` | Reproduce and fix bugs with a test-driven workflow | `01-reproduce`, `02-debug`, `03-reflect-issue` | -| `09-goalify` | Autonomous loop that replans and retries until a runnable success condition passes | `01-init-tracking`, `02-auto-accept`, `03-autonomous-loop` | +| `09-goalify` | Autonomous loop that replans and retries until a runnable success condition passes | `01-init-tracking`, `02-auto-accept`, `03-run-loop` | | `10-todo` | Split the prompt into independent todos, run one implementer agent per todo in parallel | `01-todo` | | `11-browser-qa` | Record short reviewer videos for browser-scoped happy and edge cases | `00-prerequisites`, `01-load-scope`, `02-prepare-run`, `03-run-scenarios` | diff --git a/plugins/aidd-dev/CATALOG.md b/plugins/aidd-dev/CATALOG.md index 1147d152d..70049d1a5 100644 --- a/plugins/aidd-dev/CATALOG.md +++ b/plugins/aidd-dev/CATALOG.md @@ -132,11 +132,11 @@ Auto-generated index of skills, agents, references and assets shipped by the `ai |-------|------|---| | `actions` | [01-init-tracking.md](skills/09-goalify/actions/01-init-tracking.md) | - | | `actions` | [02-auto-accept.md](skills/09-goalify/actions/02-auto-accept.md) | - | -| `actions` | [03-autonomous-loop.md](skills/09-goalify/actions/03-autonomous-loop.md) | - | +| `actions` | [03-run-loop.md](skills/09-goalify/actions/03-run-loop.md) | - | | `assets` | [autonomous-loop-worker-prompt.md](skills/09-goalify/assets/autonomous-loop-worker-prompt.md) | - | | `assets` | [plan-template.md](skills/09-goalify/assets/plan-template.md) | - | | `references` | [autonomous-loop-log-format.md](skills/09-goalify/references/autonomous-loop-log-format.md) | - | -| `-` | [SKILL.md](skills/09-goalify/SKILL.md) | `Turn a goal into an autonomous loop that replans and retries until a runnable success condition passes. Use when the user says "goalify", "keep trying until", or wants a goal verified by a command. Not for one-shot tasks or uncheckable goals.` | +| `-` | [SKILL.md](skills/09-goalify/SKILL.md) | `Runs an autonomous goal loop that replans and retries until a runnable success condition passes. Use when the user wants to goalify a task, keep trying until success, or verify a goal by command. Not for one shot tasks or uncheckable goals.` | #### `skills/10-todo` diff --git a/plugins/aidd-dev/skills/09-goalify/SKILL.md b/plugins/aidd-dev/skills/09-goalify/SKILL.md index c6d13ff23..74c9967e9 100644 --- a/plugins/aidd-dev/skills/09-goalify/SKILL.md +++ b/plugins/aidd-dev/skills/09-goalify/SKILL.md @@ -1,58 +1,90 @@ --- name: 09-goalify -description: Turn a goal into an autonomous loop that replans and retries until a runnable success condition passes. Use when the user says "goalify", "keep trying until", or wants a goal verified by a command. Not for one-shot tasks or uncheckable goals. +description: Runs an autonomous goal loop that replans and retries until a runnable success condition passes. Use when the user wants to goalify a task, keep trying until success, or verify a goal by command. Not for one shot tasks or uncheckable goals. argument-hint: task | command --- -# Skill: goalify - -Frame a checkable goal interactively, then run unattended until its success condition passes or a safety stop is reached. +# Goalify ```mermaid flowchart TD - Start[Setup or resume] --> Init[init-tracking] - Init -->|ready| Loop[autonomous-loop under auto-accept] + Setup[New task] --> Init{init-tracking: existing task status?} + Resume[Resume task] --> Init Init -->|already implemented| Done[Complete] - Init -->|unresolved prerequisite| Stop[Stop and report] - Loop --> Select{Ready independent steps?} - Select -->|parallel capability available| Batch[Delegate parallel execution] + Init -->|pending or in-progress| Loop[run-loop under auto-accept] + Init -->|no file| Collect[Collect inputs and research] + Collect --> Goal{Unambiguous goal?} + Goal -->|no| Reformulate[Ask for reformulation] + Reformulate --> Goal + Goal -->|yes| Condition{Runnable success command?} + Condition -->|no| Define[Request a checkable command] + Define --> Condition + Condition -->|yes| Preflight[Check prerequisites and collect user-only inputs] + Preflight --> Ready{Unresolved hard prerequisite?} + Ready -->|yes| Blocked[Stop for unresolved prerequisite] + Ready -->|no| Map[Present ASCII journey] + Map --> Confirm{User confirms journey?} + Confirm -->|no| Map + Confirm -->|yes| Directory{Tracking directory exists?} + Directory -->|no| CreateDirectory[Create tracking directory] + Directory -->|yes| Create[Fill tracking template and convert journey to Mermaid] + CreateDirectory --> Create + Create --> Loop + Loop --> Remaining{Unchecked steps?} + Remaining -->|no| Success{success_condition passes?} + Remaining -->|yes| Select{Ready independent steps and compatible Batch?} + Select -->|yes| Batch[Delegate parallel execution] Select -->|otherwise| Worker[Execute next step] - Batch --> Verify[Verify each result] - Worker --> Verify - Verify -->|safety stop| Stop - Verify -->|failure| Replan[Analyze and replan] + Batch --> Autonomy{auto-accept: action safe and in scope?} + Worker --> Autonomy + Autonomy -->|outside task| Skip[Skip unrelated action] + Skip --> Autonomy + Autonomy -->|payment| Payment[Report payment stop] + Autonomy -->|destructive| Destructive[Report destructive stop] + Autonomy -->|yes| Act[Act autonomously] + Act --> ActionResult{Assigned work outcome?} + ActionResult -->|self-fixable failure| Fix[Choose an in-scope fix] + Fix --> Autonomy + ActionResult -->|more actions| Autonomy + ActionResult -->|finished or other failure| Verify[Verify each result] + Payment --> Verify + Destructive --> Verify + Verify --> Record[Record each attempt and preserve partial evidence] + Record --> Safety{Reported safety outcome?} + Safety -->|payment| StopPayment[Stopped payment] + Safety -->|destructive| StopDestructive[Stopped destructive action] + Safety -->|out-of-scope| StopScope[Stopped out-of-scope] + Safety -->|none| Failed{Failed or missing result?} + Failed -->|yes| Replan[Analyze failure and amend the plan] Replan -->|retry| Loop - Verify -->|more steps| Loop - Verify -->|all steps checked| Success{success_condition passes?} + Failed -->|no| Next{Unchecked steps remain?} + Next -->|yes| Loop + Next -->|no| Success Success -->|no| Replan Success -->|yes| Done ``` ## Actions +Read [tracking setup](actions/01-init-tracking.md) first. + | Action | Does | | --- | --- | -| [init-tracking](actions/01-init-tracking.md) | frame the goal, create or resume tracking, launch the loop | -| [auto-accept](actions/02-auto-accept.md) | decide and act within the task's safety limits | -| [autonomous-loop](actions/03-autonomous-loop.md) | dispatch workers, verify evidence, replan failures, check completion | - -Run `init-tracking` interactively; it launches or resumes `autonomous-loop` under `auto-accept`. -Before running an action, read its file in `actions/`, not only the table or assets. +| init-tracking | frame the goal, create or resume tracking, launch the loop | +| auto-accept | decide and act within the task's safety limits | +| run-loop | dispatch workers, verify evidence, replan failures, check completion | ## Transversal rules -- Single source of truth: all task state lives in `aidd_docs/tasks/.md` and nowhere else. -- No repeated failures: never retry a failed approach without a meaningful change. -- Honesty over escape: never set `status: implemented` until the success condition genuinely passes. -- Auto-accept: follow the action's rules within the original task; stop on payment or destructive actions. -- The loop dispatches one leaf worker per step, directly or through a discovered parallel-execution capability in its own context. It retains the plan, safety decisions, per-item verification, and retries; it never does the work itself or adds a controller agent. -- Model policy: use a powerful available model for framing, planning, verification, and replanning, including every orchestrator launch or resume. At every worker launch or relaunch, use the smallest available model with its highest supported reasoning effort. Workers execute only their assigned step and return evidence. - -## Assets - -- `assets/plan-template.md`: the tracking file format (frontmatter, phases, acceptance criteria, Log). -- `assets/autonomous-loop-worker-prompt.md`: the prompt the loop spawns each per-step worker with. - -## References - -- `references/autonomous-loop-log-format.md`: the Log entry format the loop appends per attempt. +- Keep all task state in `aidd_docs/tasks/.md` and nowhere else. +- Never retry a failed approach without a meaningful change. +- Apply the [autonomy rules](actions/02-auto-accept.md) throughout unattended execution. +- Delegate execution to one leaf worker per step, directly or through a discovered parallel capability. + - Keep parallel dispatch in the orchestrator's context, without another controller agent. + - Retain planning, safety decisions, per-item verification, and retries. + - Never perform the workers' execution yourself. + - Workers execute only their assigned step and return evidence. +- Select models by responsibility. + - Use a powerful available model for framing, planning, verification, and replanning. + - Apply that selection to every orchestrator launch or resume. + - At every worker launch or relaunch, use the smallest available model with its highest supported reasoning effort. diff --git a/plugins/aidd-dev/skills/09-goalify/actions/01-init-tracking.md b/plugins/aidd-dev/skills/09-goalify/actions/01-init-tracking.md index 517ce2cb7..e5b8f9238 100644 --- a/plugins/aidd-dev/skills/09-goalify/actions/01-init-tracking.md +++ b/plugins/aidd-dev/skills/09-goalify/actions/01-init-tracking.md @@ -1,30 +1,50 @@ # 01 - Init tracking -Validate prerequisites, build a journey map, create the tracking file, and hand off to the autonomous loop. The last interactive step before the loop runs unattended. +Frame a checkable goal and launch its unattended loop. ## Input -The task name (required), an optional free-form description, a runnable success condition that exits 0 on success (required), and optional rules. +- Required: task name and a runnable success condition that exits 0 on success. +- Optional: free-form description and rules. ## Output -The tracking file at `aidd_docs/tasks/.md`, marked created or resumed, with any pre-flight blocker halting before the spawn. +Task tracking created or resumed at `aidd_docs/tasks/.md`. +Only ready tasks launch a loop. +Completed tasks and unresolved pre-flight blockers launch nothing. ## Process -1. **Resume.** Apply the router's model policy, then check `aidd_docs/tasks/` for a file matching the task name and read its frontmatter `status`. - - `pending` or `in-progress`: report the status (iteration, steps remaining), then skip to Spawn to resume. +1. **Resume.** Inspect any matching task file under the router's model policy. + - Match the task name in `aidd_docs/tasks/` and read its frontmatter `status`. + - `pending` or `in-progress`: report iteration and remaining steps. Continue directly to loop launch without reframing. - `implemented`: report "Task already completed" and stop. - - No file: continue to Collect. + - No file: collect the new task's inputs. 2. **Collect.** Gather the task name, description, success condition, and rules from the user. -3. **Research.** Before planning steps, read the relevant documentation (README, official guides) and identify the recommended method. Do not default to what you already know. -4. **Goal.** Ask "could I execute this with zero ambiguity?" When no, ask the user to reformulate. "Make the code better" is rejected ("what metric?"); "all tests pass after `npm test`" is accepted. -5. **Condition.** It must be a runnable command. `npm test exits 0` is valid; "the code is clean" is invalid and is pushed back to `eslint . exits 0`. -6. **Pre-flight.** For each step, list tools, secrets, API access, data, and permissions. Mark `[✓]` already satisfied, `[~]` soft (the agent self-serves), `[!]` hard (only the user can provide it). Collect every `[!]` now; when any stays unresolved, stop before the next step. -7. **Map.** Project the whole path as an ASCII map of steps, dependencies, tools, and blockers. Ask the user to confirm and iterate until they do. -8. **Scaffold.** Load [plan-template.md](../assets/plan-template.md), creating `aidd_docs/tasks/` when missing. -9. **Create.** Write `aidd_docs/tasks/.md` from the template. Fill the frontmatter (`objective`, `success_condition`, `iteration: 0`, `status: pending`), the phases with their tasks and acceptance criteria, and the journey map. -10. **Spawn.** Apply the router's model policy to launch or resume the orchestrator with [03-autonomous-loop.md](./03-autonomous-loop.md) and `` filled in. +3. **Research.** Read relevant documentation before planning steps. + - Use the README and official guides to identify the recommended method. + - Do not default to prior knowledge. +4. **Goal.** Check whether the goal can be executed without ambiguity. + - Otherwise, ask the user to reformulate. + - Reject "make the code better" until a metric is provided. + - Accept "all tests pass after `npm test`". +5. **Condition.** Require a runnable success command. + - Accept `npm test exits 0`. + - Replace "the code is clean" with a check such as `eslint . exits 0`. +6. **Pre-flight.** Check each step's tools, secrets, API access, data, and permissions. + - Mark satisfied prerequisites `[✓]`, self-service ones `[~]`, and user-only ones `[!]`. + - Collect every `[!]` now. Stop if any remains unresolved. +7. **Map.** Present the whole journey as an ASCII map. + - Include steps, dependencies, tools, and blockers. + - Iterate until the user confirms it. +8. **Scaffold.** Load the [tracking template](../assets/plan-template.md). + - Create `aidd_docs/tasks/` when missing. +9. **Create.** Write the task file from `plan-template.md`. + - Fill `objective`, `success_condition`, `iteration: 0`, and `status: pending`. + - Add phases, tasks, acceptance criteria, and the journey map. + - Convert the confirmed ASCII map into the template's Mermaid journey, preserving steps and dependencies. +10. **Spawn.** Launch or resume the orchestrator under the router's model policy. + - Use the [loop instructions](./03-run-loop.md) with the task name filled in. ## Test @@ -33,5 +53,7 @@ The tracking file at `aidd_docs/tasks/.md`, marked created or resumed | New task | The tracking file exists at `aidd_docs/tasks/.md` with `status: pending`, a runnable `success_condition`, and a journey map. | | Pending or in-progress task | The existing file is retained and the orchestrator resumes from its recorded state. | | Implemented task | "Task already completed" is reported and no agent launches. | +| Ambiguous goal or non-runnable condition | The user supplies a reformulated goal or runnable command before prerequisites and planning proceed. | +| Unconfirmed journey | The map is revised until confirmed; the saved Mermaid journey preserves its steps and dependencies. | | Unresolved hard prerequisite | No orchestrator launches until every `[!]` is resolved. | | Setup or resume | Framing and orchestrator model selections follow the router's model policy. | diff --git a/plugins/aidd-dev/skills/09-goalify/actions/02-auto-accept.md b/plugins/aidd-dev/skills/09-goalify/actions/02-auto-accept.md index ccdcb53db..5cfddadc8 100644 --- a/plugins/aidd-dev/skills/09-goalify/actions/02-auto-accept.md +++ b/plugins/aidd-dev/skills/09-goalify/actions/02-auto-accept.md @@ -1,6 +1,6 @@ # 02 - Auto-accept -Operate autonomously: do not ask for confirmation, decide and act, and stop only on money or destructive actions. +Handle confirmations autonomously within the task's safety limits. ## Input @@ -8,21 +8,26 @@ The task to handle end-to-end, a free-form description. ## Output -An exit status, completed or one of stopped-payment, stopped-destructive, or stopped-out-of-scope, with the actions taken and a one-sentence reason when stopped. +- Exit status: `completed`, `stopped-payment`, `stopped-destructive`, or `stopped-out-of-scope`. +- Actions taken, with a one-sentence reason for any stopped status. ## Process -Apply these rules in order to every prompt, dialog, checkbox, Y/n, license screen, cookie banner, or confirmation met while handling the task. - -1. **Accept.** Accept everything by default, acknowledge, and move on. -2. **Default.** When an installer offers options, pick the recommended or standard one. -3. **Self-fix.** When something fails (missing dependency, wrong version, config error), fix it and retry. Do not ask. -4. **Money.** Stop and report when an action involves payment, subscription, or an upgrade to a paid tier. -5. **Destructive.** Stop and report when an action deletes data, drops a database, removes files recursively, force-pushes, resets git history, or overwrites uncommitted work. -6. **Scope.** Skip anything leading outside the original task (unrelated tools, external signups, rabbit holes). Do only what the user asked. +1. **Check.** Apply the task's gates before handling each action or confirmation. + - Stop and report payments, subscriptions, or upgrades to paid tiers. + - Stop and report destructive actions: deleting data, dropping databases, recursive removal, force-pushes, history resets, or overwriting uncommitted work. + - Skip unrelated tools, external signups, and rabbit holes outside the original task. +2. **Act.** Handle in-scope confirmations without asking the user. + - Accept and acknowledge prompts, dialogs, checkboxes, Y/n choices, licenses, cookies, and confirmations by default. + - Choose recommended or standard installer options. + - Fix failures such as missing dependencies, wrong versions, or configuration errors, then retry. +3. **Report.** Return the actual exit status and actions taken. + - Include a one-sentence reason when stopped. ## Test -- The status matches the actual exit path. -- `completed` appears only when the task ran end-to-end with no money or destructive gate hit. -- Each stopped status carries a non-empty reason. +| Case | Pass | +| --- | --- | +| Exit | The status matches the actual exit path. | +| Completed task | `completed` appears only after end-to-end execution without a money or destructive gate. | +| Stopped task | Every stopped status includes a non-empty reason. | diff --git a/plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md b/plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md deleted file mode 100644 index 39296de35..000000000 --- a/plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md +++ /dev/null @@ -1,42 +0,0 @@ -# 03 - Autonomous loop - -Orchestrate the loop: dispatch each unchecked step, verify the result, and replan failures before retrying. One attempt is one worker and one log entry; the orchestrator never does the work itself. - -## Input - -The tracking file `aidd_docs/tasks/.md` produced by `01-init-tracking`. The loop runs with no human interaction, reading and writing through that file. - -## Output - -The success condition verified and the plan's `status` set to `implemented`, with every step checked and one Log entry per attempt. Or a safety stop with its reason reported. - -## Process - -1. **Read.** Apply the router's model policy, then read the entire file: frontmatter, journey map, steps, and full Log. -2. **Mark.** Increment `iteration` in the frontmatter, setting `status: in-progress` when still `pending`. -3. **Learn.** Read the Log to learn from prior attempts. -4. **Select.** Find the next unchecked step, or a set of ready independent steps whose prerequisites are already verified. Respect the recorded phases and journey-map dependencies. Frame each item with its existing step identifier (or checkbox position), description, acceptance criteria, dependencies, allowed write scope, and relevant context. Unknown dependencies, overlapping write scopes, or shared mutable resources require sequential execution. If every step is checked, proceed to Evaluate. -5. **Dispatch.** Apply the router's model policy to every worker launch or relaunch, using [autonomous-loop-worker-prompt.md](../assets/autonomous-loop-worker-prompt.md) with the item and context (objective, rules, prior Log entries). For two or more ready independent items, discover the parallel-execution capability (Batch) by description. Use it only if it accepts pre-framed items, concrete worker model/reasoning settings and safety instructions, spawns leaf executors in the caller's context, and returns per-item evidence without owning tracking or retries. Invoke it in this orchestrator's context, never as another agent. Supply the frames, resolved worker settings, and worker prompt. For parallel workers, explicitly override any per-unit commit policy: no Git mutations during the batch; dispatch required Git-mutating steps separately, sequentially. Immediately before the actual parallel launch, announce: `Batch: executing A, B and C in parallel. Goalify will verify each result.` Substitute the selected item identifiers; do not announce a batch for a rejected or sequential dispatch. If no compatible capability is available, spawn a single worker for the next step instead. -6. **Verify.** Collect results by item identifier. Treat missing or failed results as failures. Verify each returned result concretely by running a check command, reading a file, or testing the output; never trust a worker or Batch claim alone. Dependent steps remain blocked until their prerequisites pass this verification. -7. **Record.** For each attempted item, tick `[x]` only on verified success and append one Log entry per [autonomous-loop-log-format.md](../references/autonomous-loop-log-format.md); leave failed or missing results unchecked. Only the orchestrator writes the tracking file. If any worker stopped for payment, a destructive action, or an out-of-scope request, preserve the batch's partial evidence, stop further dispatch, request running workers to stop where supported, report the reason, and stop; never retry the stopped action. -8. **Replan.** After a failure, analyze that item's worker and verification evidence and amend its tracking step with a changed approach before retrying. Prefix each amendment with 🤖 and a brief rationale. Keep the original objective, success condition, and rules; pass the updated step and failure context to the next worker. Retain verified successes and never relaunch them merely because another batch item failed. -9. **Loop.** Move to the next unchecked step and repeat from Read. -10. **Evaluate.** Once every step is checked, run the `success_condition` command yourself. On exit 0, set `status: implemented` and stop. On failure, analyze the evidence and amend the plan as in Replan, adding unchecked steps for the root cause, then continue the loop. - -## Test - -| Case | Pass | -| --- | --- | -| Step attempt | Exactly one Log entry records the worker's attempt and the orchestrator's verification. | -| Verified step | The checked step has a `= ✓` entry citing a concrete command, file, or output. | -| Model selection | Orchestrator verification/replanning and every worker launch/relaunch follow the router's model policy. | -| Independent steps | Ready disjoint items are delegated in the orchestrator's context with its resolved worker settings and safety rules; the launch announcement names those items. | -| Unavailable capability or uncertain independence | One worker launches sequentially, without a batch announcement. | -| Dependency | A downstream step cannot launch before each prerequisite is independently verified. | -| Partial batch failure | Successful items stay checked; failed or missing items stay unchecked and are replanned before selective relaunch. | -| Batch tracking | Each attempted item has one Log entry in the existing tracking file; Batch and its executors do not write that file. | -| Shared checkout | Parallel workers do not mutate Git state; required Git-mutating steps run sequentially. | -| Failed step | Before another worker launches, the tracking plan contains a changed approach and a 🤖 rationale based on the failure evidence. | -| Safety stop | The reason is reported, the stopped action is not retried, and `status` stays `in-progress`. | -| Final condition fails | `status` stays `in-progress`; the plan gains unchecked steps addressing the failure. | -| Final condition passes | `status: implemented` is set only after the orchestrator re-runs `success_condition` and observes exit 0. | diff --git a/plugins/aidd-dev/skills/09-goalify/actions/03-run-loop.md b/plugins/aidd-dev/skills/09-goalify/actions/03-run-loop.md new file mode 100644 index 000000000..49da6842d --- /dev/null +++ b/plugins/aidd-dev/skills/09-goalify/actions/03-run-loop.md @@ -0,0 +1,86 @@ +# 03 - Run loop + +Dispatch unchecked steps, verify each result, and replan failures before retrying. + +## Input + +The task file `aidd_docs/tasks/.md` created during tracking setup. +The loop runs unattended, reading and writing through that file. + +## Output + +- Success: final condition verified, every step checked, and `status: implemented`. +- Safety stop: reason reported and `status: in-progress` retained. +- Both paths retain one Log entry per attempted step. + +## Process + +1. **Read.** Read the entire task file under the router's model policy. + - Include frontmatter, journey map, steps, and full Log. +2. **Mark.** Increment `iteration` in the frontmatter. + - Change `pending` to `in-progress` when needed. +3. **Learn.** Review prior attempts in the Log. +4. **Select.** Select the next unchecked step or ready independent steps. + - Require verified prerequisites and respect recorded phases and journey-map dependencies. + - Frame each item with its identifier or checkbox position, description, acceptance criteria, dependencies, write scope, and context. + - Use sequential execution for unknown dependencies, overlapping writes, or shared mutable resources. + - If every step is checked, proceed to the final success check. +5. **Dispatch.** Dispatch framed work under the router's model policy. + - Fill the [worker prompt](../assets/autonomous-loop-worker-prompt.md) with the item, objective, rules, and relevant prior Log entries. + - For two or more ready independent items, discover Batch by description. + - Require support for pre-framed items, concrete worker model and reasoning settings, and safety instructions. + - Require caller-context leaf executors and per-item evidence, without Batch owning tracking or retries. + - Invoke it in this orchestrator's context, never as another agent. + - Supply frames, resolved settings, and the filled worker prompt. + - Forbid worker Git mutations during the batch, overriding any per-unit commit policy. + - Dispatch required Git-mutating steps separately, sequentially. + - Announce immediately before the actual parallel launch, substituting the selected identifiers: + + ```text + Batch: executing A, B and C in parallel. Goalify will verify each result. + ``` + + - If fewer than two items are ready, or Batch is unavailable or incompatible, spawn one worker for the next step. + - Do not announce a batch for rejected or sequential dispatches. +6. **Verify.** Verify each returned result independently. + - Collect results by item identifier. Treat missing or failed results as failures. + - Run a check command, read a file, or test the output. + - Never trust a worker or Batch claim alone. + - Keep dependent steps blocked until their prerequisites pass verification. +7. **Record.** Record each attempted item's outcome in the tracking file. + - Only the orchestrator writes this state. + - Tick `[x]` only on verified success, leaving failed or missing results unchecked. + - Append each outcome using the [Log format](../references/autonomous-loop-log-format.md). + - On a payment, destructive, or out-of-scope stop: + - Preserve partial batch evidence and stop further dispatch. + - Request running workers to stop where supported. + - Report the reason and stop. Never retry the stopped action. +8. **Replan.** Replan each failed item before retrying it. + - Analyze worker and verification evidence. + - Amend the step with a changed approach, prefixed with 🤖 and a brief rationale. + - Keep the original objective, success condition, and rules. + - Pass the updated step and failure context to the next worker. + - Retain verified successes. Never relaunch them because another batch item failed. + - If unchecked steps remain, return to reading the tracking file. +9. **Evaluate.** Run `success_condition` yourself once every step is checked. + - Exit 0: set `status: implemented` and stop. + - Failure: keep `in-progress` and apply the failure-replanning process to add unchecked root-cause steps. + - Continue the loop after replanning. + +## Test + +| Case | Pass | +| --- | --- | +| Step attempt | Exactly one Log entry records the worker's attempt and the orchestrator's verification. | +| Verified step | The checked step has a `= ✓` entry citing a concrete command, file, or output. | +| Model selection | Orchestrator verification/replanning and every worker launch/relaunch follow the router's model policy. | +| Independent steps | Ready disjoint items are delegated in the orchestrator's context with its resolved worker settings and safety rules; the launch announcement names those items. | +| Unavailable capability or uncertain independence | One worker launches sequentially, without a batch announcement. | +| Dependency | A downstream step cannot launch before each prerequisite is independently verified. | +| Partial batch failure | Successful items stay checked; failed or missing items stay unchecked and are replanned before selective relaunch. | +| Batch tracking | Each attempted item has one Log entry in the existing tracking file; Batch and its executors do not write that file. | +| Shared checkout | Parallel workers do not mutate Git state; required Git-mutating steps run sequentially. | +| Failed step | Before another worker launches, the tracking plan contains a changed approach and a 🤖 rationale based on the failure evidence. | +| Safety stop | The reason is reported, the stopped action is not retried, and `status` stays `in-progress`. | +| Final condition fails | `status` stays `in-progress`; the plan gains unchecked steps addressing the failure. | +| Final condition passes | `status: implemented` is set only after the orchestrator re-runs `success_condition` and observes exit 0. | diff --git a/plugins/aidd-dev/skills/09-goalify/assets/autonomous-loop-worker-prompt.md b/plugins/aidd-dev/skills/09-goalify/assets/autonomous-loop-worker-prompt.md index f2edd5545..a6331cf47 100644 --- a/plugins/aidd-dev/skills/09-goalify/assets/autonomous-loop-worker-prompt.md +++ b/plugins/aidd-dev/skills/09-goalify/assets/autonomous-loop-worker-prompt.md @@ -1,26 +1,31 @@ - + -Execute this step. Auto-accept everything, act as the user, and make every -decision yourself (approve prompts, generate keys, install tools, click -buttons). Do not ask for permission. Just do it. +Execute this step autonomously within its safety limits. -Worker policy: execute only the assigned step within its allowed write scope and -return concrete evidence. Never spawn agents or edit the orchestrator's tracking -file. Other workers may be active; preserve their changes. The orchestrator -retains reflection, framing, and replanning. +- Accept in-scope confirmations and make decisions without asking permission. +- Act as the user: approve prompts, generate keys, install tools, and click buttons. +- Execute only the assigned step within its allowed write scope. +- Return concrete evidence. +- Never spawn agents. +- Never edit the orchestrator's tracking file. +- Preserve other workers' changes. +- Leave reflection, framing, and replanning to the orchestrator. -Signing in via an existing account (Google Sign-in, GitHub OAuth, SSO) is NOT -account creation; it uses the user's active browser session. Do it. +Use the user's active browser session to sign in through existing Google, GitHub OAuth, or SSO accounts. +This is sign-in, not account creation. -Stop and report instead, without proceeding, when an action would cost money (a -payment, subscription, or paid upgrade) or is destructive (deletes data, drops a -database, force-pushes, resets git history, removes files recursively, or -overwrites uncommitted work). Stay inside the task; skip unrelated signups. +Stop and report before proceeding with: + +- Payments, subscriptions, or paid upgrades. +- Destructive actions: deleting data, dropping databases, recursive file removal, force-pushes, history resets, or overwriting uncommitted work. + +Stay inside the task and skip unrelated signups. STEP: CONTEXT: Report: + - What you did, specifically: commands, files, URLs. - The concrete result: paste output, a screenshot, or evidence. - Whether you stopped at a money or destructive gate, and which. diff --git a/plugins/aidd-dev/skills/09-goalify/assets/plan-template.md b/plugins/aidd-dev/skills/09-goalify/assets/plan-template.md index dcf8aa3ee..217b55bbe 100644 --- a/plugins/aidd-dev/skills/09-goalify/assets/plan-template.md +++ b/plugins/aidd-dev/skills/09-goalify/assets/plan-template.md @@ -8,15 +8,17 @@ status: pending # Instruction: {title} @@ -89,15 +91,11 @@ flowchart TD ## Amendments - + ## Log - - - - - + ## Validation flow demonstration diff --git a/plugins/aidd-dev/skills/09-goalify/references/autonomous-loop-log-format.md b/plugins/aidd-dev/skills/09-goalify/references/autonomous-loop-log-format.md index 86e36a203..7504eb7d5 100644 --- a/plugins/aidd-dev/skills/09-goalify/references/autonomous-loop-log-format.md +++ b/plugins/aidd-dev/skills/09-goalify/references/autonomous-loop-log-format.md @@ -1,6 +1,8 @@ # Autonomous loop: Log entry format -The autonomous loop appends one entry to the tracking file's Log per step attempt, in this exact shape: +- Append one entry to the tracking file's Log per step attempt. +- Never rewrite history. +- Use this exact shape: ```text ### # - @@ -10,6 +12,7 @@ The autonomous loop appends one entry to the tracking file's Log per step attemp ``` - `### #` numbers the attempt. +- `` records the attempt's UTC time as `YYYY-MM-DDTHH:MM:SSZ`. - `>` records the worker's attempt. - `=` records the orchestrator's own verification (a command run or a file read), not the worker's claim. - `->` records the decision: the next step, or `RETRY` with the reason. diff --git a/scripts/__tests__/source-stays-text.test.js b/scripts/__tests__/source-stays-text.test.js index 43905af2c..7b769d80b 100644 --- a/scripts/__tests__/source-stays-text.test.js +++ b/scripts/__tests__/source-stays-text.test.js @@ -1,10 +1,12 @@ const assert = require("node:assert/strict"); const { execFileSync } = require("node:child_process"); const fs = require("node:fs"); +const os = require("node:os"); const path = require("node:path"); const { describe, it } = require("node:test"); const ROOT = path.resolve(__dirname, "../.."); +const GIT_ENV = Object.fromEntries(Object.entries(process.env).filter(([key]) => !key.startsWith("GIT_"))); /** * A source file must stay readable by the tools people actually use on it. A raw NUL byte - @@ -26,18 +28,40 @@ const TEXT_EXTENSIONS = new Set([ ".yaml", ]); -function trackedTextFiles() { - const listed = execFileSync("git", ["ls-files", "-z"], { cwd: ROOT, encoding: "buffer" }); +function workingTreeTextFiles(root = ROOT) { + const listed = execFileSync("git", ["ls-files", "-z", "--cached", "--others", "--exclude-standard"], { + cwd: root, + env: GIT_ENV, + encoding: "buffer", + }); return listed .toString("utf8") .split("\0") .filter(Boolean) - .filter((rel) => TEXT_EXTENSIONS.has(path.extname(rel))); + .filter((rel) => TEXT_EXTENSIONS.has(path.extname(rel))) + .filter((rel) => fs.existsSync(path.join(root, rel))); } -describe("every tracked source file stays greppable", () => { +describe("every source file in the working tree stays greppable", () => { + it("checks new files without trying to read deleted paths during a rename", () => { + const fixture = fs.mkdtempSync(path.join(os.tmpdir(), "aidd-source-stays-text-")); + try { + execFileSync("git", ["init", "--quiet"], { cwd: fixture, env: GIT_ENV }); + fs.writeFileSync(path.join(fixture, "unchanged.md"), "# Existing source\n"); + fs.writeFileSync(path.join(fixture, "old-name.md"), "# Renamed source\n"); + execFileSync("git", ["add", "unchanged.md", "old-name.md"], { cwd: fixture, env: GIT_ENV }); + fs.renameSync(path.join(fixture, "old-name.md"), path.join(fixture, "new-name.md")); + fs.writeFileSync(path.join(fixture, ".gitignore"), "ignored.md\n"); + fs.writeFileSync(path.join(fixture, "ignored.md"), "# Ignored file\n"); + + assert.deepEqual(workingTreeTextFiles(fixture).sort(), ["new-name.md", "unchanged.md"]); + } finally { + fs.rmSync(fixture, { recursive: true, force: true }); + } + }); + it("carries no raw NUL byte, so no search over it can fail in silence", () => { - const files = trackedTextFiles(); + const files = workingTreeTextFiles(); // A walk that found nothing would pass this test while checking nothing at all. assert.ok(files.length > 500, `expected the repository's source files, found ${files.length}`);