diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 1e92608d2..c67764ff3 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -19,7 +19,7 @@ { "name": "aidd-dev", "source": "./plugins/aidd-dev", - "description": "Code transformation: plan, implement, assert, audit, review, test, refactor, debug, for-sure. Hosts engineering agents.", + "description": "Code transformation: plan, implement, assert, audit, review, test, refactor, debug, goalify. Hosts engineering agents.", "strict": true, "metadata": { "recommended": true diff --git a/aidd_docs/README.md b/aidd_docs/README.md index 0d6bb246a..1c7313c46 100644 --- a/aidd_docs/README.md +++ b/aidd_docs/README.md @@ -40,7 +40,7 @@ Skills are grouped into plugins by domain. Install only the plugins you need. | aidd-context | Bootstrap, project init, generation of context artifacts (skills, agents, rules, commands, hooks, plugins, marketplaces), mermaid diagrams, learn, discovery | `02-project-memory`, `03-context-generate`, `09-mermaid` | | aidd-refine | Meta-cognition: brainstorm, challenge prior work, blind-spot scan, fact-check | `01-brainstorm`, `02-challenge`, `03-shadow-areas` | | aidd-pm | Product management: backlog artifacts, refinement, Product Briefs, PRD, spec | `02-user-stories`, `05-spike`, `07-epic`, `09-defect`, `10-task` | -| aidd-dev | Code transformation: plan, implement, assert, audit, review, test, refactor, debug, for-sure | `01-plan`, `02-implement`, `05-review`, `06-test` | +| aidd-dev | Code transformation: plan, implement, assert, audit, review, test, refactor, debug, goalify | `01-plan`, `02-implement`, `05-review`, `06-test` | | aidd-vcs | VCS workflows: commit, pull/merge request, release tag, issue creation | `01-commit`, `02-pull-request`, `04-issue-create` | | aidd-orchestrator | Synchronous SDLC, async issue-to-PR automation, and product backlog | `00-async-dev`, `01-sdlc`, `02-backlog` | @@ -104,7 +104,7 @@ AIDD is delivered as a plugin marketplace. Pick what you need; do not install ev | ------------ | ------------------------------------------------------------------------------------------------------------------- | | aidd-context | 00-onboard, 01-bootstrap, 02-project-memory, 03-context-generate, 09-mermaid, 10-learn, 11-explore | | aidd-refine | 01-brainstorm, 02-challenge, 03-shadow-areas, 04-fact-check | -| aidd-dev | 01-plan, 02-implement, 03-assert, 04-audit, 05-review, 06-test, 07-refactor, 08-debug, 09-for-sure, 10-todo | +| aidd-dev | 01-plan, 02-implement, 03-assert, 04-audit, 05-review, 06-test, 07-refactor, 08-debug, 09-goalify, 10-todo | | aidd-orchestrator | 00-async-dev, 01-sdlc | | aidd-vcs | 01-commit, 02-pull-request, 03-release-tag, 04-issue-create | | aidd-pm | 01-ticket-info, 02-user-stories, 03-prd, 04-spec, 05-spike, 06-product-brief, 07-epic, 08-three-amigos, 09-defect, 10-task | diff --git a/docs/CATALOG.md b/docs/CATALOG.md index 66890a81c..660281985 100644 --- a/docs/CATALOG.md +++ b/docs/CATALOG.md @@ -35,7 +35,7 @@ Bootstrap, project init, context-artifact generation, diagrams, learning, and ex ## 💻 aidd-dev -Code transformation: plan, implement, assert, audit, review, test, refactor, debug, for-sure, todo. Standalone Browser QA records short web evidence. +Code transformation: plan, implement, assert, audit, review, test, refactor, debug, goalify, todo. Standalone Browser QA records short web evidence. | Skill | Role | Actions | | --------------- | -------------------------------------------------------------------------- | ------------------------------------------------------------------------------- | @@ -47,7 +47,7 @@ Code transformation: plan, implement, assert, audit, review, test, refactor, deb | `06-test` | Write and iterate tests, validate user journeys in the browser | `01-test`, `02-test-journey` | | `07-refactor` | Improve code without changing behavior across four axes | `01-performance`, `02-security`, `03-cleanup`, `04-architecture` | | `08-debug` | Reproduce and fix bugs with a test-driven workflow | `01-reproduce`, `02-debug`, `03-reflect-issue` | -| `09-for-sure` | Iterative loop that retries until a success condition is met | `01-init-tracking`, `02-auto-accept`, `03-autonomous-loop` | +| `09-goalify` | Autonomous loop that replans and retries until a runnable success condition passes | `01-init-tracking`, `02-auto-accept`, `03-autonomous-loop` | | `10-todo` | Split the prompt into independent todos, run one implementer agent per todo in parallel | `01-todo` | | `11-browser-qa` | Record short reviewer videos for browser-scoped happy and edge cases | `00-prerequisites`, `01-load-scope`, `02-prepare-run`, `03-run-scenarios` | diff --git a/plugins/aidd-context/skills/04-skill-generate/references/skill-authoring.md b/plugins/aidd-context/skills/04-skill-generate/references/skill-authoring.md index 9170aefa1..c9fd9528a 100644 --- a/plugins/aidd-context/skills/04-skill-generate/references/skill-authoring.md +++ b/plugins/aidd-context/skills/04-skill-generate/references/skill-authoring.md @@ -16,7 +16,7 @@ The contract every generated skill satisfies. `skill-generate` obeys it too. - **R7.** The mermaid flow shows every path a run can take: the nominal chain, one entry node per case, a back-edge per loop, a terminal node per outcome. A branch stated in prose is a branch missing from the flow. - **R8.** The action table is `| Action | Does |`, one row per action file, in run order. `Action` is the bare slug, no backticks and no number. `Does` is a lowercase imperative half-line with no final period. Above the table, one sentence: what to read next, nothing more. - **R9.** `## Transversal rules` holds the rules no single action or reference owns. A rule stated there is stated nowhere else. -- **R10.** The router is loaded on every call, an action only when its turn comes: the router carries nothing an action or a reference could carry. +- **R10.** The router is loaded on every call, an action only when its turn comes: the router carries nothing an action or a reference could carry. Never link to `SKILL.md` from a skill file. ## An action diff --git a/plugins/aidd-dev/.claude-plugin/plugin.json b/plugins/aidd-dev/.claude-plugin/plugin.json index 917597521..8b0afd25e 100644 --- a/plugins/aidd-dev/.claude-plugin/plugin.json +++ b/plugins/aidd-dev/.claude-plugin/plugin.json @@ -2,7 +2,7 @@ "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "aidd-dev", "version": "2.5.0", - "description": "Code transformation: plan, implement, assert, audit, review, test, refactor, debug, for-sure, plus short standalone Browser QA evidence. Hosts engineering agents.", + "description": "Code transformation: plan, implement, assert, audit, review, test, refactor, debug, goalify, plus short standalone Browser QA evidence. Hosts engineering agents.", "author": { "name": "AI-Driven Dev", "url": "https://github.com/ai-driven-dev" @@ -16,7 +16,7 @@ "./skills/06-test", "./skills/07-refactor", "./skills/08-debug", - "./skills/09-for-sure", + "./skills/09-goalify", "./skills/10-todo", "./skills/11-browser-qa" ], diff --git a/plugins/aidd-dev/CATALOG.md b/plugins/aidd-dev/CATALOG.md index 671f54194..1147d152d 100644 --- a/plugins/aidd-dev/CATALOG.md +++ b/plugins/aidd-dev/CATALOG.md @@ -17,7 +17,7 @@ Auto-generated index of skills, agents, references and assets shipped by the `ai - [`skills/06-test`](#skills06-test) - [`skills/07-refactor`](#skills07-refactor) - [`skills/08-debug`](#skills08-debug) - - [`skills/09-for-sure`](#skills09-for-sure) + - [`skills/09-goalify`](#skills09-goalify) - [`skills/10-todo`](#skills10-todo) - [`skills/11-browser-qa`](#skills11-browser-qa) @@ -126,17 +126,17 @@ Auto-generated index of skills, agents, references and assets shipped by the `ai | `references` | [mermaid-conventions.md](skills/08-debug/references/mermaid-conventions.md) | `Rules for generating valid, high-quality Mermaid diagrams. Apply when creating or reviewing any Mermaid diagram (flowchart, state, ER, sequence, gantt).` | | `-` | [SKILL.md](skills/08-debug/SKILL.md) | `Reproduce and fix a known bug, or find an unknown root cause by hypothesis validation. Use when the user wants to fix a bug, find why something breaks, or reopen a stuck investigation. Not for building a feature or reviewing a diff.` | -#### `skills/09-for-sure` +#### `skills/09-goalify` | Group | File | Description | |-------|------|---| -| `actions` | [01-init-tracking.md](skills/09-for-sure/actions/01-init-tracking.md) | - | -| `actions` | [02-auto-accept.md](skills/09-for-sure/actions/02-auto-accept.md) | - | -| `actions` | [03-autonomous-loop.md](skills/09-for-sure/actions/03-autonomous-loop.md) | - | -| `assets` | [autonomous-loop-worker-prompt.md](skills/09-for-sure/assets/autonomous-loop-worker-prompt.md) | - | -| `assets` | [plan-template.md](skills/09-for-sure/assets/plan-template.md) | - | -| `references` | [autonomous-loop-log-format.md](skills/09-for-sure/references/autonomous-loop-log-format.md) | - | -| `-` | [SKILL.md](skills/09-for-sure/SKILL.md) | `Run an iterative agent loop that retries until a runnable success condition passes. Use when the user says "for sure", "keep trying until", or wants guaranteed completion against a success command. Not for one-shot tasks or uncheckable goals.` | +| `actions` | [01-init-tracking.md](skills/09-goalify/actions/01-init-tracking.md) | - | +| `actions` | [02-auto-accept.md](skills/09-goalify/actions/02-auto-accept.md) | - | +| `actions` | [03-autonomous-loop.md](skills/09-goalify/actions/03-autonomous-loop.md) | - | +| `assets` | [autonomous-loop-worker-prompt.md](skills/09-goalify/assets/autonomous-loop-worker-prompt.md) | - | +| `assets` | [plan-template.md](skills/09-goalify/assets/plan-template.md) | - | +| `references` | [autonomous-loop-log-format.md](skills/09-goalify/references/autonomous-loop-log-format.md) | - | +| `-` | [SKILL.md](skills/09-goalify/SKILL.md) | `Turn a goal into an autonomous loop that replans and retries until a runnable success condition passes. Use when the user says "goalify", "keep trying until", or wants a goal verified by a command. Not for one-shot tasks or uncheckable goals.` | #### `skills/10-todo` diff --git a/plugins/aidd-dev/README.md b/plugins/aidd-dev/README.md index a23bc830b..1150be1a5 100644 --- a/plugins/aidd-dev/README.md +++ b/plugins/aidd-dev/README.md @@ -8,7 +8,7 @@ Code transformation plugin for the AI-Driven Development framework. First time? Install with `/plugin install aidd-dev@aidd-framework`, then run `aidd-dev:01-plan`. -Covers code transformation: planning, implementation, assertions, audits, code review, testing, refactoring, debugging, for-sure, and parallel todo fan-out. Standalone Browser QA records short web evidence. Also hosts AI agents. +Covers code transformation: planning, implementation, assertions, audits, code review, testing, refactoring, debugging, goalify, and parallel todo fan-out. Standalone Browser QA records short web evidence. Also hosts AI agents. ## Skills @@ -22,7 +22,7 @@ Covers code transformation: planning, implementation, assertions, audits, code r | [2.6] | [test](skills/06-test/SKILL.md) | Write and iterate on tests until they pass, and validate user journeys end-to-end in the browser. | | [2.7] | [refactor](skills/07-refactor/SKILL.md) | Optimize code for performance and fix security vulnerabilities following OWASP guidelines. | | [2.8] | [debug](skills/08-debug/SKILL.md) | Reproduce and fix bugs systematically using test-driven workflow, root cause analysis, and hypothesis validation. | -| [2.9] | [for-sure](skills/09-for-sure/SKILL.md) | Iterative agent loop that tracks attempts and retries until a success condition is met. | +| [2.9] | [goalify](skills/09-goalify/SKILL.md) | Autonomous loop that replans and retries until a runnable success condition passes. | | [2.10] | [todo](skills/10-todo/SKILL.md) | Split the prompt into independent todos, run one executor agent per todo in parallel, then report a minimal table. | | [2.11] | [browser-qa](skills/11-browser-qa/SKILL.md) | Record one short named video for a locked browser happy path and each sourced browser edge case. | diff --git a/plugins/aidd-dev/skills/09-for-sure/SKILL.md b/plugins/aidd-dev/skills/09-for-sure/SKILL.md deleted file mode 100644 index 31d5be3e9..000000000 --- a/plugins/aidd-dev/skills/09-for-sure/SKILL.md +++ /dev/null @@ -1,37 +0,0 @@ ---- -name: 09-for-sure -description: Run an iterative agent loop that retries until a runnable success condition passes. Use when the user says "for sure", "keep trying until", or wants guaranteed completion against a success command. Not for one-shot tasks or uncheckable goals. -argument-hint: task | command ---- - -# Skill: for-sure - -Run an autonomous loop until a success condition is verified. An interactive pre-flight (human present) sets it up, then autonomous execution (human gone) retries until success, acting as the user and never stopping early. - -## Actions - -| # | Action | Phase | Role | -| --- | ----------------- | ---------------------- | -------------------------------------------------------------------------- | -| 01 | `init-tracking` | interactive pre-flight | validate the goal, build the journey map, create the tracking file, spawn the loop | -| 02 | `auto-accept` | autonomous | decide and act as the user, stopping only on money or destructive actions | -| 03 | `autonomous-loop` | autonomous | spawn one worker per step, verify, retry, evaluate the success condition | - -Run `01` interactively; it spawns `03`, which runs unattended under the `02` auto-accept rules until the success condition passes. -Before running an action, read its file in `actions/`, not only the table or assets. - -## Transversal rules - -- Single source of truth: all task state lives in `aidd_docs/tasks/.md` and nowhere else. -- No repeated failures: never retry a failed approach without a meaningful change. -- Honesty over escape: never set `status: implemented` until the success condition genuinely passes. -- Auto-accept: when a decision or approval is needed, act as the user (create accounts, generate keys, approve prompts, install tools), never asking. Stop only on a payment or a destructive action. -- The loop spawns one worker agent per step and never does the work itself. - -## Assets - -- `assets/plan-template.md`: the tracking file format (frontmatter, phases, acceptance criteria, Log). -- `assets/autonomous-loop-worker-prompt.md`: the prompt the loop spawns each per-step worker with. - -## References - -- `references/autonomous-loop-log-format.md`: the Log entry format the loop appends per attempt. diff --git a/plugins/aidd-dev/skills/09-for-sure/actions/03-autonomous-loop.md b/plugins/aidd-dev/skills/09-for-sure/actions/03-autonomous-loop.md deleted file mode 100644 index 853ed248d..000000000 --- a/plugins/aidd-dev/skills/09-for-sure/actions/03-autonomous-loop.md +++ /dev/null @@ -1,29 +0,0 @@ -# 03 - Autonomous loop - -Orchestrate the loop: for each unchecked step spawn a worker, verify the result, then check the box or retry. One step is one agent is one log entry, and the orchestrator never does the work itself. - -## Input - -The tracking file `aidd_docs/tasks/.md` produced by `01-init-tracking`. The loop runs with no human interaction, reading and writing through that file. - -## Output - -The success condition verified and the plan's `status` set to `implemented`, with every step checked and one Log entry per attempt. - -## Process - -1. **Read.** Read the entire file: frontmatter, journey map, steps, and full Log. -2. **Mark.** Increment `iteration` in the frontmatter, setting `status: in-progress` when still `pending`. -3. **Learn.** Read the Log to learn from prior attempts. -4. **Next.** Find the next unchecked step. -5. **Spawn.** Spawn a worker for that step with [autonomous-loop-worker-prompt.md](../assets/autonomous-loop-worker-prompt.md), passing the step description and the relevant context (objective, rules, prior Log entries for the step). -6. **Verify.** Read the worker's result, then verify concretely by running a check command, reading a file, or testing the output. Never trust the worker's claim alone. -7. **Record.** Read the worker's outcome. When it stopped at a money or destructive gate, surface the reason to the user and stop the loop; never retry, which would re-trigger the action. On verified success, tick the step `[x]`. On a plain failure, spawn another worker with the error context. Append a Log entry per [autonomous-loop-log-format.md](../references/autonomous-loop-log-format.md). -8. **Loop.** Move to the next unchecked step and repeat from Read. -9. **Evaluate.** Once every step is checked, run the `success_condition` command and verify the result yourself. On success, set `status: implemented` and stop. On failure, add new steps addressing the root cause and continue the loop. - -## Test - -- Each step attempt has exactly one Log entry. -- Every checked step has a `= ✓` entry whose verification cites a concrete command or file. -- `status: implemented` is set only after the `success_condition` command has been re-run and exits zero. diff --git a/plugins/aidd-dev/skills/09-goalify/SKILL.md b/plugins/aidd-dev/skills/09-goalify/SKILL.md new file mode 100644 index 000000000..b789034bf --- /dev/null +++ b/plugins/aidd-dev/skills/09-goalify/SKILL.md @@ -0,0 +1,55 @@ +--- +name: 09-goalify +description: Turn a goal into an autonomous loop that replans and retries until a runnable success condition passes. Use when the user says "goalify", "keep trying until", or wants a goal verified by a command. Not for one-shot tasks or uncheckable goals. +argument-hint: task | command +--- + +# Skill: goalify + +Frame a checkable goal interactively, then run unattended until its success condition passes or a safety stop is reached. + +```mermaid +flowchart TD + Start[Setup or resume] --> Init[init-tracking] + Init -->|ready| Loop[autonomous-loop under auto-accept] + Init -->|already implemented| Done[Complete] + Init -->|unresolved prerequisite| Stop[Stop and report] + Loop --> Worker[Execute next step] + Worker -->|safety stop| Stop + Worker --> Verify[Verify evidence] + Verify -->|failure| Replan[Analyze and replan] + Replan -->|retry| Loop + Verify -->|more steps| Loop + Verify -->|all steps checked| Success{success_condition passes?} + Success -->|no| Replan + Success -->|yes| Done +``` + +## Actions + +| Action | Does | +| --- | --- | +| [init-tracking](actions/01-init-tracking.md) | frame the goal, create or resume tracking, launch the loop | +| [auto-accept](actions/02-auto-accept.md) | decide and act within the task's safety limits | +| [autonomous-loop](actions/03-autonomous-loop.md) | dispatch workers, verify evidence, replan failures, check completion | + +Run `init-tracking` interactively; it launches or resumes `autonomous-loop` under `auto-accept`. +Before running an action, read its file in `actions/`, not only the table or assets. + +## Transversal rules + +- Single source of truth: all task state lives in `aidd_docs/tasks/.md` and nowhere else. +- No repeated failures: never retry a failed approach without a meaningful change. +- Honesty over escape: never set `status: implemented` until the success condition genuinely passes. +- Auto-accept: follow the action's rules within the original task; stop on payment or destructive actions. +- The loop spawns one worker agent per step and never does the work itself. +- Model policy: use a powerful available model for framing, planning, verification, and replanning, including every orchestrator launch or resume. At every worker launch or relaunch, use the smallest available model with its highest supported reasoning effort. Workers execute only their assigned step and return evidence. + +## Assets + +- `assets/plan-template.md`: the tracking file format (frontmatter, phases, acceptance criteria, Log). +- `assets/autonomous-loop-worker-prompt.md`: the prompt the loop spawns each per-step worker with. + +## References + +- `references/autonomous-loop-log-format.md`: the Log entry format the loop appends per attempt. diff --git a/plugins/aidd-dev/skills/09-for-sure/actions/01-init-tracking.md b/plugins/aidd-dev/skills/09-goalify/actions/01-init-tracking.md similarity index 70% rename from plugins/aidd-dev/skills/09-for-sure/actions/01-init-tracking.md rename to plugins/aidd-dev/skills/09-goalify/actions/01-init-tracking.md index a96224abe..517ce2cb7 100644 --- a/plugins/aidd-dev/skills/09-for-sure/actions/01-init-tracking.md +++ b/plugins/aidd-dev/skills/09-goalify/actions/01-init-tracking.md @@ -12,7 +12,7 @@ The tracking file at `aidd_docs/tasks/.md`, marked created or resumed ## Process -1. **Resume.** Check `aidd_docs/tasks/` for a file matching the task name and read its frontmatter `status`. +1. **Resume.** Apply the router's model policy, then check `aidd_docs/tasks/` for a file matching the task name and read its frontmatter `status`. - `pending` or `in-progress`: report the status (iteration, steps remaining), then skip to Spawn to resume. - `implemented`: report "Task already completed" and stop. - No file: continue to Collect. @@ -24,10 +24,14 @@ The tracking file at `aidd_docs/tasks/.md`, marked created or resumed 7. **Map.** Project the whole path as an ASCII map of steps, dependencies, tools, and blockers. Ask the user to confirm and iterate until they do. 8. **Scaffold.** Load [plan-template.md](../assets/plan-template.md), creating `aidd_docs/tasks/` when missing. 9. **Create.** Write `aidd_docs/tasks/.md` from the template. Fill the frontmatter (`objective`, `success_condition`, `iteration: 0`, `status: pending`), the phases with their tasks and acceptance criteria, and the journey map. -10. **Spawn.** Read the orchestrator recipe from [03-autonomous-loop.md](./03-autonomous-loop.md) and hand it to the Agent tool with `` filled in. +10. **Spawn.** Apply the router's model policy to launch or resume the orchestrator with [03-autonomous-loop.md](./03-autonomous-loop.md) and `` filled in. ## Test -- The tracking file exists with frontmatter `status: pending` at creation. -- Its `success_condition` is a runnable command and the journey map is present. -- Every `[!]` blocker was resolved before the spawn, and an autonomous agent was launched. +| Case | Pass | +| --- | --- | +| New task | The tracking file exists at `aidd_docs/tasks/.md` with `status: pending`, a runnable `success_condition`, and a journey map. | +| Pending or in-progress task | The existing file is retained and the orchestrator resumes from its recorded state. | +| Implemented task | "Task already completed" is reported and no agent launches. | +| Unresolved hard prerequisite | No orchestrator launches until every `[!]` is resolved. | +| Setup or resume | Framing and orchestrator model selections follow the router's model policy. | diff --git a/plugins/aidd-dev/skills/09-for-sure/actions/02-auto-accept.md b/plugins/aidd-dev/skills/09-goalify/actions/02-auto-accept.md similarity index 100% rename from plugins/aidd-dev/skills/09-for-sure/actions/02-auto-accept.md rename to plugins/aidd-dev/skills/09-goalify/actions/02-auto-accept.md diff --git a/plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md b/plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md new file mode 100644 index 000000000..5ae4bce62 --- /dev/null +++ b/plugins/aidd-dev/skills/09-goalify/actions/03-autonomous-loop.md @@ -0,0 +1,36 @@ +# 03 - Autonomous loop + +Orchestrate the loop: dispatch each unchecked step, verify the result, and replan failures before retrying. One attempt is one worker and one log entry; the orchestrator never does the work itself. + +## Input + +The tracking file `aidd_docs/tasks/.md` produced by `01-init-tracking`. The loop runs with no human interaction, reading and writing through that file. + +## Output + +The success condition verified and the plan's `status` set to `implemented`, with every step checked and one Log entry per attempt. Or a safety stop with its reason reported. + +## Process + +1. **Read.** Apply the router's model policy, then read the entire file: frontmatter, journey map, steps, and full Log. +2. **Mark.** Increment `iteration` in the frontmatter, setting `status: in-progress` when still `pending`. +3. **Learn.** Read the Log to learn from prior attempts. +4. **Next.** Find the next unchecked step. +5. **Spawn.** Apply the router's model policy to every worker launch or relaunch, using [autonomous-loop-worker-prompt.md](../assets/autonomous-loop-worker-prompt.md) with the step description and relevant context (objective, rules, prior Log entries for the step). +6. **Verify.** Read the worker's result, then verify concretely by running a check command, reading a file, or testing the output. Never trust the worker's claim alone. +7. **Record.** On verified success, tick the step `[x]`; on failure, leave it unchecked. Append a Log entry per [autonomous-loop-log-format.md](../references/autonomous-loop-log-format.md). If the worker stopped for payment, a destructive action, or an out-of-scope request, report the reason and stop; never retry the stopped action. +8. **Replan.** After a failure, analyze the worker and verification evidence and amend the tracking plan with a changed approach before retrying. Prefix each amendment with 🤖 and a brief rationale. Keep the original objective, success condition, and rules; pass the updated step and failure context to the next worker. +9. **Loop.** Move to the next unchecked step and repeat from Read. +10. **Evaluate.** Once every step is checked, run the `success_condition` command yourself. On exit 0, set `status: implemented` and stop. On failure, analyze the evidence and amend the plan as in Replan, adding unchecked steps for the root cause, then continue the loop. + +## Test + +| Case | Pass | +| --- | --- | +| Step attempt | Exactly one Log entry records the worker's attempt and the orchestrator's verification. | +| Verified step | The checked step has a `= ✓` entry citing a concrete command, file, or output. | +| Model selection | Orchestrator verification/replanning and every worker launch/relaunch follow the router's model policy. | +| Failed step | Before another worker launches, the tracking plan contains a changed approach and a 🤖 rationale based on the failure evidence. | +| Safety stop | The reason is reported, the stopped action is not retried, and `status` stays `in-progress`. | +| Final condition fails | `status` stays `in-progress`; the plan gains unchecked steps addressing the failure. | +| Final condition passes | `status: implemented` is set only after the orchestrator re-runs `success_condition` and observes exit 0. | diff --git a/plugins/aidd-dev/skills/09-for-sure/assets/autonomous-loop-worker-prompt.md b/plugins/aidd-dev/skills/09-goalify/assets/autonomous-loop-worker-prompt.md similarity index 88% rename from plugins/aidd-dev/skills/09-for-sure/assets/autonomous-loop-worker-prompt.md rename to plugins/aidd-dev/skills/09-goalify/assets/autonomous-loop-worker-prompt.md index 3e4a7e57b..f8c8ea30d 100644 --- a/plugins/aidd-dev/skills/09-for-sure/assets/autonomous-loop-worker-prompt.md +++ b/plugins/aidd-dev/skills/09-goalify/assets/autonomous-loop-worker-prompt.md @@ -4,6 +4,9 @@ Execute this step. Auto-accept everything, act as the user, and make every decision yourself (approve prompts, generate keys, install tools, click buttons). Do not ask for permission. Just do it. +Worker policy: execute only the assigned step and return concrete evidence. The +orchestrator retains reflection, framing, and replanning. + Signing in via an existing account (Google Sign-in, GitHub OAuth, SSO) is NOT account creation; it uses the user's active browser session. Do it. diff --git a/plugins/aidd-dev/skills/09-for-sure/assets/plan-template.md b/plugins/aidd-dev/skills/09-goalify/assets/plan-template.md similarity index 95% rename from plugins/aidd-dev/skills/09-for-sure/assets/plan-template.md rename to plugins/aidd-dev/skills/09-goalify/assets/plan-template.md index e5f274b24..dcf8aa3ee 100644 --- a/plugins/aidd-dev/skills/09-for-sure/assets/plan-template.md +++ b/plugins/aidd-dev/skills/09-goalify/assets/plan-template.md @@ -14,7 +14,7 @@ status: pending - Interpret comments on this file to help you fill it. - Each phase MUST have acceptance criteria. - During implementation, the AI may amend this plan. Every AI change MUST be prefixed with 🤖 and include a brief rationale. -- This file IS the live tracking file for For Sure. State lives in the `status` frontmatter field (`pending → in-progress → implemented`). +- This file IS the live tracking file for Goalify. State lives in the `status` frontmatter field (`pending → in-progress → implemented`). - `success_condition` MUST be a runnable command. The loop sets `status: implemented` only when it passes. - Log is APPEND-ONLY. One entry per step attempt. Never rewrite history. --> diff --git a/plugins/aidd-dev/skills/09-for-sure/references/autonomous-loop-log-format.md b/plugins/aidd-dev/skills/09-goalify/references/autonomous-loop-log-format.md similarity index 100% rename from plugins/aidd-dev/skills/09-for-sure/references/autonomous-loop-log-format.md rename to plugins/aidd-dev/skills/09-goalify/references/autonomous-loop-log-format.md diff --git a/scripts/__tests__/a-skill-links-only-inside-itself.test.js b/scripts/__tests__/a-skill-links-only-inside-itself.test.js index 78ce3209f..37ef7e655 100644 --- a/scripts/__tests__/a-skill-links-only-inside-itself.test.js +++ b/scripts/__tests__/a-skill-links-only-inside-itself.test.js @@ -1,9 +1,11 @@ const assert = require("node:assert/strict"); const fs = require("node:fs"); +const os = require("node:os"); const path = require("node:path"); const { describe, it } = require("node:test"); const ROOT = path.resolve(__dirname, "../.."); +const PROBE_SOURCE = path.posix.join("plugins", "probe", "skills", "01-probe", "actions", "probe.md"); /** * A skill ships two ways and a relative path survives only one of them: the tree ships flat @@ -14,12 +16,20 @@ const ROOT = path.resolve(__dirname, "../.."); * The skill's own directory is the boundary, not the plugin's. A link to a sibling skill, to * the plugin's README, or to anything in the repository is equally unreachable once a tool * has installed the skill somewhere of its own choosing; name the file in prose instead. + * The router is already loaded, so links back to SKILL.md are redundant even when they + * stay inside the skill. */ const SKILL_ROOT = /^plugins\/[^/]+\/skills\/[^/]+$/u; +const LINK_DESTINATION = /\]\(\s*(<[^>\n]+>|[^\s)]+)[^)\n]*\)/gu; -/** Markdown link targets that are relative paths — not anchors, not URLs, not mail. */ -const RELATIVE_LINK = /\]\((\.[^)\s]*)\)/gu; +function decodedPath(target) { + try { + return decodeURI(target); + } catch { + return target; + } +} function everyMarkdownUnderPlugins(directory, into) { for (const entry of fs.readdirSync(directory, { withFileTypes: true })) { @@ -38,49 +48,112 @@ function skillRootOf(relativePath) { return SKILL_ROOT.test(root) ? root : null; } -function linksLeavingTheirSkill() { - const escaping = []; - for (const file of everyMarkdownUnderPlugins(path.join(ROOT, "plugins"), [])) { - const relative = path.relative(ROOT, file).split(path.sep).join("/"); +function skillLinkViolations(repository = ROOT) { + const violations = []; + for (const file of everyMarkdownUnderPlugins(path.join(repository, "plugins"), [])) { + const relative = path.relative(repository, file).split(path.sep).join("/"); const root = skillRootOf(relative); if (root === null) continue; const text = fs.readFileSync(file, "utf8"); - for (const [, target] of text.matchAll(RELATIVE_LINK)) { + for (const [, raw] of text.matchAll(LINK_DESTINATION)) { + const target = raw.startsWith("<") && raw.endsWith(">") ? raw.slice(1, -1) : raw; + if (/^(?:[a-z][a-z\d+.-]*:|\/\/|#)/iu.test(target)) continue; const resolved = path - .normalize(path.join(path.dirname(relative), target.split("#")[0])) + .normalize(path.join(path.dirname(relative), decodedPath(target.split("#")[0]))) .split(path.sep) .join("/"); - if (resolved !== root && !resolved.startsWith(`${root}/`)) { - escaping.push(`${relative} -> ${target}`); + if ( + path.posix.basename(resolved) === "SKILL.md" + || (resolved !== root && !resolved.startsWith(`${root}/`)) + ) { + violations.push(`${relative} -> ${raw}`); } } } - return escaping; + return violations; } -describe("a skill links only inside itself", () => { - it("has no markdown link reaching out of the skill that ships it", () => { +function probeSkillLinks(markdown) { + const repository = fs.mkdtempSync(path.join(os.tmpdir(), "aidd-skill-links-")); + try { + const skill = path.join(repository, "plugins", "probe", "skills", "01-probe"); + fs.mkdirSync(path.join(skill, "actions"), { recursive: true }); + fs.writeFileSync(path.join(skill, "actions", "probe.md"), markdown, "utf8"); + for (const file of ["README.md", "CATALOG.md"]) { + fs.writeFileSync( + path.join(repository, "plugins", "probe", file), + "[skill](skills/01-probe/SKILL.md#rules)\n", + "utf8", + ); + } + return skillLinkViolations(repository); + } finally { + fs.rmSync(repository, { recursive: true, force: true }); + } +} + +describe("a skill links only inside itself, never back to its loaded router", () => { + it("has no markdown link leaving its skill or returning to SKILL.md", () => { + assert.doesNotMatch( + fs.readFileSync(__filename, "utf8"), + /(plugins|scripts)\//u, + "synthetic fixture paths must not hide this repository-wide guard from changed-test selection", + ); assert.deepEqual( - linksLeavingTheirSkill(), + skillLinkViolations(), + [], + "links must stay inside their skill and never return to SKILL.md; the router is already loaded", + ); + }); + + it("rejects inline links back to a router, with or without a fragment", () => { + const cases = [ + ["[router](../SKILL.md)", "../SKILL.md"], + ["[rules](../SKILL.md#transversal-rules)", "../SKILL.md#transversal-rules"], + ["[router](SKILL.md)", "SKILL.md"], + ["[router](./SKILL.md)", "./SKILL.md"], + ["[router](../SKILL.md \"already loaded\")", "../SKILL.md"], + ["[rules](../SKILL.md#transversal-rules \"already loaded\")", "../SKILL.md#transversal-rules"], + ["[router](<../SKILL.md>)", "<../SKILL.md>"], + ["[rules](<../SKILL.md#transversal-rules>)", "<../SKILL.md#transversal-rules>"], + ["[router](../SKILL%2Emd)", "../SKILL%2Emd"], + ["[rules](../SKILL%2Emd#transversal-rules)", "../SKILL%2Emd#transversal-rules"], + ]; + assert.deepEqual( + probeSkillLinks(cases.map(([link]) => link).join("\n")), + cases.map(([, target]) => `${PROBE_SOURCE} -> ${target}`), + ); + }); + + it("accepts actions, references, bare router mentions and README/catalog skill links", () => { + assert.deepEqual( + probeSkillLinks([ + "[action](./next.md)", + "[reference](../references/rules.md#rule)", + "[reference](<../references/rules.md> \"rules\")", + "[reference](../references/rules%2Emd#rule)", + "[reference](../references/100%coverage.md)", + "The router SKILL.md is already loaded; apply `../SKILL.md` rules.", + "[docs](https://example.com/guide)", + "[section](#process)", + ].join("\n")), [], - "a relative link out of a skill is dead in every installed copy — name the file in prose instead", ); }); // The guard has to be able to see one. A checker that resolves every path against this // repository, the way `check-markdown-links.js` does, reports nothing here at all. it("sees an escaping link when one is put in front of it", () => { - const root = path.join(ROOT, "plugins", "aidd-dev", "skills", "01-plan"); - const probe = path.join(root, "escaping-link-probe.md"); - fs.writeFileSync(probe, "[out](../../../../README.md)\n", "utf8"); - try { - const found = linksLeavingTheirSkill(); - assert.ok( - found.some((entry) => entry.includes("escaping-link-probe.md")), - `the probe was not detected; found: ${JSON.stringify(found)}`, - ); - } finally { - fs.rmSync(probe); - } + const cases = [ + ["[out](../../../../README.md)", "../../../../README.md"], + ["[out\ncontinued](../../../../README.md)", "../../../../README.md"], + ["[out [nested]](../../../../README.md)", "../../../../README.md"], + ["[out](%2E%2E/%2E%2E/%2E%2E/%2E%2E/README.md)", "%2E%2E/%2E%2E/%2E%2E/%2E%2E/README.md"], + ["[out](../../../../README%ZZ.md)", "../../../../README%ZZ.md"], + ]; + assert.deepEqual( + probeSkillLinks(cases.map(([link]) => link).join("\n")), + cases.map(([, target]) => `${PROBE_SOURCE} -> ${target}`), + ); }); });