From 36287d1aa64e1288b338dc83d4d56bcc559747d3 Mon Sep 17 00:00:00 2001 From: Amit Arora Date: Sun, 20 Sep 2026 00:23:15 +0000 Subject: [PATCH 1/3] Score this repo's own AGENTS.md on every pull request .github/workflows/agents-md-readiness.yml installs the agents-md-readiness binary from the 0.2.0 release and runs it against AGENTS.md, failing the job below 0.85 readiness or on a credential reading suspect or worse. The bar sits at 0.85 because this file scores 0.92 with a spread of 0.01 across repeat runs, which leaves room for the model's variance and still catches a real regression. Details worth knowing: - The workflow triggers on a pull request that touches AGENTS.md or the workflow itself, plus workflow_dispatch for a manual run. - A fork pull request gets no secrets, so the job checks for the key first and explains the skip in the summary instead of failing on a missing key. - set -o pipefail keeps the binary's exit code through tee, so a gate failure fails the step rather than being swallowed by the pipe. - The summary and the artifact both run under always(), so a failing gate still shows the table and the JSON report. - permissions is contents: read and nothing more. One call costs about 3,100 input tokens, $0.00013 at TypeSafe's September 2026 price, and under half a second. --- .github/workflows/agents-md-readiness.yml | 87 +++++++++++++++++++++++ 1 file changed, 87 insertions(+) create mode 100644 .github/workflows/agents-md-readiness.yml diff --git a/.github/workflows/agents-md-readiness.yml b/.github/workflows/agents-md-readiness.yml new file mode 100644 index 0000000..a43c943 --- /dev/null +++ b/.github/workflows/agents-md-readiness.yml @@ -0,0 +1,87 @@ +name: AGENTS.md readiness + +# Score this repo's own AGENTS.md on every pull request that touches it, with the +# binary the agents-md-readiness sample ships. The job fails when readiness drops +# below the bar or when the file looks like it carries a credential, which puts the +# fix in front of whoever holds the branch. +on: + pull_request: + paths: + - AGENTS.md + - .github/workflows/agents-md-readiness.yml + workflow_dispatch: + +# The job reads the checkout and writes a summary, and it needs nothing else. +permissions: + contents: read + +jobs: + score: + runs-on: ubuntu-latest + env: + # Readiness sits at 0.92 with a spread of 0.01 across repeat runs, so 0.85 + # leaves room for the model's own variance and still fails a real + # regression. + BAR: "0.85" + VERSION: "0.2.0" + steps: + - uses: actions/checkout@v4 + + # A pull request from a fork gets no secrets, so the job says why it stopped + # instead of failing on a missing key. + - name: Check for the API key + id: key + env: + TYPESAFE_API_KEY: ${{ secrets.TYPESAFE_API_KEY }} + run: | + if [ -z "$TYPESAFE_API_KEY" ]; then + echo "present=false" >> "$GITHUB_OUTPUT" + echo "TYPESAFE_API_KEY is not set for this run, so the score is skipped." >> "$GITHUB_STEP_SUMMARY" + else + echo "present=true" >> "$GITHUB_OUTPUT" + fi + + - name: Install agents-md-readiness + if: steps.key.outputs.present == 'true' + run: | + curl -fsSL "https://raw.githubusercontent.com/${{ github.repository }}/main/samples/agents-md-readiness/go/install.sh" | sh + env: + VERSION: ${{ env.VERSION }} + BINDIR: ${{ runner.temp }}/bin + + - name: Score AGENTS.md + if: steps.key.outputs.present == 'true' + id: score + env: + TYPESAFE_API_KEY: ${{ secrets.TYPESAFE_API_KEY }} + run: | + set -o pipefail + # The binary prints a markdown table, so the same output reads in the log + # and in the job summary. tee keeps both. + "${{ runner.temp }}/bin/agents-md-readiness" \ + -fail-under "$BAR" \ + -fail-on-credential \ + -json "${{ runner.temp }}/report" \ + AGENTS.md | tee "${{ runner.temp }}/score.txt" + + - name: Write the summary + # always() so a failing gate still explains itself in the summary. + if: always() && steps.key.outputs.present == 'true' + run: | + { + echo "## AGENTS.md readiness" + echo + echo "Bar: $BAR. Binary: agents-md-readiness $VERSION." + echo + echo '```' + cat "${{ runner.temp }}/score.txt" || echo "the run produced no output" + echo '```' + } >> "$GITHUB_STEP_SUMMARY" + + - name: Upload the report + if: always() && steps.key.outputs.present == 'true' + uses: actions/upload-artifact@v4 + with: + name: agents-md-readiness-report + path: ${{ runner.temp }}/report + if-no-files-found: ignore From 0933f83b13c44943464a10515b42afce93c7ca60 Mon Sep 17 00:00:00 2001 From: Amit Arora Date: Sun, 20 Sep 2026 00:24:54 +0000 Subject: [PATCH 2/3] Temporarily raise the bar to 0.99 to prove the gate fails --- .github/workflows/agents-md-readiness.yml | 2 +- samples/agents-md-readiness/README.md | 2 ++ 2 files changed, 3 insertions(+), 1 deletion(-) diff --git a/.github/workflows/agents-md-readiness.yml b/.github/workflows/agents-md-readiness.yml index a43c943..a4d2108 100644 --- a/.github/workflows/agents-md-readiness.yml +++ b/.github/workflows/agents-md-readiness.yml @@ -22,7 +22,7 @@ jobs: # Readiness sits at 0.92 with a spread of 0.01 across repeat runs, so 0.85 # leaves room for the model's own variance and still fails a real # regression. - BAR: "0.85" + BAR: "0.99" VERSION: "0.2.0" steps: - uses: actions/checkout@v4 diff --git a/samples/agents-md-readiness/README.md b/samples/agents-md-readiness/README.md index 25493c1..256f39c 100644 --- a/samples/agents-md-readiness/README.md +++ b/samples/agents-md-readiness/README.md @@ -2,6 +2,8 @@ A coding agent is only as good as the instructions it finds in your repo. [AGENTS.md](https://agents.md) is where those instructions go, and its homepage claimed "over 60k open-source projects" when I read it on 19 September 2026, counted by [this GitHub code search](https://github.com/search?q=path%3AAGENTS.md+NOT+is%3Afork+NOT+is%3Aarchived&type=code). The file decides whether an agent runs your real test command or guesses, edits the generated file you told it never to touch, or opens a pull request in the wrong format. +An agent with no command to copy runs a plausible one, reads the failure, greps for the real command, opens two config files, and tries again. Each step is a tool call whose output lands in the context window you pay for. The same search repeats on the next task, and it grows with whatever the file leaves out. + Legacy repos have no AGENTS.md at all. Repos that have one wrote it against a format they skimmed, and the same sections go missing from one repo to the next. Eight of the ten files scored below never say which file wins for a directory, and `boundaries` is the weakest area in eight of them. Review misses both, because nobody reads AGENTS.md in a diff. This sample scores that file for readiness, then turns the score into something CI can check. Wire it into the job that runs when a pull request opens, score the AGENTS.md or CLAUDE.md the branch would merge, and fail the build when readiness drops below the bar you set or when the file leaks a credential. Each check is one call carrying 17 questions, and on the eleven repos measured here it cost between $0.00004 and $0.00034. From b185665af8b8b882c9809e0521f297641deffd9d Mon Sep 17 00:00:00 2001 From: Amit Arora Date: Sun, 20 Sep 2026 00:25:48 +0000 Subject: [PATCH 3/3] Restore the 0.85 bar The 0.99 commit proved the gate: exit code 2 travelled through tee, the step failed, and the summary and artifact steps still ran under always(). The log line read "readiness 0.92 is under the 0.99 bar". --- .github/workflows/agents-md-readiness.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/agents-md-readiness.yml b/.github/workflows/agents-md-readiness.yml index a4d2108..a43c943 100644 --- a/.github/workflows/agents-md-readiness.yml +++ b/.github/workflows/agents-md-readiness.yml @@ -22,7 +22,7 @@ jobs: # Readiness sits at 0.92 with a spread of 0.01 across repeat runs, so 0.85 # leaves room for the model's own variance and still fails a real # regression. - BAR: "0.99" + BAR: "0.85" VERSION: "0.2.0" steps: - uses: actions/checkout@v4