diff --git a/.agents/plugins/marketplace.json b/.agents/plugins/marketplace.json index df73f70..e5acbe8 100644 --- a/.agents/plugins/marketplace.json +++ b/.agents/plugins/marketplace.json @@ -40,6 +40,18 @@ }, "category": "Developer Tools" }, + { + "name": "jafar-perf", + "source": { + "source": "local", + "path": "./plugins/jafar-perf" + }, + "policy": { + "installation": "AVAILABLE", + "authentication": "ON_INSTALL" + }, + "category": "Developer Tools" + }, { "name": "perf-engineer", "source": { diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index ea70d35..04b99cf 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -23,6 +23,12 @@ "version": "0.1.0", "source": "./plugins/jfr-analyzer" }, + { + "name": "jafar-perf", + "description": "JVM performance engineering with the Jafar MCP server: triage, CPU, latency, GC, memory-leak and regression playbooks, plus specialist subagents.", + "version": "0.1.0", + "source": "./plugins/jafar-perf" + }, { "name": "perf-engineer", "description": "Evidence-driven Java optimization investigations with JFR, BTrace, and JMH.", diff --git a/.github/workflows/tool-drift.yml b/.github/workflows/tool-drift.yml new file mode 100644 index 0000000..283d138 --- /dev/null +++ b/.github/workflows/tool-drift.yml @@ -0,0 +1,91 @@ +name: Tool drift + +# The skills and agents of the jfr-mcp plugins name the server's tools explicitly, and the server is +# developed in btraceio/jafar. The drift this job exists to catch therefore happens when *that* +# repository changes, not when this one does, so a push-triggered job alone would never run at the +# moment it matters. The weekly schedule is the point; the push and pull_request triggers only catch +# typos and keep the script itself working. +on: + schedule: + - cron: '17 6 * * 1' # Mondays, 06:17 UTC + push: + branches: [main] + paths: + - 'plugins/**' + - 'scripts/check-tool-references.js' + - '.github/workflows/tool-drift.yml' + pull_request: + paths: + - 'plugins/**' + - 'scripts/check-tool-references.js' + - '.github/workflows/tool-drift.yml' + workflow_dispatch: + +permissions: + contents: read + +jobs: + check: + name: Skills reference tools that exist + runs-on: ubuntu-latest + timeout-minutes: 15 + permissions: + contents: read + issues: write # only used to report a scheduled failure + steps: + - name: Check agent-plugins + uses: actions/checkout@v5 + + - name: Set up Node.js + uses: actions/setup-node@v5 + with: + node-version: 20 + + - name: Set up Java for JBang + uses: actions/setup-java@v4 + with: + distribution: temurin + java-version: '21' + + # Deliberately the same install path the jfr-mcp README tells users to run, so a break in + # that path shows up here rather than in someone's terminal. + - name: Install the published MCP server + run: | + curl -Ls https://raw.githubusercontent.com/btraceio/jafar/main/jfr-mcp/install.sh | bash + echo "$HOME/.jbang/bin" >> "$GITHUB_PATH" + + - name: Check every referenced tool exists + run: node scripts/check-tool-references.js + + # A scheduled run failing means upstream moved, and nobody is watching a cron job. An issue + # is how that reaches a human. + - name: Open an issue when the scheduled run finds drift + if: failure() && github.event_name == 'schedule' + uses: actions/github-script@v7 + with: + script: | + const title = 'Skills reference MCP tools the published server no longer exposes'; + const existing = await github.rest.issues.listForRepo({ + owner: context.repo.owner, repo: context.repo.repo, + state: 'open', labels: ['tool-drift'], + }); + const body = [ + 'The weekly tool-drift check failed: at least one skill or agent names an MCP tool', + 'that `jfr-mcp@btraceio` does not expose. That usually means a tool was renamed or', + 'removed in btraceio/jafar.', + '', + `Run: ${context.serverUrl}/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}`, + '', + 'The log names each tool and the exact file and line that references it.', + ].join('\n'); + if (existing.data.length === 0) { + await github.rest.issues.create({ + owner: context.repo.owner, repo: context.repo.repo, + title, body, labels: ['tool-drift'], + }); + } else { + await github.rest.issues.createComment({ + owner: context.repo.owner, repo: context.repo.repo, + issue_number: existing.data[0].number, body, + }); + } diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 41fe3ec..c02a2f7 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -13,3 +13,16 @@ scripts/validate-marketplace.sh For behavioral changes, add or update an eval case and include a captured response review. Use conventional commit messages and do not change plugin versions until a release is being published. + +Plugins that start the `jfr-mcp` server (`jafar-perf`, `jfr-analyzer`) name its tools in their skills +and agents, and the server lives in another repository. After changing those names, or when a new +`jfr-mcp` release lands, check that every referenced tool still exists: + +```sh +node scripts/check-tool-references.js # published server, via jbang --fresh +node scripts/check-tool-references.js --jar path/to.jar # a locally built shadow jar +``` + +It starts the server, asks it for `tools/list`, and fails naming each missing tool with its file and +line. It is not part of `validate-marketplace.sh`, because it needs the published server. + diff --git a/README.md b/README.md index 1ef4a3c..a83d56e 100644 --- a/README.md +++ b/README.md @@ -12,6 +12,7 @@ one another so the workflow instructions and supporting scripts are maintained o | `btrace-development` | Repository conventions and build guidance for BTrace development. | | [`btrace-observability`](plugins/btrace-observability/README.md) | A composable SRE skill suite for diagnosing Java applications with BTrace probes. | | [`jfr-analyzer`](plugins/jfr-analyzer/README.md) | Systematic profile analysis with optional BTrace live-probe correlation. | +| [`jafar-perf`](plugins/jafar-perf/README.md) | Guided JVM performance analysis on the Jafar MCP server: playbooks for CPU, latency, GC and memory, and specialist subagents. | | [`perf-engineer`](plugins/perf-engineer/README.md) | Evidence-driven optimization investigations with JFR, BTrace, and JMH. | ## Layout diff --git a/package.json b/package.json index 3aa8f4a..1bef9f7 100644 --- a/package.json +++ b/package.json @@ -14,6 +14,7 @@ "./plugins/btrace-development/skills", "./plugins/btrace-observability/skills", "./plugins/jfr-analyzer/skills", + "./plugins/jafar-perf/skills", "./plugins/perf-engineer/skills" ] } diff --git a/plugins/jafar-perf/.claude-plugin/plugin.json b/plugins/jafar-perf/.claude-plugin/plugin.json new file mode 100644 index 0000000..2b419d9 --- /dev/null +++ b/plugins/jafar-perf/.claude-plugin/plugin.json @@ -0,0 +1,22 @@ +{ + "name": "jafar-perf", + "version": "0.1.0", + "description": "JVM performance engineering with the Jafar MCP server: triage, CPU, latency, GC, memory-leak and regression-comparison playbooks, plus specialist analysis subagents.", + "author": { + "name": "BTrace", + "url": "https://github.com/btraceio" + }, + "repository": "https://github.com/btraceio/agent-plugins", + "license": "Apache-2.0", + "skills": "./skills/", + "mcpServers": "./.mcp.json", + "keywords": [ + "jfr", + "jvm", + "performance", + "profiling", + "heap-dump", + "pprof", + "otlp" + ] +} diff --git a/plugins/jafar-perf/.codex-plugin/plugin.json b/plugins/jafar-perf/.codex-plugin/plugin.json new file mode 100644 index 0000000..5c78f2a --- /dev/null +++ b/plugins/jafar-perf/.codex-plugin/plugin.json @@ -0,0 +1,38 @@ +{ + "name": "jafar-perf", + "version": "0.1.0", + "description": "JVM performance engineering with the Jafar MCP server: triage, CPU, latency, GC, memory-leak and regression-comparison playbooks, plus specialist analysis subagents.", + "author": { + "name": "BTrace", + "url": "https://github.com/btraceio" + }, + "repository": "https://github.com/btraceio/agent-plugins", + "license": "Apache-2.0", + "skills": "./skills/", + "mcpServers": "./.mcp.json", + "keywords": [ + "jfr", + "jvm", + "performance", + "profiling", + "heap-dump", + "pprof", + "otlp" + ], + "interface": { + "displayName": "Jafar Performance Engineer", + "shortDescription": "Guided JVM performance analysis with evidence for every claim.", + "longDescription": "Turns the Jafar MCP server into a guided JVM performance analyst: methodology skills for CPU, latency, GC, memory and heap investigations, plus specialist subagents that cite the tool call behind every claim.", + "developerName": "BTrace", + "category": "Developer Tools", + "capabilities": [ + "Guidance", + "Analysis" + ], + "defaultPrompt": [ + "Analyse this JFR recording and tell me why p99 latency doubled.", + "Compare these two recordings and tell me what regressed.", + "Use perf-lead to review this heap dump and recording." + ] + } +} diff --git a/plugins/jafar-perf/.mcp.json b/plugins/jafar-perf/.mcp.json new file mode 100644 index 0000000..a49af10 --- /dev/null +++ b/plugins/jafar-perf/.mcp.json @@ -0,0 +1,13 @@ +{ + "mcpServers": { + "jafar": { + "type": "stdio", + "command": "jbang", + "args": [ + "jfr-mcp@btraceio", + "--stdio", + "--attach" + ] + } + } +} diff --git a/plugins/jafar-perf/README.md b/plugins/jafar-perf/README.md new file mode 100644 index 0000000..3edfdc3 --- /dev/null +++ b/plugins/jafar-perf/README.md @@ -0,0 +1,97 @@ +# jafar-perf — performance engineer in a box + +A plugin (Claude Code and Codex) that turns the [Jafar MCP server](https://github.com/btraceio/jafar/blob/main/jfr-mcp/README.md) into a guided +JVM performance analyst. + +The MCP server already exposes 37 analysis tools. What it does not carry is the *methodology*: +which question to ask next, which tool answers it, what counts as evidence, and how to report. +This plugin is that layer. + +## Install + +``` +/plugin marketplace add btraceio/agent-plugins +/plugin install jafar-perf@btraceio-agent-plugins +``` + +The plugin bundles `.mcp.json`, so installing it also registers the `jafar` MCP server +(`jbang jfr-mcp@btraceio --stdio --attach`). [JBang](https://www.jbang.dev) must be on your PATH; +it fetches the server on first use. No separate `claude mcp add` is needed. + +`--attach` makes every session share one daemon, started on demand, instead of each starting its +own JVM, so a new session answers in a fraction of a second and agents working in parallel get +their own sessions on the same server. It needs a `jfr-mcp` release that has the flag; an older +one ignores it and runs a private server, which still works. See the +[daemon documentation](https://github.com/btraceio/jafar/blob/main/doc/mcp/Daemon.md#one-shared-daemon-for-many-stdio-clients---attach). + +## What is in it + +### Skills + +Invoked automatically when the work matches, or explicitly as `/jafar-perf:`. + +| Skill | Covers | +|---|---| +| `triage` | First step on any unfamiliar artifact: what it contains, what is anomalous, where to go next | +| `cpu` | Hot methods, call paths, convergence points, attributing samples to work | +| `latency` | Contention, parking, executor queue saturation, blocking I/O, per-endpoint attribution | +| `gc` | Pause distribution as a fraction of wall clock, heap behaviour, allocation hotspots | +| `memory-leak` | Retained sizes, dominators, GC root paths, leak detectors, heap-to-JFR correlation | +| `heap-diff` | Proving growth with two dumps instead of inferring it from one | +| `compare` | Before/after regression checks with a stated noise floor | +| `jfrpath` | Syntax reference for JfrPath, HdumpPath and SamplesPath | +| `report` | The output format and the evidence discipline every finding must meet | + +### Agents + +| Agent | Role | +|---|---| +| `perf-lead` | Triages, dispatches the specialists the evidence justifies, merges and ranks their findings | +| `perf-engineer` | General-purpose analyst for a single artifact, end to end | +| `cpu-analyst` | CPU-bound analysis | +| `concurrency-analyst` | Thread states, contention, queues | +| `memory-analyst` | GC and allocation | +| `heap-analyst` | Heap dumps and retention | +| `io-analyst` | File and socket I/O | + +Specialists carry narrow tool allowlists, so each one works within its dimension rather than +wandering across the whole surface. + +## Using it + +Point it at an artifact and ask: + +> Analyse `/tmp/recording.jfr` and tell me why p99 latency doubled after the last deploy. + +For a broad investigation, ask for the lead agent, which fans out to specialists and merges +their findings: + +> Use perf-lead to review `/tmp/recording.jfr`. + +For a regression check, open both recordings and compare: + +> Compare `/tmp/before.jfr` against `/tmp/after.jfr` and tell me what regressed. + +## The standard these skills enforce + +Every skill in this plugin pushes the same discipline, because it is what separates a +performance report from a guess: + +- **Every claim names the tool call that produced it.** If you cannot cite the call and the + numbers, the claim does not go in the report. +- **Rates, not counts.** Absolute counts are meaningless without the recording's duration and + misleading across recordings of different lengths. +- **Sampling is not measurement.** Sampled data is labelled as sampled, and frames below the + noise floor are not findings. +- **Absence of evidence is reported as such.** "No allocation hotspots found" is wrong when + allocation profiling was never enabled; `jfr_diagnose` returns `capabilityGaps` for exactly + this reason, and they belong in the report. +- **No claimed improvement without a measured comparison.** + +## Without Claude Code + +The methodology is also available from the server itself, so other MCP clients get it too: +prompts (`triage`, `compare`, `leak-hunt`, `latency`) and resources (`jafar://sessions`, +`jafar://help/jfrpath`, `jafar://help/hdumppath`, `jafar://help/tools`). The skills here go +further — they carry the interpretation rules and the failure modes — but the prompts cover +the sequence. diff --git a/plugins/jafar-perf/agents/concurrency-analyst.md b/plugins/jafar-perf/agents/concurrency-analyst.md new file mode 100644 index 0000000..ded7c1f --- /dev/null +++ b/plugins/jafar-perf/agents/concurrency-analyst.md @@ -0,0 +1,24 @@ +--- +name: concurrency-analyst +description: Specialist for thread and contention analysis of a JFR recording — thread states, monitor contention, parking, executor queue saturation, and per-endpoint latency attribution. Dispatch when triage shows threads blocked or waiting rather than running, or when the complaint is p99 latency rather than throughput. +tools: mcp__jafar__jfr_tsa, mcp__jafar__jfr_use, mcp__jafar__jfr_query, mcp__jafar__jfr_list_types, mcp__jafar__jfr_stackprofile, mcp__jafar__pprof_tsa, Read, Grep, Glob +skills: latency, report +model: sonnet +--- + +You analyse what threads are waiting for. Follow the `latency` skill; report in the +`report` format. + +Run `jfr_tsa` with `correlateBlocking=true` first and read `stateDistribution` before +anything else — if the recording is RUNNABLE-dominated this is a CPU question and you +should say so rather than manufacturing a contention story. + +Keep `jdk.JavaMonitorEnter` (blocked acquiring) separate from `jdk.JavaMonitorWait` +(waiting on a condition); they mean different things. Rank monitors by summed duration +relative to wall clock, never by event count. Treat executor queue saturation as +first-class: queued work cannot be recovered by faster methods. + +Two honesty requirements: JFR monitor events have a duration threshold, so absence of +events is not absence of contention — check `jdk.ActiveSetting` if it matters. And +`decorateByTime` correlations are concurrency in time, not causation; report them as +"concurrent with". diff --git a/plugins/jafar-perf/agents/cpu-analyst.md b/plugins/jafar-perf/agents/cpu-analyst.md new file mode 100644 index 0000000..dc80600 --- /dev/null +++ b/plugins/jafar-perf/agents/cpu-analyst.md @@ -0,0 +1,20 @@ +--- +name: cpu-analyst +description: Specialist for CPU-bound analysis of a JFR recording or sampling profile — hot methods, call paths, convergence points, and per-thread or per-endpoint attribution of execution samples. Dispatch when triage shows high execution-sample counts or a RUNNABLE-dominated thread state distribution. +tools: mcp__jafar__jfr_hotmethods, mcp__jafar__jfr_flamegraph, mcp__jafar__jfr_callgraph, mcp__jafar__jfr_stackprofile, mcp__jafar__jfr_query, mcp__jafar__jfr_list_types, mcp__jafar__pprof_hotmethods, mcp__jafar__pprof_flamegraph, mcp__jafar__otlp_flamegraph, Read, Grep, Glob +skills: cpu, report +model: sonnet +--- + +You analyse where CPU time goes. Follow the `cpu` skill; report in the `report` format. + +Start with `jfr_hotmethods` to learn whether the profile is concentrated or flat, then pick +the follow-up that shape calls for — bottom-up for a concentrated profile, top-down or +callgraph for a flat one. Confirm every hotspot against the three tests in the `cpu` skill +(above the noise floor, steady across time buckets, not one unrepresentative thread). + +Stay in your lane: time spent parked, blocked or waiting on I/O is not CPU cost. If the +profile shows the cost is waiting, say so and hand it back rather than analysing it here. + +Return findings with the tool call and numbers behind each, and the source location if you +can find it in the working tree. diff --git a/plugins/jafar-perf/agents/heap-analyst.md b/plugins/jafar-perf/agents/heap-analyst.md new file mode 100644 index 0000000..96f5b14 --- /dev/null +++ b/plugins/jafar-perf/agents/heap-analyst.md @@ -0,0 +1,25 @@ +--- +name: heap-analyst +description: Specialist for heap dump analysis — retained sizes, dominator tree, GC root paths, known leak detectors, graph-based clusters, collection waste, duplicate subgraphs, heap-to-heap diffs, and correlating retained objects with JFR allocation sites. Dispatch for any .hprof file, OutOfMemoryError, or memory that never comes back after GC. +tools: mcp__jafar__hdump_open, mcp__jafar__hdump_close, mcp__jafar__hdump_summary, mcp__jafar__hdump_report, mcp__jafar__hdump_query, mcp__jafar__hdump_help, mcp__jafar__jfr_open, Read, Grep, Glob +skills: memory-leak, heap-diff, report +model: sonnet +--- + +You find unintended retention. Follow the `memory-leak` skill, and `heap-diff` when two +dumps are available; report in the `report` format. + +Rank by retained size, never shallow size — a large `byte[]` or `String` population is +normal in every Java heap, and only its dominator is a finding. Run `hdump_report` first, +then the named detectors for known patterns and `clusters` for unknown ones. + +A finding is not complete without a path to a GC root. `pathToRoot()` per object, or +`retentionPaths()` merged at class level; the field named in that path is the fix. A leak +claim without a root path is a guess, and you should label it as one. + +Distinguish a leak from intended retention: a cache configured to be large is working as +designed, and the finding is then about its sizing against the container limit. + +When a JFR recording from the same interval is available, use the cross-session join to add +`allocCount`, `allocRate` and `topAllocSite`. That names the code that created the retained +objects — the single most actionable output you can produce. diff --git a/plugins/jafar-perf/agents/io-analyst.md b/plugins/jafar-perf/agents/io-analyst.md new file mode 100644 index 0000000..5d5ad75 --- /dev/null +++ b/plugins/jafar-perf/agents/io-analyst.md @@ -0,0 +1,23 @@ +--- +name: io-analyst +description: Specialist for I/O analysis of a JFR recording — slow file and socket operations, per-destination latency and throughput, and separating dependency slowness from JVM problems. Dispatch when USE analysis flags I/O, or when latency correlates with external calls rather than with locks or CPU. +tools: mcp__jafar__jfr_use, mcp__jafar__jfr_query, mcp__jafar__jfr_list_types, mcp__jafar__jfr_tsa, Read, Grep, Glob +skills: latency, report +model: sonnet +--- + +You analyse blocking I/O. Follow the `latency` skill's I/O section; report in the `report` +format. + +Start from `jfr_use resources=io`, then break down by destination: + +- `events/jdk.SocketRead[duration>10ms] | groupBy(address, agg=sum, value=duration) | top(10, by=value)` +- `events/jdk.FileRead[duration>10ms] | groupBy(path, agg=count) | top(10, by=count)` + +Normalise by the recording duration, and separate count from summed duration: many fast +reads and few slow ones are different problems with different fixes. + +Be direct about scope. Slow I/O to one address is a dependency or network problem, not a +JVM problem — say that plainly rather than proposing JVM tuning. What belongs to the +application is the *pattern*: N+1 request loops, missing batching, absent caching, +unnecessary synchronous calls on a request path. diff --git a/plugins/jafar-perf/agents/memory-analyst.md b/plugins/jafar-perf/agents/memory-analyst.md new file mode 100644 index 0000000..0434c44 --- /dev/null +++ b/plugins/jafar-perf/agents/memory-analyst.md @@ -0,0 +1,24 @@ +--- +name: memory-analyst +description: Specialist for GC and allocation analysis of a JFR recording — pause distribution as a fraction of wall clock, heap behaviour over time, allocation rate and allocation hotspots by class and site. Dispatch when triage reports GC pressure, heap growth, or questions about allocation churn. +tools: mcp__jafar__jfr_query, mcp__jafar__jfr_use, mcp__jafar__jfr_flamegraph, mcp__jafar__jfr_summary, mcp__jafar__jfr_list_types, Read, Grep, Glob +skills: gc, report +model: sonnet +--- + +You analyse GC cost and what causes it. Follow the `gc` skill; report in the `report` +format. + +Answer two questions in order: is GC hurting (pause time as a fraction of wall clock, and +the pause distribution — never the mean alone), and why is GC running (allocation rate and +the sites producing it). + +Confirm allocation profiling is enabled before drawing any allocation conclusion. If it is +not, state that the question cannot be answered from this recording and give the flag to +enable it next time. Never infer allocation from GC counts. + +If post-GC heap used climbs monotonically across the recording, stop: that is retention, +not GC tuning, and belongs to the heap-analyst. + +Rank recommendations by expected value: reduce allocation first, right-size the heap +second, change collector flags last and only with pause-distribution evidence. diff --git a/plugins/jafar-perf/agents/perf-engineer.md b/plugins/jafar-perf/agents/perf-engineer.md new file mode 100644 index 0000000..2a8de8a --- /dev/null +++ b/plugins/jafar-perf/agents/perf-engineer.md @@ -0,0 +1,32 @@ +--- +name: perf-engineer +description: General-purpose JVM performance analyst. Use for any single-artifact investigation of a JFR recording, heap dump, pprof or OTLP profile when you want one agent to triage, investigate and report end to end. For a broad investigation that should fan out across several dimensions at once, use perf-lead instead. +tools: mcp__jafar__jfr_open, mcp__jafar__jfr_close, mcp__jafar__jfr_summary, mcp__jafar__jfr_diagnose, mcp__jafar__jfr_list_types, mcp__jafar__jfr_query, mcp__jafar__jfr_help, mcp__jafar__jfr_hotmethods, mcp__jafar__jfr_flamegraph, mcp__jafar__jfr_callgraph, mcp__jafar__jfr_stackprofile, mcp__jafar__jfr_tsa, mcp__jafar__jfr_use, mcp__jafar__jfr_exceptions, mcp__jafar__jfr_compare, mcp__jafar__hdump_open, mcp__jafar__hdump_close, mcp__jafar__hdump_summary, mcp__jafar__hdump_report, mcp__jafar__hdump_query, mcp__jafar__hdump_help, mcp__jafar__pprof_open, mcp__jafar__pprof_summary, mcp__jafar__pprof_hotmethods, mcp__jafar__pprof_flamegraph, mcp__jafar__pprof_tsa, mcp__jafar__pprof_use, mcp__jafar__otlp_open, mcp__jafar__otlp_summary, mcp__jafar__otlp_flamegraph, mcp__jafar__otlp_use, Read, Grep, Glob +skills: triage, report +model: sonnet +--- + +You are a JVM performance engineer working with the Jafar analysis tools. + +Follow the `triage` skill to establish what the artifact contains before investigating, and +the `report` skill for how to present what you find. Load the more specific skill for +whatever triage points at — `cpu`, `latency`, `gc`, `memory-leak`, `heap-diff`, `compare` — +and consult `jfrpath` before composing any non-trivial query. + +Non-negotiable rules: + +- **Every claim names its tool call.** If you cannot say which call and which numbers + produced a statement, do not make the statement. +- **Rates, not counts.** Establish the recording's duration and normalise before quoting + anything. Counts from recordings of different lengths are not comparable. +- **Sampling is not measurement.** Label sampled data as sampled, and treat frames below + roughly 1% of samples as noise. +- **Report what the artifact cannot answer.** If profiling for something was not enabled, + say so explicitly rather than reporting its absence as a negative result. +- **Locate code before recommending a change.** Use Grep to find the frame in the working + tree; if you cannot find it, give the frame and say you could not locate the source. + +You may read the repository to correlate frames with source. You must not modify files. + +Finish with a ranked list of findings in the `report` format. Three well-evidenced findings +are worth more than a dozen speculative ones. diff --git a/plugins/jafar-perf/agents/perf-lead.md b/plugins/jafar-perf/agents/perf-lead.md new file mode 100644 index 0000000..5431c4d --- /dev/null +++ b/plugins/jafar-perf/agents/perf-lead.md @@ -0,0 +1,44 @@ +--- +name: perf-lead +description: Coordinator for a broad performance investigation. Triages an artifact, dispatches the specialist analysts the evidence justifies, then merges, ranks and de-duplicates their findings into one report. Use when the question is open-ended ("why is this service slow", "review this recording") rather than aimed at one dimension. +tools: mcp__jafar__jfr_open, mcp__jafar__jfr_close, mcp__jafar__jfr_summary, mcp__jafar__jfr_diagnose, mcp__jafar__jfr_list_types, mcp__jafar__jfr_compare, mcp__jafar__hdump_open, mcp__jafar__hdump_summary, mcp__jafar__hdump_report, Read, Grep, Glob, Agent(cpu-analyst, concurrency-analyst, memory-analyst, heap-analyst, io-analyst) +skills: triage, report +model: opus +--- + +You lead a performance investigation and are accountable for the final report. + +## Sequence + +1. **Triage yourself.** Open the artifact, run `jfr_summary` and `jfr_diagnose` (which runs + the USE and TSA analyses in-process and returns severity-ranked structured findings plus + `capabilityGaps`). Establish the recording duration. Do not delegate this step: the + routing decision depends on it. + +2. **Dispatch only what the evidence justifies.** Send the specialists whose dimension + triage actually flagged, and run them concurrently — one message with several Agent + calls. Give each one the artifact path, the session id, the recording duration, and the + specific finding that prompted the dispatch. Dispatching all five on every recording + wastes turns and produces padding. + +3. **Merge.** Findings carry a stable `id`, so identical conditions reported by two tools + de-duplicate cleanly; keep the more severe. Rank by impact — the share of wall clock or + of the resource at stake — not by how confident the specialist sounded. + +4. **Resolve conflicts.** When two specialists disagree, the one with the more direct + measurement wins, and you say in the report that the question was contested and why you + resolved it as you did. Do not average them, and do not report both as findings. + +## Standards you enforce + +- Every claim in the final report names the tool call and numbers behind it. +- Everything is a rate or a fraction of wall clock, with the denominator stated. +- `capabilityGaps` from triage appear in the report, separately from findings. A question + the artifact cannot answer must not be reported as a negative answer. +- Confidence is stated per finding, and sampled, heuristic or time-correlated evidence + caps it at medium. +- No recommendation without a location and an expected effect. + +Deliver one ranked report in the `report` format, plus the reproduction steps. If the +evidence does not support a conclusion, say so — "the recording does not show why" is a +legitimate and useful answer, and a fabricated cause is not. diff --git a/plugins/jafar-perf/skills/compare/SKILL.md b/plugins/jafar-perf/skills/compare/SKILL.md new file mode 100644 index 0000000..9c228c7 --- /dev/null +++ b/plugins/jafar-perf/skills/compare/SKILL.md @@ -0,0 +1,90 @@ +--- +name: compare +description: Decide whether a candidate JFR recording regressed against a baseline, and attribute the change to a frame or a metric. Use for before/after checks, "is this build slower", bisecting a performance regression, verifying that a fix actually helped, or any question involving two recordings of the same workload. +allowed-tools: mcp__jafar__jfr_open mcp__jafar__jfr_compare mcp__jafar__jfr_hotmethods mcp__jafar__jfr_stackprofile mcp__jafar__jfr_query mcp__jafar__jfr_summary +--- + +# Comparing two recordings + +The claim "this is slower" is only worth making with two measurements and a stated noise +floor. `jfr_compare` provides both. + +## Run it + +``` +jfr_open path=/abs/path/before.jfr alias=before +jfr_open path=/abs/path/after.jfr alias=after +jfr_compare baselineSessionId=before candidateSessionId=after +``` + +Optional: `eventType` to pin the execution-sample type, `minDeltaPct` to set the noise +floor in percentage points (default 1.0), `limit` for how many changed frames to return. + +## Read `comparability` first + +Before any number, the result tells you whether the comparison is sound. It flags: + +- **Different execution-sample event types** — the two recordings used different profilers + (`jdk.ExecutionSample` versus `datadog.ExecutionSample`). Frame shares remain roughly + comparable; sample counts are not comparable at all. +- **Durations differing by more than 3×** — rates are normalised, but a much shorter + recording may simply have missed periodic work such as a full GC or a cache refresh. +- **Fewer than ~1000 samples on either side** — per-frame shares are noisy; small moves mean + nothing. + +If any of these fire, say so in your answer and weaken the conclusion accordingly. A +regression claim that ignores a comparability warning is worse than no claim. + +## What the numbers mean + +**`metrics`** are per-second rates, computed with each recording's own observed span as the +denominator. Compare `baselineRate` to `candidateRate`; `baselineCount` and +`candidateCount` are shown for transparency, not for comparison. + +**`frames`** are shares of execution samples, in percentage points: + +- `baselineSelfPct` → `candidateSelfPct`, with `deltaPct` the difference in points. +- `direction` is `regression` when the share grew, `improvement` when it shrank. +- Frames moving less than `minDeltaPct` are omitted deliberately. Do not go hunting for + smaller moves and present them as findings. + +The single most common error to avoid: **a share is not a duration**. A frame growing from +3% to 9% of samples means the profile's shape changed. If total CPU work also fell, that +frame may be no slower in absolute terms — it just became a bigger slice of a smaller pie. +Cross-check the rates before calling a share change a slowdown. + +## Attribute the change + +`jfr_compare` names the frame. Finding out *why* takes one more step: + +``` +jfr_stackprofile sessionId=after buckets=10 +jfr_stackprofile sessionId=before buckets=10 +``` + +Compare the call paths reaching the changed frame, and its `timeBuckets` — a frame that +regressed only in the last two buckets points at state that accumulated (a growing +collection, a filling cache), not at a code path that got slower. + +Then locate the code with `Grep` and state the file and line. + +## When nothing changed + +The tool returns an explicit "no regression above the noise floor" finding. Report exactly +that. It is not the same as "the two builds perform identically": a change smaller than +sampling noise is invisible to this method, and you should say so rather than implying +equivalence. + +## Verifying a fix + +Same workload, same duration, same profiler settings, same JVM flags — otherwise the +comparison measures your test setup rather than the fix. Then: + +1. Record the baseline before the change. +2. Apply the change, record again with identical settings. +3. `jfr_compare` and read the frame you expected to move. + +A fix is confirmed when the frame you targeted shrank *and* the comparability block is +clean. If the targeted frame did not move but something else did, you have learned that +your model of the problem was wrong — report that, rather than claiming a win from an +unrelated improvement. diff --git a/plugins/jafar-perf/skills/cpu/SKILL.md b/plugins/jafar-perf/skills/cpu/SKILL.md new file mode 100644 index 0000000..9c0ff79 --- /dev/null +++ b/plugins/jafar-perf/skills/cpu/SKILL.md @@ -0,0 +1,93 @@ +--- +name: cpu +description: Find where CPU time goes in a JFR recording, pprof profile or OTLP profile, and attribute it to call paths and threads. Use when triage shows high execution-sample counts, when the user asks "why is the CPU pegged", "what is the hot method", "where is the time going", or asks for a flamegraph or profile of CPU usage. +allowed-tools: mcp__jafar__jfr_hotmethods mcp__jafar__jfr_stackprofile mcp__jafar__jfr_flamegraph mcp__jafar__jfr_callgraph mcp__jafar__jfr_query mcp__jafar__jfr_list_types mcp__jafar__pprof_hotmethods mcp__jafar__pprof_flamegraph mcp__jafar__otlp_flamegraph +--- + +# CPU analysis + +## Pick the right tool + +Four tools answer four different questions. Choosing wrong costs a turn and produces a +misleading answer. + +| Question | Tool | Returns | +|---|---|---| +| Which methods burn CPU? | `jfr_hotmethods` | Flat ranked list of **leaf** frames with sample counts and percentages | +| How does the code reach them? | `jfr_flamegraph` | Aggregated stack paths, folded or tree | +| Which frames are hot, when, and on which threads? | `jfr_stackprofile` | Frames with self/total percentages, time buckets, per-thread counts, `hotspot` classification | +| Which function is the convergence point? | `jfr_callgraph` | Caller→callee edges with `inDegree` | + +Start with `jfr_hotmethods`. It is one pass and it tells you whether the profile is +concentrated (one method at 40%) or flat (nothing above 3%). Those two shapes need opposite +follow-ups: + +- **Concentrated** → `jfr_flamegraph direction=bottom-up` to find who calls the hot method. +- **Flat** → `jfr_flamegraph direction=top-down` or `jfr_callgraph`, because the cost is in a + path, not a leaf. A framework that costs 30% spread over 50 leaves is invisible to + `hotmethods` and obvious in a top-down view. + +## Event type selection + +The analysis tools auto-detect the execution-sample event type and prefer a Datadog +profiler's type over the JDK's when both are present. Check what you actually have: + +``` +jfr_list_types filter=ExecutionSample +``` + +`jdk.ExecutionSample` (JDK) and `datadog.ExecutionSample` (Datadog) have different sampling +intervals. Never compare sample counts across recordings that used different profilers — +see the `compare` skill. + +## Native versus Java + +`jfr_hotmethods` returns a `categoryBreakdown` with `native` and `java` counts, and each +method carries a `type`. A profile that is 60% native frames is usually one of: JIT +compilation, GC threads, or a JNI-heavy library. Set `includeNative=false` to see the Java +picture alone, then compare the two totals. + +## Confirming a hotspot is real + +A frame is worth reporting when all three hold: + +1. Its self percentage is above the noise floor — roughly 1% of total samples, higher if the + recording is short. `jfr_stackprofile` applies this and labels frames `hotspot`. +2. It is *steady*, not a spike. `jfr_stackprofile` returns `timeBuckets[]` per frame; a frame + present in one bucket out of ten is an event, not a hotspot. The `steady-hotspot` + category means it persisted. +3. It is not an artifact of one thread doing something unrepresentative. Check + `threadCounts{}` in the same output. + +``` +jfr_stackprofile buckets=10 minPct=1.0 +``` + +## Attributing CPU to work + +Raw hotness rarely answers "why". Attribute samples to the request or endpoint that caused +them using event decoration: + +``` +jfr_query query="events/jdk.ExecutionSample | decorateByKey(datadog.Endpoint, key=localRootSpanId, decoratorKey=localRootSpanId, fields=endpoint) | groupBy($decorator.endpoint)" +``` + +For time-overlap correlation instead of a key join, use `decorateByTime` — see the +`latency` skill for the same technique applied to locks. + +## Mapping frames to source + +Once a frame is confirmed, find it in the working tree with `Grep` before recommending a +change. A method name alone is not a location: overloads, lambdas (`lambda$foo$0`), and +synthetic accessors all collapse in profiler output. Quote the file and line you found, and +say so if you could not find it. + +## What not to conclude + +- CPU samples during a GC pause are attributed to whatever thread was running; they do not + mean the sampled method is expensive. Cross-check with the `gc` skill. +- A high sample count on `Unsafe.park`, `Object.wait` or socket reads is *not* CPU cost — + those threads are not running. That is a `latency` question, not a CPU one. +- pprof and OTLP profiles infer thread state from function-name keywords, not from real + state transitions. Their `tsa` and `use` output is heuristic and must be labelled as such + in any report. diff --git a/plugins/jafar-perf/skills/gc/SKILL.md b/plugins/jafar-perf/skills/gc/SKILL.md new file mode 100644 index 0000000..c7b8d24 --- /dev/null +++ b/plugins/jafar-perf/skills/gc/SKILL.md @@ -0,0 +1,116 @@ +--- +name: gc +description: Analyse garbage collection pressure, pause times, heap sizing and allocation hotspots in a JFR recording. Use when triage reports high GC pressure, when the user asks about GC pauses, heap growth, allocation rate, OutOfMemoryError risk, or which code allocates the most. +allowed-tools: mcp__jafar__jfr_query mcp__jafar__jfr_use mcp__jafar__jfr_flamegraph mcp__jafar__jfr_list_types mcp__jafar__jfr_summary +--- + +# GC and allocation analysis + +Two distinct questions live here. Answer them in order, because the second explains the +first: + +1. **Is GC hurting?** Pause time as a fraction of wall clock, and pause distribution. +2. **Why is GC running?** Allocation rate and the code producing it. + +## 1. Is GC hurting? + +`jfr_summary` already carries a `highlights.gc` block with total collections, average pause +and total pause. Turn it into a fraction: + +``` +jfr_query query="events/jdk.GCPhasePause | stats(duration)" +jfr_query query="events/jdk.ExecutionSample | timerange()" +``` + +**Total pause ÷ wall clock** is the number that matters. 200 ms of pause in a 5-minute +recording is 0.07% and irrelevant no matter how alarming 200 ms sounds; 200 ms in a +2-second recording is 10% and dominant. + +Then look at the distribution, not the mean. A mean of 20 ms hides a 900 ms outlier that is +the actual p99 complaint: + +``` +jfr_query query="events/jdk.GCPhasePause | quantiles(0.5, 0.9, 0.99, path=duration)" +jfr_query query="events/jdk.GCPhasePause | top(10, by=duration)" +``` + +## 2. Which collector, which phase? + +``` +jfr_query query="events/jdk.GarbageCollection | groupBy(name, agg=count)" +jfr_query query="events/jdk.GCPhasePause | groupBy(name, agg=sum, value=duration) | top(10, by=value)" +``` + +Young collections that are frequent but short are usually healthy — that is the collector +doing its job. Old/full collections, or concurrent-mode failures, are the signal. G1's +`Remark` and `Cleanup` phases are stop-the-world even though the cycle is "concurrent". + +Which collector is in use, and its flags: + +``` +jfr_query query="events/jdk.ActiveSetting[name~\".*(GC|Heap).*\"] | select(name, value)" +``` + +## 3. Heap behaviour over time + +``` +jfr_query query="events/jdk.GCHeapSummary | select(startTime, heapUsed, when) | sortBy(startTime, asc=true)" +``` + +Read the *post-GC* used size (`when = "After GC"`). A sawtooth that returns to the same +floor is healthy churn. A floor that climbs monotonically across the recording is +retention — stop here and switch to the `memory-leak` skill, because no GC tuning fixes a +leak. + +## 4. Why is GC running — allocation + +Allocation profiling must be enabled or this section is unanswerable. Confirm first: + +``` +jfr_list_types filter=Alloc +``` + +- `jdk.ObjectAllocationSample` — sampled, cheap, available in the `profile` settings. +- `jdk.ObjectAllocationInNewTLAB` / `OutsideTLAB` — older, higher overhead, more detail. + +If neither is present, say "allocation profiling was not enabled in this recording" and +recommend `-XX:StartFlightRecording:settings=profile` for the next one. Do not guess at +allocation from GC counts. + +By class: + +``` +jfr_query query="events/jdk.ObjectAllocationSample | groupBy(objectClass/name, agg=sum, value=weight) | top(20, by=value)" +``` + +By allocation site — this is the actionable one, because it names the code: + +``` +jfr_flamegraph eventType=jdk.ObjectAllocationSample direction=bottom-up format=folded +``` + +`weight` on a sampled allocation event is an *estimate* of bytes represented by the sample, +not the bytes of that one object. Report it as an estimated rate (MB/s), never as an exact +total. + +## 5. What is running during GC + +``` +jfr_query query="events/jdk.ExecutionSample | decorateByTime(jdk.GCPhase, fields=name) | groupBy($decorator.name, agg=count)" +``` + +Useful for separating application cost from collector cost when a profile looks unexpectedly +hot in JVM-internal frames. + +## Recommendations worth making + +In rough order of expected value: + +1. **Reduce allocation** at the top sites found in step 4. This is the only fix that helps + every collector and every heap size. +2. **Right-size the heap** when post-GC used is close to max and collections are frequent. + Cite the `GCHeapSummary` numbers. +3. **Change collector or pause target** only with pause-distribution evidence from step 1, + and only when allocation is already understood. + +Never recommend a flag without the measurement that motivates it. "Try G1" is not a finding. diff --git a/plugins/jafar-perf/skills/heap-diff/SKILL.md b/plugins/jafar-perf/skills/heap-diff/SKILL.md new file mode 100644 index 0000000..ba88e70 --- /dev/null +++ b/plugins/jafar-perf/skills/heap-diff/SKILL.md @@ -0,0 +1,98 @@ +--- +name: heap-diff +description: Compare two or more heap dumps taken at different times to prove memory growth rather than infer it — class-level instance and retained-size deltas, newly appeared clusters, and objects that survived when they should not have. Use whenever two .hprof files of the same application are available, or when a single-dump finding needs confirmation. +allowed-tools: mcp__jafar__hdump_open mcp__jafar__hdump_query mcp__jafar__hdump_summary mcp__jafar__hdump_close +--- + +# Heap diff + +Single-snapshot leak analysis produces educated guesses: a large retained size might be a +leak or might be a correctly sized cache. Two snapshots produce facts. If `HashMap$Node` +count grew by 50,000 between t1 and t2 while the workload was steady, that is growth, not +interpretation. + +## Taking the dumps + +For the comparison to mean anything the two dumps must be separated by a workload, not by +chance. The useful pattern: + +1. Warm up, then dump — this is the baseline, after class loading and cache fill. +2. Run a known, repeated workload (N iterations of the same request mix). +3. Dump again. + +Anything that grew proportionally to N is a candidate. Both dumps should be taken after a +full GC where possible, so that uncollected garbage does not read as growth. + +## Running the diff + +Open both, then join the later against the earlier: + +``` +hdump_open path=/abs/path/dump-before.hprof alias=before +hdump_open path=/abs/path/dump-after.hprof alias=after +hdump_query query="classes | join(session=before) | sortBy(instanceCountDelta desc) | top(25)" +``` + +The current session is the one you query; `join(session=...)` names the other side. The join +key is inferred as `name` for the `classes` root; pass `by=field` to override. It is a left +join, so classes absent from the baseline appear with null baseline columns — those are +newly appeared types and deserve attention on their own. + +Rank by retained growth rather than instance count when the leak is few-and-large: + +``` +hdump_query query="classes | join(session=before) | sortBy(retainedDelta desc) | top(25)" +``` + +## Reading the result + +Three shapes, three conclusions: + +| Shape | Meaning | +|---|---| +| Count grew, retained grew proportionally | Straightforward accumulation — follow with `pathToRoot()` on the class | +| Count flat, retained grew | Existing objects growing internally — a collection or buffer growing without bound; use `waste()` | +| Count grew, retained flat | Small objects accumulating; often listener or `ThreadLocal` registrations | + +A class that grew is a symptom. The finding is the *field that holds it*, so always finish +with a root path in the later dump: + +``` +hdump_query query="classes/com.example.Entry | retentionPaths()" +``` + +## Confirming with clusters + +Cluster detection run on both dumps shows which suspicious subgraphs are new rather than +long-standing: + +``` +hdump_query query="clusters | sortBy(retainedSize desc) | top(10)" +``` + +Run against each session (switch with the `sessionId` parameter) and compare the cluster +anchors. A cluster present in both at the same size is structural, not a leak. + +## Controlling for noise + +Growth between two dumps is only evidence if the workload explains it. Before reporting: + +- Was the same workload applied, and how many iterations? +- Did the heap have a full GC before each dump? +- Is the growth larger than the variation you would see between two baseline dumps with no + workload at all? When in doubt, take that third dump and diff it against the first — that + is your noise floor. + +State the workload and the interval in the report. A delta without them is not +reproducible, and a leak claim that cannot be reproduced will not be believed. + +## Correlating growth with allocation + +Once a growing class is identified, JFR from the same interval names the code that created +the instances: + +``` +hdump_query query="classes | join(session=rec, root=\"jdk.ObjectAllocationSample\") | filter(retained > 1MB) | select(name, retained, allocCount, topAllocSite)" +``` + +See the `memory-leak` skill for the full cross-format workflow. diff --git a/plugins/jafar-perf/skills/jfrpath/SKILL.md b/plugins/jafar-perf/skills/jfrpath/SKILL.md new file mode 100644 index 0000000..99bc68a --- /dev/null +++ b/plugins/jafar-perf/skills/jfrpath/SKILL.md @@ -0,0 +1,145 @@ +--- +name: jfrpath +description: Syntax reference for the query languages behind jfr_query, hdump_query, pprof_query and otlp_query — JfrPath, HdumpPath and SamplesPath. Consult before composing any non-trivial query, and whenever a query returns a parse error, so the syntax is right on the first attempt instead of after three failures. +--- + +# Query language reference + +Four query tools, three languages. All are path-based, not SQL: you address a root, filter +it in brackets, and pipe it through operators. + +``` +[/][] ( | )* +``` + +## JfrPath — `jfr_query` + +**Roots**: `events/`, `metadata/`, `chunks`, `constants` (alias `cp`). + +### Filters go in square brackets + +``` +events/jdk.FileRead[bytes>1000] +events/jdk.FileRead[path~"/tmp/.*"] +events/jdk.FileRead[bytes>1000 and path~"/tmp/.*"] +``` + +Operators: `=` `!=` `>` `>=` `<` `<=` `~` (regex). Combine with `and`, `or`, `not` and +parentheses. Functions usable inside a filter: `contains`, `startsWith`, `endsWith`, +`matches(path,"re"[,"i"])`, `exists`, `empty`, `between(path,a,b)`, `len(path)`, and the +time predicates `before`, `after`, `on`. + +Filters can be interleaved at any segment: + +``` +events/jdk.GCHeapSummary[when/when="After GC"]/heapSpace[committedSize>1000000]/reservedSize +``` + +For list fields, choose the match mode — `any:` (default), `all:`, `none:`: + +``` +events/jdk.ExecutionSample[none:stackTrace/frames[matches(method/name/string, ".*Test.*")]] +``` + +### Numeric literals and units + +Size suffixes work and are binary: `K`/`KB` = 1024, `M`/`MB` = 1024², `G`/`GB` = 1024³. + +``` +events/jdk.FileRead[bytes>1MB] +``` + +Duration suffixes `ns`, `us`, `ms`, `s` are also accepted and convert to nanoseconds, which +is how JFR stores durations: + +``` +events/jdk.GCPhasePause[duration>10ms] +events/jdk.JavaMonitorEnter[duration>1ms] | count() +``` + +A bare number in a duration field is nanoseconds: `[duration>10000000]` is the same 10 ms. +There is deliberately no `m` suffix for minutes, because `M` already means mebibytes. + +### Pipeline operators + +| Group | Operators | +|---|---| +| Aggregate (terminal) | `count()`, `sum([path])`, `stats([path])`, `quantiles(q…[, path=])`, `sketch([path])`, `timerange([path][, duration=][, format=])`, `flamegraph([direction=])`, `stackprofile([direction=][, buckets=][, minPct=])` | +| Group and order | `groupBy(key[, agg=count\|sum\|avg\|min\|max][, value=path][, sortBy=key\|value][, asc=])`, `sortBy(field[, asc=])`, `top(n[, by=path][, asc=])`, `head(n)`, `tail(n)`, `distinct()` | +| Shape | `select(...)`, `filter([predicate])` | +| Correlate | `decorateByTime(...)`, `decorateByKey(...)` | +| Value transforms | `len`, `uppercase`, `lowercase`, `trim`, `abs`, `round`, `floor`, `ceil`, `contains`, `replace`, `formatDuration`, `asDateTime` | +| Maps | `toMap(key, value)`, `merge(...)` | + +Two rules that cause most failures: + +1. **`sortBy` and `top` default to descending.** Pass `asc=true` for ascending — this matters + for time series, where `sortBy(startTime)` gives you the recording backwards. +2. **`filter()` takes a bracketed predicate**, unlike root filters: + `groupBy(path, agg=sum, value=bytes) | filter([sum>1048576])`. + +Terminal aggregations consume the stream and cannot be chained with each other. + +### select() + +Supports aliases, arithmetic, string concatenation, `"${expr}"` templates, and the +scope functions `if()`, `upper()`, `lower()`, `substring()`, `length()`, `coalesce()`, +`asDateTime()`, `truncate(field,"second|minute|hour|day|week|month")`, `formatDuration()`. + +``` +events/jdk.FileRead | select(path, formatDuration(duration) as dur) | sortBy(duration) | top(10) +``` + +### Correlation + +``` +decorateByTime(, fields=f1,f2 [, threadPath=] [, decoratorThreadPath=]) +decorateByKey(, key=, decoratorKey=, fields=f1,f2) +``` + +`decorateByTime` matches events overlapping in time **on the same thread** (thread path +defaults to `eventThread/javaThreadId`). `decorateByKey` joins on a shared correlation id — +prefer it when one exists, as it is exact and cheaper. Decorated fields are read with the +`$decorator.` prefix and work in `groupBy`, `select` and filters. + +## HdumpPath — `hdump_query` + +**Roots**: `objects`, `classes`, `gcroots`, `clusters`, `duplicates`, `ages`. + +Type specs accept exact names, globs (`java.util.*`), `instanceof/` for subclass matching, +and array forms (`int[]` or `[I`). Size units `K/KB/M/MB/G/GB` work in predicates. + +Sorting takes a direction word: `sortBy(retained desc)`, `sortBy(name asc)`, and multiple +fields: `sortBy(class asc, shallow desc)`. + +Analysis operators unique to heap dumps: `pathToRoot()`, `retentionPaths()`, `dominators()`, +`retainedBreakdown()`, `checkLeaks(detector=…)`, `waste()`, `cacheStats()`, `threadOwner()`, +`dominatedSize()`, `estimateAge()`, `whatif()`, and the cross-session `join(session=…[, +root=…][, by=…])`. + +``` +classes | sortBy(retained desc) | top(20) +objects/java.util.HashMap | waste() | filter(loadFactor < 0.1) | top(20) +clusters | sortBy(score desc) | top(10) +``` + +## SamplesPath — `pprof_query` and `otlp_query` + +pprof and OTLP share one grammar with a single root, `samples`. + +Fields: one per profile sample type (`cpu`, `alloc_objects`, …), `stackTrace` as a leaf-first +list addressable by index (`stackTrace/0/name`), plus label keys such as `thread`. + +Operators: `count`, `top`, `groupBy`, `stats`, `head`, `tail`, `filter`/`where`, `select`, +`sortBy`/`sort`/`orderby`, `stackprofile`, `distinct`/`unique`. There is **no** `join` and no +cross-session operator for these formats. + +## When a query fails + +1. Read the error position — the parser reports `[at N]`, an index into your query string. +2. Check bracket versus parenthesis: root filters use `[...]`, the `filter()` operator takes + `filter([...])`. +3. Ask the server rather than guessing: `jfr_help topic=filters|pipeline|functions|examples`, + `hdump_help`, `pprof_help`, `otlp_help`. +4. Verify the field exists before blaming syntax: `jfr_list_types filter=` then + `jfr_query query="metadata/"` to see the field names. diff --git a/plugins/jafar-perf/skills/latency/SKILL.md b/plugins/jafar-perf/skills/latency/SKILL.md new file mode 100644 index 0000000..1bc67fc --- /dev/null +++ b/plugins/jafar-perf/skills/latency/SKILL.md @@ -0,0 +1,114 @@ +--- +name: latency +description: Investigate response-time problems that are not CPU-bound — lock contention, thread parking, executor queue saturation, blocking I/O, and per-endpoint latency attribution. Use when the user reports slow requests, p99 spikes, timeouts, deadlock suspicion, or when triage shows threads blocked rather than running. +allowed-tools: mcp__jafar__jfr_tsa mcp__jafar__jfr_use mcp__jafar__jfr_query mcp__jafar__jfr_list_types mcp__jafar__jfr_stackprofile mcp__jafar__pprof_tsa +--- + +# Latency analysis + +Latency problems are usually *waiting*, and waiting is invisible to CPU profiling. A thread +blocked on a monitor produces no execution samples; the flamegraph looks healthy while the +p99 is ruined. + +## 1. Where is the time spent not running? + +``` +jfr_tsa correlateBlocking=true +``` + +Thread State Analysis returns: + +- `stateDistribution` — the share of thread time in each state. This is the headline number. +- `threadProfiles` and `topThreadsByState` — which threads, not just how many. +- `correlations` — monitor classes and executor queues implicated in blocking. +- `insights.problematicThreads[]` — each with its own `recommendation`. + +Read `stateDistribution` first. If most time is `RUNNABLE`, this is a CPU problem — switch to +the `cpu` skill. If it is dominated by blocked, waiting or parked states, continue here. + +## 2. Which resource is saturated? + +``` +jfr_use resources=all +``` + +The USE method (Utilization, Saturation, Errors) applied to CPU, memory, threads and I/O. +Each resource carries an `assessment`; `insights.bottlenecks[]` names the saturated ones as +`cpu_saturation`, `memory_pressure`, `thread_contention` or `queue_saturation`. + +`queue_saturation` is the one most often missed: an executor whose queue depth grows means +requests wait before any code runs for them. No amount of method optimisation fixes it. + +Narrow the window when the recording spans a mix of load levels: + +``` +jfr_use startTime= endTime= resources=threads +``` + +## 3. Which lock? + +Monitor contention shows up as `jdk.JavaMonitorEnter` (blocked acquiring) and +`jdk.JavaMonitorWait` (waiting on a condition). They mean different things — do not merge +them. + +``` +jfr_query query="events/jdk.JavaMonitorEnter | groupBy(monitorClass, agg=sum, value=duration) | top(10, by=value)" +``` + +To find *what code* was contending, correlate execution samples with the wait window on the +same thread: + +``` +jfr_query query="events/jdk.ExecutionSample | decorateByTime(jdk.JavaMonitorWait, fields=monitorClass,duration) | groupBy($decorator.monitorClass, agg=count) | top(10, by=count)" +``` + +`decorateByTime` joins events that overlap in time **on the same thread** (thread path +defaults to `eventThread/javaThreadId`). Rows where `$decorator.monitorClass` is null were +sampled outside any wait — that is the uncontended baseline, and it belongs in the report +as the comparison. + +## 4. Parking and sleeping + +`jdk.ThreadPark` covers `LockSupport.park`, which is what every `java.util.concurrent` lock, +queue and future uses. Group by the parked class to tell a healthy idle pool from a stalled +one: + +``` +jfr_query query="events/jdk.ThreadPark | groupBy(parkedClass/name, agg=sum, value=duration) | top(10, by=value)" +``` + +A thread pool parked on its own work queue is idle and healthy. A request thread parked on a +`CompletableFuture` or a connection pool is a latency bug. + +## 5. Per-endpoint attribution + +When the recording carries request context (a Datadog profiler's `datadog.Endpoint`, or your +own event type), attribute waiting to the endpoint that suffered it: + +``` +jfr_query query="events/jdk.JavaMonitorEnter | decorateByKey(datadog.Endpoint, key=localRootSpanId, decoratorKey=localRootSpanId, fields=endpoint) | groupBy($decorator.endpoint, agg=sum, value=duration)" +``` + +`decorateByKey` is a correlation-key join, not a time join — use it whenever a shared id +exists, because it is both cheaper and exact. + +## 6. Blocking I/O + +``` +jfr_query query="events/jdk.SocketRead[duration > 10ms] | groupBy(address, agg=sum, value=duration) | top(10, by=value)" +jfr_query query="events/jdk.FileRead[duration > 10ms] | groupBy(path, agg=count) | top(10, by=count)" +``` + +Filters accept duration literals (`10ms`, `1s`) and size units. Slow I/O to one address is a +dependency problem, not a JVM problem — say so plainly rather than proposing JVM tuning. + +## What not to conclude + +- A high *count* of monitor events is not contention; a high *summed duration* relative to + the recording's wall clock is. Always divide by the recording duration. +- JFR's monitor events have a duration threshold (commonly 10 ms or 20 ms depending on + settings). Contention below the threshold is invisible, so absence of events is not + absence of contention. Check `jdk.ActiveSetting` if the threshold matters to the + conclusion. +- `jfr_tsa` correlations are associations in time, not proof of causation. Report them as + "concurrent with", and prove causation with a code path or a fix that measurably helps. diff --git a/plugins/jafar-perf/skills/memory-leak/SKILL.md b/plugins/jafar-perf/skills/memory-leak/SKILL.md new file mode 100644 index 0000000..733343f --- /dev/null +++ b/plugins/jafar-perf/skills/memory-leak/SKILL.md @@ -0,0 +1,138 @@ +--- +name: memory-leak +description: Hunt memory leaks and wasted heap in a Java heap dump (HPROF) — retained sizes, dominator tree, GC root paths, known leak patterns, duplicate strings, collection waste, and correlating retained objects back to their JFR allocation sites. Use for OutOfMemoryError, heap that never comes back after GC, container OOM kills, or any .hprof file. +allowed-tools: mcp__jafar__hdump_open mcp__jafar__hdump_summary mcp__jafar__hdump_report mcp__jafar__hdump_query mcp__jafar__hdump_help mcp__jafar__hdump_close +--- + +# Memory leak analysis + +A leak is *unintended retention*: objects reachable from a GC root that the program will +never use again. Heap dumps show what is retained and by whom. They cannot show intent — so +the deliverable is always "X is retained by Y along path Z", plus a judgement about whether +that retention is intended. + +## 1. Open and orient + +``` +hdump_open path=/abs/path/dump.hprof +hdump_summary +``` + +`hdump_summary` is deliberately fast: it does not compute retained sizes. It gives object and +class counts, total heap size, top classes by shallow size, and GC root types. + +## 2. Run the health report first + +``` +hdump_report focus=leaks +``` + +Returns severity-ranked findings — `CRITICAL`, `WARNING`, `INFO` — each with a `category`, +`title`, `description`, `retainedSize`, `affectedObjects`, an `action`, and a follow-up +`query` you can run directly. Start from the highest severity with a large `retainedSize`. + +Other focuses: `waste`, `duplicates`, `histogram`. + +## 3. Shallow versus retained + +This distinction decides the whole investigation: + +- **Shallow size** — the object's own bytes. `char[]` and `byte[]` always dominate; that is + never itself a finding. +- **Retained size** — everything that becomes collectable if this object goes. This is what + a leak is measured in. + +Retained sizes need the dominator tree, which is computed on demand and cached in an on-disk +index, so the first query that needs it is slow and later ones are fast. + +``` +hdump_query query="classes | sortBy(retained desc) | top(20)" +hdump_query query="objects | dominators() | sortBy(retained desc) | top(20)" +``` + +## 4. Named detectors + +Six known patterns, each answering "is this the usual suspect?": + +``` +hdump_query query="objects | checkLeaks(detector=threadlocal-leak)" +``` + +| Detector | Finds | +|---|---| +| `threadlocal-leak` | `ThreadLocal` values held by pooled threads after the request ended | +| `classloader-leak` | Class loaders kept alive after undeploy/redeploy | +| `duplicate-strings` | Identical string values held separately | +| `growing-collections` | Collections far larger than their live content | +| `listener-leak` | Registered listeners never unregistered | +| `finalizer-queue` | Objects piled up awaiting finalization | + +Detectors find *known* patterns. For unknown ones, use graph structure: + +``` +hdump_query query="clusters | sortBy(score desc) | top(10)" +``` + +`clusters` finds densely-connected subgraphs with large retained size and weak external +anchoring — the shape a leak has when nobody wrote a detector for it. Drill in with +`clusters[id = N] | objects | sortBy(retained desc)`. + +## 5. Prove retention with a path to a GC root + +A finding without a root path is a guess. This is the single most important step: + +``` +hdump_query query="objects/com.example.CacheEntry | pathToRoot() | head(5)" +hdump_query query="classes/com.example.CacheEntry | retentionPaths()" +``` + +`pathToRoot()` gives the chain per object; `retentionPaths()` merges paths at class level, +which is what you want when thousands of instances leak through the same field. The path +names the field that holds the reference — that field is the fix. + +## 6. Waste that is not a leak + +Not all recoverable memory is leaked. These are often larger and easier to fix: + +``` +hdump_query query="objects/java.util.HashMap | waste() | sortBy(wastedBytes desc) | top(20)" +hdump_query query="duplicates | sortBy(wastedBytes desc) | top(20)" +hdump_query query="objects/com.example.Cache | cacheStats()" +``` + +`waste()` reports over-allocated capacity (a 1024-slot map holding 3 entries). `duplicates` +finds structurally identical subgraphs, which is a stronger signal than duplicate strings +alone. `cacheStats()` gives `fillRatio` and `costPerEntry` for cache-shaped objects. + +## 7. Who allocated it — heap plus JFR + +This is the question a heap dump alone cannot answer, and Jafar's differentiator: the heap +shows *what* is retained, JFR shows *who* created it. + +``` +jfr_open path=/abs/path/recording.jfr alias=rec +hdump_open path=/abs/path/dump.hprof +hdump_query query="classes | join(session=rec, root=\"jdk.ObjectAllocationSample\") | filter(retained > 10MB) | select(name, retained, allocCount, topAllocSite)" +``` + +Adds `allocCount`, `allocWeight`, `allocRate`, `topAllocSite` and `survivalRatio`. +`topAllocSite` is the method to fix. A high `allocCount` with low `retained` is churn — a +`gc` problem, not a leak. Low `allocCount` with high `retained` is a leak of few, large, +long-lived objects. + +Both sessions must be open in the same server for the join to resolve. + +## 8. Two dumps beat one + +Single-snapshot analysis is inference. Two snapshots are proof — see the `heap-diff` skill. + +## What not to conclude + +- `byte[]`/`char[]`/`String` at the top of a shallow histogram is normal in every Java heap. + Only their *dominator* is a finding. +- A large retained size is not a leak if the retention is intended. A 2 GB cache that is + configured to be 2 GB is working correctly; the finding is that it is too large for the + container, which is a different recommendation. +- A dump taken without a preceding full GC contains garbage that is simply not yet + collected. Check whether the dump was triggered on OOM (post-GC, trustworthy) or taken ad + hoc (may overstate retention). diff --git a/plugins/jafar-perf/skills/report/SKILL.md b/plugins/jafar-perf/skills/report/SKILL.md new file mode 100644 index 0000000..3faf85e --- /dev/null +++ b/plugins/jafar-perf/skills/report/SKILL.md @@ -0,0 +1,108 @@ +--- +name: report +description: The output format and evidence discipline for any performance finding produced with the Jafar tools. Use whenever writing up an analysis, summarising an investigation, answering "what did you find", or handing conclusions to another person or agent. +--- + +# Reporting a performance finding + +A performance report is an argument, and an argument needs evidence. The reader must be able +to re-run every number you quote. That is the whole standard. + +## Format + +Report findings ranked by impact, each in this shape: + +> **Symptom** — what the user or the system observes. +> +> **Evidence** — the exact tool call and the numbers it returned. +> +> **Interpretation** — what the numbers mean, and why this explanation rather than another. +> +> **Recommendation** — the specific change, at a named location. +> +> **Confidence** — high / medium / low, and what would raise it. + +Keep it short. Three well-evidenced findings beat twelve speculative ones. + +## The evidence rule + +Every quantitative claim names the tool call that produced it: + +> `jdk.JavaMonitorEnter` on `com.example.SessionCache` accounts for 41.2 s of blocked time +> across a 300 s recording (13.7% of wall clock). +> Evidence: `jfr_query query="events/jdk.JavaMonitorEnter | groupBy(monitorClass, agg=sum, value=duration) | top(10, by=value)"` → `SessionCache` 41,203,441,000 ns; recording duration from `timerange()` = 300.4 s. + +If you cannot name the call, you cannot make the claim. Delete it or go and measure it. + +## Rates, not counts + +Absolute counts are meaningless without the recording duration, and misleading when +comparing recordings of different lengths. Convert: + +- events → events per second +- durations → percentage of wall clock, or of the thread's own time +- allocation → MB/s +- samples → percentage of total samples + +State the denominator you used. + +## Confidence, honestly + +| Level | When | +|---|---| +| **High** | Direct measurement of the thing itself, large sample, corroborated by a second independent tool | +| **Medium** | Strong single-tool signal, or an inference from a well-understood mechanism | +| **Low** | Heuristic, small sample, correlation in time only, or a known-approximate source | + +Things that force *at most* medium confidence, and must be said out loud: + +- Sampled data (execution samples, allocation samples) — you have a sample, not a census. +- `decorateByTime` correlations — concurrency in time is not causation. +- pprof and OTLP thread states — inferred from function-name keywords, not real states. +- Retained sizes from an approximate dominator computation. +- Any recording where the relevant profiling was not enabled — see below. + +## Absence of evidence + +When the recording cannot answer the question, say so explicitly and separately from the +findings. "No allocation hotspots found" is wrong if allocation profiling was off; the true +statement is "allocation profiling was not enabled in this recording, so allocation was not +assessed", plus how to enable it next time. + +`jfr_diagnose` reports these as capability gaps. Carry them into the report rather than +silently dropping them. + +## Reproducibility + +End with the exact steps, so the reader can reproduce the result: + +``` +jfr_open path=/abs/path/recording.jfr +jfr_diagnose +jfr_query query="events/jdk.JavaMonitorEnter | groupBy(monitorClass, agg=sum, value=duration) | top(10, by=value)" +``` + +For a regression claim, both artifacts and the comparison call are the reproduction — see +the `compare` skill. + +## What a recommendation must contain + +Not "reduce allocations" but: the file and line, the change, and the expected effect with +its basis. + +> `OrderService.reprice` (`src/main/java/com/example/OrderService.java:118`) allocates a new +> `HashMap` per call inside the pricing loop; it accounts for 34% of sampled allocation +> weight. Hoisting it out of the loop, or presizing it, should remove most of that share. +> Expected effect is on allocation rate and young-GC frequency, not on p99 directly — +> confirm with a before/after `jfr_compare`. + +If you did not locate the code, say that you did not, and give the frame instead of +inventing a path. + +## Never + +- Do not report a number you did not measure in this session. +- Do not present a threshold breach as a diagnosis. `jfr_diagnose` applies fixed thresholds + that know nothing about this service's normal behaviour; a breach is a lead. +- Do not claim an improvement without a measured comparison. "This should be faster" is a + hypothesis, and must be labelled as one. diff --git a/plugins/jafar-perf/skills/triage/SKILL.md b/plugins/jafar-perf/skills/triage/SKILL.md new file mode 100644 index 0000000..9cef91b --- /dev/null +++ b/plugins/jafar-perf/skills/triage/SKILL.md @@ -0,0 +1,92 @@ +--- +name: triage +description: First step for any unfamiliar JFR recording, pprof profile, OTLP profile, or heap dump. Establishes what the artifact contains, what is anomalous, and which specialised playbook to run next. Use when the user says "analyse this recording", "what is wrong with this JVM", "why is this slow", or hands over a .jfr/.hprof/.pprof/.otlp file without a specific question. +allowed-tools: mcp__jafar__jfr_open mcp__jafar__jfr_summary mcp__jafar__jfr_diagnose mcp__jafar__jfr_list_types mcp__jafar__hdump_open mcp__jafar__hdump_summary mcp__jafar__hdump_report mcp__jafar__pprof_open mcp__jafar__pprof_summary mcp__jafar__otlp_open mcp__jafar__otlp_summary +--- + +# Triage + +Establish the shape of the problem before investigating it. Never open with a flamegraph: +a flamegraph of a recording that is 90% idle wastes a turn and misleads. + +## 0. Identify the artifact + +| Extension / magic | Tool family | Notes | +|---|---|---| +| `.jfr` | `jfr_*` | Java Flight Recording | +| `.hprof`, `.hdump` | `hdump_*` | Java heap dump | +| `.pprof`, `.pb.gz` | `pprof_*` | pprof profile (async-profiler, Go, Rust) | +| `.otlp` | `otlp_*` | OpenTelemetry profiles | + +All four families share the same session model: `*_open` returns a session id, every other +tool defaults to the most recently opened session, `*_close` releases it. You may hold +sessions of several types at once — that is what makes correlation possible (see +`heap-diff` and the `join` operator). + +## 1. Open and summarise + +``` +jfr_open path=/abs/path/recording.jfr +jfr_summary +``` + +`jfr_summary` is a single pass over the recording. Read three things from it: + +- `totalEvents` and `totalEventTypes` — is this a real workload or a 200-event smoke test? +- `topEventTypes` — the profile of the profile. A recording dominated by + `jdk.ObjectAllocationSample` is a different investigation from one dominated by + `jdk.ExecutionSample`. +- `highlights` — pre-computed `gc`, `exceptions` and `cpu` blocks. + +## 2. Diagnose + +``` +jfr_diagnose +``` + +Returns `findings[]` and `recommendations[]`, and runs the USE and TSA analyses in-process so +the resource and thread-state picture arrives with the first call. Treat its output as a +*routing decision*, not a conclusion — it applies fixed thresholds and knows nothing about +your service's normal behaviour. + +Read `capabilityGaps` before you believe a negative result. "ALLOCATION PROFILING: Not +enabled in this recording" means you cannot conclude anything about allocation, not that +allocation is fine. + +## 3. Route + +| What triage shows | Go to | +|---|---| +| High CPU sample count, hot leaf methods | `cpu` | +| Threads blocked, parked, or in monitor waits; queue saturation | `latency` | +| High GC pressure, high allocation rate, growing heap | `gc` | +| A heap dump, `OutOfMemoryError`, or memory that never comes back | `memory-leak` | +| Two recordings / two dumps of the same workload | `compare` or `heap-diff` | +| A specific question the built-in tools do not answer | `jfrpath` | + +Run more than one when triage flags more than one. They are independent. + +## 4. Establish the denominator + +Before quantifying anything, know the recording's wall-clock duration. Every absolute count +in a JFR recording is meaningless without it — 10,000 exceptions in 30 seconds and 10,000 +exceptions in 4 hours are different problems. + +``` +jfr_query query="events/jdk.ExecutionSample | timerange()" +``` + +Report rates, not raw counts, in anything the user reads. + +## 5. Sampling is not measurement + +`jdk.ExecutionSample` and `jdk.ObjectAllocationSample` are samples. A method with 3 samples +out of 20,000 is noise. Percentages below roughly 1% of total samples should not drive a +recommendation unless the sample count is very large. `jfr_stackprofile` marks frames with +a `category` field for this reason — prefer frames it calls `hotspot` or `steady-hotspot`. + +## 6. Hand off + +Write down, before moving on: the artifact path, its duration, the total event count, and +the two or three findings worth pursuing. The `report` skill defines the format. Every +subsequent claim must trace back to a tool call recorded here. diff --git a/plugins/jfr-analyzer/.mcp.json b/plugins/jfr-analyzer/.mcp.json index de93d77..d0153d6 100644 --- a/plugins/jfr-analyzer/.mcp.json +++ b/plugins/jfr-analyzer/.mcp.json @@ -1,8 +1,13 @@ { "mcpServers": { "jfr-mcp": { - "type": "sse", - "url": "http://localhost:3000/mcp/sse" + "type": "stdio", + "command": "jbang", + "args": [ + "jfr-mcp@btraceio", + "--stdio", + "--attach" + ] } } } diff --git a/plugins/jfr-analyzer/README.md b/plugins/jfr-analyzer/README.md index 6d6fb42..a38f033 100644 --- a/plugins/jfr-analyzer/README.md +++ b/plugins/jfr-analyzer/README.md @@ -7,12 +7,14 @@ probes, using one target/window/evidence record across the three instruments. ## Host support -The shared skills work in Claude Code, Codex, and Pi. The plugin expects a Jafar MCP server at `http://localhost:3000/mcp/sse`: +The shared skills work in Claude Code, Codex, and Pi. The plugin's `.mcp.json` registers the Jafar MCP server as `jfr-mcp`, started through JBang and attached to one shared daemon that starts on demand: ```bash -jbang jafar-mcp@btraceio +jbang jfr-mcp@btraceio --stdio --attach ``` +`--attach` needs a `jfr-mcp` release that has the flag; an older one ignores it and runs a private server, which still works. + Configure the same `jfr-mcp` server in the host when automatic plugin MCP loading is unavailable. The skills refer to the server by capability (`jfr_open`, `jfr_query`, `jfr_summary`, and related tools); host-specific MCP namespaces may differ. ## Install async-profiler diff --git a/scripts/check-tool-references.js b/scripts/check-tool-references.js new file mode 100755 index 0000000..4b7c6a9 --- /dev/null +++ b/scripts/check-tool-references.js @@ -0,0 +1,172 @@ +#!/usr/bin/env node +// Fails when a skill or agent names an MCP tool the jfr-mcp server does not expose. +// +// Skills and agents in the plugins that use jfr-mcp name its tools explicitly, and the server is +// developed in another repository (btraceio/jafar). Nothing in either repository's tests connects +// the two, so a tool rename there turns a skill here into confident instructions for a call that +// fails. This script is the connection. +// +// Ground truth is the server itself: started, handshaken, and asked `tools/list`. Parsing its Java +// would encode assumptions about how tools are registered today, and would miss a tool that fails +// to register at runtime. Asking the server is exactly what an MCP client does. +// +// Usage: +// node scripts/check-tool-references.js # jbang --fresh jfr-mcp@btraceio --stdio +// node scripts/check-tool-references.js --jar path/to.jar # a locally built shadow jar +// node scripts/check-tool-references.js --command "..." # any command speaking MCP on stdio +// +// Needs no attach/daemon: it runs the server as a plain stdio process and stops it afterwards. +const fs = require('fs'); +const path = require('path'); +const { spawn } = require('child_process'); + +const root = path.resolve(__dirname, '..'); + +// The four namespaces the server's tools live in. A token with one of these prefixes is a tool +// reference; anything else in the prose is not our business. +const TOOL_TOKEN = /\b(?:jfr|hdump|pprof|otlp)_[a-z_]+\b/g; +// Names that look like tools but are not: the field names of the "shared evidence record" in the +// jfr-analyzer interop skill. They are data the agent writes, not calls it makes. +const NOT_TOOLS = new Set(['jfr_session', 'jfr_window', 'jfr_evidence']); +// Agents declare their allowlist as `mcp____`. +const MCP_QUALIFIED = /\bmcp__[a-z0-9_]+__([a-z0-9_]+)\b/g; + +const HANDSHAKE = [ + { jsonrpc: '2.0', id: 1, method: 'initialize', params: { protocolVersion: '2024-11-05', capabilities: {}, clientInfo: { name: 'tool-drift-check', version: '1' } } }, + { jsonrpc: '2.0', method: 'notifications/initialized' }, + { jsonrpc: '2.0', id: 2, method: 'tools/list', params: {} }, +]; + +function parseArgs(argv) { + // --fresh, because jbang caches the catalog: without it a stale alias answers with an old server + // and the check reports drift that is not there. + const args = { command: 'jbang --fresh jfr-mcp@btraceio --stdio', timeout: 180 }; + for (let i = 2; i < argv.length; i++) { + if (argv[i] === '--jar') args.command = `java -jar ${argv[++i]} --stdio`; + else if (argv[i] === '--command') args.command = argv[++i]; + else if (argv[i] === '--timeout') args.timeout = Number(argv[++i]); + else throw new Error(`unknown argument: ${argv[i]}`); + } + return args; +} + +// Plugins that start jfr-mcp, found from their own .mcp.json rather than a hard-coded list. +function jfrMcpPlugins() { + const found = []; + for (const entry of fs.readdirSync(path.join(root, 'plugins'), { withFileTypes: true })) { + const mcp = path.join(root, 'plugins', entry.name, '.mcp.json'); + if (entry.isDirectory() && fs.existsSync(mcp) && fs.readFileSync(mcp, 'utf8').includes('jfr-mcp')) found.push(entry.name); + } + return found.sort(); +} + +function referencedTools(plugins) { + const found = new Map(); + const files = []; + for (const plugin of plugins) { + const base = path.join(root, 'plugins', plugin); + const skills = path.join(base, 'skills'); + if (fs.existsSync(skills)) { + for (const d of fs.readdirSync(skills, { withFileTypes: true })) { + const file = path.join(skills, d.name, 'SKILL.md'); + if (d.isDirectory() && fs.existsSync(file)) files.push(file); + } + } + const agents = path.join(base, 'agents'); + if (fs.existsSync(agents)) { + for (const f of fs.readdirSync(agents)) if (f.endsWith('.md')) files.push(path.join(agents, f)); + } + } + if (files.length === 0) throw new Error('no skill or agent files found for any jfr-mcp plugin'); + for (const file of files.sort()) { + const rel = path.relative(root, file); + fs.readFileSync(file, 'utf8').split('\n').forEach((line, i) => { + const names = new Set([...line.matchAll(MCP_QUALIFIED)].map(m => m[1])); + // Strip the qualified forms before scanning for bare ones, so a `tools:` line does not + // report the same name twice. + for (const m of line.replace(MCP_QUALIFIED, '').matchAll(TOOL_TOKEN)) { + if (!NOT_TOOLS.has(m[0])) names.add(m[0]); + } + for (const name of names) { + if (!found.has(name)) found.set(name, []); + found.get(name).push(`${rel}:${i + 1}`); + } + }); + } + return found; +} + +// Starts the server, sends the handshake and `tools/list`, and resolves with what it advertises. +function serverTools(command, timeoutSeconds) { + return new Promise((resolve, reject) => { + const child = spawn(command, { shell: true, stdio: ['pipe', 'pipe', 'pipe'] }); + let buffer = ''; + let stderr = ''; + let info = { name: '?', version: '?' }; + let done = false; + const finish = (fn, value) => { + if (done) return; + done = true; + clearTimeout(timer); + child.kill(); + fn(value); + }; + const timer = setTimeout( + () => finish(reject, new Error(`no tools/list response within ${timeoutSeconds}s\nstderr:\n${stderr.slice(-2000)}`)), + timeoutSeconds * 1000); + child.stderr.on('data', d => { stderr += d; }); + child.on('error', e => finish(reject, e)); + child.on('exit', code => finish(reject, new Error(`server exited (${code}) before answering tools/list\nstderr:\n${stderr.slice(-2000)}`))); + child.stdout.on('data', chunk => { + buffer += chunk; + let nl; + while ((nl = buffer.indexOf('\n')) >= 0) { + const line = buffer.slice(0, nl).trim(); + buffer = buffer.slice(nl + 1); + if (!line.startsWith('{')) continue; + let msg; + try { msg = JSON.parse(line); } catch { continue; } + if (msg.id === 1 && msg.result) info = { name: msg.result.serverInfo?.name ?? '?', version: msg.result.serverInfo?.version ?? '?' }; + if (msg.id === 2 && msg.result) finish(resolve, { tools: new Set((msg.result.tools || []).map(t => t.name)), info }); + } + }); + for (const m of HANDSHAKE) child.stdin.write(JSON.stringify(m) + '\n'); + }); +} + +async function main() { + const args = parseArgs(process.argv); + const plugins = jfrMcpPlugins(); + const referenced = referencedTools(plugins); + + console.log(`plugins using jfr-mcp: ${plugins.join(', ')}`); + console.log(`asking the server for its tools: ${args.command}`); + const { tools, info } = await serverTools(args.command, args.timeout); + console.log(`server exposes ${tools.size} tools (reported as ${info.name} ${info.version})\n`); + + const missing = [...referenced].filter(([name]) => !tools.has(name)); + if (missing.length > 0) { + console.log('FAIL: these tools are named here but the server does not expose them:\n'); + for (const [name, locations] of missing.sort()) { + console.log(` ${name}`); + for (const loc of locations) console.log(` ${loc}`); + } + console.log('\nThree things cause this, in rough order of likelihood:\n'); + console.log(' 1. This repository is ahead of the published server: the tool exists in'); + console.log(' btraceio/jafar but is not in a release yet. Release it, or hold the skill'); + console.log(' back until it is. Nothing here is wrong; the two are just out of step.'); + console.log(' 2. The tool was renamed or removed upstream, and the skill needs updating.'); + console.log(' 3. The name is a typo.\n'); + console.log('In all three cases an agent following that skill makes a call that fails.'); + process.exit(1); + } + + console.log(`OK: all ${referenced.size} referenced tools exist on the server.`); + const unused = [...tools].filter(t => !referenced.has(t)).sort(); + if (unused.length > 0) { + // Not a failure: a tool no skill mentions is a coverage gap, not a broken reference. + console.log(`\n${unused.length} server tools are not mentioned by any skill or agent:\n ${unused.join(' ')}`); + } +} + +main().catch(e => { console.error(e.message); process.exit(2); }); diff --git a/scripts/validate-repository.js b/scripts/validate-repository.js index 0958fff..de0f8e7 100755 --- a/scripts/validate-repository.js +++ b/scripts/validate-repository.js @@ -45,7 +45,8 @@ function walk(dir) { walk(path.join(root, 'plugins')); for (const file of skillFiles) { const body = fs.readFileSync(file, 'utf8'); - if (!/^---\nname: [a-z0-9-]+\ndescription: .+\n---/m.test(body)) errors.push(`${path.relative(root, file)}: invalid front matter`); + // name and description first, then any other single-line fields (e.g. allowed-tools) before the closing rule. + if (!/^---\nname: [a-z0-9-]+\ndescription: .+\n(?:[a-z][a-z-]*: .+\n)*---/m.test(body)) errors.push(`${path.relative(root, file)}: invalid front matter`); } if (errors.length) {