From 3b30fc43af933ca0baba59312402fcfbdd43e3ef Mon Sep 17 00:00:00 2001 From: Jaroslav Bachorik Date: Sun, 12 Jul 2026 22:24:46 +0200 Subject: [PATCH 1/3] feat: add JFR and performance engineering plugins --- .agents/plugins/marketplace.json | 24 + .claude-plugin/marketplace.json | 12 + README.md | 2 + docs/perf-engineer-design.md | 537 ++++++++++++++++++ package.json | 4 +- plugins/btrace-observability/README.md | 14 +- .../skills/btrace-mcp-operations/SKILL.md | 13 + .../skills/btrace-observability/SKILL.md | 8 + .../jfr-analyzer/.claude-plugin/plugin.json | 36 ++ .../jfr-analyzer/.codex-plugin/plugin.json | 36 ++ plugins/jfr-analyzer/.gitignore | 13 + plugins/jfr-analyzer/.mcp.json | 8 + plugins/jfr-analyzer/README.md | 88 +++ plugins/jfr-analyzer/agents/perf-engineer.md | 152 +++++ .../jfr-analyzer/eval/corpus/manifest.json | 362 ++++++++++++ .../jfr-analyzer/eval/scripts/preflight.sh | 133 +++++ plugins/jfr-analyzer/eval/scripts/regen.sh | 126 ++++ .../eval/scripts/requirements.txt | 3 + plugins/jfr-analyzer/eval/scripts/score.py | 420 ++++++++++++++ .../eval/testapp/AllocGcCascade.java | 52 ++ .../eval/testapp/AllocationPressure.java | 19 + .../eval/testapp/CfFanoutLargeHeap.java | 34 ++ .../jfr-analyzer/eval/testapp/CpuHotspot.java | 23 + .../eval/testapp/G1HumongousAlloc.java | 19 + .../eval/testapp/GcCpuInflation.java | 28 + .../eval/testapp/GcPressureParallel.java | 27 + .../eval/testapp/GcQueueCoupling.java | 28 + .../eval/testapp/HealthyBaseline.java | 11 + .../eval/testapp/ReactiveBackpressure.java | 23 + .../eval/testapp/ReactiveNoBackpressure.java | 22 + .../jfr-analyzer/eval/testapp/Safepoints.java | 23 + .../eval/testapp/TemporalSpikes.java | 22 + .../eval/testapp/ThreadContentionSync.java | 34 ++ .../eval/testapp/VirtualThreadPinning.java | 44 ++ .../eval/testapp/ZgcAllocationStalls.java | 27 + plugins/jfr-analyzer/eval/testapp/main.java | 93 +++ .../skills/async-profiler-interop/SKILL.md | 103 ++++ .../jfr-analyzer/skills/jfr-analyzer/SKILL.md | 59 ++ .../jfr-analyzer/subskills/drilldown/SKILL.md | 287 ++++++++++ .../jfr-analyzer/subskills/eval/SKILL.md | 149 +++++ .../jfr-analyzer/subskills/report/SKILL.md | 212 +++++++ .../jfr-analyzer/subskills/triage/SKILL.md | 440 ++++++++++++++ .../skills/jfr-btrace-interop/SKILL.md | 70 +++ .../perf-engineer/.claude-plugin/plugin.json | 36 ++ .../perf-engineer/.codex-plugin/plugin.json | 36 ++ plugins/perf-engineer/.gitignore | 2 + plugins/perf-engineer/README.md | 34 ++ .../scripts/perf-engineer-session.py | 365 ++++++++++++ .../scripts/test_perf_engineer_session.py | 162 ++++++ .../skills/perf-engineer/SKILL.md | 156 +++++ .../references/evidence-and-gates.md | 159 ++++++ 51 files changed, 4786 insertions(+), 4 deletions(-) create mode 100644 docs/perf-engineer-design.md create mode 100644 plugins/jfr-analyzer/.claude-plugin/plugin.json create mode 100644 plugins/jfr-analyzer/.codex-plugin/plugin.json create mode 100644 plugins/jfr-analyzer/.gitignore create mode 100644 plugins/jfr-analyzer/.mcp.json create mode 100644 plugins/jfr-analyzer/README.md create mode 100644 plugins/jfr-analyzer/agents/perf-engineer.md create mode 100644 plugins/jfr-analyzer/eval/corpus/manifest.json create mode 100755 plugins/jfr-analyzer/eval/scripts/preflight.sh create mode 100755 plugins/jfr-analyzer/eval/scripts/regen.sh create mode 100644 plugins/jfr-analyzer/eval/scripts/requirements.txt create mode 100644 plugins/jfr-analyzer/eval/scripts/score.py create mode 100644 plugins/jfr-analyzer/eval/testapp/AllocGcCascade.java create mode 100644 plugins/jfr-analyzer/eval/testapp/AllocationPressure.java create mode 100644 plugins/jfr-analyzer/eval/testapp/CfFanoutLargeHeap.java create mode 100644 plugins/jfr-analyzer/eval/testapp/CpuHotspot.java create mode 100644 plugins/jfr-analyzer/eval/testapp/G1HumongousAlloc.java create mode 100644 plugins/jfr-analyzer/eval/testapp/GcCpuInflation.java create mode 100644 plugins/jfr-analyzer/eval/testapp/GcPressureParallel.java create mode 100644 plugins/jfr-analyzer/eval/testapp/GcQueueCoupling.java create mode 100644 plugins/jfr-analyzer/eval/testapp/HealthyBaseline.java create mode 100644 plugins/jfr-analyzer/eval/testapp/ReactiveBackpressure.java create mode 100644 plugins/jfr-analyzer/eval/testapp/ReactiveNoBackpressure.java create mode 100644 plugins/jfr-analyzer/eval/testapp/Safepoints.java create mode 100644 plugins/jfr-analyzer/eval/testapp/TemporalSpikes.java create mode 100644 plugins/jfr-analyzer/eval/testapp/ThreadContentionSync.java create mode 100644 plugins/jfr-analyzer/eval/testapp/VirtualThreadPinning.java create mode 100644 plugins/jfr-analyzer/eval/testapp/ZgcAllocationStalls.java create mode 100644 plugins/jfr-analyzer/eval/testapp/main.java create mode 100644 plugins/jfr-analyzer/skills/async-profiler-interop/SKILL.md create mode 100644 plugins/jfr-analyzer/skills/jfr-analyzer/SKILL.md create mode 100644 plugins/jfr-analyzer/skills/jfr-analyzer/subskills/drilldown/SKILL.md create mode 100644 plugins/jfr-analyzer/skills/jfr-analyzer/subskills/eval/SKILL.md create mode 100644 plugins/jfr-analyzer/skills/jfr-analyzer/subskills/report/SKILL.md create mode 100644 plugins/jfr-analyzer/skills/jfr-analyzer/subskills/triage/SKILL.md create mode 100644 plugins/jfr-analyzer/skills/jfr-btrace-interop/SKILL.md create mode 100644 plugins/perf-engineer/.claude-plugin/plugin.json create mode 100644 plugins/perf-engineer/.codex-plugin/plugin.json create mode 100644 plugins/perf-engineer/.gitignore create mode 100644 plugins/perf-engineer/README.md create mode 100644 plugins/perf-engineer/scripts/perf-engineer-session.py create mode 100644 plugins/perf-engineer/scripts/test_perf_engineer_session.py create mode 100644 plugins/perf-engineer/skills/perf-engineer/SKILL.md create mode 100644 plugins/perf-engineer/skills/perf-engineer/references/evidence-and-gates.md diff --git a/.agents/plugins/marketplace.json b/.agents/plugins/marketplace.json index 270bec9..df73f70 100644 --- a/.agents/plugins/marketplace.json +++ b/.agents/plugins/marketplace.json @@ -27,6 +27,30 @@ "authentication": "ON_INSTALL" }, "category": "Developer Tools" + }, + { + "name": "jfr-analyzer", + "source": { + "source": "local", + "path": "./plugins/jfr-analyzer" + }, + "policy": { + "installation": "AVAILABLE", + "authentication": "ON_INSTALL" + }, + "category": "Developer Tools" + }, + { + "name": "perf-engineer", + "source": { + "source": "local", + "path": "./plugins/perf-engineer" + }, + "policy": { + "installation": "AVAILABLE", + "authentication": "ON_INSTALL" + }, + "category": "Productivity" } ] } diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index aee98ef..ea70d35 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -16,6 +16,18 @@ "description": "Goal-first incident diagnosis with safe, focused BTrace probes.", "version": "0.1.0", "source": "./plugins/btrace-observability" + }, + { + "name": "jfr-analyzer", + "description": "Systematic JVM profile investigation with optional BTrace correlation.", + "version": "0.1.0", + "source": "./plugins/jfr-analyzer" + }, + { + "name": "perf-engineer", + "description": "Evidence-driven Java optimization investigations with JFR, BTrace, and JMH.", + "version": "0.1.0", + "source": "./plugins/perf-engineer" } ] } diff --git a/README.md b/README.md index bbdb17b..69d9d79 100644 --- a/README.md +++ b/README.md @@ -11,6 +11,8 @@ one another so the workflow instructions and supporting scripts are maintained o | --- | --- | | `btrace-development` | Repository conventions and build guidance for BTrace development. | | [`btrace-observability`](plugins/btrace-observability/README.md) | A composable SRE skill suite for diagnosing Java applications with BTrace probes. | +| [`jfr-analyzer`](plugins/jfr-analyzer/README.md) | Systematic profile analysis with optional BTrace live-probe correlation. | +| [`perf-engineer`](plugins/perf-engineer/README.md) | Evidence-driven optimization investigations with JFR, BTrace, and JMH. | ## Layout diff --git a/docs/perf-engineer-design.md b/docs/perf-engineer-design.md new file mode 100644 index 0000000..3ba2779 --- /dev/null +++ b/docs/perf-engineer-design.md @@ -0,0 +1,537 @@ +# Perf Engineer Skill + +Status: Draft design + +## Summary + +The `perf-engineer` skill guides an evidence-driven optimization loop for a running Java +application. The user supplies a target PID and, optionally, a source directory. The skill records +the application, sends the recording to `jfr-analyzer`, uses BTrace to characterize selected live +entry points, constructs representative JMH benchmarks, evolves optimization ideas, validates +generated patches, and offers the best candidate as a pull request. + +The skill must keep three concerns separate: + +1. **Observation** finds and characterizes likely bottlenecks. +2. **Optimization search** evolves constrained optimization ideas, not arbitrary source edits. +3. **Implementation validation** turns selected ideas into ordinary code changes and benchmarks. + +This separation makes the process auditable and prevents benchmark results from being mistaken for +proof that a production change is safe. + +## Goals + +- Turn a PID plus an optional source directory into a repeatable optimization investigation. +- Combine async-profiler-backed JFR, `jfr-analyzer`, and BTrace using shared target and time-window + metadata. +- Identify optimization seams and entry points from profile evidence and source inspection. +- Capture useful workload shape without copying arbitrary production object graphs. +- Generate representative, reviewable JMH benchmarks. +- Explore optimization ideas systematically and measure concrete implementations. +- Produce a PR candidate with evidence, benchmark results, limits, and reproduction instructions. +- Keep all profiling, probing, source edits, and PR creation bounded and user-visible. + +## Non-goals + +- Replaying production requests exactly by default. +- Capturing arbitrary arguments, credentials, request bodies, or object graphs. +- Claiming that a sampled profile proves causality. +- Optimizing external systems, deployment configuration, or infrastructure without a separate scope. +- Automatically merging or publishing a code change. +- Running an unbounded autonomous optimization loop. + +## User contract + +The entry point accepts: + +```text +/perf-engineer [source-directory] +``` + +Before recording, the skill confirms: + +- target PID, command line, host, Java version, and process identity; +- source directory, repository, branch, and clean/dirty working-tree state when supplied; +- available async-profiler/JFR and BTrace capabilities; +- recording duration, maximum duration, output location, and stop command; +- whether production-sensitive values may be observed; +- whether the user authorizes profiling and live instrumentation. + +These are separate authorization boundaries and are requested only when needed: + +- attach and record the target; +- deploy and stop BTrace probes; +- create benchmark/session files; +- synthesize or apply source patches; +- create a branch; +- open a draft PR. + +The default recording is bounded. “Until I say stop” means the skill waits for an explicit stop +request but still enforces a hard maximum duration. The recorder owns that timeout, rather than the +chat client. A session has an owner and lease, a durable stop operation, and a recovery path for a +disconnected client. On restart, the skill lists active sessions and offers to stop orphaned ones; +it never silently resumes them. + +The initial implementation uses a local session supervisor process. It owns recorder/probe child +processes, writes lease and state transitions to `manifest.json`, accepts an authenticated local +stop request, enforces timeouts independently of the chat client, and marks child processes as +`stopped`, `expired`, `target-exited`, or `failed`. Cleanup is idempotent and retryable after a +crash. A disconnected client does not extend the lease or hard timeout. + +## Instrument roles + +| Instrument | Primary question | Typical output | +| --- | --- | --- | +| async-profiler | Where is sampled CPU, wall-clock, allocation, lock, or native time going? | Profile artifact and stack distribution | +| JFR | What JVM events and state surround the suspected interval? | Historical event and timeline evidence | +| BTrace | What exact method/path behavior occurs on the live target? | Bounded invocation, latency, and shape summaries | +| JMH | Does a proposed implementation improve the isolated operation? | Reproducible benchmark comparison | + +No instrument is treated as authoritative for a question outside its semantics. For example, +sampled stack frequency is not treated as an exact request count, and a microbenchmark result is +not treated as proof of end-to-end improvement. + +## End-to-end workflow + +### 1. Establish the session + +Create a session directory containing: + +```text +.perf-engineer// + manifest.json + recording/ + hypotheses/ + workload/ + benchmark/ + candidates/ + report/ +``` + +The manifest records target identity, repository state, timestamps, tool versions, permissions, +and user-approved limits. Every downstream artifact references the session ID. The manifest also +stores a target fingerprint: + +```json +{ + "pid": "...", + "host": "...", + "jvm_start_time": "...", + "command_line_digest": "...", + "executable_or_container_identity": "...", + "clock": { + "wall_start": "...", + "monotonic_start_ns": "...", + "clock_source": "..." + } +} +``` + +Before every attach, record, and probe operation, revalidate the fingerprint. A reused PID, changed +command line, changed JVM start time, or changed host/container identity is a hard stop. Cleanup of +the supervisor's own child processes remains allowed after target loss; it uses recorded child +process IDs and session identity rather than attempting to attach to the target again. + +When a source directory is supplied, exploration uses a separate worktree or temporary clone. The +user’s original worktree is never modified during profiling, benchmark generation, or candidate +search. The manifest records the exact base commit and excludes unrelated dirty changes from every +candidate. + +The session workspace is outside the source worktree by default. Only an explicitly approved +benchmark or patch worktree may contain generated files. Candidate patches are computed against the +recorded base commit and are never applied to the user's original dirty worktree automatically. + +### 2. Record the target + +Use async-profiler in a JFR-compatible bounded recording mode when available. Select the event mode +from the user’s symptom or from a short diagnostic pass: + +- CPU or wall-clock for latency and hot-code questions; +- allocation for allocation pressure and GC-related hypotheses; +- lock for contention; +- native when the evidence points below Java frames. + +Record the mode, interval, stack mode, duration, output path, and exact start/end timestamps. If the +target is in a container or remote host, resolve that boundary before attaching; do not imply that a +local PID is reachable from another machine. + +If the target exits, becomes unreachable, or fails fingerprint validation, stop dependent probes and +recordings, mark the session `target-exited` or `failed`, preserve partial artifacts, and do not +attach to a replacement process automatically. + +Apply a profiling budget covering duration, expected overhead, output size, and disk space. If the +target is production-sensitive, present the expected impact and rollback/stop action before +attaching. + +Unless the user supplies stricter values, use these defaults: 30 seconds for the initial recording, +10 minutes as the hard recording maximum, 512 MiB per artifact, 1 GiB per session, 5% expected +overhead, 10 minutes of total optimization search time, and at most 12 candidate ideas or 4 +candidate patches. These are safety defaults, not performance claims. + +Before attaching, check for existing async-profiler/JFR/BTrace sessions, conflicting agents or +locks, attach permissions, and an already-active profiler collecting the requested event. Ask before +stopping or reconfiguring an existing session; otherwise select a non-conflicting mode or stop. + +Send the resulting JFR to `jfr-analyzer` for triage, drilldown, and report generation. Store the +report and the raw recording as immutable session inputs. + +### 3. Rank hypotheses + +Each hypothesis should include: + +```json +{ + "id": "h-001", + "target": "com.example.OrderService.findCandidates", + "mechanism": "repeated allocation in candidate filtering", + "evidence": ["..."], + "confidence": 0.0, + "impact_estimate": "high|medium|low", + "next_observation": "...", + "constraints": ["preserve ordering", "public API unchanged"] +} +``` + +Rank by measured impact and evidence quality, not by how easy a code change looks. Select only a +small number of hypotheses for live characterization. + +### 4. Identify the optimization seam + +For each selected hypothesis: + +1. Resolve profile frames to source symbols. +2. Inspect callers, callees, tests, constructors, and relevant configuration. +3. Identify an entry point that is both observable in the application and isolatable in a benchmark. +4. Record why the entry point is a useful seam and what the benchmark will exclude. + +The seam record also includes attribution confidence, whether frames are inlined/generated/native, +caller/callee evidence, the causal connection to the hypothesis, and alternative seams considered. +Profile frames are candidates, not automatically valid entry points. + +If the source directory is absent, the skill may still report hypotheses and recommend a BTrace +probe, but it must not synthesize a patch or claim source-level optimization readiness. + +### 5. Characterize the workload with BTrace + +BTrace is used to capture a **workload fingerprint**, not a production fixture. A default probe may +record: + +- invocation count and latency distribution; +- null/presence flags; +- runtime type names; +- collection/map sizes and nesting depth; +- string or byte-array lengths; +- numeric buckets and enum frequencies; +- selected downstream branch or method choices; +- bounded allocation or exception indicators when relevant. + +The probe must not print arbitrary arguments or object graphs. It must declare a maximum duration, +expected event rate, output destination, redaction policy, and cleanup command. If the entry point is +too hot or the shape is too complex, stop characterization and mark the workload as approximate or +non-replayable. + +Probe collection uses explicit safety tiers: + +- **Tier 0:** counters, timestamps, and latency only; +- **Tier 1:** primitive/null flags and runtime type metadata; +- **Tier 2:** bounded collection sizes, lengths, nesting, and branch choices; +- **Tier 3:** allowlisted DTO serialization with redaction and size limits, opt-in only. + +Start at Tier 0 and escalate only after confirming overhead and data safety. Do not invoke arbitrary +methods, traverse proxies, trigger lazy loading, or inspect object graphs merely to obtain a shape. + +### 6. Construct a benchmark + +Build the smallest JMH benchmark that preserves the measured dimensions of the workload fingerprint. +Prefer existing fixtures and tests. Otherwise generate deterministic builders for synthetic inputs, +with parameters named after observed shape dimensions: + +```text +@Param({"small", "typical", "large"}) collectionSize +@Param({"0", "1", "8"}) nestingDepth +@Param({"short", "long"}) keyShape +``` + +The benchmark must document: + +- the production evidence it approximates; +- what it intentionally omits, such as network, database, scheduling, or cache state; +- setup versus measured code; +- warmup, measurement, forks, JVM arguments, and profilers used; +- correctness assertions or result consumption preventing dead-code elimination. + +It must also declare its fidelity class: + +- **correlated:** baseline benchmark behavior matches the relevant production profile dimensions; +- **approximate:** selected shape dimensions match, but important production context is omitted; +- **isolated:** useful for implementation comparison only, with no production-transfer claim. + +The skill should reject a benchmark when it only measures fixture construction, cannot consume the +result correctly, has no relationship to the selected hypothesis, or cannot state what evidence it +approximates. A benchmark that cannot be correlated with production remains useful only as an +isolated implementation test. + +Before candidate search, compare the baseline benchmark with the production evidence using available +dimensions such as top stacks, allocation rate, collection sizes, branch distribution, and latency +shape. Record mismatches explicitly. Split workload inputs into a training set and a frozen holdout +set. Idea generation, scoring, and mutation may inspect only training results. The holdout is run +only after candidate patches and their ranking are fixed. Use at least one holdout shape that is +materially different from the training shapes. + +Transfer validation compares declared dimensions using project-configurable thresholds. The default +classification is `correlated` only when the baseline benchmark reproduces the top relevant stack +family and each available shape metric is within 10% of the production fingerprint; otherwise it is +`approximate` or `isolated`. Missing dimensions are reported, never treated as matching. + +The 10% default is intentionally strict. Relax it only for a specific metric or project after +repeated measurements show that the metric's natural variance or measurement method makes 10% +unachievable; record the evidence and the approved replacement threshold in the session manifest. + +### 7. Generate optimization ideas + +Ideas are represented independently from source patches. An idea genome contains semantic traits, +constraints, evidence, and a benchmark plan: + +```json +{ + "id": "idea-004", + "target": "com.example.OrderService.findCandidates", + "traits": [ + { + "operation": "allocation_reduction", + "location": "result_collection", + "preconditions": ["known_upper_bound"], + "expected_effect": "lower_alloc_rate" + } + ], + "constraints": [ + "preserve ordering", + "preserve null behavior", + "public API unchanged" + ], + "expected_mechanism": "reduce allocation and repeated normalization", + "evidence_refs": ["h-001", "shape-002"], + "risk": "medium", + "status": "candidate" +} +``` + +Ideas may be generated from profile evidence, source inspection, known library patterns, and test +constraints. Traits use typed operations, locations, preconditions, and expected effects rather +than free-form labels alone. They must state the expected mechanism and avoid vague goals such as +“make it faster.” + +### 8. Evolve ideas, not code + +The bounded search loop operates on idea genomes: + +1. Generate several independent ideas. +2. Score them for measured evidence fit, expected impact, implementation risk, and benchmarkability. +3. Recombine compatible traits while preserving constraints. +4. Mutate one trait or parameter at a time. +5. Discard ideas that violate API, correctness, safety, or benchmark constraints. +6. Synthesize ordinary source patches only for the strongest surviving ideas. +7. Compile and test each patch before benchmarking it. + +This keeps the search space semantic and reviewable. It does not mutate source text blindly and does +not assume that traits are composable; every recombined idea must survive correctness and benchmark +gates. The scorer keeps measured facts, static-analysis facts, model estimates, and benchmark +results in separate fields. Model estimates never override contradictory measurements. + +Candidates are evaluated across small, typical, large, and pathological training shapes. A +candidate that wins only on training shapes is rejected as overfit. The frozen holdout is executed +after candidate patches and their ranking are fixed; its result cannot trigger another mutation round +without starting a new search session. + +### 9. Validate candidates + +Every candidate follows this gate sequence: + +```text +source inspection → patch synthesis → formatting → compile → unit tests +→ benchmark correctness → baseline benchmark → candidate benchmark +→ regression checks → evidence report +``` + +Compare candidates against the same fresh JVM forks/processes, JVM, benchmark parameters, and +environment. Reset or recreate relevant caches, generated artifacts, and external fixtures between +runs. Never compare a warmed candidate process with a cold baseline. +Run a baseline stability check before candidate comparison, including repeated forks, warmup, +measurement count, confidence/variance reporting, GC state where relevant, and CPU/environment +metadata. Reject noisy comparisons. + +Validate behavior with compilation, unit tests, differential/property checks where applicable, and +explicit invariants for ordering, null behavior, exceptions, concurrency, caching, and numerical +semantics. A faster microbenchmark with changed semantics is rejected. + +Require transfer validation before making a production optimization recommendation: compare the +candidate against the production-correlated dimensions and the frozen holdout shape. For changes +whose cost depends on I/O, concurrency, scheduling, cache state, or framework behavior, run an +approved application-level validation or label the result `microbenchmark-only`/`approximate`. + +### 10. Prepare a PR + +Only after explicit user approval for patch application, branch creation, and PR creation, create a +branch or draft PR. By default, a candidate may change only files in the identified module and its +tests/benchmarks, with a maximum of 8 files and 400 changed lines. Expanding that scope requires a +new approval. The PR lifecycle is explicit: `candidate`, `patch-applied`, `branch-created`, +`draft-pr-opened`, and `human-reviewed`; opening a PR never implies acceptance. The PR contains: + +- the selected patch; +- tests and benchmark source; +- before/after benchmark results; +- the original profile and hypothesis references; +- workload-fingerprint limitations; +- rejected alternatives and why they lost; +- reproduction commands; +- rollback notes. +- the exact base commit and confirmation that unrelated worktree changes were excluded; +- benchmark fidelity class and transfer-validation result. + +The skill must never merge automatically. PR creation is a separate authorization boundary from +profiling and local benchmarking. + +## Handling complex parameters + +Raw parameter capture is useful only when the value is small, safe, stable, and directly relevant to +the suspected cost. Otherwise use one of these strategies: + +1. **Shape-only capture** — sizes, types, flags, buckets, and branch choices. +2. **Existing fixture reuse** — adapt test data already maintained by the project. +3. **Deterministic synthetic fixture** — construct an object with the same measured shape. +4. **Controlled serializer** — use only an allowlisted DTO/schema with redaction and size limits. +5. **Non-replayable classification** — stop at a hypothesis and recommend a macrobenchmark or load + test when the operation depends on external state or concurrency. + +The benchmark report must say which strategy was used. “Representative” means representative of the +measured cost dimensions, not identical to a production request. + +If the operation depends on external state, concurrency, scheduling, or I/O, route it to a +concurrent benchmark, macrobenchmark, load test, or system-level experiment instead of forcing it +into a single-threaded JMH benchmark. + +## Safety and stop conditions + +Stop or request confirmation when: + +- attach permissions or target identity are ambiguous; +- the probe rate or overhead exceeds the declared budget; +- sensitive data would be captured; +- the source tree has unrelated user changes that a patch could overlap; +- benchmark results are unstable or contradict the profile; +- a candidate changes public behavior, API, persistence, or concurrency semantics; +- the iteration budget, wall-clock budget, or maximum recording duration is reached. + +Every live session has an explicit cleanup operation. Raw recordings and probe output remain local +unless the user explicitly requests sharing. Artifacts are created with owner-only permissions, +enforce the 512 MiB artifact/1 GiB session limits above, and are retained for 7 days by default. +Cleanup removes raw recordings, probe output, temporary worktrees, and generated binaries while +retaining a redacted manifest and redacted final report. Failed or expired sessions leave a cleanup +record and can be removed with the session cleanup command. + +## Terminal outcomes + +The workflow may finish without a code change. Reports use one of these explicit outcomes: + +- `optimization_candidate`: a tested patch improves correlated and holdout workloads within the + declared constraints; +- `microbenchmark_only`: an implementation improvement is measured, but production transfer was not + established; +- `insufficient_evidence`: the profile or characterization did not support a confident hypothesis; +- `hypothesis_disproved`: follow-up observation contradicted the proposed mechanism; +- `non_replayable`: the cost depends on external state, concurrency, or I/O that the selected + benchmark cannot represent; +- `no_meaningful_improvement`: candidates were valid but did not beat the stable baseline; +- `too_risky`: a candidate improved a metric but violated behavior, safety, compatibility, or + operational constraints. + +The skill must not manufacture a patch or PR merely to complete the workflow. + +## Provider and transport interoperability + +The shared skills must work across Claude, Codex, and Pi. MCP is appropriate for stateful recording, +analysis, and probe sessions; a local CLI or shell command is appropriate for simple one-shot +operations. Transport-specific names must not leak into evidence records. + +Repository contents are treated as untrusted input. Source files, build files, and generated +instructions may inform analysis but must not authorize arbitrary commands. Project commands come +from an explicit allowlist or require user approval; build-file changes and dependency downloads are +always explicit. + +The minimum cross-tool handoff is: + +```json +{ + "session_id": "...", + "target": { + "pid": "...", + "host": "...", + "service": "...", + "jvm_start_time": "...", + "command_line_digest": "...", + "executable_or_container_identity": "..." + }, + "window": { + "start": "...", + "end": "...", + "timezone": "...", + "monotonic_start_ns": "...", + "monotonic_end_ns": "...", + "clock_source": "..." + }, + "profile": {"artifact": "...", "mode": "...", "evidence": ["..."]}, + "workload": {"entry_point": "...", "shape": {}, "capture_policy": "..."}, + "hypotheses": ["h-001"], + "ideas": ["idea-004"], + "next_action": "..." +} +``` + +## Initial implementation plan + +### Phase 1: orchestration and evidence + +- Add the `perf-engineer` coordinator skill. +- Reuse `async-profiler-interop`, `jfr-analyzer`, and BTrace lifecycle/data-safety skills. +- Implement session manifests, bounded recording state, hypothesis handoff, and cleanup. +- Support report-only mode when no source directory is supplied. +- Define session leases, crash recovery, target revalidation, artifact limits, and mandatory + worktree isolation. +- Implement the local session supervisor and existing-session conflict checks. + +### Phase 2: workload characterization + +- Add a small library of safe BTrace shape probes. +- Add entry-point and source-seam guidance. +- Add deterministic fixture templates for common collections, strings, DTOs, and call branches. +- Add non-replayable classification and macrobenchmark recommendations. +- Add probe safety tiers and explicit workload-fingerprint fidelity classes. + +### Phase 3: benchmark synthesis and idea search + +- Add JMH project/fixture discovery. +- Support Gradle, Maven, and Bazel project layouts first; other build systems are out of scope for + the initial implementation. +- Generate benchmark scaffolds with evidence annotations. +- Implement idea-genome schema, scoring, recombination, mutation, and bounded candidate selection. +- Require compile, tests, and baseline benchmark before candidate benchmarking. +- Require blind training/holdout shapes, baseline stability, transfer validation, and explicit + measured-vs-estimated scoring. + +### Phase 4: patch and PR workflow + +- Generate small reviewable patches from selected ideas. +- Produce candidate comparison reports. +- Add explicit user approval before patch application, branch/PR creation, and scope expansion. +- Add evaluation scenarios for allocation, CPU, lock contention, and misleading/non-replayable inputs. +- Add differential/property checks, noisy-baseline rejection, PID reuse, orphan-session recovery, + dirty-worktree isolation, and prompt-injection/untrusted-repository scenarios. + +## Open questions + +- Should generated benchmarks live in the production repository or in a temporary session workspace? +- What minimum benchmark speedup justifies a candidate when variance is high? +- How should concurrent and I/O-bound hypotheses hand off to macrobenchmarks? +- Which source transformations are safe enough for the first idea-genome library? +- What repository command allowlist is acceptable across Gradle, Maven, and Bazel? +- What evidence is required to override the default 10% fidelity threshold per metric or project? diff --git a/package.json b/package.json index e50f563..3aa8f4a 100644 --- a/package.json +++ b/package.json @@ -12,7 +12,9 @@ "pi": { "skills": [ "./plugins/btrace-development/skills", - "./plugins/btrace-observability/skills" + "./plugins/btrace-observability/skills", + "./plugins/jfr-analyzer/skills", + "./plugins/perf-engineer/skills" ] } } diff --git a/plugins/btrace-observability/README.md b/plugins/btrace-observability/README.md index 092407d..bcbe927 100644 --- a/plugins/btrace-observability/README.md +++ b/plugins/btrace-observability/README.md @@ -60,6 +60,14 @@ condition. The BTrace verifier remains enabled by default; unsafe/trusted mode i ## MCP server -Claude Code installs the bundled stdio MCP server as `btrace`. It requires JBang and JDK 11 or -newer, and it downloads the single masked `io.btrace:btrace` distribution on first use. The server -must run on the host that can attach to the target JVM. +The host plugin integration installs the bundled stdio MCP server as `btrace`. It requires JBang +and JDK 11 or newer, and it downloads the single masked `io.btrace:btrace` distribution on first +use. The server must run on the host that can attach to the target JVM. + +## JFR correlation + +When the `jfr-analyzer` plugin is installed, use it for historical profile context and pair it with +the `jfr-btrace-interop` skill. JFR identifies the resource, interval, and candidate hotspot; +BTrace performs the narrow live confirmation. Share target identity, timestamps, hypothesis, and +evidence between the two workflows, and always stop the BTrace probe when the observation window +ends. diff --git a/plugins/btrace-observability/skills/btrace-mcp-operations/SKILL.md b/plugins/btrace-observability/skills/btrace-mcp-operations/SKILL.md index ee64d05..235cd46 100644 --- a/plugins/btrace-observability/skills/btrace-mcp-operations/SKILL.md +++ b/plugins/btrace-observability/skills/btrace-mcp-operations/SKILL.md @@ -1,6 +1,7 @@ --- name: btrace-mcp-operations description: Use when an AI client should operate BTrace through the BTrace MCP server to list local JVMs, deploy probes, inspect output, or clean up diagnostic sessions. +allowed-tools: Read mcp__btrace__list_jvms mcp__btrace__list_probes --- # MCP Operations @@ -20,5 +21,17 @@ are on the same host and the operator wants an auditable conversational workflow The server uses stdio rather than opening a network listener. Keep it local to the target host and use SSH, `kubectl exec`, or an approved bastion workflow to run it beside a remote target. +## Choosing MCP versus a CLI + +Use the ordinary BTrace CLI or a host-provided wrapper for a simple one-shot action when that is +available and the operator wants a command they can copy, review, and rerun. Use this MCP server +when the workflow needs JVM discovery, multiple related operations, typed tool results, or explicit +probe-session lifecycle in one conversation. Both paths must preserve the same target PID, +observation window, probe identity, output destination, and cleanup command. + +When JFR is also involved, pass findings through the `jfr-btrace-interop` evidence record rather +than relying on provider-specific tool names or unstructured transcript text. JFR supplies the +historical interval; BTrace supplies the bounded live confirmation. + The plugin launches the bundled server with JBang. It loads the single masked BTrace distribution, so do not configure legacy `btrace-client.jar`, `btrace-agent.jar`, or `btrace-boot.jar` paths. diff --git a/plugins/btrace-observability/skills/btrace-observability/SKILL.md b/plugins/btrace-observability/skills/btrace-observability/SKILL.md index 4c963c8..c43c949 100644 --- a/plugins/btrace-observability/skills/btrace-observability/SKILL.md +++ b/plugins/btrace-observability/skills/btrace-observability/SKILL.md @@ -22,11 +22,19 @@ and combine the specialist skills below as needed. | Sensitive data, production load, permissions, or risk | `btrace-data-safety` | | Extensions, metrics exporters, or permission grants | `btrace-extensions-and-permissions` | | AI/MCP-guided local diagnostics | `btrace-mcp-operations` | +| Historical profile context or JFR/async-profiler correlation | `jfr-analyzer`, `async-profiler-interop`, and `jfr-btrace-interop` when installed | | Immutable images, no attach, or launch-time deployment | `btrace-startup-and-packaging` | For a typical production incident, combine runtime access + the relevant diagnosis skill + lifecycle + data safety. Do not force every request through every skill. +When a JFR or other profile is available, use it to establish the historical resource, time window, +and candidate hotspot before widening a live probe. If `jfr-analyzer` is installed, pair it with +`jfr-btrace-interop`: JFR supplies aggregate context and BTrace confirms a narrow live behavior. +If async-profiler is available, use it for a bounded sampled CPU, wall-clock, allocation, lock, or +native profile when that is the unresolved question. Carry target identity, timestamps, hypothesis, +and evidence between all analyses. + The skill suite is grounded in the BTrace hands-on tutorials, Quick Reference, Oneliner Guide, and provided extension examples. Prefer their verified syntax and deployment patterns over invented DSL or platform assumptions. diff --git a/plugins/jfr-analyzer/.claude-plugin/plugin.json b/plugins/jfr-analyzer/.claude-plugin/plugin.json new file mode 100644 index 0000000..a3c396a --- /dev/null +++ b/plugins/jfr-analyzer/.claude-plugin/plugin.json @@ -0,0 +1,36 @@ +{ + "name": "jfr-analyzer", + "version": "0.1.0", + "description": "Systematic performance investigation of JFR, pprof, OTLP, and HPROF profiles.", + "author": { + "name": "BTrace", + "url": "https://github.com/btraceio" + }, + "repository": "https://github.com/btraceio/agent-plugins", + "license": "Apache-2.0", + "skills": "./skills/", + "mcpServers": "./.mcp.json", + "keywords": [ + "jfr", + "java", + "performance", + "profiling", + "observability" + ], + "interface": { + "displayName": "JFR Analyzer", + "shortDescription": "Investigate JVM profiles systematically.", + "longDescription": "Analyze JFR and related profile formats with USE/TSA reasoning, structured evidence, and optional BTrace live-probe correlation.", + "developerName": "BTrace", + "category": "Developer Tools", + "capabilities": [ + "Guidance", + "Analysis" + ], + "defaultPrompt": [ + "Analyze this JFR recording and explain the biggest bottleneck.", + "Find the cause of these JVM latency spikes.", + "Correlate this profile with a live BTrace investigation." + ] + } +} diff --git a/plugins/jfr-analyzer/.codex-plugin/plugin.json b/plugins/jfr-analyzer/.codex-plugin/plugin.json new file mode 100644 index 0000000..f47b47a --- /dev/null +++ b/plugins/jfr-analyzer/.codex-plugin/plugin.json @@ -0,0 +1,36 @@ +{ + "name": "jfr-analyzer", + "version": "0.1.0", + "description": "Systematic performance investigation of JFR, pprof, OTLP, and HPROF profiles.", + "author": { + "name": "BTrace", + "url": "https://github.com/btraceio" + }, + "repository": "https://github.com/btraceio/agent-plugins", + "license": "Apache-2.0", + "keywords": [ + "jfr", + "java", + "performance", + "profiling", + "observability" + ], + "skills": "./skills/", + "mcpServers": "./.mcp.json", + "interface": { + "displayName": "JFR Analyzer", + "shortDescription": "Investigate JVM profiles systematically.", + "longDescription": "Analyze JFR and related profile formats with USE/TSA reasoning, structured evidence, and optional BTrace live-probe correlation.", + "developerName": "BTrace", + "category": "Developer Tools", + "capabilities": [ + "Guidance", + "Analysis" + ], + "defaultPrompt": [ + "Analyze this JFR recording and explain the biggest bottleneck.", + "Find the cause of these JVM latency spikes.", + "Correlate this profile with a live BTrace investigation." + ] + } +} diff --git a/plugins/jfr-analyzer/.gitignore b/plugins/jfr-analyzer/.gitignore new file mode 100644 index 0000000..d7880e4 --- /dev/null +++ b/plugins/jfr-analyzer/.gitignore @@ -0,0 +1,13 @@ +# Eval run outputs — too large and transient for git +eval/results/ + +# OTLP and pprof derived from corpus JFR at run time +eval/corpus/*.jfr +eval/corpus/*.otlp +eval/corpus/*.pprof +eval/corpus/*.pb.gz + +# Python scorer venv +eval/scripts/.venv/ +eval/scripts/__pycache__/ +eval/scripts/*.pyc diff --git a/plugins/jfr-analyzer/.mcp.json b/plugins/jfr-analyzer/.mcp.json new file mode 100644 index 0000000..de93d77 --- /dev/null +++ b/plugins/jfr-analyzer/.mcp.json @@ -0,0 +1,8 @@ +{ + "mcpServers": { + "jfr-mcp": { + "type": "sse", + "url": "http://localhost:3000/mcp/sse" + } + } +} diff --git a/plugins/jfr-analyzer/README.md b/plugins/jfr-analyzer/README.md new file mode 100644 index 0000000..8bc8516 --- /dev/null +++ b/plugins/jfr-analyzer/README.md @@ -0,0 +1,88 @@ +# JFR Analyzer + +`jfr-analyzer` turns JFR, pprof, OTLP, and HPROF recordings into a structured performance investigation. It uses USE (Utilization, Saturation, Errors) and TSA (Thread State Analysis) reasoning before moving to individual hotspots. + +Its `async-profiler-interop` skill connects sampled live profiles with JFR history and exact BTrace +probes, using one target/window/evidence record across the three instruments. + +## Host support + +The shared skills work in Claude Code, Codex, and Pi. The plugin expects a Jafar MCP server at `http://localhost:3000/mcp/sse`: + +```bash +jbang jafar-mcp@btraceio +``` + +Configure the same `jfr-mcp` server in the host when automatic plugin MCP loading is unavailable. The skills refer to the server by capability (`jfr_open`, `jfr_query`, `jfr_summary`, and related tools); host-specific MCP namespaces may differ. + +## Install async-profiler + +Async-profiler is optional for ordinary JFR-file analysis, but required for the bundled eval corpus +and for live sampled CPU, wall-clock, allocation, lock, or native profiles. Install a release from +the [async-profiler releases page](https://github.com/async-profiler/async-profiler/releases), then +point the plugin at the extracted directory: + +```sh +export ASYNC_PROFILER_HOME="$HOME/.local/lib/async-profiler" +export PATH="$ASYNC_PROFILER_HOME/bin:$PATH" +test -f "$ASYNC_PROFILER_HOME/lib/libasyncProfiler.so" \ + || test -f "$ASYNC_PROFILER_HOME/lib/libasyncProfiler.dylib" +``` + +On macOS, use the release archive containing `macos`; on Linux, use `linux-x64` or `linux-arm64` +as appropriate. The directory must contain `lib/libasyncProfiler.dylib` or +`lib/libasyncProfiler.so`. The profiler attaches to the target JVM, so the profiler and target +must run on the same host and the user must have the required JVM attach permissions. + +The eval regeneration script can perform the same installation interactively when async-profiler +is missing: + +```sh +plugins/jfr-analyzer/eval/scripts/regen.sh +``` + +It prompts before downloading and stores the tool under `$HOME/.local/lib/`. + +## Run the eval corpus + +From the repository root: + +```sh +plugins/jfr-analyzer/eval/scripts/regen.sh +``` + +The script prompts before installing JBang, async-profiler, or `jq`, then records the corpus and +updates its manifest. After the MCP server is available, run `/jfr-analyzer eval regen` to extract +event types and optionally validate oracle results. Run `/jfr-analyzer eval run` for triage passes, +then `/jfr-analyzer eval score` to generate the report. Scoring creates a local Python environment +under `plugins/jfr-analyzer/eval/scripts/.venv` when needed and may require the judge API key named +in `scripts/score.py`. + +## Transport strategy + +The analysis workflow is transport-neutral. Prefer a host-provided CLI or shell command for a +single, read-only operation when its output is trustworthy and capturable. Use the Jafar MCP +session for repeated queries, typed results, capability discovery, and multi-phase analysis. No +standalone JFR CLI adapter is bundled yet; future adapters should emit the same evidence fields +used by `jfr-btrace-interop`. + +## Commands + +```text +/jfr-analyzer recording.jfr +/jfr-analyzer triage recording.jfr +/jfr-analyzer drilldown +/jfr-analyzer report +``` + +## JFR and BTrace together + +Use JFR for historical, aggregate evidence and BTrace for a narrow live confirmation or targeted observation: + +1. Preserve the recording path, target identity, and time window. +2. Use JFR to identify the resource, time window, and candidate class/method. +3. Use `btrace-observability` to design the smallest safe live probe. +4. Correlate the probe output with the JFR window; do not treat either signal as proof in isolation. +5. Stop the probe and retain the correlation metadata with the report. + +The `jfr-btrace-interop` skill contains the handoff protocol. diff --git a/plugins/jfr-analyzer/agents/perf-engineer.md b/plugins/jfr-analyzer/agents/perf-engineer.md new file mode 100644 index 0000000..74813aa --- /dev/null +++ b/plugins/jfr-analyzer/agents/perf-engineer.md @@ -0,0 +1,152 @@ +--- +name: perf-engineer +description: Performance engineering expert. Analyzes JFR, pprof, OTLP, and HPROF profiling data using USE and TSA methodologies. Explains findings in plain language before presenting technical evidence. Usable standalone or as a spawned subagent. +model: inherit +tools: + - Read + - Glob + - Write + - Bash(find *) + - mcp__jfr-mcp__jfr_open + - mcp__jfr-mcp__jfr_close + - mcp__jfr-mcp__jfr_query + - mcp__jfr-mcp__jfr_list_types + - mcp__jfr-mcp__jfr_help + - mcp__jfr-mcp__jfr_summary + - mcp__jfr-mcp__jfr_flamegraph + - mcp__jfr-mcp__jfr_callgraph + - mcp__jfr-mcp__jfr_hotmethods + - mcp__jfr-mcp__jfr_exceptions + - mcp__jfr-mcp__jfr_use + - mcp__jfr-mcp__jfr_tsa + - mcp__jfr-mcp__jfr_diagnose + - mcp__jfr-mcp__jfr_stackprofile + - mcp__jfr-mcp__pprof_open + - mcp__jfr-mcp__pprof_close + - mcp__jfr-mcp__pprof_query + - mcp__jfr-mcp__pprof_summary + - mcp__jfr-mcp__pprof_flamegraph + - mcp__jfr-mcp__pprof_hotmethods + - mcp__jfr-mcp__pprof_use + - mcp__jfr-mcp__pprof_tsa + - mcp__jfr-mcp__pprof_stackprofile + - mcp__jfr-mcp__pprof_help + - mcp__jfr-mcp__otlp_open + - mcp__jfr-mcp__otlp_close + - mcp__jfr-mcp__otlp_query + - mcp__jfr-mcp__otlp_summary + - mcp__jfr-mcp__otlp_flamegraph + - mcp__jfr-mcp__otlp_use + - mcp__jfr-mcp__otlp_help + - mcp__jfr-mcp__hdump_open + - mcp__jfr-mcp__hdump_close + - mcp__jfr-mcp__hdump_query + - mcp__jfr-mcp__hdump_summary + - mcp__jfr-mcp__hdump_report + - mcp__jfr-mcp__hdump_help +--- + +You are a performance engineering expert specializing in JVM, Go, and polyglot applications. + +## Core principles + +Always explain findings in plain language BEFORE presenting technical evidence. Define technical +terms inline on first use. Frame everything in terms of user-visible impact: latency, throughput, +stability, cost. + +Apply USE (Utilization-Saturation-Errors) and TSA (Thread State Analysis) methodologies before +jumping to specific hotspots — never conclude "method X is the problem" without first establishing +which resource (CPU, memory, threads, I/O) is the bottleneck. + +When a source root is available, read the actual method body before forming hypotheses. + +## Glossary — define these on first use + +- **p99**: The 99th percentile latency — 99% of operations complete faster than this value. A high + p99 means a small fraction of users experience much slower responses. +- **TLAB** (Thread-Local Allocation Buffer): A private chunk of heap reserved for one thread, + allowing fast allocation without locking. When a TLAB fills up, the JVM must pause to allocate + a new one. +- **Safepoint**: A moment when all JVM threads pause briefly so the JVM can perform housekeeping + (GC, class loading, deoptimization). Frequent or long safepoints cause latency spikes. +- **Virtual thread pinning**: Virtual threads (JDK 21+) can normally be paused and reused across + requests. Pinning occurs inside `synchronized` blocks or native calls — the thread can't be + unmounted, reducing parallelism. +- **USE method**: A systematic checklist — for every resource (CPU, memory, threads, I/O) measure + Utilization (how busy it is), Saturation (how much demand is queued), and Errors. +- **TSA** (Thread State Analysis): Categorizes all threads by what they are doing — running, + blocked on a lock, waiting, sleeping — and identifies which threads drive each state. + +## JfrPath quick reference + +``` +events/ # all events of a type +events/[field > value] # filter (e.g. events/jdk.GCPhasePause[duration > 100ms]) +events/ | count() # count +events/ | stats(field) # min/max/avg/stddev +events/ | groupBy(field, agg=sum, value=other) | top(N) # group + aggregate +events/ | quantiles(0.5, 0.95, 0.99, path=duration) +``` + +Common event types: `jdk.ExecutionSample`, `jdk.GCPhasePause`, `jdk.GarbageCollection`, +`jdk.ObjectAllocationSample`, `jdk.JavaMonitorEnter`, `jdk.SafepointBegin`, and +`jdk.VirtualThreadPinned`. + +## PprofPath quick reference + +``` +samples # all samples +samples[thread='main'] # filter by label +samples | count() +samples | groupBy(thread, sum(cpu)) | head(10) +samples | groupBy(stackTrace/0/name, sum(alloc_objects)) # stackTrace/0/name = top frame method +``` + +## Workflow for a new file (standalone mode) + +1. Verify Jafar MCP is running — call `mcp__jfr-mcp__jfr_help` (or equivalent for the format). + If it fails, output: "Jafar MCP is not running. Start it with: jbang jafar-mcp@btraceio" +2. Open the file with `*_open`. +3. Run `*_summary` to understand what's in the recording. +4. Run `*_use` to identify which resource is the primary bottleneck. +5. Run `*_tsa` to understand thread behavior. +6. Then drill into the specific area the user asks about. + +## Dual-mode behavior + +**Standalone mode** (user selected this agent type at session start): +- Conduct an interactive investigation. +- Explain findings conversationally with embedded evidence. +- Suggest the next most useful query after each finding. +- Keep the session open; call `*_close` when the user says done. + +**Spawned mode** (called by the drilldown subskill): +- You will receive a structured prompt containing: sessionId, area assignment, + triage evidence summary, source root path, and output file path. +- Run the targeted queries listed in your assignment. +- Use the Write tool to write structured JSON to the output file path provided in your prompt. +- Do NOT engage conversationally — write structured JSON to the output path and exit. +- Output schema: + ```json + { + "area": "", + "summary": "", + "evidence": [ + { + "label": "", + "value": "", + "explanation": "", + "query": "" + } + ], + "hints": [ + { + "description": "", + "impact": "high|moderate|low", + "code_before": "", + "code_after": "" + } + ] + } + ``` +- If a Jafar MCP call fails, write the JSON output anyway with whatever evidence was collected, and add an `"error": ""` field at the top level of the JSON. diff --git a/plugins/jfr-analyzer/eval/corpus/manifest.json b/plugins/jfr-analyzer/eval/corpus/manifest.json new file mode 100644 index 0000000..b93feb5 --- /dev/null +++ b/plugins/jfr-analyzer/eval/corpus/manifest.json @@ -0,0 +1,362 @@ +{ + "version": "1", + "generated_at": "", + "source_hash": "sha256:PLACEHOLDER", + "scenarios": [ + { + "id": "cpu_hotspot", + "group": 1, + "description": "N_CPU threads spinning Math.sin/pow; no allocation pressure", + "jvm_flags": ["-Xmx256m", "-XX:+UseParallelGC"], + "duration_seconds": 30, + "file": "cpu_hotspot.jfr", + "jdk_version": "", + "jfr_event_types": [], + "required_event_types": ["jdk.ExecutionSample"], + "eval_runs": 5, + "expected": { + "scoring_tier": "structural", + "focusAreas": [ + {"id": "cpu-hotspots", "impact": "high", "startHere": true} + ], + "absent_area_ids": ["gc-pressure", "thread-contention"], + "crossAreaCorrelations": [] + } + }, + { + "id": "gc_pressure_parallel", + "group": 1, + "description": "Tight byte[] allocation loop; small heap forces frequent ParallelGC pauses", + "jvm_flags": ["-Xmx64m", "-XX:+UseParallelGC"], + "duration_seconds": 30, + "file": "gc_pressure_parallel.jfr", + "jdk_version": "", + "jfr_event_types": [], + "required_event_types": ["jdk.GCPhasePause"], + "eval_runs": 5, + "expected": { + "scoring_tier": "structural", + "focusAreas": [ + {"id": "gc-pressure", "impact": "high", "startHere": true} + ], + "absent_area_ids": ["cpu-hotspots"], + "crossAreaCorrelations": [] + } + }, + { + "id": "allocation_pressure", + "group": 1, + "description": "Circular buffer of 100KB objects; promotes without heavy GC", + "jvm_flags": ["-Xmx512m", "-XX:+UseParallelGC"], + "duration_seconds": 30, + "file": "allocation_pressure.jfr", + "jdk_version": "", + "jfr_event_types": [], + "required_event_types": ["jdk.ObjectAllocationSample"], + "eval_runs": 5, + "expected": { + "scoring_tier": "structural", + "focusAreas": [ + {"id": "allocation-pressure", "impact": "moderate", "startHere": true} + ], + "absent_area_ids": [], + "crossAreaCorrelations": [] + } + }, + { + "id": "thread_contention_sync", + "group": 1, + "description": "4×CPU threads competing on single synchronized method with sleep", + "jvm_flags": ["-Xmx256m"], + "duration_seconds": 30, + "file": "thread_contention_sync.jfr", + "jdk_version": "", + "jfr_event_types": [], + "required_event_types": ["jdk.JavaMonitorEnter"], + "eval_runs": 5, + "expected": { + "scoring_tier": "structural", + "focusAreas": [ + {"id": "thread-contention", "impact": "high", "startHere": true} + ], + "absent_area_ids": ["gc-pressure"], + "crossAreaCorrelations": [] + } + }, + { + "id": "safepoints", + "group": 1, + "description": "Repeated Class.forName + Method.invoke in a tight loop triggers frequent safepoints", + "jvm_flags": ["-Xmx256m"], + "duration_seconds": 30, + "file": "safepoints.jfr", + "jdk_version": "", + "jfr_event_types": [], + "required_event_types": ["jdk.SafepointBegin"], + "eval_runs": 5, + "expected": { + "scoring_tier": "structural", + "focusAreas": [ + {"id": "safepoints", "impact": "moderate", "startHere": true} + ], + "absent_area_ids": [], + "crossAreaCorrelations": [] + } + }, + { + "id": "temporal_spikes", + "group": 1, + "description": "5s CPU burst / 5s idle alternating; spike windows visible in stackgraph", + "jvm_flags": ["-Xmx256m"], + "duration_seconds": 60, + "file": "temporal_spikes.jfr", + "jdk_version": "", + "jfr_event_types": [], + "required_event_types": ["jdk.ExecutionSample"], + "eval_runs": 5, + "expected": { + "scoring_tier": "structural", + "focusAreas": [ + {"id": "temporal-spikes", "impact": "moderate", "startHere": true} + ], + "absent_area_ids": [], + "crossAreaCorrelations": [] + } + }, + { + "id": "healthy_baseline", + "group": 1, + "description": "Minimal sleep-only load; negative test — no HIGH areas expected", + "jvm_flags": ["-Xmx256m"], + "duration_seconds": 30, + "file": "healthy_baseline.jfr", + "jdk_version": "", + "jfr_event_types": [], + "required_event_types": [], + "eval_runs": 5, + "expected": { + "scoring_tier": "structural", + "focusAreas": [], + "absent_area_ids": ["cpu-hotspots", "gc-pressure", "thread-contention"], + "crossAreaCorrelations": [] + } + }, + { + "id": "alloc_gc_cascade", + "group": 2, + "description": "High alloc + bounded queue(depth=2) + 16 submitters; GC pauses cause queue backlog", + "jvm_flags": ["-Xmx128m", "-XX:+UseParallelGC"], + "duration_seconds": 30, + "file": "alloc_gc_cascade.jfr", + "jdk_version": "", + "jfr_event_types": [], + "required_event_types": ["jdk.GCPhasePause", "jdk.ObjectAllocationSample"], + "eval_runs": 5, + "expected": { + "scoring_tier": "structural", + "focusAreas": [ + {"id": "allocation-pressure", "impact": "high", "startHere": true}, + {"id": "gc-pressure", "impact": "high", "startHere": true}, + {"id": "thread-contention", "impact": "moderate", "startHere": false} + ], + "absent_area_ids": [], + "crossAreaCorrelations": [ + {"kind": "alloc-gc-latency-cascade"} + ] + } + }, + { + "id": "gc_cpu_inflation", + "group": 2, + "description": "Fast byte[512] alloc loop; G1 GC workers drive CPU above 60%; app threads trivial", + "jvm_flags": ["-XX:+UseG1GC", "-Xmx128m"], + "duration_seconds": 30, + "file": "gc_cpu_inflation.jfr", + "jdk_version": "", + "jfr_event_types": [], + "required_event_types": ["jdk.GCPhasePause", "jdk.ExecutionSample"], + "eval_runs": 5, + "expected": { + "scoring_tier": "structural", + "focusAreas": [ + {"id": "gc-pressure", "impact": "high", "startHere": true}, + {"id": "cpu-hotspots", "impact": "moderate", "startHere": false} + ], + "absent_area_ids": [], + "crossAreaCorrelations": [ + {"kind": "gc-cpu-inflation"} + ] + } + }, + { + "id": "gc_queue_coupling", + "group": 2, + "description": "Moderate alloc + 2-thread pool; trivial tasks queue up during GC stop-the-world pauses", + "jvm_flags": ["-Xmx128m", "-XX:+UseParallelGC"], + "duration_seconds": 30, + "file": "gc_queue_coupling.jfr", + "jdk_version": "", + "jfr_event_types": [], + "required_event_types": ["jdk.GCPhasePause"], + "eval_runs": 5, + "expected": { + "scoring_tier": "structural", + "focusAreas": [ + {"id": "gc-pressure", "impact": "high", "startHere": true}, + {"id": "thread-contention", "impact": "moderate", "startHere": false} + ], + "absent_area_ids": [], + "crossAreaCorrelations": [ + {"kind": "gc-queue-coupling"} + ] + } + }, + { + "id": "reactive_no_backpressure", + "group": 3, + "description": "Flux.interval(1ms) with slow subscriber (10ms/item), no backpressure operator", + "jvm_flags": ["-Xmx256m"], + "duration_seconds": 30, + "file": "reactive_no_backpressure.jfr", + "jdk_version": "", + "jfr_event_types": [], + "required_event_types": ["jdk.JavaMonitorEnter"], + "eval_runs": 5, + "expected": { + "scoring_tier": "structural", + "focusAreas": [ + {"id": "thread-contention", "impact": "high", "startHere": true} + ], + "absent_area_ids": [], + "crossAreaCorrelations": [], + "rubric": { + "bottleneck_category": "concurrency", + "severity": "high", + "required_evidence_keywords": ["producer", "subscriber", "backpressure", "queue"], + "must_not_hallucinate": ["gc-pressure"], + "judge_question": "Did the skill identify that the producer emits faster than the subscriber can process, causing thread saturation due to missing backpressure handling?" + } + } + }, + { + "id": "reactive_backpressure", + "group": 3, + "description": "Same Flux pipeline with onBackpressureBuffer(256); negative test", + "jvm_flags": ["-Xmx256m"], + "duration_seconds": 30, + "file": "reactive_backpressure.jfr", + "jdk_version": "", + "jfr_event_types": [], + "required_event_types": [], + "eval_runs": 5, + "expected": { + "scoring_tier": "structural", + "focusAreas": [], + "absent_area_ids": ["gc-pressure", "cpu-hotspots"], + "crossAreaCorrelations": [] + } + }, + { + "id": "cf_fanout_large_heap", + "group": 3, + "description": "40 CompletableFuture tasks each holding 50MB byte[] live until all complete (2GB peak)", + "jvm_flags": ["-Xmx4g", "-XX:+UseG1GC"], + "duration_seconds": 30, + "file": "cf_fanout_large_heap.jfr", + "jdk_version": "", + "jfr_event_types": [], + "required_event_types": ["jdk.ObjectAllocationSample", "jdk.GCPhasePause"], + "eval_runs": 5, + "expected": { + "scoring_tier": "structural", + "focusAreas": [ + {"id": "allocation-pressure", "impact": "high", "startHere": true}, + {"id": "gc-pressure", "impact": "high", "startHere": true}, + {"id": "thread-contention", "impact": "moderate", "startHere": false} + ], + "absent_area_ids": [], + "crossAreaCorrelations": [ + {"kind": "alloc-gc-latency-cascade"} + ], + "rubric": { + "bottleneck_category": "memory-gc", + "severity": "high", + "required_evidence_keywords": ["future", "heap", "allocation", "concurrent"], + "must_not_hallucinate": ["safepoints"], + "judge_question": "Did the skill identify that async fan-out tasks hold large byte arrays live simultaneously, creating allocation pressure and GC cascade?" + } + } + }, + { + "id": "g1_humongous_alloc", + "group": 4, + "description": "Continuous new byte[600_000]; G1HeapRegionSize=1m so objects exceed region/2 threshold", + "jvm_flags": ["-XX:+UseG1GC", "-XX:G1HeapRegionSize=1m", "-Xmx512m"], + "duration_seconds": 30, + "file": "g1_humongous_alloc.jfr", + "jdk_version": "", + "jfr_event_types": [], + "required_event_types": ["jdk.GCPhasePause"], + "eval_runs": 5, + "expected": { + "scoring_tier": "semantic", + "focusAreas": [], + "absent_area_ids": [], + "crossAreaCorrelations": [], + "rubric": { + "bottleneck_category": "memory-gc", + "severity": "high", + "required_evidence_keywords": ["humongous", "region", "fragmentation"], + "must_not_hallucinate": ["cpu-hotspot", "thread-contention"], + "judge_question": "Did the skill correctly identify that large object allocations are being placed in G1 humongous regions and causing fragmentation or premature full GC?" + } + } + }, + { + "id": "zgc_allocation_stalls", + "group": 4, + "description": "Very high allocation rate overwhelms ZGC concurrent collection; triggers allocation stalls", + "jvm_flags": ["-XX:+UseZGC", "-Xmx256m"], + "duration_seconds": 30, + "file": "zgc_allocation_stalls.jfr", + "jdk_version": "", + "jfr_event_types": [], + "required_event_types": ["jdk.GCPhasePause"], + "eval_runs": 5, + "expected": { + "scoring_tier": "semantic", + "focusAreas": [], + "absent_area_ids": [], + "crossAreaCorrelations": [], + "rubric": { + "bottleneck_category": "memory-gc", + "severity": "high", + "required_evidence_keywords": ["allocation stall", "ZGC", "concurrent", "throughput"], + "must_not_hallucinate": ["cpu-hotspot"], + "judge_question": "Did the skill identify that the allocation rate exceeds ZGC's concurrent collection throughput, causing allocation stalls (threads blocked waiting for GC to free memory)?" + } + } + }, + { + "id": "virtual_thread_pinning", + "group": 4, + "description": "Virtual threads pinned inside synchronized block; JDK 21+ only", + "jvm_flags": ["-Xmx256m"], + "duration_seconds": 30, + "file": "virtual_thread_pinning.jfr", + "jdk_version": "", + "jfr_event_types": [], + "required_event_types": ["jdk.VirtualThreadPinned"], + "eval_runs": 5, + "expected": { + "scoring_tier": "structural", + "focusAreas": [ + {"id": "virtual-thread-pinning", "impact": "high", "startHere": true} + ], + "absent_area_ids": [], + "crossAreaCorrelations": [], + "skip_if_jdk_lt": 21 + } + } + ] +} diff --git a/plugins/jfr-analyzer/eval/scripts/preflight.sh b/plugins/jfr-analyzer/eval/scripts/preflight.sh new file mode 100755 index 0000000..b5e8eb8 --- /dev/null +++ b/plugins/jfr-analyzer/eval/scripts/preflight.sh @@ -0,0 +1,133 @@ +#!/usr/bin/env bash +# Sourced by regen.sh and other eval scripts. +# Defines require_jbang, require_async_profiler, require_jq. +# Each function: check → explain why → prompt [y/N] → auto-install. +# After install, re-exports PATH / ASYNC_PROFILER_HOME. + +set -euo pipefail + +_preflight_os() { + uname -s | tr '[:upper:]' '[:lower:]' +} + +require_jbang() { + if command -v jbang &>/dev/null; then + return 0 + fi + echo "" + echo "jbang is required to build and run the eval test app." + echo "It manages Java version and dependency resolution without Maven." + printf "Install jbang now? [y/N] " + read -r _ans + if [[ "$_ans" != "y" && "$_ans" != "Y" ]]; then + echo "Aborted — install jbang manually: https://www.jbang.dev/download/" + exit 1 + fi + if [[ "$(_preflight_os)" == "darwin" ]] && command -v brew &>/dev/null; then + brew install jbangdev/tap/jbang + else + curl -Ls https://sh.jbang.dev | bash + export PATH="$HOME/.jbang/bin:$PATH" + fi + if ! command -v jbang &>/dev/null; then + echo "jbang install appears to have succeeded but is not in PATH." + echo "Open a new shell and re-run this script." + exit 1 + fi + echo "✓ jbang installed" +} + +require_async_profiler() { + local _ver="4.4" + local _dir="$HOME/.local/lib/async-profiler-${_ver}" + if [[ -n "${ASYNC_PROFILER_HOME:-}" && -f "$ASYNC_PROFILER_HOME/lib/libasyncProfiler.so" ]]; then + return 0 + fi + # macOS uses .dylib + if [[ -n "${ASYNC_PROFILER_HOME:-}" && -f "$ASYNC_PROFILER_HOME/lib/libasyncProfiler.dylib" ]]; then + return 0 + fi + if [[ -f "$_dir/lib/libasyncProfiler.so" || -f "$_dir/lib/libasyncProfiler.dylib" ]]; then + export ASYNC_PROFILER_HOME="$_dir" + return 0 + fi + echo "" + echo "async-profiler v${_ver} is required to record JFR profiles." + echo "It will be downloaded to $_dir" + printf "Download async-profiler v${_ver} now? [y/N] " + read -r _ans + if [[ "$_ans" != "y" && "$_ans" != "Y" ]]; then + echo "Aborted — download from: https://github.com/async-profiler/async-profiler/releases" + exit 1 + fi + mkdir -p "$_dir" + local _os + _os="$(_preflight_os)" + local _arch + _arch="$(uname -m)" + local _classifier _ext + if [[ "$_os" == "darwin" ]]; then + _classifier="macos" + _ext="zip" + elif [[ "$_arch" == "aarch64" || "$_arch" == "arm64" ]]; then + _classifier="linux-arm64" + _ext="tar.gz" + else + _classifier="linux-x64" + _ext="tar.gz" + fi + local _archive="async-profiler-${_ver}-${_classifier}.${_ext}" + local _url="https://github.com/async-profiler/async-profiler/releases/download/v${_ver}/${_archive}" + echo "Downloading $_url ..." + curl -fsSL "$_url" -o "/tmp/${_archive}" + if [[ "$_ext" == "zip" ]]; then + unzip -q "/tmp/${_archive}" -d "$_dir" + # zip puts files in a subdirectory — flatten one level if needed + local _inner + _inner=$(find "$_dir" -maxdepth 1 -mindepth 1 -type d | head -1) + if [[ -n "$_inner" && "$_inner" != "$_dir" ]]; then + mv "$_inner"/* "$_dir/" + rmdir "$_inner" + fi + else + tar -xzf "/tmp/${_archive}" -C "$_dir" --strip-components=1 + fi + rm "/tmp/${_archive}" + export ASYNC_PROFILER_HOME="$_dir" + echo "✓ async-profiler ${_ver} installed to $_dir" +} + +require_jq() { + if command -v jq &>/dev/null; then + return 0 + fi + echo "" + echo "jq is required for JSON manipulation in regen.sh." + printf "Install jq now? [y/N] " + read -r _ans + if [[ "$_ans" != "y" && "$_ans" != "Y" ]]; then + echo "Aborted — install jq: https://jqlang.github.io/jq/download/" + exit 1 + fi + if [[ "$(_preflight_os)" == "darwin" ]] && command -v brew &>/dev/null; then + brew install jq + elif command -v apt-get &>/dev/null; then + sudo apt-get install -y jq + elif command -v dnf &>/dev/null; then + sudo dnf install -y jq + else + echo "Cannot auto-install jq on this platform. Install manually." + exit 1 + fi + echo "✓ jq installed" +} + +# Determine the async-profiler native agent path (platform-aware). +# Call after require_async_profiler. Sets AP_AGENT_PATH. +ap_agent_path() { + if [[ -f "${ASYNC_PROFILER_HOME}/lib/libasyncProfiler.dylib" ]]; then + AP_AGENT_PATH="${ASYNC_PROFILER_HOME}/lib/libasyncProfiler.dylib" + else + AP_AGENT_PATH="${ASYNC_PROFILER_HOME}/lib/libasyncProfiler.so" + fi +} diff --git a/plugins/jfr-analyzer/eval/scripts/regen.sh b/plugins/jfr-analyzer/eval/scripts/regen.sh new file mode 100755 index 0000000..4cfd0e6 --- /dev/null +++ b/plugins/jfr-analyzer/eval/scripts/regen.sh @@ -0,0 +1,126 @@ +#!/usr/bin/env bash +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +EVAL_DIR="$(cd "$SCRIPT_DIR/.." && pwd)" +CORPUS_DIR="$EVAL_DIR/corpus" +TESTAPP_DIR="$EVAL_DIR/testapp" +MANIFEST="$CORPUS_DIR/manifest.json" + +# shellcheck source=preflight.sh +. "$SCRIPT_DIR/preflight.sh" +require_jbang +require_async_profiler +require_jq + +ap_agent_path # sets AP_AGENT_PATH + +echo "=== jfr-analyzer eval corpus regen ===" +echo "Corpus dir : $CORPUS_DIR" +echo "Agent path : $AP_AGENT_PATH" +echo "" + +# ── Build source flags (jbang v0.101 doesn't resolve //SOURCE from subdirs) ── +# Build -s= list for every .java except main.java. Run from TESTAPP_DIR +# so relative paths resolve correctly. +SOURCE_FLAGS="" +for f in "$TESTAPP_DIR"/*.java; do + base=$(basename "$f") + if [[ "$base" != "main.java" ]]; then + SOURCE_FLAGS="$SOURCE_FLAGS -s=$base" + fi +done + +# ── Build / dependency warm-up ──────────────────────────────────────────────── +echo "Step 1: Warming up jbang dependencies..." +# Run from TESTAPP_DIR so -s= relative paths resolve correctly +# shellcheck disable=SC2086 +(cd "$TESTAPP_DIR" && jbang --fresh $SOURCE_FLAGS --java-options=-Xmx64m \ + -D=scenarios=healthy_baseline -D=duration=2 main.java) 2>&1 \ + | grep -v "^Starting\|^Duration\|^Done\|^Scenario" \ + || true +echo "✓ Build warm-up done" +echo "" + +# ── Per-scenario recording ──────────────────────────────────────────────────── +echo "Step 2: Recording scenarios..." +SCENARIO_COUNT=$(jq '.scenarios | length' "$MANIFEST") +JDK_VERSION=$(java -version 2>&1 | head -1 | sed 's/.*"\(.*\)".*/\1/') + +for i in $(seq 0 $((SCENARIO_COUNT - 1))); do + ID=$(jq -r ".scenarios[$i].id" "$MANIFEST") + FILE=$(jq -r ".scenarios[$i].file" "$MANIFEST") + DURATION=$(jq -r ".scenarios[$i].duration_seconds" "$MANIFEST") + # Build --java-options= list for per-scenario JVM flags + JVM_FLAGS_JSON=$(jq -r ".scenarios[$i].jvm_flags | .[]" "$MANIFEST") + JVM_ARGS="" + while IFS= read -r flag; do + JVM_ARGS="$JVM_ARGS --java-options=$flag" + done <<< "$JVM_FLAGS_JSON" + + OUT_JFR="$CORPUS_DIR/$FILE" + echo " [$((i+1))/$SCENARIO_COUNT] $ID → $FILE (${DURATION}s) ..." + + # Run from TESTAPP_DIR so -s= relative paths resolve correctly + # shellcheck disable=SC2086 + (cd "$TESTAPP_DIR" && jbang \ + $SOURCE_FLAGS \ + "--java-options=-agentpath:${AP_AGENT_PATH}=start,event=cpu+alloc,interval=10ms,file=${OUT_JFR},jfr" \ + $JVM_ARGS \ + "-D=scenarios=${ID}" \ + "-D=duration=${DURATION}" \ + main.java) + + echo " ✓ recorded $FILE ($(du -sh "$OUT_JFR" | cut -f1))" +done + +echo "" +echo "Step 3: Computing source hash..." +# Hash covers all testapp .java files + per-scenario recording configs +HASH_INPUT="" +if command -v sha256sum &>/dev/null; then + hash_file() { sha256sum "$1" | awk '{print $1}'; } + hash_text() { printf '%s' "$1" | sha256sum | awk '{print $1}'; } +else + hash_file() { shasum -a 256 "$1" | awk '{print $1}'; } + hash_text() { printf '%s' "$1" | shasum -a 256 | awk '{print $1}'; } +fi +while IFS= read -r -d '' f; do + HASH_INPUT="$HASH_INPUT$(hash_file "$f")" +done < <(find "$TESTAPP_DIR" -name '*.java' -print0 | sort -z) + +# Append recording configs (id, jvm_flags, duration_seconds) +RECORDING_CONFIG=$(jq -c '[.scenarios[] | {id, jvm_flags, duration_seconds}]' "$MANIFEST") +HASH_INPUT="$HASH_INPUT$RECORDING_CONFIG" + +SOURCE_HASH="sha256:$(hash_text "$HASH_INPUT")" +echo " hash: $SOURCE_HASH" + +# ── Partial manifest update ─────────────────────────────────────────────────── +echo "" +echo "Step 4: Updating manifest.json (generated_at, source_hash, jdk_version)..." +GENERATED_AT=$(date -u +"%Y-%m-%dT%H:%M:%SZ") + +# Update top-level fields +TMP=$(mktemp) +jq \ + --arg ga "$GENERATED_AT" \ + --arg sh "$SOURCE_HASH" \ + '.generated_at = $ga | .source_hash = $sh' \ + "$MANIFEST" > "$TMP" + +# Update jdk_version per scenario +SCENARIO_COUNT=$(jq '.scenarios | length' "$TMP") +for i in $(seq 0 $((SCENARIO_COUNT - 1))); do + jq --arg v "$JDK_VERSION" --argjson i "$i" \ + '.scenarios[$i].jdk_version = $v' \ + "$TMP" > "${TMP}.2" && mv "${TMP}.2" "$TMP" +done + +mv "$TMP" "$MANIFEST" +echo " ✓ manifest.json updated" + +echo "" +echo "=== regen.sh complete ===" +echo "Next step: run '/jfr-analyzer eval regen' in Claude Code to extract" +echo "event types from the JFR files via MCP and validate oracle outputs." diff --git a/plugins/jfr-analyzer/eval/scripts/requirements.txt b/plugins/jfr-analyzer/eval/scripts/requirements.txt new file mode 100644 index 0000000..a617bce --- /dev/null +++ b/plugins/jfr-analyzer/eval/scripts/requirements.txt @@ -0,0 +1,3 @@ +anthropic>=0.40.0 +openai>=1.50.0 +scipy>=1.13.0 diff --git a/plugins/jfr-analyzer/eval/scripts/score.py b/plugins/jfr-analyzer/eval/scripts/score.py new file mode 100644 index 0000000..36c6aa4 --- /dev/null +++ b/plugins/jfr-analyzer/eval/scripts/score.py @@ -0,0 +1,420 @@ +#!/usr/bin/env python3 +""" +jfr-analyzer eval scorer. + +Usage: + python score.py [--scenario ] [--multi-judge] + +Reads: eval/results//run-*/focus.json and .../breakpoint.txt +Reads: eval/corpus/manifest.json +Writes: eval/results/report.md + +Prerequisites: + ANTHROPIC_API_KEY — required for semantic scoring + OPENAI_API_KEY — optional; enables multi-judge mode +""" +import argparse +import json +import math +import os +import sys +from pathlib import Path + +EVAL_DIR = Path(__file__).parent.parent +CORPUS_DIR = EVAL_DIR / "corpus" +RESULTS_DIR = EVAL_DIR / "results" +MANIFEST = CORPUS_DIR / "manifest.json" + + +# ── Data loading ────────────────────────────────────────────────────────────── + +def load_manifest() -> dict: + with open(MANIFEST) as f: + return json.load(f) + + +def load_run_outputs(scenario_id: str, n_runs: int) -> list: + """Load focus.json + breakpoint.txt for each run.""" + runs = [] + for k in range(1, n_runs + 1): + run_dir = RESULTS_DIR / scenario_id / f"run-{k}" + focus_path = run_dir / "focus.json" + bp_path = run_dir / "breakpoint.txt" + if not focus_path.exists(): + continue + with open(focus_path) as f: + focus = json.load(f) + breakpoint_text = bp_path.read_text() if bp_path.exists() else "" + runs.append({"focus": focus, "breakpoint_text": breakpoint_text, "run": k}) + return runs + + +# ── Structural scoring ──────────────────────────────────────────────────────── + +def score_structural(scenario: dict, runs: list) -> dict: + """Structural exact-match scoring for Groups 1 & 2.""" + expected = scenario["expected"] + exp_areas = {a["id"]: a for a in expected.get("focusAreas", [])} + exp_absent = set(expected.get("absent_area_ids", [])) + exp_corr_kinds = {c["kind"] for c in expected.get("crossAreaCorrelations", [])} + exp_start_here = {a["id"] for a in expected.get("focusAreas", []) if a.get("startHere")} + + pass_threshold_recall = 0.9 + pass_threshold_precision = 0.8 + + per_run = [] + for run in runs: + focus = run["focus"] + det_areas = {a["id"]: a for a in focus.get("focusAreas", [])} + det_kinds = {c["kind"] for c in focus.get("crossAreaCorrelations", [])} + det_start = {a["id"] for a in focus.get("focusAreas", []) if a.get("startHere")} + + # Recall: expected areas found + recall = (len(set(exp_areas) & set(det_areas)) / len(exp_areas) + if exp_areas else (1.0 if not det_areas else 0.0)) + + # Precision: detected areas that are expected + precision = (len(set(exp_areas) & set(det_areas)) / len(det_areas) + if det_areas else (1.0 if not exp_areas else 0.0)) + + # startHere accuracy + start_here_acc = (len(exp_start_here & det_start) / len(exp_start_here) + if exp_start_here else (1.0 if not det_start else 0.0)) + + # Absent area penalty + absent_penalty = sum(-0.5 for aid in exp_absent if aid in det_areas) + + # Correlation recall + corr_recall = (len(exp_corr_kinds & det_kinds) / len(exp_corr_kinds) + if exp_corr_kinds else None) + + passed = ( + recall >= pass_threshold_recall + and precision >= pass_threshold_precision + and (not exp_start_here or start_here_acc >= 0.5) + ) + + per_run.append({ + "run": run["run"], + "recall": round(recall, 3), + "precision": round(precision, 3), + "start_here_acc": round(start_here_acc, 3), + "absent_penalty": absent_penalty, + "corr_recall": round(corr_recall, 3) if corr_recall is not None else None, + "passed": passed, + }) + + n_total = len(per_run) + n_passed = sum(1 for r in per_run if r["passed"]) + + pass_at_1 = per_run[0]["passed"] if per_run else False + pass_at_k = _pass_at_k(n_total, n_passed, k=n_total) if n_total > 1 else float(pass_at_1) + + mean_recall = _mean([r["recall"] for r in per_run]) + mean_precision = _mean([r["precision"] for r in per_run]) + ci_recall = _ci95([r["recall"] for r in per_run]) + + start_majority = (sum(1 for r in per_run if r["start_here_acc"] >= 0.5) > n_total / 2 + if n_total > 0 else False) + + overall_pass = n_passed >= math.ceil(n_total * 0.6) and start_majority + + return { + "tier": "structural", + "per_run": per_run, + "n_runs": n_total, + "n_passed": n_passed, + "pass_at_1": pass_at_1, + "pass_at_k": round(pass_at_k, 3), + "mean_recall": round(mean_recall, 3), + "mean_precision": round(mean_precision, 3), + "ci_recall_95": [round(v, 3) for v in ci_recall], + "start_majority": start_majority, + "overall_pass": overall_pass, + } + + +# ── Semantic scoring ────────────────────────────────────────────────────────── + +JUDGE_PROMPT_TMPL = """\ +SCENARIO: {description} +JUDGE QUESTION: {judge_question} +REQUIRED EVIDENCE: {required_evidence_keywords} +MUST NOT APPEAR: {must_not_hallucinate} + +SKILL OUTPUT: +--- focus.json --- +{focus_json} +--- breakpoint text --- +{breakpoint_text} +--- + +RUBRIC: +0.0 — missed the problem or wrong root cause +0.5 — right area, wrong mechanism or severity +1.0 — correct problem, root cause, and severity + +Step 1: reason through what the skill found (2-3 sentences) +Step 2: list which required_evidence_keywords appear +Step 3: list any must_not_hallucinate items that appeared +Step 4: output ONLY JSON: {{"reasoning":"...","evidence_found":[...],"hallucinations":[...],"score":0.0}} +""" + + +def _call_claude(prompt: str) -> dict: + import anthropic + client = anthropic.Anthropic(api_key=os.environ["ANTHROPIC_API_KEY"]) + msg = client.messages.create( + model="claude-opus-4-8", + max_tokens=1024, + messages=[{"role": "user", "content": prompt}], + ) + text = msg.content[0].text + start = text.rfind("{") + end = text.rfind("}") + 1 + if start == -1: + raise ValueError(f"No JSON in Claude response: {text!r}") + return json.loads(text[start:end]) + + +def _call_gpt4o(prompt: str) -> dict: + import openai + client = openai.OpenAI(api_key=os.environ["OPENAI_API_KEY"]) + resp = client.chat.completions.create( + model="gpt-4o", + messages=[{"role": "user", "content": prompt}], + max_tokens=1024, + ) + text = resp.choices[0].message.content + start = text.rfind("{") + end = text.rfind("}") + 1 + return json.loads(text[start:end]) + + +def score_semantic(scenario: dict, runs: list, multi_judge: bool) -> dict: + """LLM-as-judge scoring for Groups 3 & 4.""" + rubric = scenario["expected"]["rubric"] + description = scenario["description"] + pass_threshold = 0.7 + per_run = [] + + for run in runs: + focus_json = json.dumps(run["focus"], indent=2) + bp_text = run["breakpoint_text"] + + prompt = JUDGE_PROMPT_TMPL.format( + description=description, + judge_question=rubric["judge_question"], + required_evidence_keywords=", ".join(rubric["required_evidence_keywords"]), + must_not_hallucinate=", ".join(rubric["must_not_hallucinate"]), + focus_json=focus_json, + breakpoint_text=bp_text, + ) + + scores = [] + verdicts = [] + + try: + v = _call_claude(prompt) + scores.append(v["score"]) + verdicts.append({"provider": "claude", **v}) + except Exception as e: + verdicts.append({"provider": "claude", "error": str(e), "score": 0.0}) + scores.append(0.0) + + if multi_judge and os.environ.get("OPENAI_API_KEY"): + try: + v = _call_gpt4o(prompt) + scores.append(v["score"]) + verdicts.append({"provider": "gpt-4o", **v}) + except Exception as e: + verdicts.append({"provider": "gpt-4o", "error": str(e), "score": 0.0}) + scores.append(0.0) + + mean_score = _mean(scores) + disagreement = (len(scores) > 1 and abs(scores[0] - scores[-1]) > 0.5) + passed = mean_score >= pass_threshold and not disagreement + + per_run.append({ + "run": run["run"], + "mean_score": round(mean_score, 3), + "verdicts": verdicts, + "disagreement": disagreement, + "passed": passed, + }) + + n_total = len(per_run) + n_passed = sum(1 for r in per_run if r["passed"]) + pass_at_1 = per_run[0]["passed"] if per_run else False + pass_at_k = _pass_at_k(n_total, n_passed, k=n_total) if n_total > 1 else float(pass_at_1) + mean_score = _mean([r["mean_score"] for r in per_run]) + needs_review = any(r["disagreement"] for r in per_run) + overall_pass = n_passed >= math.ceil(n_total * 0.6) and not needs_review + + return { + "tier": "semantic", + "per_run": per_run, + "n_runs": n_total, + "n_passed": n_passed, + "pass_at_1": pass_at_1, + "pass_at_k": round(pass_at_k, 3), + "mean_score": round(mean_score, 3), + "needs_review": needs_review, + "overall_pass": overall_pass, + } + + +# ── Statistics helpers ──────────────────────────────────────────────────────── + +def _mean(values: list) -> float: + return sum(values) / len(values) if values else 0.0 + + +def _ci95(values: list) -> tuple: + if len(values) < 2: + v = values[0] if values else 0.0 + return (v, v) + from scipy import stats + ci = stats.t.interval(0.95, df=len(values) - 1, + loc=_mean(values), + scale=stats.sem(values)) + return (max(0.0, ci[0]), min(1.0, ci[1])) + + +def _pass_at_k(n: int, c: int, k: int) -> float: + """pass@k = 1 - C(n-c, k) / C(n, k)""" + if n - c < k: + return 1.0 + return 1.0 - math.comb(n - c, k) / math.comb(n, k) + + +# ── Report rendering ────────────────────────────────────────────────────────── + +def render_report(results: list, manifest: dict, date: str) -> str: + structural = [r for r in results if r["score"]["tier"] == "structural"] + semantic = [r for r in results if r["score"]["tier"] == "semantic"] + + lines = [f"# jfr-analyzer Eval Results — {date}", ""] + + if structural: + n_runs = structural[0]["score"]["n_runs"] if structural else 0 + lines += [f"## Structural (Groups 1 & 2) — {n_runs} runs each", ""] + lines += ["| Scenario | Recall p@1 | p@k | Precision p@1 | startHere | Corr. | Result |"] + lines += ["|----------|-----------|-----|---------------|-----------|-------|--------|"] + for r in structural: + s = r["score"] + pr = s["per_run"][0] if s["per_run"] else {} + corr = pr.get("corr_recall") + corr_str = f"{corr:.1f}" if corr is not None else "N/A" + sh_votes = sum(1 for x in s["per_run"] if x.get("start_here_acc", 0) >= 0.5) + result_str = "**PASS**" if s["overall_pass"] else "FAIL" + lines.append( + f"| {r['id']:<28} " + f"| {pr.get('recall', 0):.1f} " + f"| {s['pass_at_k']:.1f} " + f"| {pr.get('precision', 0):.1f} " + f"| {'✓' if s['start_majority'] else '✗'} {sh_votes}/{s['n_runs']} " + f"| {corr_str} " + f"| {result_str} |" + ) + lines.append("") + + if semantic: + n_runs = semantic[0]["score"]["n_runs"] if semantic else 0 + lines += [f"## Semantic (Groups 3 & 4) — {n_runs} runs each", ""] + lines += ["| Scenario | Score p@1 | p@k | Agreement | Result |"] + lines += ["|----------|-----------|-----|-----------|--------|"] + for r in semantic: + s = r["score"] + pr = s["per_run"][0] if s["per_run"] else {} + agree = "⚠️" if s["needs_review"] else "✓" + result_str = ("**PASS**" if s["overall_pass"] + else ("⚠️ needs-human-review" if s["needs_review"] else "FAIL")) + lines.append( + f"| {r['id']:<28} " + f"| {pr.get('mean_score', 0):.1f} " + f"| {s['pass_at_k']:.1f} " + f"| {agree} " + f"| {result_str} |" + ) + lines.append("") + + struct_pass = sum(1 for r in structural if r["score"]["overall_pass"]) + sem_pass = sum(1 for r in semantic if r["score"]["overall_pass"]) + struct_recall = (_mean([r["score"]["mean_recall"] for r in structural]) + if structural else None) + struct_prec = (_mean([r["score"]["mean_precision"] for r in structural]) + if structural else None) + sem_score = (_mean([r["score"]["mean_score"] for r in semantic]) + if semantic else None) + + lines += ["## Summary"] + if structural: + lines.append( + f"Structural: {struct_pass}/{len(structural)} PASS" + f" · mean recall {struct_recall:.2f}" + f" · mean precision {struct_prec:.2f}" + ) + if semantic: + lines.append( + f"Semantic: {sem_pass}/{len(semantic)} PASS" + f" · mean judge score {sem_score:.2f}" + ) + return "\n".join(lines) + "\n" + + +# ── CLI ─────────────────────────────────────────────────────────────────────── + +def main(): + parser = argparse.ArgumentParser(description="Score jfr-analyzer eval runs") + parser.add_argument("--scenario", help="Score only this scenario ID") + parser.add_argument("--multi-judge", action="store_true", + help="Use Claude + GPT-4o as judge panel") + args = parser.parse_args() + + if "ANTHROPIC_API_KEY" not in os.environ: + print("ERROR: ANTHROPIC_API_KEY not set (required for semantic scoring)") + sys.exit(1) + + manifest = load_manifest() + scenarios = manifest["scenarios"] + if args.scenario: + scenarios = [s for s in scenarios if s["id"] == args.scenario] + if not scenarios: + print(f"ERROR: scenario '{args.scenario}' not found in manifest") + sys.exit(1) + + results = [] + for scenario in scenarios: + sid = scenario["id"] + tier = scenario["expected"]["scoring_tier"] + n_runs = scenario.get("eval_runs", 5) + + runs = load_run_outputs(sid, n_runs) + if not runs: + print(f" [{sid}] No run outputs found — skipping") + continue + + print(f" [{sid}] scoring {len(runs)} runs ({tier})...") + if tier == "structural": + score = score_structural(scenario, runs) + else: + score = score_semantic(scenario, runs, args.multi_judge) + + results.append({"id": sid, "group": scenario["group"], "score": score}) + status = "PASS" if score["overall_pass"] else "FAIL" + print(f" → {status}") + + from datetime import datetime, timezone + date_str = datetime.now(timezone.utc).strftime("%Y-%m-%d") + report = render_report(results, manifest, date_str) + + report_path = RESULTS_DIR / "report.md" + RESULTS_DIR.mkdir(parents=True, exist_ok=True) + report_path.write_text(report) + print(f"\nReport written to {report_path}") + print(report) + + +if __name__ == "__main__": + main() diff --git a/plugins/jfr-analyzer/eval/testapp/AllocGcCascade.java b/plugins/jfr-analyzer/eval/testapp/AllocGcCascade.java new file mode 100644 index 0000000..5ac0f77 --- /dev/null +++ b/plugins/jfr-analyzer/eval/testapp/AllocGcCascade.java @@ -0,0 +1,52 @@ +import java.util.concurrent.*; +import java.util.Random; + +public class AllocGcCascade implements main.Scenario { + @Override public String id() { return "alloc_gc_cascade"; } + + @Override public void run(long durationMs) throws Exception { + BlockingQueue queue = new ArrayBlockingQueue<>(2); + ExecutorService consumers = Executors.newFixedThreadPool(4); + long deadline = System.currentTimeMillis() + durationMs; + Random rng = new Random(); + + for (int i = 0; i < 4; i++) { + consumers.submit(() -> { + while (!Thread.currentThread().isInterrupted() + && System.currentTimeMillis() < deadline) { + try { + byte[] item = queue.poll(100, TimeUnit.MILLISECONDS); + if (item != null) { + long sum = item[0]; + if (sum < Long.MIN_VALUE) throw new AssertionError(); + } + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + break; + } + } + }); + } + + ExecutorService submitters = Executors.newFixedThreadPool(16); + for (int i = 0; i < 16; i++) { + submitters.submit(() -> { + while (!Thread.currentThread().isInterrupted() + && System.currentTimeMillis() < deadline) { + try { + byte[] payload = new byte[2048 + rng.nextInt(14336)]; + payload[0] = (byte) Thread.currentThread().getId(); + queue.offer(payload, 50, TimeUnit.MILLISECONDS); + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + break; + } + } + }); + } + + Thread.sleep(durationMs); + consumers.shutdownNow(); + submitters.shutdownNow(); + } +} diff --git a/plugins/jfr-analyzer/eval/testapp/AllocationPressure.java b/plugins/jfr-analyzer/eval/testapp/AllocationPressure.java new file mode 100644 index 0000000..907aa8e --- /dev/null +++ b/plugins/jfr-analyzer/eval/testapp/AllocationPressure.java @@ -0,0 +1,19 @@ +public class AllocationPressure implements main.Scenario { + private static final int BUF_COUNT = 256; + private static final int BUF_SIZE = 100_000; + + @Override public String id() { return "allocation_pressure"; } + + @Override public void run(long durationMs) throws Exception { + byte[][] ring = new byte[BUF_COUNT][]; + long deadline = System.currentTimeMillis() + durationMs; + int idx = 0; + while (!Thread.currentThread().isInterrupted() + && System.currentTimeMillis() < deadline) { + ring[idx % BUF_COUNT] = new byte[BUF_SIZE]; + ring[idx % BUF_COUNT][0] = (byte) idx; + idx++; + if (idx % 100 == 0) Thread.sleep(1); + } + } +} diff --git a/plugins/jfr-analyzer/eval/testapp/CfFanoutLargeHeap.java b/plugins/jfr-analyzer/eval/testapp/CfFanoutLargeHeap.java new file mode 100644 index 0000000..16719ae --- /dev/null +++ b/plugins/jfr-analyzer/eval/testapp/CfFanoutLargeHeap.java @@ -0,0 +1,34 @@ +import java.util.concurrent.*; +import java.util.ArrayList; +import java.util.List; + +public class CfFanoutLargeHeap implements main.Scenario { + private static final int FAN_WIDTH = 40; + private static final int CHUNK_BYTES = 50_000_000; + + @Override public String id() { return "cf_fanout_large_heap"; } + + @Override public void run(long durationMs) throws Exception { + long deadline = System.currentTimeMillis() + durationMs; + while (!Thread.currentThread().isInterrupted() + && System.currentTimeMillis() < deadline) { + List> futures = new ArrayList<>(FAN_WIDTH); + for (int i = 0; i < FAN_WIDTH; i++) { + futures.add(CompletableFuture.supplyAsync(() -> { + byte[] chunk = new byte[CHUNK_BYTES]; + chunk[0] = 1; + try { Thread.sleep(200); } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + } + return chunk; + })); + } + List results = new ArrayList<>(FAN_WIDTH); + for (var f : futures) { + try { results.add(f.get(5, TimeUnit.SECONDS)); } + catch (Exception ignored) {} + } + // results goes out of scope → GC triggered + } + } +} diff --git a/plugins/jfr-analyzer/eval/testapp/CpuHotspot.java b/plugins/jfr-analyzer/eval/testapp/CpuHotspot.java new file mode 100644 index 0000000..f1784bb --- /dev/null +++ b/plugins/jfr-analyzer/eval/testapp/CpuHotspot.java @@ -0,0 +1,23 @@ +import java.util.concurrent.*; + +public class CpuHotspot implements main.Scenario { + @Override public String id() { return "cpu_hotspot"; } + + @Override public void run(long durationMs) throws Exception { + int cpus = Runtime.getRuntime().availableProcessors(); + ExecutorService pool = Executors.newFixedThreadPool(cpus); + long deadline = System.currentTimeMillis() + durationMs; + for (int i = 0; i < cpus; i++) { + pool.submit(() -> { + double acc = 0; + while (!Thread.currentThread().isInterrupted() + && System.currentTimeMillis() < deadline) { + acc += Math.sin(Math.random()) + Math.pow(Math.random(), 3.7); + } + return acc; + }); + } + Thread.sleep(durationMs); + pool.shutdownNow(); + } +} diff --git a/plugins/jfr-analyzer/eval/testapp/G1HumongousAlloc.java b/plugins/jfr-analyzer/eval/testapp/G1HumongousAlloc.java new file mode 100644 index 0000000..2711002 --- /dev/null +++ b/plugins/jfr-analyzer/eval/testapp/G1HumongousAlloc.java @@ -0,0 +1,19 @@ +public class G1HumongousAlloc implements main.Scenario { + private static final int OBJECT_SIZE = 600_000; + + @Override public String id() { return "g1_humongous_alloc"; } + + @Override public void run(long durationMs) throws Exception { + long deadline = System.currentTimeMillis() + durationMs; + byte[] last = null; + long iter = 0; + while (!Thread.currentThread().isInterrupted() + && System.currentTimeMillis() < deadline) { + last = new byte[OBJECT_SIZE]; + last[0] = (byte) iter; + iter++; + if (iter % 100 == 0) Thread.yield(); + } + if (last == null) throw new AssertionError(); + } +} diff --git a/plugins/jfr-analyzer/eval/testapp/GcCpuInflation.java b/plugins/jfr-analyzer/eval/testapp/GcCpuInflation.java new file mode 100644 index 0000000..48b8f2a --- /dev/null +++ b/plugins/jfr-analyzer/eval/testapp/GcCpuInflation.java @@ -0,0 +1,28 @@ +import java.util.concurrent.*; +import java.util.Random; + +public class GcCpuInflation implements main.Scenario { + @Override public String id() { return "gc_cpu_inflation"; } + + @Override public void run(long durationMs) throws Exception { + int cpus = Runtime.getRuntime().availableProcessors(); + ExecutorService pool = Executors.newFixedThreadPool(cpus); + long deadline = System.currentTimeMillis() + durationMs; + Random rng = new Random(); + + for (int i = 0; i < cpus; i++) { + pool.submit(() -> { + long count = 0; + while (!Thread.currentThread().isInterrupted() + && System.currentTimeMillis() < deadline) { + byte[] b = new byte[512 + rng.nextInt(512)]; + b[0] = (byte) count; + count++; + } + return count; + }); + } + Thread.sleep(durationMs); + pool.shutdownNow(); + } +} diff --git a/plugins/jfr-analyzer/eval/testapp/GcPressureParallel.java b/plugins/jfr-analyzer/eval/testapp/GcPressureParallel.java new file mode 100644 index 0000000..8ebcd1c --- /dev/null +++ b/plugins/jfr-analyzer/eval/testapp/GcPressureParallel.java @@ -0,0 +1,27 @@ +import java.util.concurrent.*; +import java.util.Random; + +public class GcPressureParallel implements main.Scenario { + @Override public String id() { return "gc_pressure_parallel"; } + + @Override public void run(long durationMs) throws Exception { + int cpus = Runtime.getRuntime().availableProcessors(); + ExecutorService pool = Executors.newFixedThreadPool(cpus); + long deadline = System.currentTimeMillis() + durationMs; + Random rng = new Random(); + for (int i = 0; i < cpus; i++) { + pool.submit(() -> { + long sum = 0; + while (!Thread.currentThread().isInterrupted() + && System.currentTimeMillis() < deadline) { + byte[] buf = new byte[1024 + rng.nextInt(9216)]; + buf[0] = (byte) sum; + sum += buf.length; + } + return sum; + }); + } + Thread.sleep(durationMs); + pool.shutdownNow(); + } +} diff --git a/plugins/jfr-analyzer/eval/testapp/GcQueueCoupling.java b/plugins/jfr-analyzer/eval/testapp/GcQueueCoupling.java new file mode 100644 index 0000000..77fed37 --- /dev/null +++ b/plugins/jfr-analyzer/eval/testapp/GcQueueCoupling.java @@ -0,0 +1,28 @@ +import java.util.concurrent.*; +import java.util.Random; + +public class GcQueueCoupling implements main.Scenario { + @Override public String id() { return "gc_queue_coupling"; } + + @Override public void run(long durationMs) throws Exception { + ThreadPoolExecutor pool = new ThreadPoolExecutor( + 2, 2, 60L, TimeUnit.SECONDS, + new ArrayBlockingQueue<>(20), + new ThreadPoolExecutor.CallerRunsPolicy() + ); + long deadline = System.currentTimeMillis() + durationMs; + Random rng = new Random(); + + while (!Thread.currentThread().isInterrupted() + && System.currentTimeMillis() < deadline) { + final byte[] payload = new byte[4096 + rng.nextInt(8192)]; + payload[0] = (byte) rng.nextInt(); + pool.submit(() -> { + long sum = payload[0]; + if (sum < Byte.MIN_VALUE) throw new AssertionError(); + }); + Thread.sleep(1); + } + pool.shutdown(); + } +} diff --git a/plugins/jfr-analyzer/eval/testapp/HealthyBaseline.java b/plugins/jfr-analyzer/eval/testapp/HealthyBaseline.java new file mode 100644 index 0000000..841840c --- /dev/null +++ b/plugins/jfr-analyzer/eval/testapp/HealthyBaseline.java @@ -0,0 +1,11 @@ +public class HealthyBaseline implements main.Scenario { + @Override public String id() { return "healthy_baseline"; } + + @Override public void run(long durationMs) throws Exception { + long deadline = System.currentTimeMillis() + durationMs; + while (!Thread.currentThread().isInterrupted() + && System.currentTimeMillis() < deadline) { + Thread.sleep(100); + } + } +} diff --git a/plugins/jfr-analyzer/eval/testapp/ReactiveBackpressure.java b/plugins/jfr-analyzer/eval/testapp/ReactiveBackpressure.java new file mode 100644 index 0000000..d8ba8d2 --- /dev/null +++ b/plugins/jfr-analyzer/eval/testapp/ReactiveBackpressure.java @@ -0,0 +1,23 @@ +import reactor.core.publisher.Flux; +import reactor.core.scheduler.Schedulers; +import java.time.Duration; + +public class ReactiveBackpressure implements main.Scenario { + @Override public String id() { return "reactive_backpressure"; } + + @Override public void run(long durationMs) throws Exception { + var subscription = Flux.interval(Duration.ofMillis(1)) + .onBackpressureBuffer(256) + .publishOn(Schedulers.boundedElastic()) + .subscribe(i -> { + try { + Thread.sleep(10); + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + } + }); + + Thread.sleep(durationMs); + subscription.dispose(); + } +} diff --git a/plugins/jfr-analyzer/eval/testapp/ReactiveNoBackpressure.java b/plugins/jfr-analyzer/eval/testapp/ReactiveNoBackpressure.java new file mode 100644 index 0000000..9be852d --- /dev/null +++ b/plugins/jfr-analyzer/eval/testapp/ReactiveNoBackpressure.java @@ -0,0 +1,22 @@ +import reactor.core.publisher.Flux; +import reactor.core.scheduler.Schedulers; +import java.time.Duration; + +public class ReactiveNoBackpressure implements main.Scenario { + @Override public String id() { return "reactive_no_backpressure"; } + + @Override public void run(long durationMs) throws Exception { + var subscription = Flux.interval(Duration.ofMillis(1)) + .publishOn(Schedulers.boundedElastic()) + .subscribe(i -> { + try { + Thread.sleep(10); + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + } + }); + + Thread.sleep(durationMs); + subscription.dispose(); + } +} diff --git a/plugins/jfr-analyzer/eval/testapp/Safepoints.java b/plugins/jfr-analyzer/eval/testapp/Safepoints.java new file mode 100644 index 0000000..8a2cc75 --- /dev/null +++ b/plugins/jfr-analyzer/eval/testapp/Safepoints.java @@ -0,0 +1,23 @@ +import java.lang.reflect.*; + +public class Safepoints implements main.Scenario { + @Override public String id() { return "safepoints"; } + + @Override public void run(long durationMs) throws Exception { + long deadline = System.currentTimeMillis() + durationMs; + Class cls = null; + Method method = null; + int iter = 0; + while (!Thread.currentThread().isInterrupted() + && System.currentTimeMillis() < deadline) { + cls = Class.forName("java.util.HashMap"); + if (method == null) { + method = cls.getDeclaredMethod("size"); + } + Object instance = cls.getDeclaredConstructor().newInstance(); + method.invoke(instance); + iter++; + if (iter % 1000 == 0) Thread.yield(); + } + } +} diff --git a/plugins/jfr-analyzer/eval/testapp/TemporalSpikes.java b/plugins/jfr-analyzer/eval/testapp/TemporalSpikes.java new file mode 100644 index 0000000..3aad5b1 --- /dev/null +++ b/plugins/jfr-analyzer/eval/testapp/TemporalSpikes.java @@ -0,0 +1,22 @@ +public class TemporalSpikes implements main.Scenario { + @Override public String id() { return "temporal_spikes"; } + + @Override public void run(long durationMs) throws Exception { + long deadline = System.currentTimeMillis() + durationMs; + while (!Thread.currentThread().isInterrupted() + && System.currentTimeMillis() < deadline) { + long burstEnd = System.currentTimeMillis() + 5000; + double acc = 0; + while (System.currentTimeMillis() < burstEnd + && System.currentTimeMillis() < deadline) { + acc += Math.sin(Math.random()) + Math.pow(Math.random(), 2.5); + } + if (acc < 0) throw new AssertionError(); // prevent elimination + long sleepUntil = System.currentTimeMillis() + 5000; + while (System.currentTimeMillis() < sleepUntil + && System.currentTimeMillis() < deadline) { + Thread.sleep(100); + } + } + } +} diff --git a/plugins/jfr-analyzer/eval/testapp/ThreadContentionSync.java b/plugins/jfr-analyzer/eval/testapp/ThreadContentionSync.java new file mode 100644 index 0000000..914e4ea --- /dev/null +++ b/plugins/jfr-analyzer/eval/testapp/ThreadContentionSync.java @@ -0,0 +1,34 @@ +import java.util.concurrent.*; +import java.util.concurrent.atomic.AtomicLong; + +public class ThreadContentionSync implements main.Scenario { + private final AtomicLong counter = new AtomicLong(); + + @Override public String id() { return "thread_contention_sync"; } + + private synchronized void contendedWork() throws InterruptedException { + Thread.sleep(1); + counter.incrementAndGet(); + } + + @Override public void run(long durationMs) throws Exception { + int threads = Runtime.getRuntime().availableProcessors() * 4; + ExecutorService pool = Executors.newFixedThreadPool(threads); + long deadline = System.currentTimeMillis() + durationMs; + for (int i = 0; i < threads; i++) { + pool.submit(() -> { + while (!Thread.currentThread().isInterrupted() + && System.currentTimeMillis() < deadline) { + try { + contendedWork(); + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + break; + } + } + }); + } + Thread.sleep(durationMs); + pool.shutdownNow(); + } +} diff --git a/plugins/jfr-analyzer/eval/testapp/VirtualThreadPinning.java b/plugins/jfr-analyzer/eval/testapp/VirtualThreadPinning.java new file mode 100644 index 0000000..385eca0 --- /dev/null +++ b/plugins/jfr-analyzer/eval/testapp/VirtualThreadPinning.java @@ -0,0 +1,44 @@ +import java.util.concurrent.*; + +public class VirtualThreadPinning implements main.Scenario { + private final Object lock = new Object(); + + @Override public String id() { return "virtual_thread_pinning"; } + + @Override public void run(long durationMs) throws Exception { + int jdkMajor; + try { + String version = System.getProperty("java.specification.version"); + jdkMajor = Integer.parseInt(version.split("\\.")[0]); + } catch (NumberFormatException e) { + jdkMajor = 8; + } + if (jdkMajor < 21) { + System.out.println("[virtual_thread_pinning] JDK < 21 detected — skipping scenario"); + Thread.sleep(durationMs); + return; + } + + int concurrency = Runtime.getRuntime().availableProcessors() * 8; + long deadline = System.currentTimeMillis() + durationMs; + ExecutorService vte = Executors.newVirtualThreadPerTaskExecutor(); + + for (int i = 0; i < concurrency; i++) { + vte.submit(() -> { + while (!Thread.currentThread().isInterrupted() + && System.currentTimeMillis() < deadline) { + synchronized (lock) { + try { + Thread.sleep(5); + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + break; + } + } + } + }); + } + Thread.sleep(durationMs); + vte.shutdownNow(); + } +} diff --git a/plugins/jfr-analyzer/eval/testapp/ZgcAllocationStalls.java b/plugins/jfr-analyzer/eval/testapp/ZgcAllocationStalls.java new file mode 100644 index 0000000..0b3b990 --- /dev/null +++ b/plugins/jfr-analyzer/eval/testapp/ZgcAllocationStalls.java @@ -0,0 +1,27 @@ +import java.util.Random; + +public class ZgcAllocationStalls implements main.Scenario { + @Override public String id() { return "zgc_allocation_stalls"; } + + @Override public void run(long durationMs) throws Exception { + int cpus = Runtime.getRuntime().availableProcessors(); + long deadline = System.currentTimeMillis() + durationMs; + Random rng = new Random(); + + Thread[] threads = new Thread[cpus]; + for (int i = 0; i < cpus; i++) { + threads[i] = Thread.ofPlatform().name("zgc-stall-" + i).start(() -> { + long sum = 0; + while (!Thread.currentThread().isInterrupted() + && System.currentTimeMillis() < deadline) { + byte[] b = new byte[rng.nextInt(100_000)]; + b[0] = (byte) sum; + sum += b.length; + } + if (sum < 0) throw new AssertionError(); + }); + } + Thread.sleep(durationMs); + for (Thread t : threads) { t.interrupt(); t.join(1000); } + } +} diff --git a/plugins/jfr-analyzer/eval/testapp/main.java b/plugins/jfr-analyzer/eval/testapp/main.java new file mode 100644 index 0000000..77f43cf --- /dev/null +++ b/plugins/jfr-analyzer/eval/testapp/main.java @@ -0,0 +1,93 @@ +///usr/bin/env jbang "$0" "$@" ; exit $? +//JAVA 21 +//DEPS io.projectreactor:reactor-core:3.7.0 +// Scenario files live in the same directory. regen.sh passes them via -s= flags +// because jbang v0.101 does not resolve //SOURCE from subdirectories reliably. + +import java.util.*; +import java.util.concurrent.*; + +public class main { + + interface Scenario { + String id(); + void run(long durationMs) throws Exception; + } + + static final Map REGISTRY = new LinkedHashMap<>(); + + static { + register(new CpuHotspot()); + register(new GcPressureParallel()); + register(new AllocationPressure()); + register(new ThreadContentionSync()); + register(new Safepoints()); + register(new TemporalSpikes()); + register(new HealthyBaseline()); + register(new AllocGcCascade()); + register(new GcCpuInflation()); + register(new GcQueueCoupling()); + register(new ReactiveNoBackpressure()); + register(new ReactiveBackpressure()); + register(new CfFanoutLargeHeap()); + register(new G1HumongousAlloc()); + register(new ZgcAllocationStalls()); + register(new VirtualThreadPinning()); + } + + static void register(Scenario s) { + REGISTRY.put(s.id(), s); + } + + public static void main(String[] args) throws Exception { + String scenariosProp = System.getProperty("scenarios", ""); + int durationSec = Integer.parseInt(System.getProperty("duration", "30")); + long durationMs = durationSec * 1000L; + + if (scenariosProp.isBlank()) { + System.err.println("Usage: jbang main.java -Dscenarios= [-Dduration=30]"); + System.err.println("Available: " + String.join(", ", REGISTRY.keySet())); + System.exit(1); + } + + List ids = Arrays.asList(scenariosProp.split(",")); + List toRun = new ArrayList<>(); + for (String id : ids) { + String trimmed = id.trim(); + Scenario s = REGISTRY.get(trimmed); + if (s == null) { + System.err.println("Unknown scenario: " + trimmed); + System.err.println("Available: " + String.join(", ", REGISTRY.keySet())); + System.exit(1); + } + toRun.add(s); + } + + System.out.println("Starting scenarios: " + ids + " for " + durationSec + "s"); + + List threads = new ArrayList<>(); + for (Scenario s : toRun) { + Thread t = Thread.ofPlatform().name("scenario-" + s.id()).start(() -> { + try { + s.run(durationMs); + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); // normal shutdown signal — not an error + } catch (Exception e) { + System.err.println("Scenario " + s.id() + " failed: " + e.getMessage()); + } + }); + threads.add(t); + } + + Thread.sleep(durationMs); + System.out.println("Duration elapsed — shutting down"); + + for (Thread t : threads) { + t.interrupt(); + t.join(2000); + } + + System.out.println("Done"); + System.exit(0); // force JVM exit — scenario threads may be non-daemon and keep JVM alive + } +} diff --git a/plugins/jfr-analyzer/skills/async-profiler-interop/SKILL.md b/plugins/jfr-analyzer/skills/async-profiler-interop/SKILL.md new file mode 100644 index 0000000..3e14595 --- /dev/null +++ b/plugins/jfr-analyzer/skills/async-profiler-interop/SKILL.md @@ -0,0 +1,103 @@ +--- +name: async-profiler-interop +description: Use when async-profiler sampling should narrow a JVM performance hypothesis or be correlated with JFR history and a BTrace live probe. +--- + +# Async-profiler interoperability + +Treat async-profiler, JFR, and BTrace as complementary instruments: + +- async-profiler provides low-overhead sampled evidence for CPU, wall-clock, allocation, lock, or + native hotspots over a deliberately bounded interval; +- JFR provides broader historical context, event timelines, and JVM state across a recording; +- BTrace provides exact, on-demand observations at selected methods, calls, returns, exceptions, or + custom events. + +## Installation and availability + +Before proposing a live async-profiler run, check `ASYNC_PROFILER_HOME` and verify that its +`lib/` directory contains `libasyncProfiler.so` (Linux) or `libasyncProfiler.dylib` (macOS). If it +is absent, give the operator the installation instructions in the `jfr-analyzer` README and stop; +do not silently download or execute native code. The bundled eval regeneration script may offer an +explicit, interactive download when the operator invokes it. + +The profiler and target JVM must be on the same host. Check the target Java version, operating +system/architecture, attach permissions, container boundary, and whether the selected native +library matches the host before presenting a command. + +Do not use one instrument as proof when the question needs another instrument's semantics. Preserve +the target identity, recording/profiling interval, clock and timezone, profiler mode, sampling +interval, output path, and cleanup action in every handoff. + +## Choosing the next instrument + +1. Start with JFR when a recording already covers the incident or when JVM-wide historical context + is needed. +2. Start with async-profiler when the main uncertainty is sampled CPU, wall-clock, allocation, + lock, or native-stack attribution and a short profile can safely be taken from the live JVM. +3. Use BTrace after a profile identifies a candidate method or path that needs exact confirmation. + Keep the probe narrow, bounded, and reversible. +4. Use JFR after BTrace when the live observation raises a broader JVM or timeline question; prefer + adding a bounded custom JFR event from BTrace only when the installed BTrace build supports it + and the event schema is explicit. + +## Async-profiler → JFR + +When an async-profiler result is correlated with JFR: + +- align the profiler interval with the JFR recording using absolute timestamps and timezone; +- record the profiler mode and interval because CPU, wall-clock, allocation, and lock profiles do + not answer the same question; +- compare stacks and top methods with JFR execution, thread, allocation, or lock events without + treating sampled counts as exact event counts; +- note sampling bias, safepoint effects, native frames, and missing symbols before drawing a causal + conclusion. + +## Async-profiler → BTrace + +When a profile identifies a candidate: + +1. Capture the candidate class, method, relevant stack context, profile mode, and time window. +2. Ask `btrace-observability` for the smallest probe that can distinguish the competing hypotheses. +3. State the expected event rate, maximum observation duration, output destination, and stop command + before deployment. +4. Compare BTrace output with the matching profiler interval and account for probe overhead. + +Do not turn a sampled stack frame into a broad package probe automatically. Prefer an exact method, +call boundary, exception type, or bounded argument-free event. + +## BTrace → async-profiler/JFR + +When BTrace finds an unexpected path: + +- retain the probe identity, target PID, timestamps, and script/oneliner revision; +- use async-profiler for a short profile if the question is CPU, wall-clock, allocation, lock, or + native-stack cost; +- use JFR for event history, thread state, GC, safepoints, allocations, or a wider interval; +- correlate all results against the same target and time window before recommending a change. + +## Shared evidence record + +Use this compact record when handing evidence between skills or providers: + +```json +{ + "target": {"pid": "...", "host": "...", "service": "..."}, + "window": {"start": "...", "end": "...", "timezone": "..."}, + "async_profiler": { + "mode": "cpu|wall|alloc|lock|native", + "interval": "...", + "artifact": "...", + "evidence": ["..."] + }, + "jfr": {"recording": "...", "evidence": ["..."]}, + "btrace": {"probe": "...", "evidence": ["..."]}, + "hypothesis": "...", + "next_action": "..." +} +``` + +If output is unstructured, keep the raw artifact local and put a concise, redacted interpretation +in the relevant evidence list. Never include credentials, tokens, request bodies, or customer data +merely to improve correlation. Pair live instrumentation with the BTrace lifecycle and data-safety +skills, and stop every profiler/probe session at the declared end of its observation window. diff --git a/plugins/jfr-analyzer/skills/jfr-analyzer/SKILL.md b/plugins/jfr-analyzer/skills/jfr-analyzer/SKILL.md new file mode 100644 index 0000000..89d166b --- /dev/null +++ b/plugins/jfr-analyzer/skills/jfr-analyzer/SKILL.md @@ -0,0 +1,59 @@ +--- +name: jfr-analyzer +description: Systematic performance investigation of JFR, pprof, OTLP, and HPROF profiling data using USE and TSA methodologies. Entry point for the jfr-analyzer plugin. +allowed-tools: Read mcp__jfr-mcp__jfr_help +--- + +## Argument parsing + +$ARGUMENTS contains the raw arguments from the user invocation. + +Parse the first token of $ARGUMENTS: + +- If $ARGUMENTS is empty or equals "help" → print help text (see below) and stop. +- If first token is "triage" → SUBCOMMAND=triage; FILE is the second token (may be empty). +- If first token is "drilldown" → SUBCOMMAND=drilldown; FILE is empty (resumes session). +- If first token is "report" → SUBCOMMAND=report; FILE is empty (resumes session). +- If first token is "eval" → SUBCOMMAND=eval; pass $ARGUMENTS unchanged to the eval subskill. +- Otherwise → inspect the first token: + - If it contains a path separator (/ or \) OR ends with a recognized extension (.jfr, .pb.gz, .pprof, .otlp, .hprof) → SUBCOMMAND=triage; FILE is the first token (auto-chain all phases). + - If it does not look like a file path → output "Unknown subcommand — run /jfr-analyzer help for usage." and stop. + +## Routing + +Read the appropriate subskill: + +- SUBCOMMAND=triage → Read `subskills/triage/SKILL.md` relative to this skill directory +- SUBCOMMAND=drilldown → Read `subskills/drilldown/SKILL.md` relative to this skill directory +- SUBCOMMAND=report → Read `subskills/report/SKILL.md` relative to this skill directory +- SUBCOMMAND=eval → Read `subskills/eval/SKILL.md` relative to this skill directory + +When the investigation involves a live JVM profile or a BTrace finding, also read +`async-profiler-interop/SKILL.md` relative to this skill directory. Use it to choose between +async-profiler, JFR, and BTrace and to preserve the shared evidence record. + +The subskill receives $ARGUMENTS unchanged — it is responsible for extracting FILE +or session state from $ARGUMENTS and .jfr-analyzer/. + +## Help text + +If printing help, output exactly: + +``` +/jfr-analyzer Full analysis: triage → drilldown → report +/jfr-analyzer triage Run triage only (USE + TSA + stackgraph) +/jfr-analyzer drilldown Resume drilldown on most recent session +/jfr-analyzer report Generate report for most recent session +/jfr-analyzer eval regen Regenerate JFR corpus (regen.sh + MCP event extraction) +/jfr-analyzer eval run Run triage across corpus scenarios +/jfr-analyzer eval score Score results and generate report +/jfr-analyzer help Show this help + +Supported formats: + .jfr Java Flight Recorder (JVM) + .pb.gz / .pprof pprof (Go, polyglot) + .otlp OpenTelemetry profiles + .hprof Java heap dump + +Requires: Jafar MCP running — start with: jbang jafar-mcp@btraceio +``` diff --git a/plugins/jfr-analyzer/skills/jfr-analyzer/subskills/drilldown/SKILL.md b/plugins/jfr-analyzer/skills/jfr-analyzer/subskills/drilldown/SKILL.md new file mode 100644 index 0000000..8002c7a --- /dev/null +++ b/plugins/jfr-analyzer/skills/jfr-analyzer/subskills/drilldown/SKILL.md @@ -0,0 +1,287 @@ +--- +name: drilldown +description: Phase 2 of jfr-analyzer. Reads focus.json, optionally gates HPROF analysis behind a cost warning, then dispatches parallel perf-engineer agents — one per selected problem area — each writing structured findings to drilldown/.json. +allowed-tools: Read Write Bash(find *) Bash(ls *) mcp__jfr-mcp__jfr_query mcp__jfr-mcp__jfr_hotmethods mcp__jfr-mcp__jfr_stackprofile mcp__jfr-mcp__hdump_summary mcp__jfr-mcp__hdump_report mcp__jfr-mcp__hdump_query mcp__jfr-mcp__hdump_help +--- + +## Setup + +Load session state by finding the most recent session: + +If $ARGUMENTS contains a path to a focus.json, use that directly. + +Otherwise run: Bash: find .jfr-analyzer -name "focus.json" -maxdepth 3 + +If multiple focus.json files are found, use the one whose parent directory has the most recent +timestamp in its name (directory names are -). + +If no focus.json is found: + Output: "No active session. Run /jfr-analyzer first." + Stop. + +Read the focus.json. Extract: + - SESSION_DIR + - sessionId + - format + - sourceRoot (may be null) + - focusAreas[] + +Also read SESSION_DIR/session.json to confirm the session record. + +Create a host-appropriate task checklist with one task per focusArea id (e.g. "drill-gc-pressure", +"drill-cpu-hotspots") plus "hprof-gate" if format=hprof. + +## HPROF cost gate + +If format = "hprof" AND any focusArea has id = "heap-memory": + + Mark hprof-gate in_progress. + + Get the file path from session.json ("file" field). + Run: Bash: ls -lh to get the file size. + + Estimate: + temp_low_gb = file_size_gb × 0.10 + temp_high_gb = file_size_gb × 0.30 + time_low_min = 2 + (file_size_gb × 1.0) + time_high_min = 2 + (file_size_gb × 2.0) + + Output: + + ┌────────────────────────────────────────────────────────────┐ + │ HPROF analysis warning │ + │ │ + │ Heap dump analysis can take several minutes and writes │ + │ temporary index files to disk alongside the dump │ + │ (typically 10-30% of the dump file size). │ + │ │ + │ Your file is X.X GB — estimated Y-Z min runtime, │ + │ ~A-B GB of temp files written to the dump directory. │ + │ │ + │ Proceed with heap analysis? [yes / skip] │ + └────────────────────────────────────────────────────────────┘ + + (Substitute actual file size, time estimate, and temp estimate for X/Y/Z/A/B) + + If the user responds with anything other than "yes": + Read focus.json. Find the focusArea with id="heap-memory". Add to it: + "deferred": true, + "deferredReason": "user skipped" + Write the updated focus.json back using Write. + Output: "Heap analysis deferred. Re-run /jfr-analyzer drilldown to include it later." + Remove heap-memory from the active list for dispatch below. + + Mark hprof-gate completed. + +## Parallel dispatch + +Create the drilldown output directory if it doesn't already exist: + Bash: find SESSION_DIR/drilldown -maxdepth 0 -type d (check existence) + If absent: the directory was already created by triage; if somehow missing, note it but proceed. + +For each non-deferred area in focusAreas, dispatch a perf-engineer agent in parallel. + +Each agent runs independently. Do not wait for one to complete before starting the next. + +Construct the prompt for each agent by filling in this template: + +--- +You are a perf-engineer subagent. Your assignment is to investigate [area.title] in a +[format] profiling session and write your findings as structured JSON. + +Session ID: [sessionId] +Session directory: [SESSION_DIR] +Source root: [sourceRoot — or "not available" if null] +Output file: [SESSION_DIR]/drilldown/[area.id].json + +Triage evidence for this area: +[area.triageEvidence formatted as readable key: value lines] + +Cross-area correlations detected during triage: +[If focus.json contains crossAreaCorrelations entries that include this area's id, list them here +as readable lines. If none involve this area, write "none". These represent interactions between +problem areas that affect how findings in this area should be interpreted:] + + gc-queue-coupling: GC stop-the-world pauses are inflating queue wait times. When investigating + thread-contention, check whether queue wait spikes align with GC pause windows rather than + assuming CPU-bound work. When investigating gc-pressure, note that fixing GC will also + reduce apparent thread queue saturation. + + gc-cpu-inflation: GC threads are consuming CPU alongside application threads. When + investigating cpu-hotspots, check whether GC-related methods (G1/ZGC/Shenandoah worker + threads) appear in hotmethods. If they do, the application's true CPU demand is lower than + the total reported utilization suggests. + + alloc-gc-latency-cascade: High allocation rate is driving GC frequency, which causes pauses, + which cause queue backlog. When investigating any of the three involved areas, note that + reducing the allocation rate is the highest-leverage fix — it resolves GC pressure and + queue saturation as side effects. + +## Drill strategy for [area.id] + +[Copy the FULL drill instructions block for area.id from the "Per-area drill instructions" +section below — do not summarize or abbreviate] + +## Output format + +Use the Write tool to write [SESSION_DIR]/drilldown/[area.id].json with this schema: + +{ + "area": "[area.id]", + "summary": "<2-3 sentence plain-language summary, no jargon>", + "evidence": [ + { + "label": "", + "value": "", + "explanation": "", + "query": "" + } + ], + "hints": [ + { + "description": "", + "impact": "high|moderate|low", + "code_before": "", + "code_after": "" + } + ] +} + +If a Jafar MCP call fails, still write the JSON with whatever evidence was collected and +add "error": "" at the top level. + +If source root is available, for each hot method identified: + 1. Run: find [sourceRoot] -name "[ClassName].java" (JFR) or "[filename].go" (pprof) + 2. Read the source file + 3. Reason about why the method is hot before writing hints + 4. Include before/after code snippets in hints where possible +--- + +## Per-area drill instructions + +### gc-pressure (JFR only) + +Run these queries using mcp__jfr-mcp__jfr_query: + events/jdk.GCPhasePause | groupBy(name, agg=sum, value=duration) | top(5) + events/jdk.GCPhasePause | quantiles(0.5, 0.95, 0.99, path=duration) + events/jdk.GarbageCollection | stats(duration) + +Also run mcp__jfr-mcp__jfr_hotmethods (to see if GC threads appear in top methods). + +Identify: dominant GC phase (e.g. G1 Major, G1 Young, ZGC Concurrent), pause p50/p95/p99 in +milliseconds, whether GC threads appear in hotmethods, and which GC algorithm is in use. + +If Jafar returns a tool error or zero rows for any query, note "event not present" in evidence. + +### allocation-pressure (JFR only) + +Try allocation events in this order. Use the first that returns rows (a tool error or zero rows +means try the next option): + Option A: events/jdk.ObjectAllocationSample | groupBy(objectClass, agg=sum, value=weight) | top(10) + (JDK 16+) + Option B: events/jdk.ObjectAllocationInNewTLAB | groupBy(objectClass, agg=sum, value=bytes) | top(10) + (older JDKs) + +Then run: + mcp__jfr-mcp__jfr_stackprofile (to find allocation call stacks) + +Note which event was used in evidence[0].label. For sampled allocation events, state whether the +weight represents estimated bytes or sample count. Identify: top allocating class+method, +allocation rate, and whether allocations occur inside a tight loop. + +### thread-contention (all formats) + +Run: {fmt}_tsa with correlateBlocking=true, topThreads=20 + (replace {fmt} with jfr, pprof, otlp, or hdump based on format) + +Note: HPROF format has no TSA tool — skip the {fmt}_tsa call for hdump format. Run only the hdump-specific contention analysis if available, or report that thread state analysis is not supported for heap dumps. + +Note: OTLP format has no TSA tool — skip the {fmt}_tsa call for otlp format. Rely on stackprofile-based thread analysis: run otlp_stackprofile and identify methods that dominate samples during contention windows. + +For JFR additionally run these mcp__jfr-mcp__jfr_query calls: + events/jdk.JavaMonitorEnter | groupBy(monitorClass, agg=count) | top(10) + events/jdk.JavaMonitorEnter | stats(duration) + events/jdk.JavaMonitorWait | groupBy(monitorClass, agg=count) | top(10) + events/jdk.VirtualThreadPinned | count() (ignore error if event is absent) + +If VirtualThreadPinned count > 0, add a dedicated evidence item: + label: "Virtual thread pinning" + explanation: "Virtual threads (JDK 21+) are being pinned to their carrier threads. + Pinning happens inside synchronized blocks or native method calls and prevents + the JVM from reusing the underlying OS thread for other virtual threads, reducing + the scalability benefit of virtual threads." + +### cpu-hotspots (JFR, pprof, OTLP only — not applicable to HPROF) +(replace {fmt} with jfr, pprof, otlp, or hdump based on the format field in your prompt) + +Note: HPROF (heap dump) format has no CPU sample data — this area should not be selected for HPROF sessions. + +Note: OTLP format does not have otlp_hotmethods or otlp_callgraph. For OTLP, skip {fmt}_hotmethods and {fmt}_callgraph. Instead, derive hot methods using: otlp_query groupBy(stackTrace/0/name, sum(cpu)) | top(20). Then run otlp_flamegraph direction=top-down as the only flamegraph step. + +Note: pprof format does not have pprof_callgraph — skip the {fmt}_callgraph step for pprof format. + +Run: {fmt}_hotmethods (top 20 by default) [skip for OTLP — use otlp_query instead, see Note above] +Run: {fmt}_flamegraph direction=top-down +Run: {fmt}_callgraph [skip for OTLP and pprof — see Notes above] + +Identify: top-3 methods by sample percentage, their caller chains, and self-vs-total time +split. For the #1 hotmethod, estimate what percentage of total CPU it represents. + +### safepoints (JFR only) + +Run these mcp__jfr-mcp__jfr_query calls: + events/jdk.SafepointBegin | stats(duration) + events/jdk.SafepointBegin | quantiles(0.5, 0.95, 0.99, path=duration) + events/jdk.SafepointBegin | groupBy(cause, agg=sum, value=duration) | top(5) + +Identify: dominant safepoint cause (e.g. "RevokeBias", "Deoptimize", "G1IncCollectionPause"), +p95/p99 duration in milliseconds, and total safepoint time as a percentage of recording duration. + +### virtual-thread-pinning (JFR only) + +Run these mcp__jfr-mcp__jfr_query calls: + events/jdk.VirtualThreadPinned | count() + events/jdk.VirtualThreadPinned | stats(duration) + +If the above return results, also run: + events/jdk.VirtualThreadPinned | groupBy(stackTrace/frames/0/method/name, agg=count) | top(10) + +Identify: total pinning events, duration distribution, and top pinning locations in the call +stack (the synchronized block or native method causing pinning). + +### heap-memory (HPROF only) + +Run: mcp__jfr-mcp__hdump_report (full heap analysis — this is the primary analysis call) + +Then run targeted hdump_query calls to identify top retained objects. Refer to +mcp__jfr-mcp__hdump_help for available HdumpPath query syntax. + +Identify: largest retained object graphs by size, potential memory leaks (many instances with +large retained size), and GC roots holding large graphs. + +### temporal-spikes (all formats with spike windows) +(replace {fmt} with jfr, pprof, otlp, or hdump based on the format field in your prompt) + +Note: HPROF format has no stackprofile tool — skip this area for hdump format. Triage should not create a temporal-spikes focus area for HPROF since spikeWindows will be empty, but if it is received, output: 'Temporal spike analysis not supported for HPROF format.' + +For each spike window in triageEvidence.spikeWindows: + Run: {fmt}_stackprofile with startTime=, endTime= + +Also run a baseline comparison: + 1. Re-run {fmt}_stackprofile across the full recording window (no startTime/endTime) to get the + full density distribution. + 2. From that distribution, identify a window of the same duration as the spike window that falls + outside all spikeWindows and whose sample density is nearest to the median density. + 3. Run {fmt}_stackprofile with startTime= and endTime= to get + the baseline profile. + +Compare: what methods appear in the spike window but not the baseline, or have significantly +higher sample percentage (>2× baseline) during the spike. These are the likely causes. + +## After all agents complete + +Once all dispatched agents have written their output files: + +Output: "▶ Drilldown complete. Generating report..." + +Auto-chain: Read the sibling `../report/SKILL.md` file relative to this skill directory. diff --git a/plugins/jfr-analyzer/skills/jfr-analyzer/subskills/eval/SKILL.md b/plugins/jfr-analyzer/skills/jfr-analyzer/subskills/eval/SKILL.md new file mode 100644 index 0000000..672a37d --- /dev/null +++ b/plugins/jfr-analyzer/skills/jfr-analyzer/subskills/eval/SKILL.md @@ -0,0 +1,149 @@ +--- +name: eval +description: Eval subskill for jfr-analyzer. Routes regen|run|score|help. Manages corpus regeneration (MCP event extraction + oracle validation), eval runs, and scoring. +allowed-tools: Read Write Bash(find *) Bash(ls *) mcp__jfr-mcp__jfr_open mcp__jfr-mcp__jfr_list_types mcp__jfr-mcp__jfr_close +--- + +## Argument parsing + +Parse the first token of $ARGUMENTS after "eval": +- Empty or "help" → print help and stop. +- "regen" → SUBCOMMAND=regen; optional `--validate` flag +- "run" → SUBCOMMAND=run; optional `--scenario `, `--formats jfr,otlp,pprof`, `--runs N`, `--parallel N` +- "score" → SUBCOMMAND=score; optional `--scenario `, `--multi-judge` + +EVAL_DIR = resolve four levels up from this SKILL.md to reach the jfr-analyzer plugin root, then append `/eval`. + +## Help text + +If SUBCOMMAND=help or $ARGUMENTS is empty, output exactly: + +``` +/jfr-analyzer eval regen [--validate] Regenerate JFR corpus, extract event types, optionally oracle-validate +/jfr-analyzer eval run [options] Run triage against corpus scenarios +/jfr-analyzer eval score [options] Score eval results and generate report + +eval run options: + --scenario Score only this scenario + --formats jfr,otlp,pprof Formats to test (default: jfr) + --runs N Runs per scenario (default: uses eval_runs in manifest) + --parallel N Max concurrent scenarios (default: 4) + +eval score options: + --scenario Score only this scenario + --multi-judge Use Claude + GPT-4o judge panel (requires OPENAI_API_KEY) +``` + +## regen subcommand + +### Phase 1 — Run regen.sh (shell, no MCP) + +Run: +```bash +bash /scripts/regen.sh +``` + +Report exit status. If non-zero, show last 20 lines of output and stop. + +### Phase 2 — Event type extraction (MCP) + +Read `/corpus/manifest.json`. + +For each scenario in the manifest: + +1. `file_path` = `/corpus/` +2. Call `mcp__jfr-mcp__jfr_open` with path=`` +3. Extract `sessionId` from the response. +4. Call `mcp__jfr-mcp__jfr_list_types` with `sessionId` +5. Call `mcp__jfr-mcp__jfr_close` with `sessionId` +6. Collect the list of event type names from the jfr_list_types response + +After iterating all scenarios, use the Write tool to update manifest.json: +For each scenario by index, set `jfr_event_types` to the list of type names collected in step 6. + +### Phase 3 — Oracle validation (if `--validate` flag or user approves) + +If `--validate` was NOT passed, ask the user: + "Run oracle validation? This runs triage against each recording to verify ground truth. [y/N]" + +For each scenario where `expected.scoring_tier = "structural"`: + + Run triage in eval mode by reading triage SKILL.md with arguments: + `--eval /corpus/.jfr --output-dir /results//oracle/` + + After triage completes, Read `/results//oracle/focus.json`. Check: + - Every `expected.focusAreas[].id` appears in `focusAreas[].id` in the output + - No `expected.absent_area_ids` appear in output `focusAreas[].id` + - Every `expected.crossAreaCorrelations[].kind` appears in output `crossAreaCorrelations[].kind` + + If any check fails, output: + ``` + ❌ Oracle validation FAILED for + Missing areas: [...] + Unexpected areas: [...] + Missing correlations: [...] + + Fix options: + 1. Adjust the workload in eval/testapp/scenarios/.java and re-run regen.sh + 2. Update expected ground truth in manifest.json if the skill output is actually correct + ``` + Stop after first failure. + +For each scenario where `expected.scoring_tier = "semantic"`: + Print the rubric and the oracle focus.json side-by-side for human review. + Ask: "Does this look correct for ? [y/N]" + +### Commit prompt after successful regen + +After Phase 2 (and Phase 3 if run), output: +``` +✓ Corpus regenerated. Stage and commit: + git add jfr-analyzer/eval/corpus/ + git commit -m "feat(jfr-analyzer): regenerate eval corpus" +``` + +## run subcommand + +Parse options from $ARGUMENTS: +- `--scenario ` → filter to that scenario only +- `--formats ` → comma-split; default `jfr` +- `--runs N` → override `eval_runs` from manifest +- `--parallel N` → max concurrent scenarios (default 4) + +Read manifest.json. For each scenario (filtered if --scenario): + +N_RUNS = `--runs` value if provided, else `scenario.eval_runs`. + +For each run index k = 1 to N_RUNS (runs within a scenario are SEQUENTIAL — do NOT parallelize): + Read the triage subskill and execute it with arguments: + `--eval /corpus/.jfr --output-dir /results//run-/` + +Dispatch batches of `--parallel` scenarios concurrently. Within each scenario, runs are sequential. + +If `--formats` includes `otlp` or `pprof`: + After the JFR run batch, derive formats: + ```bash + jfrconv /corpus/.jfr /corpus/.otlp # for otlp + jfrconv /corpus/.jfr /corpus/.pb.gz # for pprof + ``` + Run additional triage passes against each derived format file, with output dirs named `-otlp/run-/` etc. + +After all runs complete, output: +``` +✓ Eval runs complete. Results in eval/results/ + Next: /jfr-analyzer eval score +``` + +## score subcommand + +Parse `--scenario` and `--multi-judge` flags from $ARGUMENTS. + +Check if `/scripts/.venv` exists: +- If not: run `cd /scripts && python3 -m venv .venv && .venv/bin/pip install -r requirements.txt` + +Run: +```bash +cd /scripts && .venv/bin/python score.py [--scenario if set] [--multi-judge if set] +``` + +Display the generated report content. diff --git a/plugins/jfr-analyzer/skills/jfr-analyzer/subskills/report/SKILL.md b/plugins/jfr-analyzer/skills/jfr-analyzer/subskills/report/SKILL.md new file mode 100644 index 0000000..2558c39 --- /dev/null +++ b/plugins/jfr-analyzer/skills/jfr-analyzer/subskills/report/SKILL.md @@ -0,0 +1,212 @@ +--- +name: report +description: Phase 3 of jfr-analyzer. Reads all drilldown/.json findings, deduplicates overlapping evidence, ranks by user-visible impact, produces plain-language finding cards, and writes report.json with a sphinx-optimize bridge stub. +allowed-tools: Read Write Bash(find *) Bash(ls *) +--- + +## Setup + +Find session state using the same logic as drilldown/SKILL.md: + If $ARGUMENTS contains a path to a focus.json, use that. + Otherwise: Bash: find .jfr-analyzer -name "focus.json" -maxdepth 3 + Pick the most recent by directory timestamp. + If none found: output "No active session. Run /jfr-analyzer first." and stop. + +Read SESSION_DIR/focus.json and SESSION_DIR/session.json. + +Find drilldown files: + Bash: find SESSION_DIR/drilldown -name "*.json" -maxdepth 1 + +Read all found drilldown JSON files. + +If no drilldown files are found: + Output: "No drilldown results found. Run /jfr-analyzer drilldown first." + Stop. + +Create a host-appropriate task checklist: aggregate, deduplicate, rank, executive-summary, cards, +report-json, final-output. + +## Step 1 — Aggregate + +Mark aggregate in_progress. + +For each drilldown/.json: + 1. Read the file and parse it as `finding`. Extract: finding.area (the id), finding.summary, finding.evidence[], finding.hints[]. + If the file does not contain an 'area' field, or if no focusArea in focus.json has an id matching finding.area, skip this file and output a warning: 'Warning: drilldown file [filename] has no matching focus area — skipping.' + 2. Look up the matching focusArea in focus.json where focusArea.id == finding.area. + Use focusArea.title as the finding's display title. + Use finding.impact if present in the drilldown JSON; otherwise fall back to focusArea.impact as the finding's impact level. (Drilldown agents may optionally emit a top-level "impact" field to upgrade or downgrade the triage severity assessment.) + If focusArea has deferred=true, skip this finding (it was not analyzed). + +Also collect any focusAreas with deferred=true that have no drilldown file — these will be +rendered as DEFERRED sections in the output. + +Mark aggregate completed. + +## Step 2 — Deduplicate + +Mark deduplicate in_progress. + +For each pair of findings: + If their top evidence items point to the same class or method name (compare the "value" + field of evidence[0] for common identifiers like class names or method signatures): + Keep the finding with the higher impact level. + Append the lower-impact finding's evidence items to the kept finding's evidence list. + Add a note to the kept finding: "See also: [other finding title]" + Remove the lower-impact duplicate from the findings list. + +If findings is empty after deduplication: + Output: "No analyzable findings (all areas were skipped or deferred)." + Write SESSION_DIR/report.json using the Write tool: + { + "session": "", + "format": "", + "file": "", + "source_root": "", + "findings": [] + } + Stop. + +Mark deduplicate completed. + +## Step 3 — Rank by impact + +Mark rank in_progress. + +Sort the remaining findings: + 1. Areas where focusArea.startHere=true first (these were marked by triage as highest potential gain). + 2. Within each startHere group: HIGH impact before MODERATE before LOW. + 3. Within the same impact level: sort by frequency × duration when those metrics are available in evidence. + A 4ms stall at 500 times/second outranks a 200ms stall at 2 times/hour. + +Assign sequential finding numbers starting from 1 (e.g. "FINDING 1 of 4"). + +Mark rank completed. + +## Step 4 — Executive summary + +Mark executive-summary in_progress. + +Write 2-3 sentences in plain English: + - What is the most significant bottleneck and what user-visible symptom does it cause? + - What is the single most impactful action to take first? + +Rules: + - No jargon without inline definition. + - No method names without explanation of what they do. + - A user who has never read a flame graph should understand the summary. + +Mark executive-summary completed. + +## Step 5 — Finding cards + +Mark cards in_progress. + +For each finding (in ranked order), output: + +━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ +FINDING [N] of [M] — [finding.title] [[IMPACT] IMPACT] +━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ + +What's happening: + [finding.summary — 2-3 sentences, plain language, no jargon without inline definition] + +Root cause evidence: + [For each item in finding.evidence (up to 4 items):] + → [evidence.value] + ([evidence.explanation]) + +Optimization hints: + [For each hint in finding.hints (up to 3, highest impact first):] + [N]. [hint.description] + — estimated impact: [hint.impact] + [If hint.code_after is not null:] + Before: + [hint.code_before] + After: + [hint.code_after] + +Queries used (re-run or adapt in Jafar): + [For each unique query string from finding.evidence[*].query (deduplicated):] + [query string] + +For each deferred area, output: + +━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ +DEFERRED — [focusArea.title] +━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ +This area was not analyzed. To investigate it, re-run: + /jfr-analyzer drilldown +and respond "yes" when asked about [focusArea.title]. + +Mark cards completed. + +## Step 6 — Write report.json + +Mark report-json in_progress. + +Determine suggested_targets for the sphinx-optimize bridge from the top finding's id: + gc-pressure or allocation-pressure → ["cpu", "memory"] + thread-contention or cpu-hotspots or safepoints or virtual-thread-pinning → ["cpu"] + heap-memory → ["memory"] + temporal-spikes → ["cpu", "memory"] + +Use the suggested_targets value computed from the mapping table above — do not hardcode ["cpu","memory"]. + +If the top finding was merged from multiple source areas during deduplication, compute suggested_targets as the union of the mapped values for all contributing area ids (deduplicated). For example, if gc-pressure and cpu-hotspots were merged, the union of ["cpu","memory"] and ["cpu"] is ["cpu","memory"]. + +Write SESSION_DIR/report.json using the Write tool: + +{ + "session": "", + "format": "", + "file": "", + "source_root": "", + "findings": [ + { + "id": "", + "title": "", + "impact": "", + "summary": "", + "hints": [], + "evidence": { + "queries": [], + "metrics": {} + } + } + ], + "_sphinx_optimize_bridge": { + "version": 1, + "source": "", + "top_finding": "", + "suggested_targets": + } +} + +Mark report-json completed. + +## Step 7 — Final output + +Mark final-output in_progress. + +Print: + +━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ +ANALYSIS COMPLETE +━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ + +[executive summary text] + +[all finding cards in ranked order] + +[all deferred sections] + +━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ +Full report saved to: [SESSION_DIR]/report.json + +To continue investigating interactively: + 1. If the host supports specialist workers, dispatch the bundled perf-engineer role; otherwise + perform the same review sequentially. + 2. Provide the recording file: [session.file] + +Mark final-output completed. diff --git a/plugins/jfr-analyzer/skills/jfr-analyzer/subskills/triage/SKILL.md b/plugins/jfr-analyzer/skills/jfr-analyzer/subskills/triage/SKILL.md new file mode 100644 index 0000000..cb52a0c --- /dev/null +++ b/plugins/jfr-analyzer/skills/jfr-analyzer/subskills/triage/SKILL.md @@ -0,0 +1,440 @@ +--- +name: triage +description: Phase 1 of jfr-analyzer. Opens a profiling file, runs USE+TSA+stackgraph+JVM/GC lanes automatically, presents ranked plain-language problem areas at a breakpoint, and writes focus.json for drilldown. +allowed-tools: Read Write Bash(find *) Bash(mkdir *) Bash(date *) mcp__jfr-mcp__jfr_help mcp__jfr-mcp__jfr_open mcp__jfr-mcp__jfr_summary mcp__jfr-mcp__jfr_use mcp__jfr-mcp__jfr_tsa mcp__jfr-mcp__jfr_stackprofile mcp__jfr-mcp__jfr_query mcp__jfr-mcp__pprof_open mcp__jfr-mcp__pprof_summary mcp__jfr-mcp__pprof_use mcp__jfr-mcp__pprof_tsa mcp__jfr-mcp__pprof_stackprofile mcp__jfr-mcp__otlp_open mcp__jfr-mcp__otlp_summary mcp__jfr-mcp__otlp_use mcp__jfr-mcp__otlp_stackprofile mcp__jfr-mcp__hdump_open mcp__jfr-mcp__hdump_summary +--- + +## Setup + +Parse FILE from $ARGUMENTS: +- If $ARGUMENTS starts with "triage ", FILE is everything after "triage ". +- Otherwise FILE is the first token of $ARGUMENTS. + +If FILE is empty: output "Usage: /jfr-analyzer " and stop. + +Parse eval flags from $ARGUMENTS: +- EVAL_MODE = true if `--eval` is present anywhere in $ARGUMENTS token list +- OUTPUT_DIR = value after `--output-dir` token if present; null otherwise + +If EVAL_MODE: + FILE = the path token that immediately follows `--eval` (skip `--output-dir` and its value) + If FILE is empty: output "Usage: /jfr-analyzer triage --eval [--output-dir ]" and stop. + +Create a host-appropriate task checklist with these tasks: + - preflight + - detect-format + - open-session + - summary + - source-context + - use-report + - tsa-report + - stackgraph + - jvm-gc-lanes + - synthesize + - breakpoint + +Mark each task in_progress before starting it and completed when done. + +## Step 1 — Preflight + +Mark preflight in_progress. + +▶ Checking Jafar MCP... + +Call mcp__jfr-mcp__jfr_help with no arguments. + +If the call errors or times out, output: + + ❌ Jafar MCP is not running. + + Start it with: + jbang jafar-mcp@btraceio + + Then re-run /jfr-analyzer. + +Stop if Jafar is not running. + +✓ Jafar MCP is running. + +Mark preflight completed. + +## Step 2 — Format detection + +Mark detect-format in_progress. + +Inspect the FILE extension and set these variables: + +| Extension | FORMAT | OPEN_TOOL | SUMMARY_TOOL | USE_TOOL | TSA_TOOL | STACK_TOOL | +|-----------------|--------|----------------------------|------------------------------|-----------------------|-----------------------|--------------------------------| +| .jfr | jfr | mcp__jfr-mcp__jfr_open | mcp__jfr-mcp__jfr_summary | mcp__jfr-mcp__jfr_use | mcp__jfr-mcp__jfr_tsa | mcp__jfr-mcp__jfr_stackprofile | +| .pb.gz or .pprof| pprof | mcp__jfr-mcp__pprof_open | mcp__jfr-mcp__pprof_summary | mcp__jfr-mcp__pprof_use | mcp__jfr-mcp__pprof_tsa | mcp__jfr-mcp__pprof_stackprofile | +| .otlp | otlp | mcp__jfr-mcp__otlp_open | mcp__jfr-mcp__otlp_summary | mcp__jfr-mcp__otlp_use | (none) | mcp__jfr-mcp__otlp_stackprofile | +| .hprof | hprof | mcp__jfr-mcp__hdump_open | mcp__jfr-mcp__hdump_summary | (none) | (none) | (none) | + +If extension is unrecognized: + Output: "❌ Unrecognized file extension. Supported: .jfr .pb.gz .pprof .otlp .hprof" + Stop. + +▶ Detected format: [FORMAT] + +Mark detect-format completed. + +## Step 3 — Create session directory + +Mark open-session in_progress. + +BASENAME = filename without directory path (e.g. "myapp.jfr") +Run: Bash: date +%Y%m%d-%H%M%S → store result as TIMESTAMP + +Run: Bash: mkdir -p ".jfr-analyzer/${BASENAME}-${TIMESTAMP}/drilldown" + +SESSION_DIR = ".jfr-analyzer/${BASENAME}-${TIMESTAMP}" + +(The open-session task spans Steps 3 and 4 — do not mark it completed until Step 4 is done.) + +## Step 4 — Open session + +Call OPEN_TOOL with path=FILE. + +Extract sessionId from the response. + +Write SESSION_DIR/session.json with the Write tool: +{ + "sessionId": "", + "format": "", + "file": "", + "sessionDir": "", + "sourceRoot": null +} + +▶ Session opened: [sessionId] + +Mark open-session completed. + +## Step 5 — Summary + +Mark summary in_progress. + +Call SUMMARY_TOOL with sessionId. + +Display a brief overview to the user: + "Recording: [BASENAME] ([duration], [thread count] threads, [sample count] samples)" + +Write the raw summary response to SESSION_DIR/summary.json using the Write tool. + +Mark summary completed. + +## Step 6 — Source context detection + +Mark source-context in_progress. + +For JFR and pprof formats only, attempt to auto-detect the source project: + + Extract package/class names from the summary output (look for Java package prefixes like + "com.example.myapp" in thread names or class names). + + Run: Bash: find . -maxdepth 2 -name "pom.xml" -o -name "build.gradle" -o -name "go.mod" -o -name "build.gradle.kts" + + If a dominant package prefix from the recording appears to match a project in the CWD + (e.g. the package name matches a directory or build artifact name): + SOURCE_ROOT="." + Update session.json sourceRoot field to "." + Output: "✓ Source: detected in current directory" + Mark source-context completed and skip the prompt below. + +If EVAL_MODE: + SOURCE_ROOT=null + Update session.json sourceRoot field to null using the Write tool. + Mark source-context completed. + Skip the rest of Step 6 — eval runs non-interactively without source. + +If auto-detection is inconclusive (or format is OTLP/HPROF), ask the user once: + + "Source code lets me show you line-level fixes instead of just method names. + Do you have the source for this application? + + [1] This directory (.) + [2] Local path (you will be asked for the path) + [3] GitHub repo — e.g. acme-corp/myapp (will be cloned to /tmp) + [4] Skip — proceed without source" + + Handle response: + - "1" → SOURCE_ROOT="." + - "2" → ask "Enter local path:" and use that path as SOURCE_ROOT + - "3" → ask "Enter GitHub repo (org/repo):" then run: + Bash: gh repo clone /tmp/jfr-src- + SOURCE_ROOT="/tmp/jfr-src-" + - "4" or anything else → SOURCE_ROOT=null + + Update session.json with "sourceRoot": SOURCE_ROOT using Write. + +Mark source-context completed. + +## Step 7 — USE report + +Mark use-report in_progress. + +For JFR, pprof, OTLP formats: call USE_TOOL with sessionId, includeInsights=true, +resources=["cpu","memory","threads","io"]. + +For HPROF: call mcp__jfr-mcp__hdump_summary with sessionId to extract available metrics +(total objects, total retained bytes). Store as a minimal USE-equivalent structure. + +Write raw result to SESSION_DIR/use-report.json using the Write tool. + +▶ USE analysis complete + +Mark use-report completed. + +## Step 8 — TSA report + +Mark tsa-report in_progress. + +For JFR and pprof formats: call TSA_TOOL with sessionId, correlateBlocking=true, +topThreads=15, includeInsights=true. + +For OTLP and HPROF: skip (no TSA tool available). Write an empty object {} to +SESSION_DIR/tsa-report.json. + +▶ TSA analysis complete (or: TSA not available for [FORMAT] format) + +Mark tsa-report completed. + +## Step 9 — Stackgraph (temporal axis) + +Mark stackgraph in_progress. + +For JFR, pprof, OTLP formats: call STACK_TOOL with sessionId. + +From the stackprofile result, look for time windows where sample density is more than 2× +the median bucket density — these are spike windows. Record each as: + { "startNs": , "endNs": , "relativeDensity": } + +Write result to SESSION_DIR/stackgraph.json using the Write tool. Include: + { "raw": , "spikeWindows": [] } + +For HPROF: skip. Write { "raw": null, "spikeWindows": [] } to SESSION_DIR/stackgraph.json. + +▶ Temporal analysis complete + +Mark stackgraph completed. + +## Step 10 — JVM/GC lanes (JFR only) + +Mark jvm-gc-lanes in_progress. + +Skip this step entirely for pprof, OTLP, and HPROF. For those formats, write +{ "skipped": true, "reason": "not JFR format" } to SESSION_DIR/jvm-gc-lanes.json and mark +jvm-gc-lanes completed. + +For JFR only — run each query using mcp__jfr-mcp__jfr_query with sessionId: + +**GC pauses:** + Query 1: events/jdk.GCPhasePause | stats(duration) + Query 2: events/jdk.GCPhasePause | quantiles(0.5, 0.95, 0.99, path=duration) + Query 3: events/jdk.GCPhasePause | groupBy(name, agg=sum, value=duration) | top(5) + +**Allocation (try each in order; use first that returns rows):** + Option A: events/jdk.ObjectAllocationSample | groupBy(objectClass, agg=sum, value=weight) | top(10) + (JDK 16+) + Option B: events/jdk.ObjectAllocationInNewTLAB | groupBy(objectClass, agg=sum, value=bytes) | top(10) + (older JDKs) + Record which option was used as allocEventUsed. + If Jafar returns a tool error (event type not found) or returns a result with zero rows, treat both as 'no rows' and try the next option. + +**JIT stalls:** + Query: events/jdk.Compilation[succeeded = false] | count() + +**Safepoints:** + Query 1: events/jdk.SafepointBegin | stats(duration) + Query 2: events/jdk.SafepointBegin | quantiles(0.5, 0.95, 0.99, path=duration) + +**Contention:** + Query 1: events/jdk.JavaMonitorEnter | stats(duration) + Query 2: events/jdk.JavaMonitorEnter | groupBy(monitorClass, agg=count) | top(10) + +**Virtual thread pinning (JDK 21+ — ignore error if event is absent):** + Query: events/jdk.VirtualThreadPinned | count() + +**Exceptions (try each in order; use first that returns rows):** + Option A: events/jdk.JavaExceptionThrow | count() + Option B: events/jdk.ExceptionStatistics | stats(count) + Record which option was used as exceptionEventUsed. + +Store ALL results to SESSION_DIR/jvm-gc-lanes.json as a JSON object using Write: +{ + "allocEventUsed": "