diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 9f9f5eb..53c8ee0 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -95,3 +95,44 @@ jobs: run: | ./gradlew build --offline -x :tiercache-tck:test -x :tiercache-transport-redis:test -x :examples:demo-spring:test ./gradlew :tiercache-core:shadowJar --offline + + server-contracts: + runs-on: ubuntu-latest + timeout-minutes: 25 + strategy: + fail-fast: false + matrix: + profile: [redis62, redis74, redis8, valkey] + steps: + - uses: actions/checkout@v7 + - uses: actions/setup-java@v6 + with: + distribution: temurin + java-version: '17' + - uses: gradle/actions/setup-gradle@v6.3.0 + - run: python3 compatibility/run-server-matrix.py ${{ matrix.profile }} + - uses: actions/upload-artifact@v7 + if: always() + with: + name: server-${{ matrix.profile }} + path: build/compatibility-evidence/ + if-no-files-found: error + + spring-consumers: + runs-on: ubuntu-latest + timeout-minutes: 25 + steps: + - uses: actions/checkout@v7 + - uses: actions/setup-java@v6 + with: + distribution: temurin + java-version: '17' + - uses: gradle/actions/setup-gradle@v6.3.0 + - run: ./gradlew stageCompatibilityArtifacts + - run: python3 compatibility/run-consumers.py + - uses: actions/upload-artifact@v7 + if: always() + with: + name: spring-consumers + path: build/compatibility-evidence/ + if-no-files-found: error diff --git a/.github/workflows/nightly.yml b/.github/workflows/nightly.yml index 4efe80f..2be0294 100644 --- a/.github/workflows/nightly.yml +++ b/.github/workflows/nightly.yml @@ -105,27 +105,40 @@ jobs: fi echo "reproducible build OK: $HASH_A" - # CVE scan (supply-chain design D4): daily Trivy filesystem scan. - # Non-blocking by design (exit-code 0) - release gating comes later. + # Resolve the published runtime/classifier graphs; source-only scanning is insufficient. cve-scan: runs-on: ubuntu-latest + timeout-minutes: 40 + permissions: + contents: read steps: - uses: actions/checkout@v7 - - name: Trivy filesystem scan (HIGH,CRITICAL) - uses: aquasecurity/trivy-action@v0.36.0 - with: - scan-type: fs - scan-ref: . - severity: HIGH,CRITICAL - format: table - output: trivy-report.txt - exit-code: '0' + with: + persist-credentials: false + - uses: actions/setup-java@v6 + with: + distribution: temurin + java-version: '17' + - uses: gradle/actions/setup-gradle@v6.3.0 + - run: python3 -m unittest discover -s scripts/release -p 'test_*.py' -v + - run: ./gradlew releaseEvidenceInputs --max-workers=2 + - run: python3 scripts/release/install_trivy.py "$RUNNER_TEMP/trivy" + - run: python3 scripts/release/evidence.py scan --trivy "$RUNNER_TEMP/trivy/trivy" --output build/release-scan --exceptions scripts/release/exceptions.json - name: Append report to job summary - run: cat trivy-report.txt >> "$GITHUB_STEP_SUMMARY" + if: always() + run: | + if [ -f build/release-scan/summary.txt ]; then cat build/release-scan/summary.txt >> "$GITHUB_STEP_SUMMARY"; fi - uses: actions/upload-artifact@v7 + if: always() with: - name: trivy-report - path: trivy-report.txt + name: dependency-scan + path: | + build/release-inputs/ + build/release-scan/*.json + build/release-scan/*.log + build/release-scan/*.txt + **/build/reports/cyclonedx/*-sbom.json + if-no-files-found: error retention-days: 90 native-smoke: @@ -179,3 +192,23 @@ jobs: name: native-smoke-log path: native-smoke.log retention-days: 30 + + sentinel-regression: + runs-on: ubuntu-latest + timeout-minutes: 20 + steps: + - uses: actions/checkout@v7 + - uses: actions/setup-java@v6 + with: + distribution: temurin + java-version: '17' + - uses: gradle/actions/setup-gradle@v6.3.0 + - run: ./gradlew :tiercache-spring-boot-starter:sentinelRuntime + - run: docker pull eclipse-temurin:17-jre + - run: python3 compatibility/run-sentinel.py + - uses: actions/upload-artifact@v7 + if: always() + with: + name: sentinel-regression + path: build/compatibility-evidence/ + if-no-files-found: error diff --git a/.github/workflows/release-candidate.yml b/.github/workflows/release-candidate.yml index f64091b..62384e6 100644 --- a/.github/workflows/release-candidate.yml +++ b/.github/workflows/release-candidate.yml @@ -1,87 +1,145 @@ name: release-candidate -# Manual trigger only: builds the publishable module jars, signs them with -# Sigstore keyless (GitHub Actions OIDC identity, no stored keys), and creates -# GitHub Artifact Attestations (SLSA-style build provenance). on: workflow_dispatch: + inputs: + source_ref: + description: 'Source commit/ref; final mode requires refs/tags/v' + required: true + type: string + version: + description: 'Must equal the checked-out gradle.properties version' + required: true + type: string + mode: + description: 'Trial accepts development snapshots; final attaches evidence to an existing release' + required: true + default: trial + type: choice + options: [trial, final] -permissions: - contents: read - id-token: write - attestations: write +permissions: {} jobs: - build-sign-attest: + build-scan: + outputs: + source_commit: ${{ steps.source.outputs.commit }} runs-on: ubuntu-latest + timeout-minutes: 45 + permissions: + contents: read steps: - uses: actions/checkout@v7 + with: + ref: ${{ inputs.source_ref }} + fetch-depth: 0 + persist-credentials: false - uses: actions/setup-java@v6 with: distribution: temurin java-version: '17' - uses: gradle/actions/setup-gradle@v6.3.0 - - # Tests are covered by ci.yml; this workflow only assembles the - # publishable artifact set. shadowJar is listed explicitly because the - # shaded jar is the main core artifact (the plain jar keeps the - # `unshaded` classifier). - - name: Build module jars - run: ./gradlew build -x test :tiercache-core:shadowJar - - # Publishable set: the seven library module jars (core shaded + unshaded) - # plus the TCK compliance-suite jar (`tests` classifier), excluding - # jmh/test-fixtures jars, the plain TCK jar, and the demo app. - - name: Stage artifacts + - name: Check source and version identity + id: source + env: + SOURCE_REF: ${{ inputs.source_ref }} + INTENDED_VERSION: ${{ inputs.version }} + EVIDENCE_MODE: ${{ inputs.mode }} run: | - set -euo pipefail - mkdir -p dist - cp tiercache-core/build/libs/tiercache-core-*.jar dist/ - rm -f dist/*-jmh.jar dist/*-test-fixtures.jar - for module in tiercache-invalidation tiercache-transport-redis tiercache-spring-boot-starter tiercache-micrometer tiercache-kotlin tiercache-reactor tiercache-micronaut; do - cp "$module/build/libs/$module"-*.jar dist/ - done - cp tiercache-tck/build/libs/tiercache-tck-*-tests.jar dist/ - ls -l dist/ - - # SLSA-style build provenance as GitHub Artifact Attestations. - - uses: actions/attest-build-provenance@v4 + python3 scripts/release/evidence.py identity --ref "$SOURCE_REF" --version "$INTENDED_VERSION" --mode "$EVIDENCE_MODE" + echo "commit=$(git rev-parse HEAD)" >> "$GITHUB_OUTPUT" + - name: Evidence regression tests + run: python3 -m unittest discover -s scripts/release -p 'test_*.py' -v + - name: Build fresh publication inputs and resolved SBOMs + run: ./gradlew clean releaseEvidenceInputs --no-build-cache --max-workers=2 + - name: Install checksum-pinned scanner + run: python3 scripts/release/install_trivy.py "$RUNNER_TEMP/trivy" + - name: Scan and enforce acceptance + run: python3 scripts/release/evidence.py scan --trivy "$RUNNER_TEMP/trivy/trivy" --output build/release-scan --exceptions scripts/release/exceptions.json + - name: Stage exact current-run evidence + env: + SOURCE_REF: ${{ inputs.source_ref }} + INTENDED_VERSION: ${{ inputs.version }} + EVIDENCE_MODE: ${{ inputs.mode }} + run: python3 scripts/release/evidence.py stage --scan build/release-scan --output dist --ref "$SOURCE_REF" --version "$INTENDED_VERSION" --mode "$EVIDENCE_MODE" + - uses: actions/upload-artifact@v7 with: - subject-path: 'dist/*.jar' - - - uses: sigstore/cosign-installer@v4.1.2 - - # Keyless signing via the runner's OIDC identity. Cosign v3 requires - # --bundle: each jar gets a .sigstore.json bundle holding the - # signature, certificate, and transparency-log proof. - - name: Sign jars (cosign keyless) + name: candidate-inputs + path: dist/ + if-no-files-found: error + retention-days: 7 + - name: Keep scan diagnostics even on rejection + uses: actions/upload-artifact@v7 + if: always() + with: + name: dependency-scan + path: | + build/release-inputs/ + build/release-scan/*.json + build/release-scan/*.log + build/release-scan/*.txt + **/build/reports/cyclonedx/*-sbom.json + if-no-files-found: error + retention-days: 90 + - name: Scan summary + if: always() run: | - set -euo pipefail - cd dist - for jar in *.jar; do - cosign sign-blob --yes --bundle "$jar.sigstore.json" "$jar" - done + if [ -f build/release-scan/summary.txt ]; then cat build/release-scan/summary.txt >> "$GITHUB_STEP_SUMMARY"; fi + sign-attest: + needs: build-scan + runs-on: ubuntu-latest + timeout-minutes: 15 + permissions: + contents: read + id-token: write + attestations: write + steps: + - uses: actions/checkout@v7 + with: + ref: ${{ needs.build-scan.outputs.source_commit }} + persist-credentials: false + - uses: actions/download-artifact@v8 + with: + name: candidate-inputs + path: dist + - run: python3 scripts/release/evidence.py verify dist + - name: Package the complete manifest-covered bundle + run: | + mkdir signed + tar -czf signed/tiercache-evidence.tar.gz -C dist . + - uses: sigstore/cosign-installer@v4.1.2 + - name: Sign all evidence bytes as one bundle + run: timeout 300 cosign sign-blob --yes --bundle signed/tiercache-evidence.sigstore.json signed/tiercache-evidence.tar.gz + - uses: actions/attest-build-provenance@v4 + with: + subject-path: signed/tiercache-evidence.tar.gz - uses: actions/upload-artifact@v7 with: name: release-candidate - path: dist/ + path: signed/ + if-no-files-found: error retention-days: 90 - - name: Print verification commands + retain-with-release: + if: inputs.mode == 'final' + needs: sign-attest + runs-on: ubuntu-latest + timeout-minutes: 10 + permissions: + contents: write + steps: + - uses: actions/download-artifact@v8 + with: + name: release-candidate + path: signed + # Requires an existing release (draft is allowed); creates no release, + # changes no published artifact, and does not upload to Maven Central. + - name: Attach immutable evidence to the existing release + env: + GH_TOKEN: ${{ github.token }} + GH_REPO: ${{ github.repository }} + INTENDED_VERSION: ${{ inputs.version }} run: | - cat <.sigstore.json \\ - --certificate-identity-regexp '^https://github.com/${{ github.repository }}/\\.github/workflows/release-candidate\\.yml@.*' \\ - --certificate-oidc-issuer 'https://token.actions.githubusercontent.com' \\ - - - Verify a jar's build provenance (GitHub Artifact Attestations): - - gh attestation verify --repo ${{ github.repository }} - EOF + gh release view "v$INTENDED_VERSION" + gh release upload "v$INTENDED_VERSION" signed/tiercache-evidence.tar.gz signed/tiercache-evidence.sigstore.json diff --git a/.gitignore b/.gitignore index 8312088..851c781 100644 --- a/.gitignore +++ b/.gitignore @@ -17,6 +17,7 @@ out/ .classpath .project .settings/ +.agents # OS .DS_Store diff --git a/AGENTS.md b/AGENTS.md index d32bc7a..5d666a0 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -20,7 +20,7 @@ Everything in this repository is **English only**: code, comments, commit messag ## Repository state -Gradle (Kotlin DSL) multi-module build, Java 17 toolchain. Modules present: `tiercache-core` (cascade read path, singleflight, cluster-wide rebuild coordination, null caching, config validation, stale-while-revalidate/XFetch, shaded Caffeine L1, L1/L2/lock SPI), `tiercache-invalidation` (invalidation protocol: versioned messages, journal, replay, last-write-wins), `tiercache-transport-redis` (Lettuce-backed L2 + lock provider, Pub/Sub and Streams invalidation profiles, Redis 6.2+/Valkey contract-tested), `tiercache-spring-boot-starter` (Spring Boot 3.5.x auto-config, Cache SPI adapters, Spring Cache migration gate), `tiercache-micrometer` (Micrometer metrics + OTel tracing, JMX inspection), `tiercache-kotlin` (Kotlin coroutines API: `KTierCache` suspend facade, invalidation `Flow`, `tierCache { }` config DSL), `tiercache-reactor` (Reactor API: `ReactorTierCache` Mono facade, invalidation `Flux`), `tiercache-micronaut` (Micronaut CacheManager/SyncCache/AsyncCache adapter, `tiercache.*` config, conditional metrics), `tiercache-tck` (chaos harness: stampede full form, avalanche, penetration, degradation, reconnect storm), `examples/demo-spring` (quick-start demo). +Gradle (Kotlin DSL) multi-module build, Java 17 toolchain. Modules present: `tiercache-core` (cascade read path, singleflight, cluster-wide rebuild coordination, null caching, config validation, stale-while-revalidate/XFetch, shaded Caffeine L1, L1/L2/lock SPI), `tiercache-invalidation` (invalidation protocol: versioned messages, journal, replay, last-write-wins), `tiercache-transport-redis` (Lettuce-backed L2 + lock provider, Pub/Sub and Streams invalidation profiles, Redis 6.2+/Valkey contract-tested), `tiercache-spring-boot-starter` (Spring Boot 3.5/4.1 consumer-tested auto-config, Cache SPI adapters, Spring Cache migration gate), `tiercache-micrometer` (Micrometer metrics + OTel tracing, JMX inspection), `tiercache-kotlin` (Kotlin coroutines API: `KTierCache` suspend facade, invalidation `Flow`, `tierCache { }` config DSL), `tiercache-reactor` (Reactor API: `ReactorTierCache` Mono facade, invalidation `Flux`), `tiercache-micronaut` (Micronaut CacheManager/SyncCache/AsyncCache adapter, `tiercache.*` config, conditional metrics), `tiercache-tck` (chaos harness: stampede full form, avalanche, penetration, degradation, reconnect storm), `examples/demo-spring` (quick-start demo). ## Build & test commands @@ -45,9 +45,9 @@ Gradle (Kotlin DSL) multi-module build, Java 17 toolchain. Modules present: `tie ## CI (GitHub Actions) - `.github/workflows/ci.yml` — push to `main` and PRs. JDK matrix 17/21/25: the 17-leg runs the full `./gradlew build` (all tests, TCK, coverage gates); the 21/25 legs run a lean build (`-x :tiercache-tck:test`) plus `:tiercache-tck:vtStressTest`. TCK runs on the 17-leg only — it validates runtime behavior, not compiler compatibility, so tripling its wall time buys nothing. The 17-leg also publishes the CycloneDX `sbom` artifact. The `offline-build` job proves `build --offline` works on a primed cache (Docker tasks excluded — image pulls are outside the offline artifact-build scope). -- `.github/workflows/nightly.yml` — scheduled (03:47 UTC) and manual (`workflow_dispatch`). Jobs: `pitest` (≥75% gate), `reproducible-build` (core shaded jar must be byte-identical across two independent checkouts), `cve-scan` (Trivy HIGH+CRITICAL, non-blocking report), `native-smoke`, `benchmarks`, `jmh` (informational; hosted-runner numbers never gate). The soak gate is deliberately NOT in CI: hosted runners kill long jobs, so soak runs locally/on-premises via `./gradlew :tiercache-tck:soakTest`. -- `.github/workflows/release-candidate.yml` — manual dispatch: builds the seven module jars plus the TCK compliance-suite jar (`tests` classifier), creates SLSA build provenance (`actions/attest-build-provenance`) and Sigstore keyless signatures (`cosign sign-blob`, `.sigstore.json` bundles). Verification commands are printed by the workflow and mirrored in SECURITY.md. -- Blocking gates: build+tests+coverage (per push), PIT, reproducible-build (nightly), soak (local, before releases). Informational: JMH, benchmarks, CVE scan. A red nightly does not block merges but must be investigated the same day. +- `.github/workflows/nightly.yml` — scheduled (03:47 UTC) and manual (`workflow_dispatch`). Jobs: `pitest` (≥75% gate), `reproducible-build` (core shaded jar must be byte-identical across two independent checkouts), `cve-scan` (resolved SBOM completeness plus blocking unexcepted HIGH/CRITICAL acceptance), `native-smoke`, `benchmarks`, `jmh` (informational; hosted-runner numbers never gate). The soak gate is deliberately NOT in CI: hosted runners kill long jobs, so soak runs locally/on-premises via `./gradlew :tiercache-tck:soakTest`. +- `.github/workflows/release-candidate.yml` — manual explicit ref/version dispatch. Trial permits development snapshots; final requires a matching stable version tag and clean checkout. Builds all nine publications including core fixtures and TCK tests, checks resolved SBOMs and CVEs, and signs/attests the complete checksum-manifest bundle. Final mode attaches evidence to an existing release without creating/publishing a release or uploading to Central. See docs/release-evidence.md. +- Blocking gates: build+tests+coverage (per push), PIT, reproducible-build (nightly), soak (local, before releases). Informational: JMH and benchmarks. Incomplete dependency evidence and unexcepted HIGH/CRITICAL findings reject release acceptance. A red nightly does not block merges but must be investigated the same day. - Dependency hygiene: Dependabot runs weekly for Gradle and GitHub Actions (grouped minor/patch PRs). - To rerun nightly manually: `gh workflow run nightly.yml`. @@ -89,10 +89,10 @@ A user pulls in exactly **one starter module**. Never let framework or client de These are the product's identity. Violating them is a bug, not a trade-off: -1. **Correct by default.** All protections — singleflight, distributed rebuild coordination with double-check and watchdog lease, TTL jitter 5–10%, TTL ordering `TTL_L1_effective ≤ TTL_L2` with fail-fast startup validation on violation, null-caching policy, L2 timeouts below business timeout — are ON without configuration. Disabling requires explicit opt-in and is logged as a risk. Known DIY traps (missing double-check after lock acquisition, explicit `leaseTime` killing the watchdog, L1 never warmed from L2) must be impossible **by API construction**, not by documentation. +1. **Correct by default.** Singleflight, distributed rebuild coordination with double-check and watchdog lease, TTL jitter, TTL-ordering validation and bounded default L2 timeouts are enabled by default. Null caching, stale serving and XFetch require opt-in. Disabling requires explicit opt-in and is logged as a risk. Known DIY traps (missing double-check after lock acquisition, explicit `leaseTime` killing the watchdog, L1 never warmed from L2) must be impossible **by API construction**, not by documentation. 2. **Never claim strong consistency.** The cache is eventually consistent by design. All artifacts — docs, logs, exceptions, marketing text — describe a bounded, measurable staleness window only. -3. **Observability as a feature.** Every failure mode has a metric (`tiercache.requests{result=...}`, `tiercache.latency{level=...}`, `tiercache.invalidation{direction=...}`, `tiercache.journal.size`, `tiercache.degraded`, `tiercache.breaker.state`, ...). Every chaos test must be diagnosable from metrics alone. -4. **Degradation is honest.** On L2 failure: circuit breaker → L1-only mode, zero infrastructure exceptions escaping into business code; loss of cross-instance `putIfAbsent` atomicity is surfaced via `tiercache.degraded=1` metric + log + docs. Recovery: journal replay → rate-limited warm-up → breaker close; **instant full L1 flush on reconnect is forbidden**. +3. **Observability as a feature.** Supported cache outcomes and recovery activity have metrics (`tiercache.requests{result=...}`, `tiercache.latency{level=...}`, `tiercache.invalidation{direction=...}`, `tiercache.journal.size`, `tiercache.degraded`, `tiercache.breaker.state`, ...). Use metrics together with logs and application signals; lock-release cleanup is currently log-only. +4. **Degradation is honest.** On L2 failure: circuit breaker → L1-only mode, zero infrastructure exceptions escaping into business code; loss of cross-instance `putIfAbsent` atomicity is surfaced via `tiercache.degraded=1` metric + log + docs. Recovery uses bounded asynchronous journal work outside state monitors. Verified history preserves unaffected L1 entries; unconfirmable history may require a baseline-before-clear fallback. A failed baseline clears conservatively but retains the cursor and remains pending. The breaker closes only after safe same-epoch recovery; no universal no-flush or source-load multiplier is promised. 5. **Zero migration threshold.** Migration from standard Spring Cache = swap the starter + one config line; existing `@Cacheable` code unchanged. ## Explicit non-goals (fixed scope boundary) @@ -124,7 +124,7 @@ Quality gates (enforced in CI once set up): - Branch coverage of `core` ≥ 90%; PIT mutation score ≥ 75% on invalidation/degradation paths. - JMH benchmarks as regression gates: L1-hit overhead ≤ +50% over raw Caffeine, zero steady-state allocations; > 10% regression blocks merge. Throughput benchmarks (mixed workload, cascade) are trend measurements without absolute budgets — absolute throughput is environment-dependent. - All observability metrics asserted by tests. -- Test matrix: Redis 6.2+ and Valkey, standalone/Sentinel/Cluster, JDK 17/21/25. +- Test matrix: pinned Redis 6.2/7.4/8.x and Valkey standalone contracts; Redis 7.4 Sentinel regressions with starter-managed connections; separate Boot 3.5/4.1 consumer BOMs on Java 17. JDK 17/21/25 build checks remain separate. Redis Cluster is unsupported by the stock transport. See docs/compatibility.md. ## Performance targets (for orientation) @@ -155,7 +155,7 @@ Work should land in roadmap order — do not build phase 2+ features before the ## Competitive context (why decisions look this way) - **Spring Cache** has no multi-level support — `CompositeCacheManager` never warms L1 on L2 hit; this defect must be impossible by construction here. -- **JetCache** is frozen; **Redisson** paywalls near-cache eviction and per-entry TTL (PRO); **Hazelcast** is a platform, not a library over existing Redis; **caffeinated-redis** lacks stampede protection, degradation handling, and null semantics. Our differentiators: completeness of failure-mode protection, observability, zero-config migration, and infrastructural reliability. +- **Redisson** paywalls near-cache eviction and per-entry TTL (PRO); **Hazelcast** is a platform, not a library over existing Redis; **caffeinated-redis** lacks stampede protection, degradation handling, and null semantics. Our differentiators: completeness of failure-mode protection, observability, zero-config migration, and infrastructural reliability. - Framework market is fragmented (Spring Boot ~42%, Micronaut ~39%) — hence framework-independent core + thin adapters. Post-GA candidates: Micronaut/Quarkus modules via the same SPI, without bloating core. # Coding guide diff --git a/CHANGELOG.md b/CHANGELOG.md index 85b920f..a891a3a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,55 @@ All notable changes to this project are documented in this file. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). +## [Unreleased] + +## [2.0.0] - 2026-09-23 + +### Changed + +- Documentation clarification (2026-09-23): historical claims below that whole-cache clear and journal append are atomic, that reconnect never flushes L1, or that every failure is diagnosable from metrics alone are superseded. SCAN clear and EVICT_ALL append are separate; unverifiable history (including read failure) can require fallback clear, while unstored events cannot be replayed. See docs/configuration.md, docs/recovery.md and docs/troubleshooting.md. This clarification does not retroactively claim that older releases passed current gates. + +- Raised the Micrometer baseline to 1.15.12 and aligned the Redis transport's Netty family through the published 4.2.17.Final BOM to address HIGH/CRITICAL dependency findings. Consumer-enforced BOMs can override these defaults; scan the application's resolved graph. + +- Release dependency evidence now compares resolved runtime/classifier inventories with module and aggregate SBOMs, including shaded Caffeine and core test fixtures. Trivy SBOM scans fail closed for incomplete evidence and unexcepted HIGH/CRITICAL findings. Candidate manifests bind exact publication bytes, SBOMs and scan reports to an explicit ref/version; complete bundles are signed/attested, with final evidence attached to an existing release. No automatic Central publication is added; see docs/release-evidence.md. + +- Added pinned Redis 6.2/7.4/8.x and Valkey contract profiles, isolated Spring Boot 3.5/4.1 published-artifact consumer checks, and a Docker Sentinel regression for planned promotion, abrupt primary loss and full outage. Evidence records dependency graphs, image identities, topology and observed value versions. Redis Cluster remains unsupported; see docs/compatibility.md. + +- **BREAKING:** the built-in Redis keyspace moves to v2. Complete cache names are encoded without delimiter/glob ambiguity, and data, tags, journals, channels, consumer groups and rebuild locks have separate address families. Clearing one cache can no longer delete another cache's data or library control state. Old/new active deployments require a coordinated cold cutover; no legacy reads, event bridge or automatic cleanup are provided. See UPGRADING.md and docs/redis-keyspace-v2.md. Value frames and application serializers are unchanged; whole-cache clear is still non-transactional. + +### Fixed + +- The strict soak gate now measures post-GC heap and process RSS independently, verifies explicit GC completion, and observes every worker Future, including captured Errors. Missing measurements, insufficient samples, cancelled/early/hung workers and independent memory growth above the unchanged 5% budget fail the gate. Linux/macOS readers and per-sample JSON evidence replace the combined heap-plus-virtual-memory proxy; see docs/tck.md. + + +- JMX inspection now owns only its successful registration and consumes that ownership once on close. Skipped/failed registrations and repeated close cannot remove another factory's or a foreign MBean; concurrent lifecycle calls are ordered without changing the fixed ObjectName. All enum-derived Micrometer labels now use Locale.ROOT, preserving JMX hit ratios and metric identities under Turkish or changing default locales. Restart affected processes; existing malformed time series are not renamed in place. + + +- Invalidation publication now observes native Redis completion without delaying writes or replacing committed write results with publication errors. Compatible transport/listener defaults distinguish acknowledged, failed, legacy-unconfirmed and Streams-not-required outcomes. A bounded observer worker exports `tiercache.invalidation.publish` batches with rate-limited sanitized diagnostics; SENT remains the submission-attempt counter. Late completion cannot restart closed observers; see docs/observability.md. + + +- Built-in invalidation journals now reject capacities <=64 before connection use, with one shared 64-event cadence and minimum-capacity definition across Redis, Spring and Micronaut. The minimum is 65 (cursor row plus 64 later events), the default remains 10000, and disabled journals ignore unused capacity settings. This prevents avoidable cursor-baseline loss; sufficient outage retention still requires workload sizing. + + +- Spring Cache now implements both asynchronous retrieve overloads through the owning factory's bounded async view. Lookups no longer perform L2 I/O on the retrieval caller thread; synchronized CompletableFuture/Mono loaders use existing coalescing, null policies and cancellation/rejection semantics. The legacy internal adapter constructor remains synchronous-only and returns explicit failed futures for retrieval; see UPGRADING.md. + + +- L1 freshness now shares the lifetime of its value or null marker, independent of bounded invalidation fencing. Fresh access preserves the store-time retention floor and atomically replaces the current holder, preventing resurrection after L1 eviction. Custom L1 providers need the compatible atomic-replacement extension only for degradation stale serving combined with access expiry. Spring and Micronaut preserve explicit zero overrides and identify invalid cache/default settings. + + +- Lock providers now dispose their owned connections exactly once across lazy initialization and shutdown races, without closing borrowed clients/connections. Derived providers stay lazy. Factory close gates auxiliary work, retires discarded refresh claims and preserves usable synchronous views with local coalescing. Watchdog shutdown releases acquired tokens once before loader fallback; closed-provider outcomes retire breaker permits without recording a Redis result. See docs/resource-lifecycle.md. + + +- Streams no longer strands the rest of a delivered batch after a corrupt row or failed ACK. Readers drain own pending work, claim only within their own group, retain per-row apply/ACK state and require a current safe reset baseline before settling corrupt/missing history. Shared stream/journal validation skips decoding an already-accounted opaque anchor, including poison at the reset tail. Failed clears now complete as bounded recovery failures rather than immediate obsolete-pass retries. Added bounded failure diagnostics and stream failure counters; see docs/streams-recovery.md for stable/random group lifecycle and internal SPI migration. + +- Journal recovery now uses two owned workers with bounded per-cache scheduling, pass budgets and exponential retries. Redis I/O and observer callbacks run outside cache-state and breaker monitors. Successful probes return without waiting for replay; CLOSED is gated by the same recovery epoch's verified replay or safe reset. Failed baseline reads still clear L1 but retain the cursor and remain pending. Shutdown cancels queued work and detaches retired coherence hooks. Added `tiercache.invalidation.recovery.pending{cache}` and real-Redis JFR coverage; see docs/recovery.md. + +- Tagged data and its reverse index now share one absolute expiry instant. On Redis 6.2, separate relative TTL commands could produce different expiration times inside the same Lua script, especially with many tags. Extend-only tag-set TTLs and losing-write behavior remain unchanged. + +- Tagged writes preserve the remote acceptance result: rejected candidates no longer warm L1, publish UPDATE events, or alter tag indexes. Lettuce atomically updates accepted values, replacement tags, reverse indexes and journal bookkeeping. Losing writers perform one bounded convergence read; degraded writes stay local. Custom versioned tagged SPI providers must implement the new outcome method and capability query; see UPGRADING.md. + +- Foreground misses that join a skipped SWR/XFetch refresh no longer receive a false null. In-flight claims distinguish real results from skipped coordination and promote foreground demand through the existing bounded load path. Races with an already-skipped or replacement refresh retain singleflight ownership, the original coordination deadline and the two-loader-execution limit; genuine nulls and loader failures remain distinct. + ## [1.4.0] - 2026-09-21 ### Added diff --git a/README.md b/README.md index 880164a..99fc9bd 100644 --- a/README.md +++ b/README.md @@ -12,7 +12,7 @@ The cache is eventually consistent by design; no strong-consistency guarantees a - **Two-level read cascade** — L1 (shaded Caffeine, zero-allocation hit path) → L2 (Redis/Valkey) → your loader. An L2 hit always warms L1, so the next read of the same key is served in-process. - **Correct by default** — singleflight per instance plus cluster-wide rebuild coordination (distributed lock with watchdog lease extension and mandatory double-check), TTL jitter, fail-fast TTL-ordering validation, atomic `putIfAbsent` — all on without configuration; disabling requires an explicit opt-in and is logged as a risk. Null caching is opt-in (`null-policy: allow`) with a tri-state `lookup` to distinguish miss from cached-null. - **Cross-instance invalidation** — versioned events with last-write-wins ordering, a bounded journal with replay on reconnect, and two transport profiles: lightweight Pub/Sub or durable Redis Streams. -- **Honest degradation** — a circuit breaker switches the cache to L1-only when Redis fails; business code never sees infrastructure exceptions. Recovery replays recorded invalidations when journal history is readable and intact. Missing or unverifiable history, a replay-read failure, or a missing journal can trigger a full L1 flush and a source-load burst. Writes that never reached Redis cannot be reconstructed by replay; see [recovery semantics](docs/configuration.md#circuit-breaker-and-degradation) for the fallback signals and limits. +- **Honest degradation** — a circuit breaker switches protected cache operations to local fallback when Redis fails. Loader errors, invalid configuration and async executor rejection remain visible. Recovery replays recorded invalidations when journal history is readable and intact. Missing or unverifiable history, a replay-read failure, or a missing journal can trigger a full L1 flush and a source-load burst. Writes that never reached Redis cannot be reconstructed by replay; see [recovery semantics](docs/configuration.md#circuit-breaker-and-degradation) for the fallback signals and limits. - **Stale serving** — stale-while-revalidate and XFetch early refresh keep hot keys fast while values refresh in the background. - **Observability as a feature** — Micrometer metrics for cache outcomes, degradation and invalidation, OpenTelemetry tracing, JMX inspection, and a reference Grafana dashboard with alert rules in [`docs/grafana/`](docs/grafana/). Some cleanup failures are log-only; see the [observability catalog](docs/observability.md). - **Kotlin coroutines** — `suspend` API, invalidation `Flow`, and a `tierCache { }` config DSL in `tiercache-kotlin`. A suspending loader runs on the caller's coroutine dispatcher, so a blocking loader blocks that dispatcher — offload blocking work with `withContext(Dispatchers.IO)`. @@ -25,7 +25,7 @@ The cache is eventually consistent by design; no strong-consistency guarantees a ```kotlin // build.gradle.kts -implementation("io.github.cramen:tiercache-spring-boot-starter:1.4.0") +implementation("io.github.cramen:tiercache-spring-boot-starter:2.0.0") ``` ```yaml @@ -41,7 +41,7 @@ The starter replaces the standard cache manager: `@Cacheable` / `@CachePut` / `@ ```kotlin // build.gradle.kts -implementation("io.github.cramen:tiercache-micronaut:1.4.0") +implementation("io.github.cramen:tiercache-micronaut:2.0.0") ``` ```yaml @@ -57,8 +57,8 @@ The module replaces Micronaut's `DefaultCacheManager`: `@Cacheable` / `@CachePut ```kotlin // build.gradle.kts -implementation("io.github.cramen:tiercache-core:1.4.0") -implementation("io.github.cramen:tiercache-transport-redis:1.4.0") +implementation("io.github.cramen:tiercache-core:2.0.0") +implementation("io.github.cramen:tiercache-transport-redis:2.0.0") ``` ```java @@ -85,7 +85,7 @@ Reads cascade L1 → L2 → loader; concurrent loads of the same key share one l | Module | What it gives you | |---|---| -| `tiercache-spring-boot-starter` | Spring Boot 3 auto-configuration — the one dependency most Spring apps need | +| `tiercache-spring-boot-starter` | Spring Boot 3.5 / 4.1 consumer-tested auto-configuration — the one dependency most Spring apps need | | `tiercache-micronaut` | Micronaut CacheManager/SyncCache/AsyncCache adapter — the one dependency Micronaut apps need | | `tiercache-core` | The framework-independent cache engine: cascade, singleflight, rebuild coordination, degradation | | `tiercache-transport-redis` | Lettuce-backed Redis/Valkey L2 and invalidation transport | @@ -97,11 +97,16 @@ Reads cascade L1 → L2 → loader; concurrent loads of the same key share one l ## Requirements - Java 17 or newer -- Redis 6.2+ or Valkey (for L2 and cross-instance features) +- Redis 6.2+ or Valkey (for L2 and cross-instance features). See the [tested platform matrix](docs/compatibility.md) for exact versions and Sentinel scope; Redis Cluster is unsupported by the stock transport. - Docker, to run the integration tests and TCK chaos suite locally ## Compatibility and versioning +TierCache 2.0 introduces Redis keyspace v2, a breaking operational change. +It requires a coordinated cold-cache cutover; +old/new instances are not rolling-compatible. See the +[v2 migration guide](docs/redis-keyspace-v2.md) before upgrading from 1.x. + All published Maven artifacts follow **semantic versioning**: patch releases for backwards-compatible fixes, minor releases for backwards-compatible additions, major releases for breaking changes. **Supported public API** — the documented entry points only: @@ -129,6 +134,8 @@ Everything else — builders, transport internals, metrics helpers, and any type - [Migration from Spring Cache](docs/migration-from-spring-cache.md) - [Migration from Redisson](docs/migration-from-redisson.md) - [Migration from JetCache](docs/migration-from-jetcache.md) +- [Troubleshooting](docs/troubleshooting.md) +- [Release verification evidence](docs/release-evidence.md) - [Sizing and TTL guidance](docs/sizing-and-ttl.md) - [Observability: metrics, tracing, dashboards](docs/observability.md) - [Running the TCK chaos suite](docs/tck.md) @@ -139,3 +146,5 @@ Everything else — builders, transport internals, metrics helpers, and any type ## License [Apache License 2.0](LICENSE) + +Recovery completion, HALF_OPEN admission and fallback limits are described in [the recovery guide](docs/recovery.md). diff --git a/SECURITY.md b/SECURITY.md index c004745..3117d1a 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -24,25 +24,33 @@ You will receive an acknowledgment within 3 business days. ## Scope -This policy covers the published library modules (`tiercache-core`, -`tiercache-invalidation`, `tiercache-transport-redis`, -`tiercache-spring-boot-starter`, `tiercache-micrometer`, -`tiercache-kotlin`). The `examples/` applications are demonstration code -and are out of scope. +This policy covers the published library modules and TCK, including their shipped +classifiers. The `examples/` applications are demonstration code and are out of scope. ## Supply-Chain Verification -Release-candidate artifacts are built on GitHub Actions with -SLSA-style build provenance and Sigstore keyless signatures. To verify -an artifact: - -```bash -gh attestation verify --repo cramen/tier_cache -cosign verify-blob --bundle .sigstore.json \ - --certificate-identity-regexp '^https://github.com/cramen/tier_cache/.github/workflows/release-candidate.yml@.*' \ - --certificate-oidc-issuer 'https://token.actions.githubusercontent.com' -``` - -Each CI build of `main` also publishes a CycloneDX SBOM as the `sbom` -workflow artifact, and the nightly pipeline checks that the core jar -builds reproducibly (byte-identical double build). +The release-candidate workflow resolves the actual publication runtime graphs, +checks SBOM completeness (including shaded Caffeine and published test fixtures), +and scans them with a checksum-pinned Trivy binary. Unexcepted HIGH/CRITICAL +findings and incomplete or failed scans reject acceptance. Any checked-in +exception must name the exact vulnerability/artifact/version, owner, rationale and +future expiration date; exceptions require maintainer review. + +The complete candidate evidence bundle is signed with Sigstore and receives +GitHub build provenance. Its manifest binds binaries, classifiers, publication +metadata, SBOMs and scan results to their exact SHA-256 bytes and source commit. +Final-mode evidence is retained with an existing GitHub release; ordinary +SNAPSHOT development and trial workflows remain allowed. + +See [release evidence](docs/release-evidence.md) for the precise scope, scanner +self-test, exception policy, commands, signature verification and comparison with +Central downloads. A separately rebuilt artifact is not covered by the candidate +attestation unless its checksum matches. No historical release is retroactively +claimed to have passed this new evidence gate. + + +The [recorded trial](docs/release-evidence.md#recorded-trial-evidence--2026-09-22) +is dated evidence for a specific commit, not acceptance of a future release. +Application BOMs can override library versions; functional compatibility checks +in the [platform matrix](docs/compatibility.md) do not replace scanning the final +application dependency graph. diff --git a/UPGRADING.md b/UPGRADING.md index ccd6dd2..747fb7b 100644 --- a/UPGRADING.md +++ b/UPGRADING.md @@ -7,6 +7,178 @@ target versions. For the full list of additions and fixes, see [CHANGELOG.md](CHANGELOG.md). +## 2.0.0: observable invalidation publication + +The void transport method remains available; legacy implementations adapt to +UNCONFIRMED through the new default publishAsync method. Native Pub/Sub completion +supplies actual Redis acknowledgement or a failed stage, while Streams reports +NOT_REQUIRED. Existing metrics listeners remain compatible through a default +batched publication callback. No mandatory Micrometer dependency was added. + +SENT records attempted submission, not successful delivery. Use the new +`tiercache.invalidation.publish` outcome counters and dashboard panels. Terminal +callbacks run on a bounded observer worker; counts may appear after the cache call +returns. Observer errors and failed publication do not replace committed write +results. Direct void Pub/Sub callers should use publishAsync to observe submission +errors as well as late failures. No automatic republish or stronger delivery +recovery guarantee is introduced. Close drains available observations for at most +one second, with best-effort export after backend shutdown. + +## 2.0.0: invalidation journal capacity floor + +Enabled built-in journals now reject `tiercache.invalidation.journal-capacity` +values <=64, including the exact boundary 64. Configure at least 65; the confirmed +cursor row must survive alongside the next 64 delivered events. The default remains +10000. Direct RedisStreamJournal construction and Spring/Micronaut wiring validate +before journal commands or owned connection creation; disabled invalidation leaves +its unused capacity setting alone. Values are rejected, never silently clamped. + +65 is only the protocol floor. Choose a larger window for bursts, disconnected +receivers and scheduled recovery delays. Redis trimming remains approximate, and +read failures or real history loss can still require a conservative L1 reset. +See [journal sizing](docs/sizing-and-ttl.md#journal-capacity). + +## 2.0.0: complete Spring async retrieval + +Managed Spring caches now use the factory's bounded async view for both retrieve +methods introduced in Spring 6.1. Single-argument retrieval no longer blocks its +caller on L2. Supplier retrieval supports synchronized CompletableFuture and Mono +caching through existing engine coalescing, with unchanged null policy and async +shutdown/rejection semantics. Ordinary sync=false annotations gain no coalescing +promise; synchronous writes/evictions remain synchronous. + +The internal two-argument TierCacheSpringCache constructor remains usable for sync +operations, but both retrieve overloads return failed futures without an async view. +Direct async users must use TierCacheManager or the new constructor accepting both +views. This changes its former blocking single-argument retrieval behavior. +See [Spring async retrieval](docs/migration-from-spring-cache.md#asynchronous-retrieval) +for wrappers, cached nulls, bounded execution and cancellation. + +## 2.0.0: value-bound local freshness + +Degradation freshness now lives with the retained L1 entry instead of an +independently expiring metadata map. Freshness and allowed stale serving no longer +end early when a version fence expires or is evicted. Access refresh preserves +the original retention floor and cannot reinsert an entry removed by Caffeine. + +Existing LocalCache methods and defaults remain compatible. Custom providers +must preserve opaque StoredEntry holders. Only providers combining a positive +`degradationStaleTtl` with `l1ExpireAfterAccess` need to implement the new default +capability and atomic identity-replacement methods; otherwise engine-cache +creation rejects that combination. Built-in Caffeine already supports it. +Redis frames and remote timestamps are unchanged. The window remains off by +default and does not heal writes performed while Redis was unavailable. + +## 2.0.0: lock-provider and factory shutdown + +Client-backed lock providers now close the dedicated connections they create; +caller-supplied clients and connections remain caller-owned. Derived providers +are lazy and owned by their factory. Direct acquisition after provider close +fails immediately instead of accidentally reopening coordination resources. + +Already-obtained synchronous caches remain usable with a usable supplied L2, +but factory close disables coordination, refresh, invalidation publication and +recovery. Continued cluster coherence is not promised. Async shutdown semantics, +constructor signatures, stored data and lease/compensation settings are unchanged. +See [resource ownership and shutdown](docs/resource-lifecycle.md). + +## 2.0.0: Redis keyspace v2 (breaking) + +The built-in Redis/Valkey transport now uses separate v2 data, tag, journal, +channel, group and lock addresses. This fixes cross-cache deletion by clear +for hierarchical names (`user` / `user:roles`) and glob-containing names +(`a?` / `a1`). Java cache APIs and value frames are unchanged, but active +old/new instances are **not rolling-compatible**. + +Version 2.0.0 carries this operational break under the published major-release +compatibility policy. It is not a compatible 1.x patch/minor upgrade. + +Follow the [v2 migration guide](docs/redis-keyspace-v2.md): quiesce traffic or +source mutations, drain and stop all old requests/loaders/publishers, start +with empty v2 namespaces, then resume with capacity for cold-cache loads. +V2 never reads, copies, subscribes to or deletes legacy state. Rollback also +requires a drained cutover and clean isolated cache storage. Whole-cache +clear remains a non-transactional scan followed by a separate journal append. + +## 2.0.0: asynchronous invalidation recovery + +Replay no longer runs under state monitors or inline in the successful probe +or reconnect callback. The breaker stays HALF_OPEN until its recovery epoch +has verified replay or a safe baseline-and-clear result; CLOSED/recovered +notifications can therefore arrive later. A failed baseline still clears L1 +but keeps the confirmed cursor and recovery pending. Repeated failures retry +with bounded backoff instead of waiting for another live event. + +Factories own two recovery workers and unregister pending gauges on close. +Closed coherence hooks cannot be restarted by surviving synchronous caches; +real probes may still establish caller-owned L2 availability. Existing SPI +methods remain, with additive asynchronous completion and local-clear epoch +hooks. Custom callers must await the completion stage when they require +settled recovery; returning from the old void callback is no longer that +boundary. See [the recovery contract](docs/recovery.md). + +## 2.0.0: Streams pending recovery + +Streams now drains its own pending work before new rows and resumes only +inside the receiver's own group. Poison/missing rows require a committed +baseline-before-clear result before covered ACKs. Failed application and ACK +attempts retain the delivered batch and use bounded retries. A reader without +a capable gap handler remains pending rather than silently skipping the gap. + +Default random-identity groups are retired best-effort on graceful close; +explicit stable-UUID groups persist for an **exclusive** restart. Registration +clears the fresh/resumed target against a captured baseline, so covered old +UPDATE payloads cannot warm a new L1. Do not run two live owners of one UUID. + +`CheckedRange` keeps its record signature but may omit a raw-validated cursor +anchor. Consumers must use its integrity flag rather than requiring the first +event to be the cursor row. Existing custom journals may retain a valid typed +anchor. Corrupt unconsumed rows now raise sanitized typed failures. See +[Streams recovery and operator procedures](docs/streams-recovery.md). + +## 2.0.0: custom transport migration + +### Custom transports: versioned tagged writes + +The public `TierCache.put(key, value, tags)` signature is unchanged. Custom +`RemoteCache` providers and decorators must implement both +`supportsTaggedWriteOutcomes()` (a side-effect-free, no-I/O capability query) +and `putTaggedIfNewer(...)` before accepting **versioned tagged writes**. +The existing void `putTagged(...)` method alone is no longer sufficient. +Old providers still compile and link, but core now throws an actionable +`CacheConfigurationException` before any mutation for this unsupported +operation, including when the breaker is OPEN. Unsupported capability is +not recorded as an infrastructure failure. + +Return `WON` only after accepting the candidate, and `LOST` when a newer +stored value or tombstone rejects it. Couple the acceptance decision with +data, replacement tag memberships, reverse index and any configured journal +append. A losing candidate must leave all of them unchanged. Advertise the +capability only when this contract is implemented; decorators must forward +both methods. The default extension returns `UNSUPPORTED` without I/O for +versioned entries. Unversioned entries still delegate to the legacy void +method with its existing unconditional semantics. + +The built-in Lettuce transport implements this in one Lua operation on +Redis/Valkey. This tagged-write correction alone does not change value frames, +key names or journal formats; the separate v2 keyspace change above does +change addresses and requires its coordinated migration. +Its existing limitation remains: without a journal, or for unversioned +entries, tagged writes are unconditional. Retagging replaces old memberships; +it does not accumulate every tag ever assigned to the key. During a mixed +rollout, old writers can still create incorrect memberships or publish a +losing candidate. Upgrade all writers; already-corrupt indexes are not +repaired automatically by the new protocol. + +A confirmed loss performs at most one convergence read and never publishes +the losing value. A refused breaker probe or an admitted infrastructure +failure uses local-only fallback without publishing or persisting tags. +This does not make an unacknowledged timeout a confirmed loss: Redis may +have accepted the write before the client timed out. There is no guaranteed +rollback, exactly-once retry, or reconciliation after such an uncertain +outcome. Lua excludes interleaving commands, but does not roll back commands +already executed if a later Redis runtime error occurs. + ## 1.4.0 - New opt-in knob `tiercache.degradation-stale-ttl` (per cache, default `0` @@ -178,3 +350,22 @@ Modules: - `tiercache-tck` — public chaos-test suite (Testcontainers) and benchmarks. Baseline requirements: JDK 17+, Redis 6.2+ or Valkey. + + +## Documentation clarification — 2026-09-23 + +Whole-cache SCAN clear and its following EVICT_ALL journal append are separate, +non-transactional steps; neither clear nor tag/batch eviction fences concurrent +writes. Verified replay preserves unaffected L1 entries, but missing/unverifiable +history, including read failure inside nominal retention, can require a clear. +A failed fallback baseline still clears conservatively while retaining the cursor +and pending recovery. A silent gap needs a recovery trigger; journal existence +alone does not heal it. Changes that never reached Redis cannot be reconstructed. + +Budget source staleness using [TTL and outage conditions](docs/sizing-and-ttl.md), +not an unconditional one-L1-TTL bound. A fallback across a fleet can cause extra +source traffic. Keep Spring `sync=true` for loader coalescing; null and stale +serving remain opt-in. See [diagnosis](docs/troubleshooting.md), the +[tested platform matrix](docs/compatibility.md) and [release evidence](docs/release-evidence.md). +These are scope clarifications, not a new format or API migration; the separately +documented Redis v2 cold cutover and major-version requirement still apply. diff --git a/build.gradle.kts b/build.gradle.kts index 50e56e8..72e3b90 100644 --- a/build.gradle.kts +++ b/build.gradle.kts @@ -18,6 +18,9 @@ plugins { kotlin("jvm") version "2.2.21" apply false } +group = "io.github.cramen" +version = providers.gradleProperty("version").get() + // Whole-repo aggregate SBOM, named like the module ones so CI can collect a // flat directory of *-sbom.json files. tasks.named("cyclonedxBom") { @@ -41,8 +44,14 @@ subprojects { // shaded into the core jar, so it is part of the shipped artifact. apply(plugin = "org.cyclonedx.bom") tasks.named("cyclonedxDirectBom") { + // Demo applications are not shipped Maven artifacts. + if (project.path.startsWith(":examples")) enabled = false projectType.set(Component.Type.LIBRARY) - includeConfigs.set(listOf("runtimeClasspath")) + includeConfigs.set(when (project.name) { + "tiercache-core" -> listOf("runtimeClasspath", "testFixturesRuntimeClasspath") + "tiercache-tck" -> listOf("runtimeClasspath", "testRuntimeClasspath") + else -> listOf("runtimeClasspath") + }) } tasks.named("cyclonedxBom") { projectType.set(Component.Type.LIBRARY) @@ -85,3 +94,24 @@ subprojects { } } } + +// Isolated repository consumed by the compatibility builds; never publishes remotely. +val consumerModules = setOf("tiercache-core", "tiercache-invalidation", "tiercache-transport-redis", + "tiercache-spring-boot-starter", "tiercache-micrometer") +subprojects { + if (name in consumerModules) { + pluginManager.withPlugin("maven-publish") { + extensions.configure { + repositories.maven { + name = "compatibility" + url = rootProject.layout.buildDirectory.dir("compatibility-repository").get().asFile.toURI() + } + } + } + } +} +tasks.register("stageCompatibilityArtifacts") { + dependsOn(consumerModules.map { ":$it:publishAllPublicationsToCompatibilityRepository" }) +} + +apply(from = "gradle/release-evidence.gradle") diff --git a/compatibility/.gitignore b/compatibility/.gitignore new file mode 100644 index 0000000..c18dd8d --- /dev/null +++ b/compatibility/.gitignore @@ -0,0 +1 @@ +__pycache__/ diff --git a/compatibility/fixture_config.py b/compatibility/fixture_config.py new file mode 100644 index 0000000..c1086d4 --- /dev/null +++ b/compatibility/fixture_config.py @@ -0,0 +1,4 @@ +from pathlib import Path +ROOT = Path(__file__).resolve().parents[1] +PROFILES = dict(line.split('=', 1) for line in (ROOT/'compatibility/platforms.properties').read_text().splitlines() + if line and not line.startswith('#')) diff --git a/compatibility/platforms.properties b/compatibility/platforms.properties new file mode 100644 index 0000000..fff6dea --- /dev/null +++ b/compatibility/platforms.properties @@ -0,0 +1,7 @@ +# Exact versions; image IDs/digests and resolved dependency graphs are saved per run. +redis62=redis:6.2.24-alpine +redis74=redis:7.4.11-alpine +redis8=redis:8.10.2-alpine +valkey=valkey/valkey:9.1.2-alpine +boot35=3.5.16 +boot4=4.1.1 diff --git a/compatibility/run-consumers.py b/compatibility/run-consumers.py new file mode 100644 index 0000000..2a5f5fe --- /dev/null +++ b/compatibility/run-consumers.py @@ -0,0 +1,29 @@ +#!/usr/bin/env python3 +"""Run isolated Boot consumer graphs against one ephemeral standalone Redis.""" +import json, subprocess, time, uuid +from pathlib import Path +from fixture_config import ROOT, PROFILES +name = "tiercache-consumer-" + uuid.uuid4().hex[:12] +version = next(x.split("=",1)[1].strip() for x in (ROOT/"gradle.properties").read_text().splitlines() if x.startswith("version=")) +def run(args, **kw): return subprocess.run(args, check=True, timeout=900, **kw) +try: + run(["docker", "run", "-d", "--name", name, "-p", "127.0.0.1::6379", PROFILES["redis62"]]) + port = subprocess.check_output(["docker", "port", name, "6379"], text=True).strip().rsplit(":",1)[1] + deadline=time.monotonic()+30 + while subprocess.run(["docker","exec",name,"redis-cli","PING"],capture_output=True,text=True,timeout=5).stdout.strip()!="PONG": + if time.monotonic()>deadline: raise RuntimeError("Consumer Redis readiness timed out") + time.sleep(.1) + for boot in [PROFILES["boot35"], PROFILES["boot4"]]: + args=[str(ROOT/"gradlew"), "-p", str(ROOT/"compatibility/spring-consumer"), "clean", "test", "--max-workers=2", + "-PbootVersion="+boot, "-PtiercacheVersion="+version, + "-PartifactRepository="+str(ROOT/"build/compatibility-repository"), "-PredisUri=redis://127.0.0.1:"+port] + result=subprocess.run(args, timeout=900) + import shutil + target=ROOT/"build/compatibility-evidence"/("boot-"+boot) + if target.exists(): shutil.rmtree(target) + shutil.copytree(ROOT/"compatibility/spring-consumer/build", target) + (target/"command.json").write_text(json.dumps({"command":args,"exitCode":result.returncode},indent=2)) + (target/"redis-image.json").write_text(subprocess.check_output(["docker","image","inspect",PROFILES["redis62"]],text=True,timeout=30)) + result.check_returncode() +finally: + subprocess.run(["docker", "rm", "-f", "-v", name], timeout=30, check=False) diff --git a/compatibility/run-sentinel.py b/compatibility/run-sentinel.py new file mode 100644 index 0000000..b460e0b --- /dev/null +++ b/compatibility/run-sentinel.py @@ -0,0 +1,144 @@ +#!/usr/bin/env python3 +"""Bounded Sentinel regressions. Only fixture-owned Docker resources are removed.""" +import ipaddress, json, subprocess, time, uuid +from fixture_config import ROOT, PROFILES +IMAGE=PROFILES['redis74'] +def command(args, timeout=30): + p=subprocess.run(args, text=True, capture_output=True, timeout=timeout) + if p.returncode: raise RuntimeError(f'{args}: {p.stderr} {p.stdout}') + return p.stdout.strip() +def until(description, predicate, seconds=90): + end=time.monotonic()+seconds + last=None + while time.monotonic()=offset for n in nodes[1:]) + until('retained history replicated',replicated) + event('history-replicated') + if mode=='planned': + if cli(sentinels[0],'-p','26379','SENTINEL','failover','mymaster')!='OK': + raise AssertionError('Sentinel did not accept planned failover') + elif mode=='abrupt':command(['docker','kill','--signal','KILL',nodes[0]]) + else: + for n in nodes:command(['docker','stop','-t','2',n]) + (control/'offline').write_text('ready') + until('offline source update',lambda:(control/'offline-written').exists()) + for n in nodes:command(['docker','start',n]) + event('fault-injected',mode=mode) + def elected_topology(): + primary=topology() + if primary is not None and (mode=='outage' or primary!=0): + return {'master':primary} + return None + elected=until('agreed elected primary',elected_topology) + # Index zero is a valid master after an outage. + elected=elected['master'] + event('elected',master=elected) + (control/'topology-ready').write_text(str(elected)) + code=command(['docker','wait',probe],timeout=130) + logs=subprocess.run(['docker','logs',probe],capture_output=True,text=True,timeout=10) + (output/'probe.log').write_text(logs.stdout+logs.stderr) + if code!='0' or not (control/'passed').exists():raise AssertionError('Probe failed, exit='+code) + event('passed') + finally: + cleanup_errors=[] + for name in names: + try: + p=subprocess.run(['docker','logs',name],capture_output=True,text=True,timeout=10) + (output/(name+'.log')).write_text(p.stdout+p.stderr) + except Exception as e: cleanup_errors.append(str(e)) + for name in reversed(names): + try: + p=subprocess.run(['docker','rm','-f','-v',name],capture_output=True,text=True,timeout=30) + if p.returncode and 'No such container' not in p.stderr: cleanup_errors.append(p.stderr) + except Exception as e: cleanup_errors.append(str(e)) + try: + p=subprocess.run(['docker','network','rm',token],capture_output=True,text=True,timeout=30) + if p.returncode and 'not found' not in p.stderr: cleanup_errors.append(p.stderr) + except Exception as e: cleanup_errors.append(str(e)) + if cleanup_errors: + (output/'cleanup-errors.json').write_text(json.dumps(cleanup_errors,indent=2)) + raise RuntimeError('Fixture cleanup incomplete: '+str(cleanup_errors)) +if __name__=='__main__': + import sys + for mode in sys.argv[1:] or ['planned','abrupt','outage']:scenario(mode) diff --git a/compatibility/run-server-matrix.py b/compatibility/run-server-matrix.py new file mode 100644 index 0000000..9f73e9c --- /dev/null +++ b/compatibility/run-server-matrix.py @@ -0,0 +1,20 @@ +#!/usr/bin/env python3 +"""Run identical transport contracts, retaining reports for every server profile.""" +import json, shutil, subprocess, sys +from pathlib import Path +from fixture_config import ROOT, PROFILES as ALL_PROFILES +PROFILES={k:ALL_PROFILES[k] for k in ("redis62", "redis74", "redis8", "valkey")} +for profile in sys.argv[1:] or PROFILES: + image=PROFILES[profile] + output=ROOT/"build/compatibility-evidence"/profile;output.mkdir(parents=True,exist_ok=True) + args=[str(ROOT/"gradlew"), ":tiercache-transport-redis:serverContractTest", "--max-workers=2", "-PserverImage="+image] + reports=ROOT/"tiercache-transport-redis/build/test-results/serverContractTest" + if reports.exists(): shutil.rmtree(reports) + if (output/"test-results").exists(): shutil.rmtree(output/"test-results") + with (output/"gradle.log").open("w") as log: + result=subprocess.run(args,cwd=ROOT,stdout=log,stderr=subprocess.STDOUT,timeout=1200) + if reports.exists(): shutil.copytree(reports,output/"test-results",dirs_exist_ok=True) + inspection=subprocess.run(["docker","image","inspect",image],capture_output=True,text=True,timeout=30) + (output/"image.json").write_text(inspection.stdout or inspection.stderr) + (output/"run.json").write_text(json.dumps({"command":args,"exitCode":result.returncode,"image":image},indent=2)) + result.check_returncode() diff --git a/compatibility/spring-consumer/build.gradle.kts b/compatibility/spring-consumer/build.gradle.kts new file mode 100644 index 0000000..b6c1f46 --- /dev/null +++ b/compatibility/spring-consumer/build.gradle.kts @@ -0,0 +1,44 @@ +import java.time.Duration + +plugins { java } +val bootVersion = providers.gradleProperty("bootVersion").get() +val tiercacheVersion = providers.gradleProperty("tiercacheVersion").get() +val artifactRepository = providers.gradleProperty("artifactRepository").get() +repositories { + exclusiveContent { + forRepository { maven { url = uri(artifactRepository) } } + filter { includeGroup("io.github.cramen") } + } + mavenCentral() +} +// Local SNAPSHOT publications are rebuilt between compatibility runs. +configurations.configureEach { resolutionStrategy.cacheChangingModulesFor(0, "seconds") } +java { toolchain { languageVersion = JavaLanguageVersion.of(17) } } +dependencies { + testImplementation(enforcedPlatform("org.springframework.boot:spring-boot-dependencies:$bootVersion")) + testImplementation("io.github.cramen:tiercache-spring-boot-starter:$tiercacheVersion") + testImplementation("io.github.cramen:tiercache-micrometer:$tiercacheVersion") + testImplementation("org.junit.jupiter:junit-jupiter") + testRuntimeOnly("org.junit.platform:junit-platform-launcher") +} +tasks.withType { options.compilerArgs.add("-parameters") } +tasks.test { + useJUnitPlatform() + timeout.set(Duration.ofMinutes(10)) + systemProperty("tiercache.test.redisUri", providers.gradleProperty("redisUri").get()) + systemProperty("tiercache.test.bootVersion", bootVersion) + inputs.property("bootVersion", bootVersion) + outputs.upToDateWhen { false } +} +tasks.register("dependencyEvidence") { + doLast { + val artifacts = configurations.testRuntimeClasspath.get().resolvedConfiguration.resolvedArtifacts + val report = layout.buildDirectory.file("reports/runtime-dependencies.txt").get().asFile + report.parentFile.mkdirs() + report.writeText("Boot BOM: $bootVersion\nTierCache: $tiercacheVersion\nGradle JVM: ${System.getProperty("java.version")}\n" + + artifacts.sortedBy { it.moduleVersion.id.toString() }.joinToString("\n") { + "${it.moduleVersion.id} classifier=${it.classifier ?: "main"} file=${it.file.name}" + } + "\n") + } +} +tasks.test { dependsOn("dependencyEvidence") } diff --git a/compatibility/spring-consumer/settings.gradle.kts b/compatibility/spring-consumer/settings.gradle.kts new file mode 100644 index 0000000..6870e8d --- /dev/null +++ b/compatibility/spring-consumer/settings.gradle.kts @@ -0,0 +1 @@ +rootProject.name = "tiercache-spring-consumer" diff --git a/compatibility/spring-consumer/src/test/java/io/tiercache/compatibility/ConsumerTest.java b/compatibility/spring-consumer/src/test/java/io/tiercache/compatibility/ConsumerTest.java new file mode 100644 index 0000000..d5e81db --- /dev/null +++ b/compatibility/spring-consumer/src/test/java/io/tiercache/compatibility/ConsumerTest.java @@ -0,0 +1,86 @@ +package io.tiercache.compatibility; + +import io.micrometer.core.instrument.MeterRegistry; +import io.micrometer.core.instrument.simple.SimpleMeterRegistry; +import io.tiercache.spring.TierCacheManager; +import org.junit.jupiter.api.Test; +import org.springframework.boot.SpringApplication; +import org.springframework.boot.SpringBootVersion; +import org.springframework.boot.WebApplicationType; +import org.springframework.boot.autoconfigure.EnableAutoConfiguration; +import org.springframework.cache.CacheManager; +import org.springframework.cache.annotation.Cacheable; +import org.springframework.cache.concurrent.ConcurrentMapCacheManager; +import org.springframework.context.ConfigurableApplicationContext; +import org.springframework.context.annotation.*; +import java.util.Map; +import java.util.UUID; +import java.util.concurrent.*; +import java.util.concurrent.atomic.AtomicInteger; +import static org.junit.jupiter.api.Assertions.*; + +class ConsumerTest { + @Configuration(proxyBeanMethods=false) + @EnableAutoConfiguration + static class App { + @Bean MeterRegistry registry() { return new SimpleMeterRegistry(); } + @Bean Service service() { return new Service(); } + } + @Configuration(proxyBeanMethods=false) + static class CustomManager { + @Bean CacheManager customManager() { return new ConcurrentMapCacheManager("sync"); } + } + public static class Service { + final AtomicInteger syncCalls = new AtomicInteger(), nullCalls = new AtomicInteger(), asyncCalls = new AtomicInteger(); + final CompletableFuture result = new CompletableFuture<>(); + final CountDownLatch entered = new CountDownLatch(1); + @Cacheable(cacheNames="sync", sync=true) + public String sync(String key) { syncCalls.incrementAndGet(); return "value:" + key; } + @Cacheable(cacheNames="nullable", sync=true) + public String nullable(String key) { nullCalls.incrementAndGet(); return null; } + @Cacheable(cacheNames="async", sync=true) + public CompletableFuture async(String key) { asyncCalls.incrementAndGet(); entered.countDown(); return result; } + } + ConfigurableApplicationContext start(boolean enabled, Class... extras) { + SpringApplication app = new SpringApplication(App.class); + app.addPrimarySources(java.util.List.of(extras)); + app.setWebApplicationType(WebApplicationType.NONE); + app.setDefaultProperties(Map.of("tiercache.enabled", enabled, + "tiercache.redis-uri", System.getProperty("tiercache.test.redisUri"), + "tiercache.caches.nullable.null-policy", "allow", + "spring.main.banner-mode", "off")); + return app.run(); + } + @Test void shippedStarterActivatesAndExercisesAnnotationsAndMetrics() throws Exception { + assertEquals(System.getProperty("tiercache.test.bootVersion"), SpringBootVersion.getVersion()); + assertEquals(17, Runtime.version().feature()); + System.out.println("CONSUMER_JVM " + System.getProperty("java.runtime.version")); + try (var context = start(true)) { + assertInstanceOf(TierCacheManager.class, context.getBean(CacheManager.class)); + Service service = context.getBean(Service.class); + // CGLIB proxies do not expose target fields; obtain counters via the target bean methods below. + String key = UUID.randomUUID().toString(); + assertEquals("value:"+key, service.sync(key)); + assertEquals("value:"+key, service.sync(key)); + assertNull(service.nullable(key)); assertNull(service.nullable(key)); + var first = service.async(key); var second = service.async(key); + Service target = (Service) ((org.springframework.aop.framework.Advised)service).getTargetSource().getTarget(); + assertTrue(target.entered.await(5, TimeUnit.SECONDS)); + target.result.complete("async-value"); + assertEquals("async-value", first.get(5, TimeUnit.SECONDS)); + assertEquals("async-value", second.get(5, TimeUnit.SECONDS)); + assertEquals("async-value", service.async(key).get(5, TimeUnit.SECONDS)); + assertEquals(1, target.syncCalls.get()); assertEquals(1, target.nullCalls.get()); assertEquals(1, target.asyncCalls.get()); + assertFalse(context.getBean(MeterRegistry.class).find("tiercache.requests").counters().isEmpty()); + } + } + @Test void disabledStarterBacksOff() { + try (var context = start(false)) { assertTrue(context.getBeansOfType(TierCacheManager.class).isEmpty()); } + } + @Test void applicationCacheManagerTakesPrecedence() { + try (var context = start(true, CustomManager.class)) { + assertEquals(1, context.getBeansOfType(CacheManager.class).size()); + assertInstanceOf(ConcurrentMapCacheManager.class, context.getBean(CacheManager.class)); + } + } +} diff --git a/docs/compatibility.md b/docs/compatibility.md new file mode 100644 index 0000000..bfc4c94 --- /dev/null +++ b/docs/compatibility.md @@ -0,0 +1,122 @@ +# Tested platforms + +The library baseline is Java 17. Redis 6.2 remains the minimum server line; newer +server profiles are additional compatibility evidence, not a reason to raise that +floor. These tests do not establish support for every patch, OS or deployment. + +## Reproducible profiles + +Exact versions live in [`compatibility/platforms.properties`](../compatibility/platforms.properties). +The transport contracts select the server through a Gradle task input and retain +JUnit XML, the command, selected image and Docker image identity. All profiles run +the same storage, conditional/versioned writes, expiration, tags, namespace, +journal, Pub/Sub and Streams tests. + +| Profile | Image | Scope | +|---|---|---| +| Redis compatibility floor | `redis:6.2.24-alpine` | Standalone transport contracts | +| Redis 7.4 | `redis:7.4.11-alpine` | Standalone contracts and Sentinel | +| Redis 8 | `redis:8.10.2-alpine` | Standalone transport contracts | +| Valkey | `valkey/valkey:9.1.2-alpine` | Standalone transport contracts | + +**Redis Cluster is unsupported by the stock transport.** Multiple application +instances and Sentinel failover do not imply Redis Cluster slot routing, key +placement or multi-key script compatibility. Sentinel evidence here uses Redis +7.4; it does not establish Valkey Sentinel compatibility. + +## Spring consumers + +Independent builds in `compatibility/spring-consumer` consume locally published +TierCache artifacts. They do not use Gradle project substitution or reuse the +library's compilation graph. Each run imports its own enforced Boot BOM and saves +the resulting runtime graph, including the actual Lettuce version. + +| Consumer BOM | Spring Framework | Resolved Lettuce | Netty handler | Micrometer core | Test JVM | +|---|---|---|---|---|---| +| Boot 3.5.16 | 6.2.19 | 6.6.0.RELEASE | 4.1.135.Final | 1.15.12 | Java 17 | +| Boot 4.1.1 | 7.0.9 | 7.5.2.RELEASE | 4.2.17.Final | 1.17.1 | Java 17 | + +The transport publishes the Netty 4.2.17.Final alignment BOM; the metrics module +uses Micrometer 1.15.12. The Boot 3.5 fixture intentionally enforces its own BOM +and therefore still resolves Netty 4.1.135.Final. Its functional PASS is **not** a +clean dependency scan: that version is below the 4.1.137.Final fix for +[CVE-2026-75595](https://github.com/netty/netty/security/advisories/GHSA-c4c3-7fpv-j4q5). +Applications that enforce that BOM must apply a compatible patched Netty override +or migrate their BOM and scan the resulting application graph. A library's +ordinary platform constraints cannot override an application's enforced platform. + +The library itself compiles against Lettuce 7.7.0.RELEASE. A consumer BOM can select +a different client: the table records what was actually tested, not what the +library's own dependency catalog suggests. The consumer checks cover automatic +starter discovery, custom CacheManager backoff, disabled activation, synchronous +annotations, allowed null caching, metrics and `CompletableFuture` caching with +`sync=true` and a coalesced loader. + +Boot 3.5 is retained as a compatibility baseline; its OSS maintenance has ended. +Compatibility is not a security-maintenance promise. Boot 4.1 is the modern +consumer profile. Check upstream [supported versions](https://github.com/spring-projects/spring-boot/wiki/Supported-Versions) +when choosing a production release. + +## Sentinel scenarios and limits + +The fixture creates a unique Docker network with one primary, two replicas, +three Sentinel processes and a JVM containing two independent Spring contexts. +Both contexts use the actual starter-managed data, lock, journal and Pub/Sub +connections. Running the JVM inside that network makes the advertised addresses +reachable without a test-only address translator or replacement RemoteCache. +Readiness requires agreed primary addresses, a reachable primary and Sentinel +quorum. Retained-history assertions wait for both replicas to catch up first. + +Three bounded scenarios run separately: + +1. Planned promotion through `SENTINEL FAILOVER`. +2. Abrupt primary loss through Docker `SIGKILL`. +3. Complete Redis outage, a source-version change and an eviction that cannot + reach Redis, followed by restart from the retained AOF history. + +The first two deliberately omit one publication while keeping the versioned +update and journal row. The second context must replay that history, observe +`v2` and retain an unaffected L1 entry. Both breakers must finish recovery before +new distributed writes are asserted. Successful Sentinel election alone is not +application recovery; writes made while degraded can remain local. + +The outage scenario records the observed value independently of call success +and checks that the missed eviction did not appear in the restored journal. +Replay cannot reconstruct an event never stored in Redis. There is no promised +one-second freshness bound, zero loader burst, or lossless asynchronous Redis +replication. The prepared replication barrier narrows the first two scenarios +to retained-history recovery; it does not test loss of unreplicated writes. + +Protected cache calls are checked against a five-second scenario bound, workers +and topology waits have finite deadlines, and only fixture-owned resources are +removed in cleanup. Full logs and topology events remain on disk on failure. +These finite regressions do not cover every partition or failover interleaving. + +## Running the checks + +Run from the repository root with Docker, Python 3 and a Java 17 toolchain: + +```bash +# All standalone profiles, or pass redis62 / redis74 / redis8 / valkey. +python3 compatibility/run-server-matrix.py + +# Independent consumer builds against the locally built artifacts. +./gradlew stageCompatibilityArtifacts +python3 compatibility/run-consumers.py + +# Planned promotion, primary kill, full outage; individual names are accepted. +./gradlew :tiercache-spring-boot-starter:sentinelRuntime +docker pull eclipse-temurin:17-jre +python3 compatibility/run-sentinel.py +``` + +Artifacts go to `build/compatibility-evidence/`: per-profile JUnit results and +image identity, consumer dependency graphs, and Sentinel topology events and +container logs. `sentinelRuntime` includes the test module's runtime; it is not +used as a substitute for the separately published-artifact consumer tests. +The fixture JVM image is a Java 17 runtime; its exact runtime version is logged. + +PR CI runs the compact standalone matrix and both Spring consumers on Java 17. +Nightly runs the three Sentinel scenarios with a bounded job timeout. Existing +JDK 21/25 build checks remain separate; this is deliberately not a Cartesian +product of all JDKs, servers and topology failures. diff --git a/docs/configuration.md b/docs/configuration.md index ef10ed8..e0a174a 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -36,7 +36,7 @@ Programmatic configuration mirrors this exactly: `CacheOverride` fields left | `tiercache.metrics.enabled` | `true` | Binds the metrics listener when a Micrometer `MeterRegistry` bean exists. `false` opts out. Note: this is a conditional property of the metrics auto-configuration, not a bound field of `TiercacheProperties`. See [observability](observability.md). | | `tiercache.invalidation.enabled` | `true` | Wires the cross-instance invalidation engine (journal + transport) when the default Redis transport is used. `false` opts out: caches become single-node, nothing is published or subscribed. | | `tiercache.invalidation.profile` | `pubsub` | Invalidation transport profile: `pubsub` (default) or `streams`. See [Invalidation profiles](#invalidation-profiles). | -| `tiercache.invalidation.journal-capacity` | `10000` | Maximum journal entries kept per cache stream. The journal backs replay of missed invalidations after reconnects. | +| `tiercache.invalidation.journal-capacity` | `10000` | Approximate retained-row target per cache stream. Enabled built-in journals require at least `65`; `<=64` fails initialization before connection use. Size for event rate and recovery lag, not merely this protocol floor. | | `tiercache.async-executor-threads` | `max(4, availableProcessors)` | Maximum threads serving `AsyncTierCache` operations (the bounded async executor). When the pool and its bounded queue (10,000) are saturated, submissions fail their `CompletionStage` with `RejectedExecutionException` rather than growing threads without bound. Raise when IO-bound async loaders starve throughput; background SWR/XFetch revalidation runs on its own small bounded pool and never competes for these threads. | ## Per-cache knobs @@ -62,15 +62,14 @@ property name. Properties are shown relative to a level prefix — use | `xfetch-beta` | `xfetchBeta` | `1s` | duration | XFetch tuning factor. Must be positive when XFetch is enabled. Smaller values refresh earlier/more aggressively. | | `degradation-stale-ttl` | `degradationStaleTtl` | `0` (disabled) | duration | Extra L1 retention window served stale while the L2 circuit breaker rejects calls (OPEN, or HALF_OPEN with no probe permit). Must be `>= 0`. See [Degradation stale window](#degradation-stale-window). | -Per-cache programmatic equivalents: `CacheOverride` exposes the same eleven -knobs as nullable builder-style setters (`l1MaxSize(long)`, +Per-cache programmatic equivalents: `CacheOverride` exposes the settings from the table above +through nullable builder-style setters (`l1MaxSize(long)`, `l1ExpireAfterWrite(Duration)`, …). `null` means "inherit the defaults". ## TTL ordering invariant -An L1 entry must never outlive its L2 counterpart, otherwise L1 could serve -data after the L2 entry expired. Enforced at startup for every cache -(`CacheConfigValidator`): +The configured fresh L1 duration must not exceed the configured L2 duration. +Startup validation checks this relationship for every cache (`CacheConfigValidator`): - `l1-expire-after-write <= l2-ttl` - `l1-expire-after-access <= l2-ttl` (when set) @@ -79,7 +78,10 @@ data after the L2 entry expired. Enforced at startup for every cache A violation throws `CacheConfigurationException` naming the cache, both values, and the fix. Jitter never breaks this invariant because it only shortens TTLs (see below), so the effective L1 TTL equals the configured -value in the worst case. +value in the worst case. This compares configured durations, not the remaining +TTL of a particular Redis entry. Warming L1 from L2 starts a new local lifetime; +it can outlast that Redis entry. Access expiry and opt-in stale retention need +separate budgeting. See [end-to-end staleness](sizing-and-ttl.md#l1-ttl-vs-l2-ttl). ## TTL jitter @@ -127,11 +129,14 @@ transport is used; `tiercache.invalidation.enabled=false` opts out. Lowest propagation latency; a disconnected instance misses events and catches up via the journal on recovery. - `streams` — events go through Redis Streams. Select with - `tiercache.invalidation.profile=streams`. + `tiercache.invalidation.profile=streams`. Own pending rows are drained before + new rows; corrupt/missing history requires safe L1 reset before ACK. Its + retention is the same shared journal retention. See [Streams recovery](streams-recovery.md). Both profiles share the journal (`tiercache.invalidation.journal-capacity`, default 10000 entries per cache stream), which records recent invalidations -so a recovering instance replays what it missed instead of flushing L1. +so a recovering instance can replay verified retained history. Unverifiable +history may require a per-cache L1 clear; see [recovery behavior](recovery.md). ## Invalidation modes @@ -144,9 +149,12 @@ Per cache, `invalidation-mode` selects what an invalidation event carries: larger than `payload-cap-bytes` (default 64 KiB, minimum 1024) fall back to plain `invalidate` events automatically. -Note: with the starter, the Redis transport's UPDATE-mode payload cap is -taken from the **defaults level** (`tiercache.defaults.payload-cap-bytes`), -even when the mode is overridden per cache. +Both standard starters honor the **resolved per-cache** mode and payload cap. +An omitted `tiercache.caches..payload-cap-bytes` inherits +`tiercache.defaults.payload-cap-bytes`; an explicit override replaces it. +For example, defaults of 65536 and a `catalog` override of 8192 allow larger +UPDATE payloads in other caches while `catalog` falls back to INVALIDATE above +8192 serialized payload bytes. The cap is not a Java object-size estimate. ## Stale window semantics @@ -164,8 +172,14 @@ An L2 hit is classified by write age: **without** warming L1 (so an in-flight refresh is never overwritten by the stale copy), and one asynchronous revalidation per key per instance is triggered through the same coordinated load path as a miss. Revalidation - failures never reach readers; the stale entry keeps serving until its - window ends. + failures do not replace an already served stale result; the stale entry + keeps serving until its window ends. A foreground hard miss joining that + refresh receives its actual result or failure. If the refresh skips a + busy distributed lock, foreground demand continues through ordinary + bounded coordination/loading instead of treating the skip as a missing + value. This transition preserves the original coordination deadline and + existing loader-retry budget; a genuine loader null still follows the + configured null policy. - **past the window** — treated as a hard miss. `stale-ttl` must be `>= 0` (fail-fast validation). Stale hits and @@ -206,8 +220,13 @@ window (after at least 5 calls). While open, the cache runs **L1-only**: no infrastructure exceptions escape into business code, and cross-instance atomicity (`putIfAbsent`, rebuild coordination) degrades to per-instance — surfaced via the `tiercache.degraded=1` metric and a log line. After 5 -seconds the breaker half-opens and admits up to 3 probe calls; it closes -when all probes succeed and reopens on any probe failure. +seconds the breaker half-opens and admits up to 3 probe calls. With a recovery +handler it remains HALF_OPEN after successful probes until asynchronous +replay or a safe reset completes. Probes return their results without waiting +for replay; additional L2 calls are rejected while recovery is pending. Any +probe failure reopens the breaker. This protection concerns runtime L2 calls +with the breaker enabled. It does not suppress application loader failures, +invalid configuration, startup connection failures or async executor rejection. On recovery, the engine attempts journal replay **before** reporting recovery. Successful replay retains entries it does not invalidate. If the @@ -216,9 +235,15 @@ replay read — the affected cache's L1 is flushed. This can happen even within the journal's capacity window and can cause a source-load burst. The fallback emits a log, `tiercache.invalidation{direction="dropped"}` and the `onJournalOverflow` callback; despite its name, that callback also -reports failed replay verification. A hand-built engine without a journal -instead logs and flushes every registered L1; that path has no journal -overflow metric or callback. +reports failed replay verification. A configured invalidation handler without a journal +instead logs and flushes its registered L1 caches; that path has no journal +overflow metric or callback. A factory without an invalidation handler has no +coherence-recovery hook and cannot claim journal-backed recovery. + +Recovery uses two owned workers per factory, coalesced per cache, with bounded +passes and 1–30 second exponential failure retries. A failed baseline read +still clears L1 but retains the confirmed cursor and leaves recovery pending; +it cannot close the breaker. See [completion, shutdown, metrics and limits](recovery.md). ## Degradation stale window @@ -235,6 +260,23 @@ close the breaker. Fresh accesses with `l1-expire-after-access` configured slide freshness and the stale horizon without ever shortening the store-time retention floor; stale accesses never extend anything. +The authoritative deadlines live in an immutable local copy of the value or +null marker, using the same monotonic clock as Caffeine expiry. They are removed +with that entry, not by a separate ten-minute metadata expiry or fencing-map +size cap. Size eviction can still remove the entire entry. A fresh access uses +`logical = now + access TTL`, `stale until = logical + window`, and +`retention until = max(store-time retention floor, stale until)`. Retention +beyond the stale cutoff does not authorize serving the value after that cutoff. + +For a custom `LocalCache`, preserve each opaque `StoredEntry` unchanged. Combining +a positive degradation window with access expiry additionally requires +`supportsAtomicReplace()` and `replaceIfSame(...)`: identity-checked replacement +of value and TTL without reinserting a removed/expired entry. Caffeine implements +this operation. An unsupported provider fails at cache creation with the cache +name and required capability; other configurations remain compatible. These +local deadlines are never serialized into Redis frames. + + Trade-offs to weigh before enabling: entries live longer in L1 (memory bounded by `window / L1 TTL x working set`, still capped by `l1-max-size`), and the knob changes nothing in normal mode — it only serves staleness @@ -328,3 +370,108 @@ only: Fail-fast validation is core's and applies identically: an invalid combination (for example an L1 TTL above the L2 TTL) aborts application startup with an actionable error. + +## Built-in Redis namespaces (v2) + +Each complete physical cache namespace is encoded as a delimiter-safe +Base64URL token: Spring `users` uses the token for `spring:users`, Micronaut +uses `micronaut:users`, and programmatic wiring uses the configured cache +name. Colons, glob syntax, empty names where supported, and valid Unicode +retain their literal identities. Malformed Unicode fails before mutation. +Data and control keys occupy separate `tiercache:v2:*` families. The logical +journal identity remains `users` in both framework examples. + +There is no legacy/v2 compatibility switch. Version 2.0.0 requires +a coordinated cold cutover; an ordinary mixed-version +rolling upgrade is unsafe. See [exact layouts, clear limits, migration and +rollback](redis-keyspace-v2.md). Clear removes only data in its own namespace, +using a non-transactional scan; its journal append is a separate operation. + +## Resource lifecycle + +Factories own auxiliary workers and derived lock providers, while supplied L2, +clients, connections and explicit providers retain their caller ownership. +Close factories when their owning component stops. See [resource ownership and +shutdown](resource-lifecycle.md) for lazy connection cleanup, surviving synchronous +views, async cancellation and coordination limits during shutdown. + + +## Tags, batches and whole-cache clear + +```java +TierCache products = factory.getCache("products"); +products.put("p1", "first", "category:books", "campaign:summer"); +products.put("p2", "second", "category:books"); +products.evictByTag("category:books"); +products.evictAll(java.util.List.of("p3", "p4")); +products.evictAll(); +``` + +With the built-in transport, an accepted versioned tagged write changes data, +tag membership and its journal record in one Redis operation. A losing write +changes none of them. Tag eviction enumerates matching keys and evicts them +individually; batch eviction also operates per key. Neither operation is a +transaction over the whole set or a fence against concurrent writes. + +Whole-cache `evictAll()` clears its own data namespace using bounded SCAN work, +then separately appends EVICT_ALL to the journal. Concurrent writes/loads and +partial failures can interleave with these steps. Receivers converge through +live delivery or triggered replay of retained events, not instant synchronous +acknowledgement from every instance. During degradation, tag membership stored +only in Redis cannot provide a complete offline local tag index. See +[namespace boundaries](redis-keyspace-v2.md) and [recovery](recovery.md). + +## Custom Spring Redis client and timeouts + +The default client uses a 100 ms connect timeout and a 250 ms command timeout. +There are no `tiercache.*` timeout properties. A primary application-owned +`RedisClient` can customize both without replacing the `RemoteCache` wiring: + +```java +@Configuration(proxyBeanMethods = false) +class CacheClientConfiguration { + @Bean(destroyMethod = "shutdown") + @Primary + RedisClient applicationRedisClient(@Value("${tiercache.redis-uri}") String uri) { + RedisClient client = RedisClient.create(uri); + client.setOptions(ClientOptions.builder() + .socketOptions(SocketOptions.builder() + .connectTimeout(Duration.ofMillis(150)).build()) + .timeoutOptions(TimeoutOptions.enabled(Duration.ofMillis(350))) + .build()); + return client; + } +} +``` + +Imports are `io.lettuce.core.{RedisClient,ClientOptions,SocketOptions,TimeoutOptions}`, +`java.time.Duration`, Spring's `context.annotation.{Configuration,Bean,Primary}` +and `beans.factory.annotation.Value`. Keep `tiercache.enabled=true` and +`tiercache.redis-uri` configured. Choose timeouts below the application's request +budget; the numbers above demonstrate customization, not universal sizing. + +The primary client is used for data, lock, journal and invalidation connections. +The current auto-configuration also creates its extra default client bean: its +condition checks for a custom RemoteCache, not another RedisClient. Both clients +are closed by their owning Spring context. This wiring is exercised by +[CustomRedisClientDocumentationTest](../tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/CustomRedisClientDocumentationTest.java). +It is not a claim that an equivalent custom Micronaut-client example was tested. + +## Serialization and Native Image + +The standard transport uses JDK serialization: keys and values must be +serializable, class evolution must remain compatible, and untrusted serialized +data must not be accepted. Redis access and payload provenance are therefore +part of the application's trust boundary. + +There is no starter property that selects a serializer. Programmatic transport +builders expose key/value serializer extension points, but these transport +classes and SPIs are internal, outside the stable API listed in the README. +Manual wiring must keep L2, journal and UPDATE transport serializers compatible. +Providing only a custom `RemoteCache` skips the starter's automatic invalidation +wiring; it does not magically create a compatible journal or transport. + +For GraalVM Native Image, applications using JDK serialization must register +**their own key/value DTOs** for serialization. The library's String registration +does not cover application classes. The native demo verifies its own payloads, +not every application's serializer or DTO graph. diff --git a/docs/grafana/alerts.yml b/docs/grafana/alerts.yml index 6274887..138ba26 100644 --- a/docs/grafana/alerts.yml +++ b/docs/grafana/alerts.yml @@ -14,16 +14,18 @@ groups: labels: severity: warning annotations: - summary: "Cache misses growing at stable traffic" + summary: "Cache miss rate growing without a comparable increase in total traffic" + runbook_url: "https://github.com/cramen/tier_cache/blob/main/docs/troubleshooting.md#increased-source-load" - # Any dropped invalidation is a staleness incident. + # A fallback clear signals unverifiable history, not a count of lost messages. - alert: TiercacheDroppedInvalidations expr: sum(rate(tiercache_invalidation_total{direction="dropped"}[5m])) > 0 for: 1m labels: severity: critical annotations: - summary: "Invalidation events dropped (journal window overflow)" + summary: "L1 fallback clear: inspect trimmed or unverifiable journal history" + runbook_url: "https://github.com/cramen/tier_cache/blob/main/docs/troubleshooting.md#dropped-invalidations-and-fallback-clears" # Degraded mode longer than 5 minutes. - alert: TiercacheDegraded @@ -33,8 +35,9 @@ groups: severity: critical annotations: summary: "Cache in L1-only degraded mode > 5 minutes" + runbook_url: "https://github.com/cramen/tier_cache/blob/main/docs/troubleshooting.md#degraded-or-pending-recovery" - # Sustained revalidation failures: stale entries are served but never refreshed. + # A sustained positive failure rate can coexist with successful refreshes. - alert: TiercacheRevalidationFailures expr: sum(rate(tiercache_l2_revalidation_failures_total[5m])) > 0 for: 10m @@ -42,3 +45,14 @@ groups: severity: warning annotations: summary: "SWR revalidations failing for > 10 minutes" + runbook_url: "https://github.com/cramen/tier_cache/blob/main/docs/troubleshooting.md#loader-and-refresh-failures" + + # Failed PUBLISH is observable but may still represent an ambiguous command. + - alert: TiercachePublicationFailures + expr: sum by (cache) (rate(tiercache_invalidation_publish_total{outcome="failed"}[5m])) > 0 + for: 1m + labels: + severity: warning + annotations: + summary: "Invalidation publication failures; inspect Redis connectivity and recovery" + runbook_url: "https://github.com/cramen/tier_cache/blob/main/docs/troubleshooting.md#publication-failures" diff --git a/docs/grafana/tiercache-dashboard.json b/docs/grafana/tiercache-dashboard.json index 89aaa35..29cf2d8 100644 --- a/docs/grafana/tiercache-dashboard.json +++ b/docs/grafana/tiercache-dashboard.json @@ -98,6 +98,22 @@ "targets": [ {"expr": "sum by (cache) (rate(tiercache_l2_revalidation_failures_total[$5m]))", "legendFormat": "{{cache}}"} ] + }, + { + "id": 13, + "title": "Publication outcomes", + "type": "timeseries", + "targets": [ + {"expr": "sum by (cache, outcome) (rate(tiercache_invalidation_publish_total[5m]))", "legendFormat": "{{cache}} {{outcome}}"} + ] + }, + { + "id": 14, + "title": "Pub/Sub publication failure fraction", + "type": "timeseries", + "targets": [ + {"expr": "sum by (cache) (rate(tiercache_invalidation_publish_total{outcome=\"failed\"}[5m])) / clamp_min(sum by (cache) (rate(tiercache_invalidation_publish_total{outcome=~\"acknowledged|failed\"}[5m])), 1e-9)", "legendFormat": "{{cache}}"} + ] } ] } diff --git a/docs/migration-from-jetcache.md b/docs/migration-from-jetcache.md index 4cfe072..a931f6a 100644 --- a/docs/migration-from-jetcache.md +++ b/docs/migration-from-jetcache.md @@ -6,13 +6,11 @@ attributes into configuration. Cache method bodies do not change. ## Why migrate -JetCache is effectively frozen: the upstream project shows no active -maintenance (the last releases date to the 2.7.x line), and known gaps — -limited observability, no first-class degradation handling — will not be -addressed. Tiercache covers the same ground (in-process + Redis two-level -cascade behind annotations) and adds the failure-mode protections, -fail-fast config validation, and metrics a two-level cache needs in -production. See [README](../README.md) for the feature list. +Tiercache provides an in-process + Redis two-level cache through standard +Spring Cache annotations, with built-in failure-mode protections, +fail-fast configuration validation, and metrics. Migration can be useful +when these features match your application's requirements. See +[README](../README.md) for the feature list. ## Annotation mapping diff --git a/docs/migration-from-redisson.md b/docs/migration-from-redisson.md index 6c5d2e5..ed60f35 100644 --- a/docs/migration-from-redisson.md +++ b/docs/migration-from-redisson.md @@ -5,7 +5,7 @@ This guide covers moving a Redisson near-cache (`RLocalCachedMap` with distributed-objects toolkit; Tiercache is a cache library. Tiercache replaces the **local-cached map** use case only. If you also use Redisson for locks, queues, topics, or other distributed objects, keep Redisson for those — see -[What Tiercache does not replace](#what-tiercache-does-not-replace). +[What Tiercache does not replace](#honest-gaps). Redisson API names below follow the public Redisson documentation; check the javadoc of your exact Redisson version for details. diff --git a/docs/migration-from-spring-cache.md b/docs/migration-from-spring-cache.md index 09e788c..336b1c1 100644 --- a/docs/migration-from-spring-cache.md +++ b/docs/migration-from-spring-cache.md @@ -84,14 +84,57 @@ Miss coalescing engages only on the value-loader path. With the default `@Cacheable(sync = false)`, Spring performs a get-then-put: a miss reads the cache, invokes your method, and stores the result — concurrent misses each invoke the method, with no coalescing (standard Spring Cache behavior, not -specific to Tiercache). Setting `sync = true` routes misses through -`Cache.get(key, Callable)` instead, which the starter's adapter +specific to Tiercache). For synchronous return values, setting `sync = true` +routes misses through `Cache.get(key, Callable)`, which the starter's adapter (`TierCacheSpringCache`) implements via the core's `getOrCompute` — engaging singleflight (one loader execution per key per instance) plus distributed rebuild coordination (one loader per key across the cluster when the Redis transport is present). Use `sync = true` on `@Cacheable` methods whose loader is expensive or whose keys are hot. +## Asynchronous retrieval + +The manager implements both Spring `Cache.retrieve` overloads, available since +Spring Framework 6.1. Cache I/O and supplier invocation run on the owning +TierCacheFactory's bounded async executor, rather than the retrieval caller thread. +L2 hits still warm L1 and concurrent loader calls share the engine's existing claim. + +| Call | Value hit | Cached null | Miss | +| --- | --- | --- | --- | +| `retrieve(key)` | Future of `ValueWrapper(value)` | Future of a non-null wrapper holding null | Future completing with null | +| `retrieve(key, supplier)` | Future of the value; supplier is not invoked | Future of null; supplier is not invoked | Invoke supplier through coalescing; return its value or null | + +Both methods always return a non-null future. The supplier overload never returns +a Spring wrapper or its internal NullValue sentinel. Loader nulls follow the cache's +allow/deny policy; supplier exceptions and failed/cancelled stages remain failed or +cancellation-type future outcomes with their original cause. + +`@Cacheable(sync = true)` methods returning CompletableFuture use the supplier +overload. Spring's synchronized Mono adaptation is covered too when Reactor is +present. Ordinary `sync = false` annotations retain separate read/invoke/write +behavior and gain no coalescing guarantee. Synchronous values, CachePut and +CacheEvict keep their existing semantics. + +The core currently occupies an async worker while awaiting the supplier's stage. +Slow suppliers can therefore saturate the shared bounded pool. Overflow returns a +failed future with RejectedExecutionException; work never falls back to the caller +or a new executor. Factory close settles queued/running retrieval futures with the +existing cancellation-type outcome and rejects later submissions; completed results +stay unchanged. Cancelling one caller's future does not cancel the shared load or +an application-owned supplier stage. Shutdown cannot undo loader side effects. + +This repairs retrieval, not every Spring reactive operation: cache initialization +and synchronous writes/evictions can still perform blocking work. + +### Direct construction of the internal adapter + +The old `TierCacheSpringCache(name, syncCache)` constructor remains available for +synchronous use. Both retrieve methods now return actionable failed futures with +UnsupportedOperationException when no async view was supplied. Its single-argument +retrieve previously blocked and returned a completed result; that behavior changes. +Use TierCacheManager, or pass the synchronous and asynchronous views of the same +factory cache to the new constructor. No unmanaged/common-pool fallback is created. + ## What changes semantically - **Eventual consistency.** L1 copies on other instances are refreshed by @@ -105,8 +148,8 @@ loader is expensive or whose keys are hot. `null-marker-ttl`; subsequent calls do not invoke the method until the marker expires. Under the default `deny` policy nothing changes versus `ConcurrentMapCache`: nulls are not cached. -- **Loader coalescing.** With singleflight, concurrent callers for the same - missing key share one loader execution. Loader side effects therefore run +- **Loader coalescing.** On the `sync = true` value-loader path, concurrent + callers for the same missing key share one loader execution. Loader side effects therefore run once per rebuild round, not once per caller — which is the point, but matters if your loader performed per-call bookkeeping. - **Cache names not in configuration.** A name absent from diff --git a/docs/observability.md b/docs/observability.md index 1995762..22adc61 100644 --- a/docs/observability.md +++ b/docs/observability.md @@ -35,14 +35,17 @@ All meters are created by |---|---|---|---| | `tiercache.requests` | Counter | `cache`, `result` | Cache lookups by outcome. `result` is one of `l1_hit`, `l2_hit`, `miss`, `load`, `coalesced` (waited on another caller's in-flight load), or `stale_degraded` (retained L1 served while L2 admission is rejected). | | `tiercache.latency` | Timer | `cache`, `level` | Latency of cache operations by level. The level enum defines `l1`/`l2`; core times only L2-touching operations, so `level="l2"` is what you will see in practice (L1 hits are deliberately not timed — zero clock reads on the hot path). | -| `tiercache.invalidation` | Counter | `cache`, `direction` | Invalidation events by direction: `sent`, `received`, `replayed` (from the journal on recovery), `dropped` (a journal-backed L1 flush because replay history could not be verified, for example after trimming or a read failure). The latter records fallback events, not a count of individually lost messages. | +| `tiercache.invalidation` | Counter | `cache`, `direction` | Invalidation events by direction: `sent` (admitted publication attempts, not confirmed delivery), `received`, `replayed` (from the journal on recovery), `dropped` (a journal-backed L1 flush because replay history could not be verified, for example after trimming or a read failure). The latter records fallback events, not a count of individually lost messages. | | `tiercache.degraded` | Gauge | — | `1` while the L2 circuit breaker is open (L1-only mode), else `0`. | -| `tiercache.breaker.state` | Gauge | — | Breaker machine state: `0` = closed, `1` = half-open (recovery probing), `2` = open. During half-open `tiercache.degraded` is already back at `0`. | +| `tiercache.breaker.state` | Gauge | — | Breaker machine state: `0` = closed, `1` = half-open (probing or awaiting coherence recovery), `2` = open. During half-open `tiercache.degraded` is already back at `0`. | +| `tiercache.invalidation.recovery.pending` | Gauge | `cache` | `1` while triggered recovery or its follow-up/retry is pending, `0` when settled. Unregistered on close. Read alongside breaker state and logs; a failed-baseline clear does not reset it to success. | +| `tiercache.invalidation.stream` | Counter | `cache`, `result` | Stream/journal failures: `decode_failed`, `apply_failed`, `ack_failed`, `resync_failed`. Counts failed attempts, not individual lost invalidations. | +| `tiercache.invalidation.publish` | Counter | `cache`, `outcome` | Terminal publication results: `acknowledged`, `failed`, `unconfirmed`, `not_required`. Counts complete asynchronously in batches; acknowledgement means Redis accepted PUBLISH, not that receivers applied it. | | `tiercache.journal.size` | Gauge | `cache` | Invalidation journal entries currently held for the cache. | -| `tiercache.last.load.age` | Gauge | `cache` | Milliseconds since the last load event for the cache, tracked from store events, not per-entry metadata. | +| `tiercache.last.load.age` | Gauge | `cache` | Milliseconds since the last load event for the cache, tracked from `LOAD` outcomes, not every put or per-entry metadata. | | `tiercache.null.entries` | Counter | `cache` | Null-markers stored under the `allow` null policy. | | `tiercache.l2.stale.hits` | Counter | `cache` | L2 entries served stale (past logical TTL, inside the stale window). | -| `tiercache.l2.revalidation.triggers` | Counter | `cache` | Asynchronous revalidations claimed and submitted (stale-while-revalidate and XFetch). | +| `tiercache.l2.revalidation.triggers` | Counter | `cache` | Refresh claims before executor submission (stale-while-revalidate and XFetch); rejected submission also increments failures. | | `tiercache.l2.revalidation.completions` | Counter | `cache` | Revalidations that finished without error. | | `tiercache.l2.revalidation.failures` | Counter | `cache` | Revalidations that failed; the stale entry keeps serving until its window ends. | @@ -64,7 +67,9 @@ and produces OpenTelemetry spans (tracer name `io.tiercache`): - `tiercache.l2.` — one span per L2 operation, with attributes `cache.name`, `db.system=redis`, and `cache.hit` set at span end. -- `tiercache.invalidation.apply` — wraps inbound invalidation processing. +- `tiercache.invalidation.apply` — observes an already-committed inbound invalidation. + Observer dispatch runs outside state monitors; this span does not measure + time spent inside the L1 state commit. There are deliberately **no spans on the L1-hit path** — core never calls the tracing hooks there, so hot reads pay nothing. Wire it alongside the @@ -96,6 +101,39 @@ call `register()`, and `close()` to unregister. There is deliberately no top-N-keys operation: per-key counting would tax the hot path, so key-level inspection is refused by design. +### Ownership with multiple factories + +The fixed name `io.tiercache:type=Inspection` exposes at most one inspection per +JVM; it does not aggregate multiple factories. The first successful registration +owns that name. Another inspection that finds it occupied is a no-op and cannot +unregister the owner when it closes. Repeating register on the owner preserves +ownership; repeating close cannot remove a later factory's registration. + +Register and close are ordered per inspection. An explicit register after close +can establish a new ownership period if the name is free; a skipped factory does +not automatically take over when the name becomes vacant. An absent name during +cleanup is benign. Other unregister failures are logged and ownership is consumed, +so cleanup is not retried against a potential replacement. Such failures may +require operator cleanup of the fixed name. + +This protects registrations managed through TiercacheInspection and pre-existing +foreign MBeans. JMX offers no atomic compare-and-unregister here: arbitrary external +replacement of an owned MBean by management tools is outside this ownership guarantee. +The ObjectName, attributes, operations and hit-ratio denominators are unchanged. + +### Locale-independent metric identifiers + +Enum-derived result, level and direction tags use Locale.ROOT, independently of +the JVM default locale. For example, `l1_hit`, `l2_hit`, `miss` and `received` keep +the same spelling under Turkish locale and after a locale change. Application cache +names retain their original spelling and case. Dashboards and JMX continue to use +the existing lowercase ASCII identifiers. + +Restart affected processes with the fix if they previously emitted locale-specific +spellings. Already-created malformed meters are not renamed or deleted in place; +old time series can remain in the monitoring backend until normal retention removes +them. This does not add new meters or change their meaning. + ## Grafana dashboard `docs/grafana/tiercache-dashboard.json` is a self-contained dashboard @@ -105,30 +143,108 @@ Panels: | Panel | Query (Prometheus) | |---|---| -| Request outcomes | `sum by (result) (rate(tiercache_requests_total[$5m]))` | -| L2 latency (p99) | `histogram_quantile(0.99, sum by (le, cache) (rate(tiercache_latency_seconds_bucket{level="l2"}[$5m])))` | -| Invalidation flow | `sum by (direction) (rate(tiercache_invalidation_total[$5m]))` | +| Request outcomes | `sum by (result) (rate(tiercache_requests_total[5m]))` | +| L2 latency (p99) | `histogram_quantile(0.99, sum by (le, cache) (rate(tiercache_latency_seconds_bucket{level="l2"}[5m])))` | +| Invalidation flow | `sum by (direction) (rate(tiercache_invalidation_total[5m]))` | | Degraded | `max(tiercache_degraded)` | | Breaker state | `max(tiercache_breaker_state)` | | Journal size | `tiercache_journal_size` | | Last load age (ms) | `tiercache_last_load_age` | -| Null entries | `sum by (cache) (rate(tiercache_null_entries_total[$5m]))` | -| L2 stale hits | `sum by (cache) (rate(tiercache_l2_stale_hits_total[$5m]))` | -| Revalidation triggers / completions / failures | `sum by (cache) (rate(tiercache_l2_revalidation_{triggers,completions,failures}_total[$5m]))` | +| Null entries | `sum by (cache) (rate(tiercache_null_entries_total[5m]))` | +| L2 stale hits | `sum by (cache) (rate(tiercache_l2_stale_hits_total[5m]))` | +| Revalidation triggers / completions / failures | Three separate queries: `rate(tiercache_l2_revalidation_triggers_total[5m])`, `rate(tiercache_l2_revalidation_completions_total[5m])`, `rate(tiercache_l2_revalidation_failures_total[5m])` | Every query maps onto a meter in the [catalog above](#metrics-catalog). ## Alerts `docs/grafana/alerts.yml` contains a Prometheus rule group (`tiercache`) -with four alerts: +with five alerts: | Alert | Severity | Fires when | |---|---|---| -| `TiercacheMissGrowth` | warning | The miss rate more than doubled over 30 minutes while total traffic stayed flat (a hot-key or eviction problem, not a traffic spike). | +| `TiercacheMissGrowth` | warning | The 10-minute miss rate exceeds twice its value 30 minutes earlier, while the total request rate is below 1.2 times its earlier value; this is a signal to investigate, not a diagnosis. | | `TiercacheDroppedInvalidations` | critical | A journal-backed L1 flush occurred in the last 5 minutes (`direction="dropped"`). Inspect logs to distinguish trimmed/unavailable history from a replay-read failure; the signal does not by itself prove messages were lost. | | `TiercacheDegraded` | critical | `tiercache_degraded` has been above 0 for 5 minutes: the cache is running L1-only and cross-instance guarantees are degraded. | -| `TiercacheRevalidationFailures` | warning | Stale-while-revalidate revalidations have been failing for over 10 minutes: stale entries are served but never refreshed. | - -Load the file into your Prometheus `rule_files` (or drop it into an -Alertmanager/Grafana-managed rule provisioning directory). +| `TiercacheRevalidationFailures` | warning | Stale-while-revalidate the failure rate stays positive for 10 minutes; successful refreshes may coexist with failures. | +| `TiercachePublicationFailures` | warning | Publication failures persist for one minute. Inspect Redis connectivity and recovery; this is not proof that every failed command was undelivered. | + +Load the file into your Prometheus `rule_files` or a compatible Prometheus-rule +provisioning system. Alertmanager routes notifications; it does not evaluate +these expressions. + +See [triggered recovery](recovery.md) for HALF_OPEN admission, baseline failure, bounded retries and shutdown semantics. + +Streams settlement and gap waits also contribute to the pending gauge. See [Streams diagnostics and PEL inspection](streams-recovery.md#diagnostics-and-operator-checks); Redis backlog is not bounded by the local batch size. + +## Publication outcomes + +`sent` keeps its legacy meaning as an attempted submission. Use +`tiercache.invalidation.publish` to distinguish what happened afterward: + +- `acknowledged`: the actual Redis PUBLISH command completed successfully, even + with zero subscribers. It is not acknowledgement by receiving cache instances. +- `failed`: submission/serialization threw, or the command completed exceptionally + or was cancelled. Ambiguous failure does not prove Redis never executed it. +- `unconfirmed`: a legacy transport returned from void publish without an outcome + API. The library cannot infer Redis acknowledgement from that return. +- `not_required`: Streams uses the durable journal row and issues no separate + PUBLISH. This outcome does not certify an arbitrary caller's journal write. + +Writes do not await publication completion or retry ambiguous messages. A publish +failure cannot replace an already successful data-write result. Oversized UPDATEs +still fall back to INVALIDATE and are not failures by themselves. Metrics and logs +do not repair a receiver that missed the last message: existing recovery triggers, +retention limits and conservative reset rules still apply. + +Completion callbacks only update bounded internal accounting. One owned observer +worker exports batched terminal counts outside transport I/O threads and library +monitors. There is at most one pending token per registered cache, plus the fixed +`__tiercache_unregistered__` metrics bucket for direct, unregistered publications. +Reserve that label when naming caches for metrics. The observer keeps four outcome +accumulators and bounded failure-class summaries, not a list of messages or futures. +The existing SENT callback retains its submission-path timing; terminal counters +arrive asynchronously and need not match SENT in an instantaneous scrape. + +Failure diagnostics contain cache and failure class, never serialized keys/payloads +or exception text, and are limited to one warning per bucket per 30 seconds. +Observer exceptions do not change the recorded publication outcome; a failing custom +metrics backend is not guaranteed complete exported counts and is not blindly retried. + +Close stops admission and allows at most one second for queued observer work, then +stops notification. Already-admitted outcomes still settle internally; late completion +cannot restart workers or notify closed observers. An already-running arbitrary +listener may outlive the drain timeout. Export after backend shutdown is best effort. +A post-close write through a surviving synchronous view creates no publish attempt, +so it increments neither SENT nor a terminal publication counter. + +For Pub/Sub, compare the failed rate to the acknowledged-plus-failed rate. Keep +unconfirmed and not-required series visible separately; they are capability/profile +outcomes, not network errors. The dashboard includes these series and a failure +fraction panel. Investigate failures alongside pending recovery and connection logs. + + +## Coverage and diagnosis boundaries + +Request and activity counters are created as events occur, including for caches +created on demand. The standard starters register `journal.size`, +`last.load.age` and the JMX cache-name list for **configured** cache names; a +runtime-created name need not appear there. Programmatic wiring supplies that +list explicitly. Recovery-pending gauges follow active recovery registrations, +including runtime-created caches, and are removed when those registrations close. +Factory-level degraded/breaker gauges have no cache tag and do not aggregate +multiple independent factories with the same meter identity. + +`last.load.age` is zero before a LOAD outcome has been observed; it is not an +entry freshness gauge. `null.entries` is cumulative stores, not current occupancy. +A revalidation completion does not by itself prove a new value was stored (a +coordinated refresh may skip). Revalidation failures do not cover all foreground +loader failures: inspect the application's error/latency signals as well. +Neither zero publication failures nor Redis acknowledgement proves every +subscriber applied every event. L1 hits have neither a timer nor a tracing span. + +Use the [troubleshooting runbook](troubleshooting.md) with these counters and +logs. The sample dashboard's histogram query requires your registry to publish +histogram buckets; the binder does not enable them itself. The sample miss alert +uses its written rate/traffic expression, not a statistical proof of constant +traffic or a diagnosis of the underlying cause. diff --git a/docs/recovery.md b/docs/recovery.md new file mode 100644 index 0000000..0f636e8 --- /dev/null +++ b/docs/recovery.md @@ -0,0 +1,120 @@ +# Triggered invalidation recovery + +## Requests do not run replay + +A detected reconnect, enough successful HALF_OPEN probes, or journal +tracking pressure schedules recovery. Live invalidations still apply under +the per-cache in-memory state gate; journal reads and observer callbacks run +outside cache-state, L1 and breaker monitors. A successful probe returns its +actual cache result without waiting for the journal backlog. These changes +do not make cache reads strongly consistent. + +Each factory with an invalidation handler owns one scheduler with two daemon +workers. A directly constructed built-in invalidation service lazily owns +its own two workers unless a factory supplies them. Recovery never waits +for another job submitted to the same pool. There is at most one queued +token and one active pass per registered cache, without a per-key/event +backlog. A pass reads at most 16 batches of at most 256 rows, then yields so +another ready cache can progress. Version tracking remains bounded to 512 +unconfirmed versions and 256 recently confirmed versions plus one read batch. +Live events received during a read remain tracked until that read or a later +read actually accounts for them. + +After a failed pass, the cache retains one pending job and retries after +1, 2, 4, 8, 16, then at most 30 seconds. New triggers do not bypass that delay. +A safe reset also delays its follow-up read, so a persistently unreadable +journal cannot cause an immediate repeated-flush loop. Healthy caches are +not periodically polled. A silently missed event can remain stale until a +recovery trigger; having a journal alone does not detect every gap. + +## Completion and fallback + +With a recovery handler, successful probes leave the breaker HALF_OPEN +while coherence recovery is pending. Additional L2 attempts are rejected +and use the existing local/degradation paths. CLOSED and the recovered +notification occur only after that episode's replay or safe reset completes. +An old completion cannot close a newer outage. A failed attempt returns the +breaker to OPEN; another attempt respects both the probe wait and recovery +backoff, never retrying faster than once per second. + +Verified retained history preserves unaffected L1 entries. If required +history is trimmed or cannot be verified, recovery may clear that cache's +L1. It captures the journal-end baseline **before** clearing, validates the +reserved generation, then commits the clear and cursor together. Rows after +the baseline remain eligible for subsequent replay. This is not a source +transaction, and a fleet-wide fallback can create a source-load burst. + +If the baseline read fails, L1 is still cleared conservatively, the previous +confirmed cursor is retained, and recovery stays pending. Such a clear is +not successful recovery and cannot authorize skipping a poison row. The +per-cache reset result carries a safe baseline and generation only when +those were established. An engine explicitly configured without a journal +can clear registered L1 caches, but does not report journal replay/overflow. +A factory with no invalidation handler has no coherence-recovery hook. + +Callbacks run after state commits and outside state monitors. An observer +exception is logged and cannot undo a committed row, cursor or clear. This +does not make arbitrary user callbacks fast or provide durable observer +delivery. There is no fixed global recovery deadline; total duration depends +on backlog, server latency and concurrent writes. Commands, per-pass work, +concurrency, retained state and retry frequency are bounded separately. + +## Signals and shutdown + +`tiercache.invalidation.recovery.pending{cache}` is 1 while triggered work is +unsettled, including failure retries, and 0 when settled. A safe reset may +allow the breaker to close while its follow-up read is still pending, so +use the gauge together with breaker state and logs. `tiercache.degraded` +still describes OPEN only; its value 0 during HALF_OPEN does not mean that +new L2 calls are admitted while recovery is pending. + +Closing the factory cancels queued jobs, invalidates in-flight generations, +unregisters pending gauges and shuts down its workers. Unregistration does +not report successful recovery. The breaker detaches the coherence hook +without inventing a successful probe; previously obtained synchronous caches +can later probe an available caller-owned L2 without restarting closed +coherence services. They have no post-close cross-instance coherence promise. +This does not redesign watchdog/lock-provider shutdown behavior. + +## Internal SPI migration + +Existing handler signatures remain. `configureRecoveryExecutor` and +`recoverAsync` are additive defaults: legacy recovery hooks run on the owned +workers and must throw if they fail; a normal legacy return is treated as +completion. Built-in recovery composes asynchronous cache passes instead of +blocking a worker on queued work. Code that needs completion must observe +the returned stage, not assume that returning from `onL2Recovery` or a +transport reconnect callback means replay has finished. + +Targets with concurrent local clears implement the additive recovery epoch +and conditional-application hooks. The built-in target checks those epochs +inside its existing per-key commit protocol. Legacy target defaults preserve +source compatibility but cannot invent atomic clear ordering for a custom +implementation. No stored value or invalidation wire format changes here. + +## Verification + +Deterministic gates cover parked reads versus delivery, clear, repeated +recovery and close; baseline-before-clear and failed-baseline outcomes; +throwing observers; one-token-per-cache scheduling, fairness and backoff; +and epoch-guarded breaker completion. The real Redis JFR case holds journal +recovery while a virtual-thread HTTP request and registered reconnect +callback must return. It then delays actual Redis reads and verifies no +library-attributed `jdk.VirtualThreadPinned` events on JDK 21. Repeat on a +newer JDK for functional ordering; lack of pinning on newer JVMs alone does +not prove absence of monitor serialization. + +```sh +./gradlew :tiercache-tck:vtStressTest --tests '*RecoveryJfrTest' +./gradlew :tiercache-tck:vtStressTest --tests '*RecoveryJfrTest' -PtiercacheVtJdk=25 +``` + +Recordings are retained in `tiercache-tck/build/reports/recovery-jdk*.jfr`. + +The Streams transport consumes per-cache reset proofs through the compatible +gap handler. A proof must still belong to the current registration/reset +and may cover only IDs at or below its baseline. See [Streams pending and +ACK handling](streams-recovery.md). Its reader contributes a separate pending +source while settlement/resync is outstanding, including without a capable +handler. A failed clear is reported as failure with backoff and never grants +safe-reset authorization. diff --git a/docs/redis-keyspace-v2.md b/docs/redis-keyspace-v2.md new file mode 100644 index 0000000..545eb98 --- /dev/null +++ b/docs/redis-keyspace-v2.md @@ -0,0 +1,100 @@ +# Redis keyspace v2 and coordinated migration + +## Compatibility decision + +This is an operational and wire/keyspace **breaking change**. Although Java +method signatures and value frames stay unchanged, old and new instances +use different cache data, journals, channels and locks. They cannot safely +serve shared cached data during an ordinary rolling upgrade. + +The published [compatibility policy](../README.md#compatibility-and-versioning) +reserves incompatible documented behavior for a **major release**. This change +ships in **2.0.0** and requires the migration below; it is not a compatible +1.x patch or minor update. + +## Exact addresses + +`token(S)` is unpadded Base64URL of the strict UTF-8 bytes of the entire +string `S`. Unpaired UTF-16 surrogates are rejected before mutation. Case, +Unicode normalization and delimiters are preserved, not interpreted. The +empty string has an empty token where the API accepts an empty name. +`byteToken(K)` is unpadded Base64URL of the original bytes, not a hash. + +`D` is the physical data namespace, including a framework prefix. `L` is +the logical cache name used by invalidation. `K` is the application's +serialized key, and `T` is a tag. + +| Purpose | Address | +| --- | --- | +| Data | `tiercache:v2:data::` followed by raw `K` bytes | +| Tag set | `tiercache:v2:tags::` | +| Reverse index | `tiercache:v2:tagkeys::` | +| Journal stream | `tiercache:v2:journal:` | +| Trim counter | `tiercache:v2:journal-trims:` | +| Pub/Sub channel | `tiercache:v2:inv:` | +| Streams consumer group | `tiercache:v2:cg::` | +| Built-in rebuild lock | `tiercache:v2:rebuild:` | + +Tag sets contain complete v2 data keys. Reverse-index sets contain encoded +tag tokens. Data and its reverse index share one absolute server expiry instant, including +on Redis 6.2 where time can advance during a Lua script. Tag-set expiry remains +extend-only, and bounded member reclamation rules are unchanged. +Resulting addresses must fit Redis's 512 MiB key limit; invalid lengths fail +before the mutating command. Value frames, application serializers, message +payloads, stream row fields and version comparison rules are unchanged. + +For a Spring cache named `users`, `D = spring:users` and `L = users`. +Micronaut uses `D = micronaut:users` with the same logical-name rule. +Producers, replay and consumers all use the journal token for `L`. Different +physical data namespaces do not by themselves establish safe coordination +between independently configured applications: all participants must agree +on cache identity and data contracts. + +## What clear guarantees + +Clear scans only the literal `tiercache:v2:data::` prefix followed +by `*`, then unlinks those data keys in batches. Names such as `user`, +`user:roles`, `a?` and `tiercache` cannot expand that scan into another data +namespace or a control-key family. Arbitrary serialized application key +bytes do not change the namespace boundary. + +Whole-cache clear remains **non-transactional**. Concurrent writes can +survive the scan. Versioned clear appends its journal row after the scan; +a failure between deletion and append can leave a partial outcome. Tag +indexes survive a clear and expire or are reclaimed under the existing +TTL/janitor rules. They do not make absent data live. No global snapshot, +source transaction or Redis Cluster support is introduced. + +## Cold cutover + +1. Inventory every cache participant, including application replicas, + background workers, loaders and publishers. Establish source capacity + for cold-cache misses and an optional bounded warm-up. +2. Pause source mutations or remove all traffic from the cached path. + Drain old requests, loaders and publishers, then stop all old-format + instances. An old publisher must not remain active during activation. +3. Start the v2 deployment with empty v2 namespaces. If an earlier v2 + attempt left state, clear the known v2 namespaces while all writers are + stopped. Use isolated storage when ownership cannot be established. +4. Confirm all participants use v2 and their logical cache identities + agree. Resume traffic/mutations; authoritative loading fills the cold + cache. Apply any warm-up gradually within the source's capacity. + +Binaries may be staged ahead of time, but activation requires this drained +cutover. Do not run old and v2 instances against active shared cached data +and call it a compatible rolling update. There are no dual reads/writes, +legacy event subscriptions, automatic copying or implicit migration flags. + +## Legacy cleanup and rollback + +V2 ignores and preserves legacy data/control keys, channels and groups. +Old data expires under its existing TTL. Journals or indexes that remain +need an operator-approved inventory of exact keys with proven ownership, +or an explicitly dedicated retired cache database. The library provides no +legacy-prefix SCAN/UNLINK, automatic cleanup or FLUSHDB migration step. + +Rollback also requires pausing traffic/mutations, draining and stopping all +v2 participants, and preparing clean, explicitly isolated cache storage +before starting old binaries. Never reactivate unverified legacy cached +values merely because they still exist. Resuming either format against +stale leftover state is not a supported rollback. diff --git a/docs/release-evidence.md b/docs/release-evidence.md new file mode 100644 index 0000000..9ebd07f --- /dev/null +++ b/docs/release-evidence.md @@ -0,0 +1,167 @@ +# Release dependency evidence + +Release acceptance checks the dependencies of the actual built publications. A +filesystem scan of Kotlin DSL sources, dependency checksums, and a successful +build are not substitutes for a vulnerability scan. + +## What is covered + +`./gradlew releaseEvidenceInputs` builds the files declared by the Maven +publications and generates an independent resolved-artifact inventory at +`build/release-inputs/inventory.json`. It includes all nine library/TCK modules, +main, sources and documentation JARs, core `unshaded`, `test-fixtures` and +`test-fixtures-sources`, TCK `tests`, and matching POM/Gradle module metadata. +It never collects arbitrary `build/libs/*.jar` leftovers. + +Module CycloneDX SBOMs describe the selected runtime graph. Core additionally +includes its fixture runtime; TCK includes its published compliance-suite runtime. +Embedded Caffeine must be present in both the core inventory and SBOM, and the +shaded class must be present in the main JAR. Demo applications are excluded from +the release aggregate. Root coordinates are `io.github.cramen:tiercache:`. + +Completeness checks compare every module's resolved group/name/version set with +its SBOM, verify classifier coverage, check the aggregate, and compare publication +metadata and artifact hashes. Dependencies selected by conflict resolution are +recorded, not inferred from version catalog declarations. Optional application +integrations and consumer BOM overrides still require scanning the final +application's own graph. + +## Vulnerability acceptance + +The scanner is Trivy 0.74.0, installed from an upstream archive using checked-in +SHA-256 values. Updating the scanner requires reviewing that version and updating +its hashes. The scanner and database versions, timestamps, database hash, exact +SBOM hashes, raw JSON results and a readable summary are retained. + +Each scan first validates a synthetic SBOM containing Log4j 2.14.1 and requires +recognition of CVE-2021-44228. No vulnerable JAR is added to runtime dependencies. +That self-test result is separate from shipped dependency findings. + +Every module SBOM and the aggregate are scanned in SBOM mode. All package results +are requested, and reported package identities are compared with SBOM contents. +Missing/empty documents, omitted dependencies, unparsed scanner output, failed +scanner processes, unavailable or expired databases and unexcepted HIGH/CRITICAL +findings reject acceptance. Lower severities remain visible. Infrastructure +failure is reported as an incomplete scan, never a clean result. + +`scripts/release/exceptions.json` starts empty. An exception requires exactly: + +```json +{ + "id": "CVE-YYYY-NNNN", + "package": "pkg:maven/exact.group/exact-artifact@exact-version", + "owner": "accountable-owner", + "reason": "Reviewed exposure, compensating control and remediation plan", + "expires": "YYYY-MM-DD" +} +``` + +The expiration date must be strictly after the evaluation date in UTC. Wildcard, +empty, duplicate, expired and unmatched entries fail validation. Exceptions apply +only to the exact vulnerability/package/version and require maintainer review; +they are not generated automatically when a scan fails. + +## Trial and final evidence + +The manual `release-candidate` workflow requires an explicit source ref, intended +version and mode. Both modes resolve the ref to the checked-out commit and require +the intended version to equal `gradle.properties`. + +- **trial** permits SNAPSHOT development and creates downloadable candidate + evidence without publishing anything or attaching assets to a release. +- **final** requires a clean checkout, stable non-SNAPSHOT `X.Y.Z` version and + `refs/tags/vX.Y.Z`. It attaches the signed evidence bundle to an **existing** + GitHub release, including an existing draft. It creates/publishes no release + and performs no Maven Central upload. Existing assets are not overwritten. + +The current Redis keyspace v2 change still requires a major release and coordinated +cold cutover; a successful scan does not waive that migration/version policy. + +CI uses a clean build with the selected ref's release settings. Staging creates a +new directory and checks the scan input hashes again, re-evaluates findings and +exceptions, and refuses expired database evidence. `manifest.json` records the +immutable source commit, version, mode and each staged file's coordinate (where +applicable) and SHA-256. A dirty local trial is explicitly labeled as such and is +not evidence of a reproducible build of an unmodified commit. + +The complete bundle, including the manifest, binaries, classifiers, publication +metadata, SBOMs, inventory and scan reports, is archived as +`tiercache-evidence.tar.gz`. Cosign signs that archive; GitHub attests its exact +bytes. Build/scanning jobs have read-only repository permissions, signing has +OIDC/attestation permissions, and only the final release-attachment job can write +release assets. Tool and job deadlines are finite. + +## Running locally + +Docker is not required for dependency evidence generation. The build requires +its configured Java toolchain, Python 3.11+ and network access for scanner/database +installation. The checked-in installer supports Linux x86_64 and macOS arm64. + +```bash +python3 -m unittest discover -s scripts/release -p 'test_*.py' -v +./gradlew releaseEvidenceInputs +python3 scripts/release/install_trivy.py /tmp/tiercache-trivy +python3 scripts/release/evidence.py scan \ + --trivy /tmp/tiercache-trivy/trivy \ + --output build/release-scan \ + --exceptions scripts/release/exceptions.json +python3 scripts/release/evidence.py stage \ + --scan build/release-scan --output dist \ + --ref --version --mode trial +python3 scripts/release/evidence.py verify dist +``` + +Use a fresh output directory for each attempt; the scanner and stager refuse to +reuse an existing one. Failure diagnostics remain available. Local checks cannot +exercise GitHub OIDC signing or prove that a GitHub workflow run succeeded. + +## Verifying downloaded and published bytes + +First verify the downloaded archive's signature and provenance. Specify the +trusted **workflow ref** that dispatched the run; it is distinct from the selected +source ref stored in the manifest. + +```bash +cosign verify-blob \ + --bundle tiercache-evidence.sigstore.json \ + --certificate-identity 'https://github.com/cramen/tier_cache/.github/workflows/release-candidate.yml@refs/heads/main' \ + --certificate-oidc-issuer 'https://token.actions.githubusercontent.com' \ + tiercache-evidence.tar.gz +gh attestation verify tiercache-evidence.tar.gz --repo cramen/tier_cache +mkdir verified-evidence +tar -xzf tiercache-evidence.tar.gz -C verified-evidence +python3 scripts/release/evidence.py verify verified-evidence +``` + +Check `manifest.json` for the intended source commit/version and accepted scan. +Final bundles and signatures are retained as release assets, rather than relying +only on the 90-day workflow artifact. Trial artifacts are not release acceptance. + +If the maintainer separately builds/publishes to Central, download all matching +publication files (JARs/classifiers, POM and `.module`) into a directory, then run: + +```bash +python3 scripts/release/evidence.py compare-published verified-evidence downloaded-central-files +``` + +Missing or differing bytes fail verification. RC provenance does not automatically +cover separately rebuilt Central artifacts, even if version strings match. +Checksums bind bytes, signatures authenticate evidence, provenance describes the +build invocation, and the CVE scan evaluates a particular dependency graph against +a dated database. None alone proves the others or guarantees no future CVEs. + + +## Recorded trial evidence — 2026-09-22 + +[Trial run 35773803950](https://github.com/cramen/tier_cache/actions/runs/35773803950) +on commit `ad364b4893ccc9d435abd6b194665e5d4da6fd22` passed clean build, +SBOM completeness, the scanner self-test, dependency acceptance, staging, +Cosign signing and GitHub attestation. Independent attestation verification and +all 90 manifest-file checks passed, including 49 publication files and ten SBOMs. +The dated scan had no HIGH/CRITICAL findings, four MEDIUM findings and no exceptions. + +This was a `1.5.0-SNAPSHOT` trial. Release attachment was skipped and no Central +publication occurred. It does not certify later commits, final-mode release upload, +or a consumer graph changed by an enforced BOM. See [tested consumers and their +Netty override boundary](compatibility.md#spring-consumers). Generate fresh evidence +for the actual release candidate; the Redis v2 major-release requirement remains. diff --git a/docs/resource-lifecycle.md b/docs/resource-lifecycle.md new file mode 100644 index 0000000..ccb0a4f --- /dev/null +++ b/docs/resource-lifecycle.md @@ -0,0 +1,65 @@ +# Resource ownership and shutdown + +Close the factory when its owning application component stops. Each factory owns +its async, refresh, watchdog and recovery workers. Closing it is idempotent and +stops admission to those auxiliary services before disposing their resources. + +## Who closes what + +| Resource | Owner | +| --- | --- | +| RedisClient supplied by the application | Application | +| Existing connection passed to LettuceLockProvider | Application | +| Dedicated connection opened by LettuceLockProvider(RedisClient) | Provider | +| Provider derived from LettuceRemoteCache by a factory | That factory | +| Provider explicitly passed to builder.lockProvider(...) | Caller | +| Provider created as a Spring or Micronaut bean | Framework bean lifecycle | +| L2 supplied to a factory | Caller or its framework lifecycle | + +Derived lock providers are lazy. Two factories sharing one L2 have independent +provider connections; closing one does not close the shared client or the other +provider. Closing an unused provider opens no connection. A connection attempt +already in progress can finish after close, but its owned connection is disposed +instead of published. Connect, Redis commands and resource cleanup run outside +intrinsic lifecycle monitors. + +## Using a cache after factory close + +An already-obtained synchronous cache can still read L1/L2, write, and invoke its +loader with local singleflight **while the supplied L2 remains usable**. It no +longer starts distributed rebuild coordination, lease renewal, background refresh, +invalidation publication or recovery. This is not continued cluster coherence: +other nodes may retain stale values after a post-close write. In a framework +shutdown the L2 itself may also be disposed; the surviving-view guarantee does +not keep it open. + +Discarded queued refreshes release their claims, allowing later foreground reads +to load. Running application loaders retain the existing interruption limits; +shutdown does not undo their side effects. Existing async views keep their +cancellation/rejection contract, and completed stages keep their results. They +do not switch to successful synchronous execution after close. + +A pending coherence recovery is retired with its breaker epoch. Later real L2 +probes may restore data-path availability without restarting recovery workers or +emitting a coherence-recovered notification. Closing alone is not a successful +probe and does not bypass the wait of an OPEN breaker. + +## Locks during shutdown + +Direct acquisition on a closed built-in provider now fails immediately with an +internal lifecycle exception. The engine treats that outcome as unavailable +coordination and loads locally; it does not wait for a supposed lock holder or +count closure as Redis failure/success. + +If shutdown rejects watchdog scheduling after acquisition, the token is released +once before the fallback loader runs. A live scheduler's unrelated rejection +remains an error. Cleanup cannot replace a successful loader result or its original +exception. Renewal admitted before shutdown may stop, so exclusivity beyond the +surviving lease is not guaranteed. + +Existing handles and compensation tasks retain the commands used for acquisition; +they never reopen a provider. Release remains token-checked and cannot delete a +new owner's lock. Shutdown retires compensation bookkeeping and stops rescheduling. +An already-dispatched acquire can still have executed remotely. Cleanup is best +effort; if connection shutdown prevents it, the orphan expires within its lease. +The compensation cap, retry window and lease settings are unchanged. diff --git a/docs/sizing-and-ttl.md b/docs/sizing-and-ttl.md index 3515402..1d38f17 100644 --- a/docs/sizing-and-ttl.md +++ b/docs/sizing-and-ttl.md @@ -124,13 +124,21 @@ metric, and the `TiercacheDroppedInvalidations` alert rule in [docs/grafana/](grafana/). If you see drops, raise the capacity before reaching for longer L1 TTLs. -Hard floor: the capacity must comfortably exceed the live cursor cadence -(64 events). The replay cursor advances over confirmed-applied contiguous -rows, and each cadence tick needs the cursor's own row to still be in the -stream; with a journal smaller than the cadence, that row is always -trimmed by tick time, prefix integrity is unconfirmable, and the service -takes the flush path BY DESIGN — even with nothing actually lost. A few -hundred entries is the practical minimum; the default is far above it. +The enforced protocol floor is **65 entries**: the fixed live cursor cadence is +64 delivered events, and the inclusive integrity check also needs the previously +confirmed cursor row. Capacity 64 loses that row by the next full tick under exact +retention, so values <=64 are rejected rather than silently increased. Direct +Redis journal construction and both starters share the same validator. The default +remains 10000; a disabled journal does not validate its unused capacity setting. + +Accepting 65 is not a no-flush guarantee or an outage budget. Bursts, time between +scheduling and executing recovery, failed reads and disconnected receivers need +headroom beyond the floor. Size for the invalidation rate times maximum expected +recovery/read lag, with operational margin; a few hundred or more can be needed +even for short delays. Redis MAXLEN trimming is approximate: the stream can retain +more rows than the configured target, but that temporary over-retention is not a +correctness mechanism. This property is neither a byte cap nor an exact row maximum. +Conservative reset still applies whenever history cannot be verified. ## Per-cache overrides vs global defaults diff --git a/docs/streams-recovery.md b/docs/streams-recovery.md new file mode 100644 index 0000000..69e9f4d --- /dev/null +++ b/docs/streams-recovery.md @@ -0,0 +1,122 @@ +# Streams pending and gap recovery + +## One group per receiver + +The Streams profile consumes the same bounded journal used by replay. Its +retention window is the actual journal window; choosing Streams does not +create longer history or exactly-once delivery. Each receiver has its own +v2 consumer group so invalidations are broadcast rather than distributed +between application instances. + +The default transport constructor generates a random identity for a new +volatile L1. Its own group is destroyed best-effort on graceful close. +Crashes or an unavailable server can leave that group behind. The constructor +with an explicit UUID represents a **stable identity** and retains its group +on close. The operator must ensure exactly one live owner of that identity; +concurrent owners are unsupported. A resumed owner can claim an older consumer +only inside its own group. It never claims or destroys another receiver's group. + +Service registration establishes the journal baseline and clears the target +before Streams starts dispatch. Historical pending UPDATE rows covered by +that baseline are acknowledged without installing old payloads into the new +L1. Replacing a Streams target retires the old subscription and establishes +a fresh baseline; callbacks from its old registration cannot apply to the +replacement. An ordinary readable reconnect of the same registered target +can drain pending without an unconditional L1 clear. + +## The processing boundary + +The reader drains its own pending IDs before reading `>`. It then resumes +older consumers inside the same exclusively owned group. Missing pending +payloads are detected before a claim could remove their PEL entries, including +on Redis 6.2 without a deleted-ID field in XAUTOCLAIM replies. + +Each cache retains at most one batch of 50 rows plus one current row/reset +stage. A row has separate decode, application and ACK outcomes. Failure does +not abandon the delivered batch's remainder. Application failures have at +most three attempts, spaced by 1 and 2 seconds, before requesting recovery. +If application succeeded but its ACK failed, only settlement is retried; +an unknown ACK outcome is not proof that Redis failed to acknowledge it. +Cross-process redelivery is still possible and relies on version/idempotency +rules, not an exactly-once claim. + +## No ACK based on an unsafe reset + +Corrupt or missing rows are unconfirmable history. The reader requests the +service's asynchronous gap handler and waits on its own dedicated reader +thread, outside state monitors and outside business/Lettuce I/O threads. +Other caches can continue. It does not repeatedly decode the known poison +while waiting for recovery. + +The service captures the journal-end baseline **before** clearing L1. Only +a completed clear with a still-current baseline/generation authorizes ACK +of rows at or below that baseline. A row appended after it is processed +normally. There is no blind XGROUP SETID to the latest tail. Failed baseline +reads, failed clears and superseded/closed reset completions authorize no +poison ACK. A failed baseline still clears L1 conservatively, retains the old +cursor and remains pending; a failed clear is a failed attempt with backoff, +not a reason to spin immediately on another reset. + +The live reader and journal replay use the same required-field, version, +type and serializer-result validation. Corruption fails a checked range as +a whole, before its decoded prefix can advance recovery. After a successful +reset, a corrupt tail can be the confirmed anchor: checked reads validate +its raw ID atomically, omit its already-accounted payload and decode only +later rows. Missing anchors still report unconfirmable history. + +Failed connection, ACK and resync attempts back off through 1, 2, 4, 8, 16 and +30 seconds; coordinator retries retain their own bounded schedule. A cache +may remain visibly pending when safe recovery is unavailable. A standalone +Streams transport without a capable gap handler does not fabricate a clear +or silently ACK poison. If a missing pending ID is beyond the available +reset baseline, it remains pending until a safe baseline covers it. + +Closing a reader prevents new dispatch/ACK admission, invalidates its reset +wait and unregisters its pending gauge source. A command already admitted +before close can still have executed remotely; close is not a rollback of +an in-flight ACK. Late reset callbacks cannot initiate ACK through a closed +reader generation. + +## Diagnostics and operator checks + +`tiercache.invalidation.stream{cache,result}` counts `decode_failed`, +`apply_failed`, `ack_failed` and `resync_failed`. Decoding failures encountered +by journal replay also use `decode_failed`. Labels contain no row ID, key, +payload or exception text. Cache/row/failure-class diagnostics are rate-limited +to one warning per cache per 30 seconds. Serializer exceptions are replaced +by sanitized typed corruption metadata without retaining their messages/causes. + +`tiercache.invalidation.recovery.pending{cache}` includes reader settlement/resync retries as well as coordinator work. A failed-baseline clear is not +reported as recovery success. A committed safe reset also emits the existing +flush signal. These metrics require the metrics listener/binder to be wired; +without one, diagnostics remain available through logs. + +Redis PEL/group memory is not bounded by the in-process batch limit. Inspect +backlog and retention explicitly, using the exact logical cache token and +receiver UUID from [keyspace v2](redis-keyspace-v2.md). For logical `users`: + +```text +XPENDING tiercache:v2:journal:dXNlcnM tiercache:v2:cg:dXNlcnM: +XINFO GROUPS tiercache:v2:journal:dXNlcnM +XINFO STREAM tiercache:v2:journal:dXNlcnM +``` + +Pending count zero alone does not prove there is no unread history. Before +retiring a group, prove the owning instance is stopped and will not resume, +verify the exact cache/group ownership, and then explicitly destroy only +that retired group. Do not infer ownership by scanning other groups or their +consumer counts. Provision journal retention for actual traffic/outages. + +## Compatibility + +Value frames and row fields are unchanged. The v2 coordinated cold cutover +still applies; no v1 cursor translation or bridge is added. Existing void +transport methods remain. Additive gap/metrics hooks are ignored by legacy +transports. Custom gap handlers must provide a committed reset result and a +side-effect-free current-generation check; merely returning a successful +future is insufficient authorization. + +The `CheckedRange` record shape is unchanged, but an intact non-beginning +range may omit its raw-validated anchor. Custom journals may still return a +valid typed anchor; the service skips it by ID. Consumers must use +`startIntact` and must not assume that `rows[0]` always equals the cursor. diff --git a/docs/tck.md b/docs/tck.md index a0b3a72..1d78b4d 100644 --- a/docs/tck.md +++ b/docs/tck.md @@ -1,8 +1,9 @@ # TierCache TCK (compliance suite) `tiercache-tck` is the public chaos-test suite. It runs the cache against real -Redis/Valkey containers and proves the failure-mode protections the library -promises. The suite classes live in the module's test source set; they ship as +Redis/Valkey containers and checks the shipped Lettuce-based stack under specific failure scenarios. +It is not a generic certification suite for arbitrary custom transports or +serializers; those need their own SPI contract and integration tests. The suite classes live in the module's test source set; they ship as a dedicated jar with the `tests` classifier: ``` @@ -69,12 +70,12 @@ tasks.test { | `MultiInstanceStampedeTest` | With distributed rebuild coordination, the loader runs exactly once cluster-wide on a shared Redis; with coordination disabled, at most once per instance (harness sensitivity). | | `AvalancheTest` | Mass writes with one base TTL get effective L1 TTLs spread over the jitter band, so entries do not expire simultaneously. | | `PenetrationTest` | Repeated requests for nonexistent keys are absorbed by null markers; the loader sees only a tiny fraction of the traffic. | -| `DegradationChaosTest` | Redis paused under read load: business operations continue at L1-only latency, no infrastructure exceptions escape, the degraded signal fires; after recovery L1 survives (no reconnect flush) and several instances recover without a loader spike (reconnect storm). | +| `DegradationChaosTest` | Redis paused under read load: business operations continue at L1-only latency, no infrastructure exceptions escape, the degraded signal fires; verified retained history preserves unaffected L1 entries. The reconnect-storm fixture checks its configured workload, not a universal source-load multiplier. | | `PubSubLossTest` | A disconnected receiver heals missed invalidations via journal replay on reconnect within the journal window; beyond the window it flushes L1 entirely. | | `InvalidationRaceTest` | Concurrent put/evict races across instances converge every L1 to the L2 content (versioned writes, last-write-wins) — no resurrected or stale values after quiescence. | | `TagAndUpdateTest` | Tag and batch invalidation across instances, UPDATE-mode cross-instance warm-up, and oversized-payload fallback on a real server. | -| `MetricsDiagnosabilityTest` | Every chaos scenario above is visible in the published metrics. | -| `SoakTest` (tag `soak`) | Sustained churn against a real L2: post-GC memory growth ≤ 5%, journal size bounded. Not run by the default suite. | +| `MetricsDiagnosabilityTest` | Selected scenario outcomes are asserted through published metrics; this does not cover every failure or replace diagnostic logs. | +| `SoakTest` (tag `soak`) | Sustained churn against a real L2: Independent post-GC heap and process RSS growth ≤ 5%, observed workers, bounded journal and JSON evidence. Not run by the default suite. | ## Running the suite from the repository @@ -90,7 +91,7 @@ All of the following are excluded from `check`; run them explicitly. | Command | What it does | Budget | |---|---|---| -| `./gradlew :tiercache-tck:soakTest` | Churn soak against a real L2 container. Default duration PT10M; override with `-Dtiercache.soak.duration=PT24H` for the full profile. Not run in CI (hosted runners kill long jobs) — run it locally or on your own hardware before releases. | Memory growth ≤ 5%, journal size bounded | +| `./gradlew :tiercache-tck:soakTest` | Churn soak against a real L2 container. Default duration PT10M; override with `-Dtiercache.soak.duration=PT24H` for the full profile. Not run in CI (hosted runners kill long jobs) — run it locally or on your own hardware before releases. | Each memory series ≤ 5%, complete workload, bounded journal | | `./gradlew :tiercache-tck:vtStressTest` | 100k virtual threads over the read path; fails on any `jdk.VirtualThreadPinned` event on library frames. Requires a JDK 21+ toolchain; skipped loudly otherwise. | Zero pinning events | | `./gradlew :tiercache-tck:jmhBenchmark` | Throughput benchmark against a Redis container: `mixedWorkload` (95% hot L1 hits / 5% cold cascade reads, reference profile), `cascadeRead` (pure cascade, worst-case reference), `l1Hit` (attribution control). Results in `tiercache-tck/build/results/jmh-benchmark/results.txt`. | No absolute budget — trend/regression measurement (throughput is environment-dependent) | | `./gradlew :tiercache-tck:propagationBenchmark` | Invalidation propagation latency harness (two Pub/Sub instances, 10k events by default; override with `-Dtiercache.propagation.events`). Results in `tiercache-tck/build/results/propagation/results.txt`. | p99 ≤ 5 ms publish-to-applied (single AZ) | @@ -98,3 +99,77 @@ All of the following are excluded from `check`; run them explicitly. The propagation harness (`io.tiercache.tck.PropagationBenchmark`) is a plain `main` class inside the `tests` jar, so consumers can also run it from the artifact on a classpath assembled as shown above. + +### Real-journal recovery and virtual threads + +`RecoveryJfrTest` supplements the in-memory 100k-thread read gate with a real +Redis journal, a virtual-thread HTTP probe and the registered reconnect +callback. Deterministic gates verify callbacks return before replay; JFR +checks library-attributed monitor pinning on JDK 21. Run it with +`./gradlew :tiercache-tck:vtStressTest --tests '*RecoveryJfrTest'`; use +`-PtiercacheVtJdk=25` for newer-JDK functional coverage. Recordings remain in +`tiercache-tck/build/reports/recovery-jdk*.jfr`. See [recovery](recovery.md). + +### Streams pending and corruption + +`RedisStreamsRecoveryTest` and `ValkeyStreamsRecoveryTest` cover pending batch +remainder, missing payloads on Redis 6.2/newer reply behavior, safe baseline +and clear gates, ACK reply loss, same-group stable resume, other-group +isolation, closed/superseded callbacks and corrupt replay anchors. Shared +decoder tests verify sanitized failures and typed payload compatibility; +metric tests verify fixed labels under non-English JVM locales. + +## Strict soak measurements and evidence + +`./gradlew :tiercache-tck:soakTest` always performs a new run; an earlier Gradle +up-to-date result cannot satisfy this gate. The default remains PT10M with eight +workers and a 30-second sample interval. Use +`-Dtiercache.soak.duration=PT24H` for the full profile. The strict minimum is PT1M30S: +at least four total samples are needed to retain three after warm-up. Shorter or +incomplete diagnostic runs cannot pass the release gate. + +Two independent memory series are measured in bytes: + +- Post-GC heap comes from heap-pool usage in a completed explicit `System.gc()` + notification. Completion must be observed within five seconds. A fixed sleep, + an unrelated young GC, or ignored explicit GC is not evidence of a collected + live set. Remove `-XX:+DisableExplicitGC` and use a collector exposing the required + completion notifications if preflight reports the measurement inconclusive. +- Process RSS is read from Linux `/proc/self/status` (`VmRSS`) or macOS + `/bin/ps -o rss= -p ` with a fixed locale and two-second command timeout. + Both sources report KiB, converted to bytes. Missing, zero, negative, malformed, + timed-out or unsupported measurements fail the strict gate. Virtual/committed + memory is never substituted for RSS or added to live heap. + +Preflight validates measurement availability before starting Redis/workload traffic. +The first `ceil(sample count × 0.20)` samples are discarded. Each remaining series +uses its own fixed first steady-state baseline; its peak must be no more than 5% +higher. Adjacent increments below 5% do not hide cumulative growth above the budget. +The journal checks remain: peak ≤ twice configured capacity (4000 rows here), and +second-half mean ≤ first-half mean + 25% of capacity (500 rows here). Approximate +Redis trimming is unchanged. + +Every submitted worker Future is inspected, including failures captured by +FutureTask. Workers must start, perform successful operations and reach the intended +deadline. Errors, exceptions, early completion, cancellation or failure to terminate +within 60 seconds fail the workload. Cleanup always cancels/interrupts remaining work +and waits at most one additional second; uninterruptible work is reported, never +converted into a successful run. + +The report is `tiercache-tck/build/reports/soak/report.json` by default; override with +`-Dtiercache.soak.report=/absolute/path/report.json`. It is updated after every +sample and finalized for PASS or FAIL, including acquisition/workload failures. +It contains duration, JDK/collector, heap configuration, sources/units, independent +memory/journal series, explicit-GC completion counts, per-worker operation counts, +Future outcomes and fixed-baseline assessments. Keep the JSON alongside Gradle logs, +not just the BUILD SUCCESSFUL line. A ten-minute pass is evidence for that observed +run, not proof against every leak or a substitute for a day-long soak. + + +## Platform compatibility and Sentinel + +The [platform matrix](compatibility.md) describes the pinned standalone server +contracts, isolated Boot 3.5/4.1 consumer builds and real starter-managed Sentinel +failover regressions. Run commands and evidence locations are documented there. +These checks complement the churn/chaos suite; they do not imply Redis Cluster +support, instantaneous failover freshness or recovery of unstored invalidations. diff --git a/docs/troubleshooting.md b/docs/troubleshooting.md new file mode 100644 index 0000000..1898583 --- /dev/null +++ b/docs/troubleshooting.md @@ -0,0 +1,109 @@ +# Troubleshooting + +Start with one cache and one affected instance: record time, observed value/source +version, profile, breaker state, pending recovery and Redis connectivity. Successful +requests are an availability signal, not proof of freshness. Do not log serialized +keys, values, credentials or entire Redis URIs. Avoid FLUSHDB or broad SCAN deletion +on shared Redis; use the [namespace migration procedure](redis-keyspace-v2.md). + +## Degraded or pending recovery + +**Signals:** `tiercache.degraded=1`, `tiercache.breaker.state=2`, or +`tiercache.invalidation.recovery.pending{cache}=1`. HALF_OPEN is state 1 while the +degraded gauge is already zero; additional L2 calls can still be rejected during +recovery. Check Lettuce connection messages and `io.tiercache.TierCacheFactory` +breaker logs together with `io.tiercache.invalidation.InvalidationService` logs. + +**Check:** availability, credentials, TLS/DNS, command latency, Sentinel's agreed +primary and reachable advertised addresses; then journal readability and retention. +Successful election is not proof that all application connections recovered. + +**Act:** restore connectivity and let bounded probes/replay proceed. Repeated +triggers do not bypass retry backoff. A failed baseline retains the confirmed +cursor and pending state even after a conservative local clear. Investigate +persistent failures before changing timeouts or capacity. See [recovery](recovery.md). + +## Dropped invalidations and fallback clears + +**Signal:** increasing `tiercache.invalidation{direction="dropped"}`. This counts +journal-backed fallback-clear events, not individually lost messages. Inspect the +reason in invalidation logs: trimmed cursor/history, unreadable journal, malformed +rows or unconfirmable integrity. Even an apparently short outage can require a clear. + +**Check:** event rate multiplied by outage/recovery lag versus journal capacity. +The protocol minimum 65 is not a production sizing recommendation. For Streams, +inspect your own consumer group's pending rows and [settlement diagnostics](streams-recovery.md#diagnostics-and-operator-checks). +Do not ACK/delete pending data manually just to silence a counter. + +**Act:** repair the reader/connectivity or size retention for observed lag. A +verified baseline precedes a fallback clear; failed baseline acquisition keeps +recovery pending. A no-journal recovery handler logs its fallback but has no +journal-overflow counter, so zero `dropped` does not prove no cache was cleared. + +## Publication failures + +**Signals:** `tiercache.invalidation.publish{outcome="failed"}` and the bounded +publication diagnostics from `io.tiercache.invalidation.PublicationObserver`. +`sent` counts attempts. Redis acknowledgement can occur with zero subscribers; +neither counter proves that receivers applied the event. + +**Check:** serialization failures, connection/command failures and receiver +reconnect/pending state. A timeout can be ambiguous: Redis may have executed the +command. `unconfirmed` indicates a legacy outcome API; `not_required` is normal +for Streams and is not a publication error. + +**Act:** fix connectivity or serializer compatibility, then inspect recovery and +observed value versions. Do not blindly retry ambiguous application writes. A +publication failure cannot roll back a committed data write, and recovery cannot +reconstruct an invalidation that never entered retained history. + +## Loader and refresh failures + +**Signals:** application errors/latency and source load, plus +`tiercache.l2.revalidation.failures` for background refresh. That counter does not +cover every foreground loader exception. A completion counter is not proof of a +new store; a coordinated refresh can skip. Successful and failed refreshes may +coexist under the alert expression. + +**Check:** source errors and timeout budgets, async executor saturation, null +policy, and whether the caller uses `getOrCompute` or Spring `@Cacheable(sync=true)`. +Ordinary `sync=false` annotations invoke the method outside the coordinated loader +path. Under `null-policy=deny`, repeated missing keys can repeatedly hit the source. + +**Act:** fix source failures and apply application-side concurrency/rate limits. +Enable null/stale caching only when its semantics fit the data. Stale windows end; +they do not guarantee indefinite availability or suppress all source traffic. + +## Increased source load + +Compare request outcomes, traffic, L1 capacity/expiry, fallback-clear events and +L2 expiry patterns. `miss` alone does not count every loader invocation; inspect +`load`, application metrics and coalesced outcomes. L2 TTLs are not jittered, so +bulk writes can create an expiry wave. XFetch is optional and needs workload sizing. + +Fleet-wide fallback clears, cold starts and unrecoverable history can all produce +bursts. There is no universal two-times source-load ceiling. Protect the source +with application capacity controls and staggered warm-up where appropriate. See +[TTL sizing](sizing-and-ttl.md) and [stale settings](configuration.md#stale-window-semantics). + +## Lock cleanup delays and missing metrics + +Rebuild-lock release failures are logged at DEBUG by +`io.tiercache.internal.DefaultTierCache`, without a dedicated counter or a breaker +failure. Enable that logger temporarily, correlate timings and lease expiry, and +check transport connectivity. Cleanup preserves the loaded value/original loader +failure; it does not turn a failed release into proof the lock was removed. + +If a cache is absent from JMX or per-cache gauges, check whether its name is listed +in starter configuration. Activity counters and recovery registrations can exist +for dynamically created caches that are absent from the configured-cache list. +JMX's fixed name belongs to the first successful registration and does not +aggregate factories. See [observability scope](observability.md#coverage-and-diagnosis-boundaries). + +## Verification and release decisions + +Use the [platform matrix](compatibility.md) for tested combinations and the +[release evidence procedure](release-evidence.md) for dependency acceptance. +A functional consumer PASS is not a clean application CVE scan; an enforced +application BOM can override the library's patched defaults. A signed trial bundle +is not a final release or proof that separately published Central bytes match. diff --git a/gradle.properties b/gradle.properties index 11fa4ed..97d695b 100644 --- a/gradle.properties +++ b/gradle.properties @@ -1 +1 @@ -version=1.5.0-SNAPSHOT +version=2.0.0 diff --git a/gradle/libs.versions.toml b/gradle/libs.versions.toml index d7663cb..b1294d1 100644 --- a/gradle/libs.versions.toml +++ b/gradle/libs.versions.toml @@ -2,7 +2,8 @@ slf4j = "2.0.16" caffeine = "3.2.0" lettuce = "7.7.0.RELEASE" -micrometer = "1.14.14" +micrometer = "1.15.12" +netty = "4.2.17.Final" otel = "1.49.0" spring-boot = "3.5.16" micronaut-platform = "4.10.18" @@ -12,6 +13,7 @@ testcontainers = "1.21.4" [libraries] slf4j-api = { module = "org.slf4j:slf4j-api", version.ref = "slf4j" } caffeine = { module = "com.github.ben-manes.caffeine:caffeine", version.ref = "caffeine" } +netty-bom = { module = "io.netty:netty-bom", version.ref = "netty" } lettuce-core = { module = "io.lettuce:lettuce-core", version.ref = "lettuce" } micrometer-core = { module = "io.micrometer:micrometer-core", version.ref = "micrometer" } otel-api = { module = "io.opentelemetry:opentelemetry-api", version.ref = "otel" } diff --git a/gradle/release-evidence.gradle b/gradle/release-evidence.gradle new file mode 100644 index 0000000..676fa3e --- /dev/null +++ b/gradle/release-evidence.gradle @@ -0,0 +1,60 @@ +import groovy.json.JsonOutput +import org.gradle.api.publish.maven.MavenPublication +import java.security.MessageDigest + +// Enumerate publication artifacts and resolved configurations, not build/libs globs. +gradle.projectsEvaluated { + def modules = subprojects.findAll { it.plugins.hasPlugin('maven-publish') } + tasks.register('releaseEvidenceInputs') { + group = 'verification' + description = 'Build publication files, SBOMs and an independent resolved-artifact inventory.' + dependsOn rootProject.tasks.named('cyclonedxBom') + modules.each { module -> + dependsOn module.tasks.named('cyclonedxBom') + module.publishing.publications.withType(MavenPublication).each { pub -> + dependsOn pub.artifacts + dependsOn module.tasks.named("generatePomFileFor${pub.name.capitalize()}Publication") + dependsOn module.tasks.named("generateMetadataFileFor${pub.name.capitalize()}Publication") + } + } + outputs.upToDateWhen { false } + doLast { + def sha = { File f -> MessageDigest.getInstance('SHA-256').digest(f.bytes).encodeHex().toString() } + def relative = { File f -> rootProject.relativePath(f) } + def inventory = [schemaVersion: 1, group: rootProject.group.toString(), + name: rootProject.name, version: rootProject.version.toString(), modules: []] + modules.sort { it.name }.each { module -> + def names = ['runtimeClasspath'] + if (module.name == 'tiercache-core') names += 'testFixturesRuntimeClasspath' + if (module.name == 'tiercache-tck') names += 'testRuntimeClasspath' + def configs = names.collectEntries { name -> + def resolved = module.configurations.getByName(name).resolvedConfiguration.resolvedArtifacts + [(name): resolved.collect { a -> + [group: a.moduleVersion.id.group, name: a.moduleVersion.id.name, + version: a.moduleVersion.id.version, classifier: a.classifier ?: '', + extension: a.extension, sha256: sha(a.file)] + }.sort { "${it.group}:${it.name}:${it.version}:${it.classifier}" }] + } + def publications = module.publishing.publications.withType(MavenPublication).collect { pub -> + def files = pub.artifacts.findAll { it.extension == 'jar' }.collect { a -> + if (!a.file.isFile()) throw new GradleException("Missing publication file: ${a.file}") + [path: relative(a.file), classifier: a.classifier ?: '', extension: a.extension, sha256: sha(a.file)] + } + ['pom': "pom-default.xml", 'module': "module.json"].each { extension, filename -> + def file = module.layout.buildDirectory.file("publications/${pub.name}/${filename}").get().asFile + if (!file.isFile()) throw new GradleException("Missing publication metadata: $file") + files += [path: relative(file), classifier: '', extension: extension, sha256: sha(file)] + } + [group: pub.groupId, name: pub.artifactId, version: pub.version, + files: files.sort { "${it.extension}:${it.classifier}" }] + } + inventory.modules += [name: module.name, configurations: configs, publications: publications, + sbom: relative(module.layout.buildDirectory.file("reports/cyclonedx/${module.name}-sbom.json").get().asFile)] + } + inventory.aggregateSbom = "build/reports/cyclonedx/${rootProject.name}-sbom.json" + def output = layout.buildDirectory.file('release-inputs/inventory.json').get().asFile + output.parentFile.mkdirs() + output.text = JsonOutput.prettyPrint(JsonOutput.toJson(inventory)) + '\n' + } + } +} diff --git a/scripts/release/.gitignore b/scripts/release/.gitignore new file mode 100644 index 0000000..c18dd8d --- /dev/null +++ b/scripts/release/.gitignore @@ -0,0 +1 @@ +__pycache__/ diff --git a/scripts/release/evidence.py b/scripts/release/evidence.py new file mode 100644 index 0000000..2b9b6ff --- /dev/null +++ b/scripts/release/evidence.py @@ -0,0 +1,293 @@ +#!/usr/bin/env python3 +"""Fail-closed dependency evidence and exact-byte release manifests (stdlib only).""" +import argparse +import datetime as dt +import contextlib +import hashlib +import json +import os +import re +import shutil +import subprocess +import sys +import urllib.parse +import xml.etree.ElementTree as ET +import zipfile +from pathlib import Path + +MODULES = {'tiercache-'+name for name in ('core','invalidation','transport-redis','spring-boot-starter', + 'micrometer','kotlin','reactor','micronaut','tck')} +GROUP = 'io.github.cramen' +class EvidenceError(RuntimeError): pass + +def require(condition, message): + if not condition: raise EvidenceError(message) +def read(path): + with Path(path).open() as source: return json.load(source) +def write(path, value): + path=Path(path); path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(value, indent=2, sort_keys=True)+'\n') +def sha(path): + with Path(path).open('rb') as source: return hashlib.file_digest(source, 'sha256').hexdigest() +def coordinate(item): return (item['group'], item['name'], item['version']) +def purl(coord): return 'pkg:maven/'+urllib.parse.quote(coord[0],safe='.')+'/'+urllib.parse.quote(coord[1],safe='')+'@'+urllib.parse.quote(coord[2],safe='.-') +def from_purl(value): + require(isinstance(value,str) and value.startswith('pkg:maven/'), 'Missing Maven package identity') + body=value.removeprefix('pkg:maven/').split('?',1)[0].split('#',1)[0] + name, version=body.rsplit('@',1); group,name=name.rsplit('/',1) + return tuple(urllib.parse.unquote(x) for x in (group,name,version)) +def local(root, name): + path=(root/name).resolve() + require(path.is_relative_to(root.resolve()) and path.is_file(), 'Missing or unsafe input: '+str(name)) + return path + +def validate_sbom(document, expected, identity): + require(document.get('bomFormat')=='CycloneDX', 'Not a CycloneDX document') + require(coordinate(document['metadata']['component'])==identity, 'SBOM root identity mismatch') + components=document.get('components') + require(isinstance(components,list) and components, 'Empty relevant SBOM components') + observed={coordinate(c) for c in components if all(k in c for k in ('group','name','version'))} + observed.add(identity) + missing=set(expected)-observed + require(not missing, 'SBOM omits resolved components: '+str(sorted(missing))) + for c in components: + if c.get('group') and c.get('name') and c.get('version'): + require(from_purl(c.get('purl'))==coordinate(c), 'SBOM PURL/coordinate mismatch') + return observed + +def validate_inventory(root, inventory): + require(inventory.get('schemaVersion')==1 and inventory.get('group')==GROUP, 'Invalid inventory identity') + version=inventory['version'] + require(isinstance(version,str) and re.fullmatch(r'[A-Za-z0-9][A-Za-z0-9.\-]*',version), 'Unsafe version') + modules=inventory.get('modules',[]) + require(len(modules)==len(MODULES) and {m['name'] for m in modules}==MODULES, 'Incomplete module inventory') + all_components=set(); sboms={}; expected_packages={} + for module in modules: + name=module['name']; configs=module['configurations'] + expected_configs={'runtimeClasspath'} + if name=='tiercache-core': expected_configs.add('testFixturesRuntimeClasspath') + if name=='tiercache-tck': expected_configs.add('testRuntimeClasspath') + require(set(configs)==expected_configs, 'Missing classifier runtime graph: '+name) + deps={coordinate(a) for artifacts in configs.values() for a in artifacts} + require(deps, 'Empty resolved inventory: '+name) + all_components.update(deps); all_components.add((GROUP,name,version)) + publications=module['publications']; require(len(publications)==1, 'Unexpected publication count') + publication=publications[0] + require(coordinate(publication)==(GROUP,name,version), 'Publication coordinates do not match project') + expected_classifiers={'','sources','javadoc'} + if name=='tiercache-core': expected_classifiers|={'unshaded','test-fixtures','test-fixtures-sources'} + if name=='tiercache-tck': expected_classifiers.add('tests') + files=publication['files'] + require(len(files)==len(expected_classifiers)+2, 'Unexpected publication artifact count: '+name) + require({f['classifier'] for f in files if f['extension']=='jar'}==expected_classifiers, 'Missing/extra classifier: '+name) + require({f['extension'] for f in files}=={'jar','pom','module'}, 'Missing publication metadata') + by_name={} + for file in files: + path=local(root,file['path']); require(sha(path)==file['sha256'], 'Input checksum changed: '+str(path)) + if file['extension']=='jar': by_name[path.name]=file + elif file['extension']=='pom': + pom=ET.parse(path).getroot(); ns='{http://maven.apache.org/POM/4.0.0}' + require(tuple(pom.findtext(ns+k) for k in ('groupId','artifactId','version'))==(GROUP,name,version), 'POM identity mismatch') + metadata=read(local(root,next(f['path'] for f in files if f['extension']=='module'))) + comp=metadata['component'] + require((comp['group'],comp['module'],comp['version'])==(GROUP,name,version), 'Gradle metadata identity mismatch') + for variant in metadata['variants']: + for file in variant.get('files',[]): + require(file['name'] in by_name, 'Metadata references unstaged artifact: '+file['name']) + require(file.get('sha256')==by_name[file['name']]['sha256'], 'Metadata checksum mismatch') + if name=='tiercache-core': + caffeine={c for c in deps if c[:2]==('com.github.ben-manes.caffeine','caffeine')} + require(len(caffeine)==1, 'Missing/ambiguous embedded Caffeine inventory') + require(any(c[:2]==('org.slf4j','slf4j-api') for c in deps), 'Missing SLF4J inventory') + main=next(f for f in files if f['extension']=='jar' and not f['classifier']) + with zipfile.ZipFile(local(root,main['path'])) as jar: + require('io/tiercache/internal/caffeine/cache/Cache.class' in jar.namelist(), 'Shaded Caffeine missing from main jar') + if name=='tiercache-transport-redis': + require(any(c[:2]==('io.lettuce','lettuce-core') for c in deps), 'Missing transport client inventory') + doc=read(local(root,module['sbom'])) + observed=validate_sbom(doc,deps,(GROUP,name,version)) + sboms[name]=module['sbom']; expected_packages[name]={c for c in observed if c[0]!=GROUP} + aggregate=read(local(root,inventory['aggregateSbom'])) + observed=validate_sbom(aggregate,all_components,(GROUP,inventory['name'],version)) + sboms['aggregate']=inventory['aggregateSbom'] + expected_packages['aggregate']={c for c in observed if c[0]!=GROUP} + return sboms,expected_packages + +def validate_scan(document, expected): + require(document.get('SchemaVersion')==2 and document.get('ArtifactType')=='cyclonedx', 'Invalid/unparsed Trivy SBOM report') + results=document.get('Results'); require(isinstance(results,list) and results, 'Empty Trivy results') + packages=set(); findings=[] + for result in results: + require(result.get('Class')=='lang-pkgs' and result.get('Type')=='jar', 'Unexpected Trivy result type') + for package in result.get('Packages',[]): + packages.add(from_purl(package.get('Identifier',{}).get('PURL'))) + for vuln in result.get('Vulnerabilities',[]) or []: + severity=vuln.get('Severity'); require(severity in {'UNKNOWN','LOW','MEDIUM','HIGH','CRITICAL'}, 'Invalid vulnerability severity') + identity=from_purl(vuln.get('PkgIdentifier',{}).get('PURL')) + require(identity in packages, 'Finding refers to an unreported package') + require(vuln.get('VulnerabilityID'), 'Missing vulnerability ID') + findings.append({'id':vuln['VulnerabilityID'],'package':purl(identity),'severity':severity, + 'fixedVersion':vuln.get('FixedVersion','')}) + require(set(expected)<=packages, 'Scanner omitted SBOM packages: '+str(sorted(set(expected)-packages))) + return findings + +def apply_policy(findings, exceptions, today=None): + today=today or dt.datetime.now(dt.timezone.utc).date() + require(isinstance(exceptions,list), 'Exception list must be an array') + keys=set(); used=set(); accepted=[]; blocked=[] + for item in exceptions: + require(set(item)=={'id','package','owner','reason','expires'}, 'Invalid exception fields') + require(all(isinstance(v,str) and v.strip() for v in item.values()), 'Empty exception field') + require(item['package']==purl(from_purl(item['package'])), 'Exception needs an exact canonical Maven package/version') + require(dt.date.fromisoformat(item['expires'])>today, 'Expired exception: '+item['id']) + key=(item['id'],item['package']); require(key not in keys, 'Duplicate exception'); keys.add(key) + for finding in findings: + if finding['severity'] not in {'HIGH','CRITICAL'}: continue + key=(finding['id'],finding['package']) + if key in keys: accepted.append(finding); used.add(key) + else: blocked.append(finding) + require(used==keys, 'Unmatched exceptions: '+str(sorted(keys-used))) + return {'status':'FAIL' if blocked else 'PASS','blockingFindings':blocked,'exceptedFindings':accepted, + 'exceptions':exceptions,'evaluatedOn':str(today)} + +def execute(args, log, timeout=600, env=None, stderr_log=None): + with contextlib.ExitStack() as stack: + output=stack.enter_context(Path(log).open('w')) + errors=stack.enter_context(Path(stderr_log).open('w')) if stderr_log else subprocess.STDOUT + result=subprocess.run(args,stdout=output,stderr=errors,timeout=timeout,env=env) + require(result.returncode==0, 'Scanner/tool failed; see '+str(log)) + +def scan(root, trivy, output, exceptions_path): + output.mkdir(parents=True,exist_ok=False) + require(exceptions_path.resolve()==(root/'scripts/release/exceptions.json').resolve(), 'Use the reviewed checked-in exception list') + inventory=read(root/'build/release-inputs/inventory.json') + sboms,packages=validate_inventory(root,inventory) + write(output/'input-check.json',{'status':'PASS','inventorySha256':sha(root/'build/release-inputs/inventory.json'), + 'exceptionsSha256':sha(exceptions_path), + 'sboms':{name:sha(root/path) for name,path in sboms.items()}}) + cache=root/'build/trivy-cache'; cache.mkdir(parents=True,exist_ok=True) + empty=output/'empty-config.yaml'; empty.write_text('{}\n') + ignore=output/'empty-ignore'; ignore.write_text('') + env={k:v for k,v in os.environ.items() if not k.startswith('TRIVY_')} + base=[str(trivy),'--cache-dir',str(cache),'--config',str(empty)] + execute(base+['--version','--format','json'],output/'scanner.json',env=env,timeout=30,stderr_log=output/'scanner.log') + scanner=read(output/'scanner.json') + require(scanner.get('Version')=='0.74.0', 'Unexpected scanner version') + write(output/'scanner-identity.json',{'version':scanner['Version'],'binarySha256':sha(trivy), + 'startedAt':dt.datetime.now(dt.timezone.utc).isoformat(),'databaseRepository':'ghcr.io/aquasecurity/trivy-db:2'}) + execute(base+['image','--download-db-only','--db-repository','ghcr.io/aquasecurity/trivy-db:2','--timeout','5m','--no-progress'],output/'database.log',env=env,timeout=330) + db=read(cache/'db/metadata.json') + require(db.get('Version') and db.get('UpdatedAt') and db.get('NextUpdate'), 'Missing database identity') + require(dt.datetime.fromisoformat(db['NextUpdate'].replace('Z','+00:00'))>dt.datetime.now(dt.timezone.utc), 'Expired vulnerability database') + write(output/'database.json',dict(db,sha256=sha(cache/'db/trivy.db'))) + fixture=root/'scripts/release/fixtures/vulnerable.cdx.json' + fixture_report=output/'scanner-self-test.json' + execute(base+['sbom','--skip-db-update','--skip-version-check','--scanners','vuln','--list-all-pkgs', + '--ignorefile',str(ignore),'--ignore-unfixed=false','--format','json','--output',str(fixture_report), + '--timeout','5m',str(fixture)],output/'scanner-self-test.log',env=env) + known=validate_scan(read(fixture_report),{('org.apache.logging.log4j','log4j-core','2.14.1')}) + require(any(f['id']=='CVE-2021-44228' and f['severity']=='CRITICAL' for f in known), 'Scanner failed known-vulnerable fixture') + findings=[] + for name,path in sboms.items(): + report=output/(name+'.json') + execute(base+['sbom','--skip-db-update','--skip-version-check','--scanners','vuln','--list-all-pkgs', + '--ignorefile',str(ignore),'--ignore-unfixed=false','--severity','UNKNOWN,LOW,MEDIUM,HIGH,CRITICAL', + '--format','json','--output',str(report),'--timeout','5m',str(root/path)],output/(name+'.log'),env=env) + findings.extend(validate_scan(read(report),packages[name])) + findings=list({(f['id'],f['package'],f['severity']):f for f in findings}.values()) + policy=apply_policy(findings,read(exceptions_path)) + write(output/'acceptance.json',dict(policy,findings=findings,scannerVersion=scanner['Version'],database=db)) + (output/'summary.txt').write_text(f"Status: {policy['status']}\nScanned SBOMs: {len(sboms)}\nFindings: {len(findings)}\nBlocking: {len(policy['blockingFindings'])}\n"+ + '\n'.join(f"{f['severity']} {f['id']} {f['package']} fixed={f['fixedVersion']}" for f in findings)+'\n') + require(policy['status']=='PASS', 'Unexcepted HIGH/CRITICAL vulnerabilities; see acceptance.json') + +def source_identity(root, ref, intended, mode): + def git(*args): return subprocess.check_output(['git','-C',str(root),*args],text=True).strip() + commit=git('rev-parse','HEAD'); require(git('rev-parse','--verify',ref+'^{commit}')==commit, 'Ref does not identify checkout HEAD') + version=next(line.split('=',1)[1].strip() for line in (root/'gradle.properties').read_text().splitlines() if line.startswith('version=')) + require(intended==version, 'Intended version differs from checked-out gradle.properties') + dirty=bool(git('status','--porcelain','--untracked-files=normal')) + if mode=='final': + require(not dirty, 'Final evidence needs a clean checkout') + require(re.fullmatch(r'\d+\.\d+\.\d+',version), 'Final evidence requires a stable non-SNAPSHOT version') + require(ref=='refs/tags/v'+version, 'Final ref must match refs/tags/v') + return {'commit':commit,'ref':ref,'version':version,'mode':mode,'dirty':dirty} + +def verify_manifest(folder): + manifest=read(folder/'manifest.json') + require(manifest.get('schemaVersion')==1 and manifest.get('files'), 'Empty/invalid manifest') + paths=[entry['path'] for entry in manifest['files']] + require(len(paths)==len(set(paths)), 'Duplicate manifest path') + require(set(paths)=={str(p.relative_to(folder)) for p in folder.rglob('*') if p.is_file() and p!=folder/'manifest.json'}, 'Manifest does not cover exact bundle file set') + for entry in manifest['files']: require(sha(local(folder,entry['path']))==entry['sha256'], 'Bundle checksum mismatch: '+entry['path']) + return manifest + +def stage(root, scan_dir, output, ref, version, mode): + identity=source_identity(root,ref,version,mode) + inventory=read(root/'build/release-inputs/inventory.json'); sboms,packages=validate_inventory(root,inventory) + require(inventory['version']==version, 'Inventory version mismatch') + checks=read(scan_dir/'input-check.json') + require(checks['inventorySha256']==sha(root/'build/release-inputs/inventory.json'), 'Scan is for a different inventory') + require(checks['sboms']=={name:sha(root/path) for name,path in sboms.items()}, 'Scan is for different SBOM bytes') + acceptance=read(scan_dir/'acceptance.json') + require(acceptance['status']=='PASS','Rejected scan cannot become accepted evidence') + require(checks['exceptionsSha256']==sha(root/'scripts/release/exceptions.json'), 'Exception list changed after scan') + require(acceptance['exceptions']==read(root/'scripts/release/exceptions.json'), 'Unreviewed exceptions in scan report') + findings=[] + for name in sboms: findings.extend(validate_scan(read(scan_dir/(name+'.json')),packages[name])) + findings=list({(f['id'],f['package'],f['severity']):f for f in findings}.values()) + require(apply_policy(findings,acceptance['exceptions'])['status']=='PASS', 'Scan policy no longer passes') + require(not (scan_dir/'failure.json').exists(), 'Failed scan cannot become accepted evidence') + require(read(scan_dir/'scanner.json').get('Version')=='0.74.0', 'Missing scanner identity') + database=read(scan_dir/'database.json') + require(database.get('sha256') and database.get('UpdatedAt'), 'Missing database evidence') + require(dt.datetime.fromisoformat(database['NextUpdate'].replace('Z','+00:00'))>dt.datetime.now(dt.timezone.utc), 'Database expired before staging') + output.mkdir(parents=True,exist_ok=False); entries=[] + def copy(source, relative, coord=None): + dest=output/relative; dest.parent.mkdir(parents=True,exist_ok=True); shutil.copyfile(source,dest) + entries.append({'path':str(relative),'sha256':sha(dest),'coordinate':coord}) + for module in inventory['modules']: + for pub in module['publications']: + for item in pub['files']: + suffix='-'+item['classifier'] if item['classifier'] else '' + filename=f"{pub['name']}-{version}{suffix}.{item['extension']}" + coord=dict(group=pub['group'],name=pub['name'],version=version,classifier=item['classifier'],extension=item['extension']) + copy(local(root,item['path']),Path('artifacts')/filename,coord) + for name,path in sboms.items(): copy(root/path,Path('sbom')/(name+'.json')) + copy(root/'build/release-inputs/inventory.json',Path('inventory.json')) + for path in sorted(scan_dir.iterdir()): + if path.is_file() and path.suffix in {'.json','.txt','.log'}: copy(path,Path('scan')/path.name) + write(output/'manifest.json',dict(schemaVersion=1,source=identity,files=entries)) + verify_manifest(output) + +def main(): + parser=argparse.ArgumentParser(); sub=parser.add_subparsers(dest='command',required=True) + p=sub.add_parser('scan'); p.add_argument('--root',type=Path,default=Path.cwd()); p.add_argument('--trivy',type=Path,required=True) + p.add_argument('--output',type=Path,required=True); p.add_argument('--exceptions',type=Path,required=True) + p=sub.add_parser('stage'); p.add_argument('--root',type=Path,default=Path.cwd()); p.add_argument('--scan',type=Path,required=True) + p.add_argument('--output',type=Path,required=True); p.add_argument('--ref',required=True); p.add_argument('--version',required=True) + p.add_argument('--mode',choices=['trial','final'],default='trial') + p=sub.add_parser('verify'); p.add_argument('folder',type=Path) + p=sub.add_parser('identity'); p.add_argument('--ref',required=True); p.add_argument('--version',required=True) + p.add_argument('--mode',choices=['trial','final'],required=True) + p=sub.add_parser('compare-published'); p.add_argument('folder',type=Path); p.add_argument('published',type=Path) + args=parser.parse_args() + try: + if args.command=='scan': scan(args.root.resolve(),args.trivy.resolve(),args.output.resolve(),args.exceptions) + elif args.command=='stage': stage(args.root.resolve(),args.scan.resolve(),args.output.resolve(),args.ref,args.version,args.mode) + elif args.command=='identity': print(json.dumps(source_identity(Path.cwd(),args.ref,args.version,args.mode))) + elif args.command=='compare-published': + manifest=verify_manifest(args.folder.resolve()) + for entry in manifest['files']: + if entry.get('coordinate'): + artifact=local(args.published,Path(entry['path']).name) + require(sha(artifact)==entry['sha256'], 'Published artifact differs: '+artifact.name) + print('All published artifact bytes match the candidate manifest') + else: verify_manifest(args.folder.resolve()) + except (EvidenceError,ValueError,KeyError,OSError,subprocess.SubprocessError) as error: + if args.command=='scan': + args.output.mkdir(parents=True,exist_ok=True) + write(args.output/'failure.json',{'status':'FAIL','error':str(error)}) + print(str(error),file=sys.stderr); return 1 + return 0 +if __name__=='__main__': sys.exit(main()) diff --git a/scripts/release/exceptions.json b/scripts/release/exceptions.json new file mode 100644 index 0000000..fe51488 --- /dev/null +++ b/scripts/release/exceptions.json @@ -0,0 +1 @@ +[] diff --git a/scripts/release/fixtures/vulnerable.cdx.json b/scripts/release/fixtures/vulnerable.cdx.json new file mode 100644 index 0000000..bed0f7e --- /dev/null +++ b/scripts/release/fixtures/vulnerable.cdx.json @@ -0,0 +1,7 @@ +{ + "bomFormat": "CycloneDX", + "specVersion": "1.6", + "version": 1, + "metadata": {"component": {"type": "application", "name": "scanner-self-test", "version": "1"}}, + "components": [{"type": "library", "group": "org.apache.logging.log4j", "name": "log4j-core", "version": "2.14.1", "purl": "pkg:maven/org.apache.logging.log4j/log4j-core@2.14.1", "bom-ref": "pkg:maven/org.apache.logging.log4j/log4j-core@2.14.1"}] +} diff --git a/scripts/release/install_trivy.py b/scripts/release/install_trivy.py new file mode 100644 index 0000000..800c2b4 --- /dev/null +++ b/scripts/release/install_trivy.py @@ -0,0 +1,21 @@ +#!/usr/bin/env python3 +"""Install the pinned scanner, verifying a checked-in release archive checksum.""" +import hashlib, io, platform, sys, tarfile, urllib.request +from pathlib import Path +VERSION = '0.74.0' +ARCHIVES = { + ('Linux', 'x86_64'): ('Linux-64bit', '2ae6fe3ee734b7fdf11335663e18c75ea12dccc76062f09f164a3b0f8be4371a'), + ('Darwin', 'arm64'): ('macOS-ARM64', '1caada5e0e2091909357c7525d3aa76f4b660b13821bc143b190c7483e31cc11'), +} +def main(): + target = Path(sys.argv[1]).resolve(); target.mkdir(parents=True, exist_ok=True) + suffix, expected = ARCHIVES[(platform.system(), platform.machine())] + url = f'https://github.com/aquasecurity/trivy/releases/download/v{VERSION}/trivy_{VERSION}_{suffix}.tar.gz' + data = urllib.request.urlopen(url, timeout=120).read() + if hashlib.sha256(data).hexdigest() != expected: raise RuntimeError('Trivy archive checksum mismatch') + with tarfile.open(fileobj=io.BytesIO(data), mode='r:gz') as archive: + executable = archive.extractfile('trivy') + if executable is None: raise RuntimeError('Missing scanner executable') + binary = target/'trivy'; binary.write_bytes(executable.read()); binary.chmod(0o755) + print(binary) +if __name__ == '__main__': main() diff --git a/scripts/release/test_evidence.py b/scripts/release/test_evidence.py new file mode 100644 index 0000000..088ed0c --- /dev/null +++ b/scripts/release/test_evidence.py @@ -0,0 +1,167 @@ +import copy +import datetime as dt +import json +import subprocess +import sys +import tempfile +import unittest +import zipfile +from pathlib import Path +from unittest.mock import patch +import evidence as e + +CAFFEINE=('com.github.ben-manes.caffeine','caffeine','3.2.0') +SLF4J=('org.slf4j','slf4j-api','2.0.16') +LETTUCE=('io.lettuce','lettuce-core','7.7.0.RELEASE') + +def component(coord): + return dict(zip(('group','name','version'),coord),purl=e.purl(coord)) +def sbom(identity, components): + return {'bomFormat':'CycloneDX','metadata':{'component':component(identity)},'components':[component(c) for c in components]} +def report(packages, findings=()): + return {'SchemaVersion':2,'ArtifactType':'cyclonedx','Results':[{'Class':'lang-pkgs','Type':'jar', + 'Packages':[{'Identifier':{'PURL':e.purl(p)}} for p in packages], + 'Vulnerabilities':list(findings)}]} +def finding(severity='HIGH'): + return {'VulnerabilityID':'CVE-2099-0001','PkgIdentifier':{'PURL':e.purl(CAFFEINE)},'Severity':severity,'FixedVersion':'9.9'} + +class EvidenceTests(unittest.TestCase): + def setUp(self): + self.temp=tempfile.TemporaryDirectory(); self.root=Path(self.temp.name) + self.addCleanup(self.temp.cleanup) + def inventory(self): + inv={'schemaVersion':1,'group':e.GROUP,'name':'tiercache','version':'2.0.0','modules':[], 'aggregateSbom':'aggregate.json'} + all_deps={CAFFEINE,SLF4J,LETTUCE} + for name in sorted(e.MODULES): + deps={SLF4J} + if name=='tiercache-core':deps.add(CAFFEINE) + if name=='tiercache-transport-redis':deps.add(LETTUCE) + configs={'runtimeClasspath':[component(c) for c in deps]} + if name=='tiercache-core':configs['testFixturesRuntimeClasspath']=[component(SLF4J)] + if name=='tiercache-tck':configs['testRuntimeClasspath']=[component(SLF4J)] + classifiers={'','sources','javadoc'} + if name=='tiercache-core':classifiers|={'unshaded','test-fixtures','test-fixtures-sources'} + if name=='tiercache-tck':classifiers.add('tests') + files=[]; module_files=[] + for classifier in sorted(classifiers): + filename=name+('-'+classifier if classifier else '')+'.jar'; path=self.root/filename + with zipfile.ZipFile(path,'w') as jar: jar.writestr('io/tiercache/internal/caffeine/cache/Cache.class',b'test') + files.append(dict(path=filename,classifier=classifier,extension='jar',sha256=e.sha(path))) + module_files.append(dict(name=filename,sha256=e.sha(path))) + pom=self.root/(name+'.pom');pom.write_text(f'{e.GROUP}{name}2.0.0') + metadata=self.root/(name+'.module');e.write(metadata,{'component':{'group':e.GROUP,'module':name,'version':'2.0.0'},'variants':[{'files':module_files}]}) + for path,ext in [(pom,'pom'),(metadata,'module')]:files.append(dict(path=path.name,classifier='',extension=ext,sha256=e.sha(path))) + e.write(self.root/(name+'.json'),sbom((e.GROUP,name,'2.0.0'),deps)) + inv['modules'].append(dict(name=name,configurations=configs,sbom=name+'.json', + publications=[dict(group=e.GROUP,name=name,version='2.0.0',files=files)])) + e.write(self.root/'aggregate.json',sbom((e.GROUP,'tiercache','2.0.0'),all_deps|{(e.GROUP,n,'2.0.0') for n in e.MODULES})) + return inv + def test_complete_realistic_publication_inventory(self): + docs,packages=e.validate_inventory(self.root,self.inventory());self.assertEqual(10,len(docs));self.assertIn(CAFFEINE,packages['tiercache-core']) + def test_missing_fixture_classifier_is_rejected(self): + inv=self.inventory();core=next(m for m in inv['modules'] if m['name']=='tiercache-core') + core['publications'][0]['files']=[f for f in core['publications'][0]['files'] if f['classifier']!='test-fixtures'] + with self.assertRaises(e.EvidenceError):e.validate_inventory(self.root,inv) + def test_missing_classifier_runtime_graph_is_rejected(self): + inv=self.inventory();next(m for m in inv['modules'] if m['name']=='tiercache-tck')['configurations'].pop('testRuntimeClasspath') + with self.assertRaises(e.EvidenceError):e.validate_inventory(self.root,inv) + def test_missing_shaded_dependency_in_sbom(self): + inv=self.inventory();path=self.root/'tiercache-core.json';doc=e.read(path);doc['components']=[component(SLF4J)];e.write(path,doc) + with self.assertRaisesRegex(e.EvidenceError,'omits resolved'):e.validate_inventory(self.root,inv) + def test_missing_shaded_dependency_in_inventory_also_fails(self): + inv=self.inventory();core=next(m for m in inv['modules'] if m['name']=='tiercache-core') + core['configurations']['runtimeClasspath']=[component(SLF4J)] + with self.assertRaisesRegex(e.EvidenceError,'Caffeine'):e.validate_inventory(self.root,inv) + def test_missing_module_fails(self): + inv=self.inventory();inv['modules'].pop() + with self.assertRaises(e.EvidenceError):e.validate_inventory(self.root,inv) + def test_empty_component_list_fails(self): + with self.assertRaises(e.EvidenceError):e.validate_sbom(sbom(('x','y','1'),[]),{SLF4J},('x','y','1')) + def test_scanner_coverage_cannot_be_replaced_by_clean_status(self): + with self.assertRaises(e.EvidenceError):e.validate_scan(report([SLF4J]),{SLF4J,CAFFEINE}) + def test_unparsed_report_fails(self): + with self.assertRaises(e.EvidenceError):e.validate_scan({'SchemaVersion':2,'Results':[]},{SLF4J}) + def test_clean_scan_passes(self): + findings=e.validate_scan(report([CAFFEINE]),{CAFFEINE});self.assertEqual('PASS',e.apply_policy(findings,[])['status']) + def test_known_vulnerable_synthetic_fixture_blocks(self): + findings=e.validate_scan(report([CAFFEINE],[finding()]),{CAFFEINE});self.assertEqual('FAIL',e.apply_policy(findings,[])['status']) + def exception(self, expires='2100-01-01'): + return dict(id='CVE-2099-0001',package=e.purl(CAFFEINE),owner='security-owner',reason='Synthetic scoped test only',expires=expires) + def test_scoped_exception_allows_only_exact_finding(self): + findings=e.validate_scan(report([CAFFEINE],[finding()]),{CAFFEINE}) + self.assertEqual('PASS',e.apply_policy(findings,[self.exception()])['status']) + wrong=self.exception();wrong['package']=e.purl(('other','artifact','1')) + with self.assertRaises(e.EvidenceError):e.apply_policy(findings,[wrong]) + def test_expired_and_unmatched_exceptions_fail(self): + for exceptions in [[self.exception('2000-01-01')],[self.exception()]]: + with self.assertRaises(e.EvidenceError):e.apply_policy([],exceptions) + def test_exception_expiring_today_fails(self): + with self.assertRaises(e.EvidenceError):e.apply_policy([],[self.exception('2026-09-22')],dt.date(2026,9,22)) + def test_json_stdout_stays_parseable_when_scanner_logs_to_stderr(self): + e.execute([sys.executable,'-c', 'import sys,json; print("diagnostic",file=sys.stderr); print(json.dumps(dict(Version="0.74.0")))'], + self.root/'version.json',stderr_log=self.root/'version.log') + self.assertEqual('0.74.0',e.read(self.root/'version.json')['Version']) + self.assertIn('diagnostic',(self.root/'version.log').read_text()) + def test_scanner_process_failure_fails(self): + with self.assertRaises(e.EvidenceError):e.execute([sys.executable,'-c','raise SystemExit(2)'],self.root/'scanner.log') + def test_corrupted_input_artifact_fails(self): + inv=self.inventory();file=inv['modules'][0]['publications'][0]['files'][0];(self.root/file['path']).write_bytes(b'changed') + with self.assertRaisesRegex(e.EvidenceError,'checksum'):e.validate_inventory(self.root,inv) + def test_manifest_detects_changed_bytes_and_extras(self): + jar=self.root/'artifact.jar';jar.write_bytes(b'original') + e.write(self.root/'manifest.json',{'schemaVersion':1,'files':[{'path':'artifact.jar','sha256':e.sha(jar)}]}) + e.verify_manifest(self.root) + jar.write_bytes(b'changed') + with self.assertRaises(e.EvidenceError):e.verify_manifest(self.root) + jar.write_bytes(b'original');(self.root/'leftover.jar').write_bytes(b'old') + with self.assertRaises(e.EvidenceError):e.verify_manifest(self.root) + def prepared_scan(self): + inv=self.inventory();e.write(self.root/'build/release-inputs/inventory.json',inv) + e.write(self.root/'scripts/release/exceptions.json',[]) + docs,packages=e.validate_inventory(self.root,inv) + scans=self.root/'scan';scans.mkdir() + for name,pkgs in packages.items():e.write(scans/(name+'.json'),report(pkgs)) + e.write(scans/'input-check.json',{'inventorySha256':e.sha(self.root/'build/release-inputs/inventory.json'), + 'exceptionsSha256':e.sha(self.root/'scripts/release/exceptions.json'), + 'sboms':{name:e.sha(self.root/path) for name,path in docs.items()}}) + e.write(scans/'acceptance.json',{'status':'PASS','exceptions':[]}) + e.write(scans/'scanner.json',{'Version':'0.74.0'}) + e.write(scans/'database.json',{'UpdatedAt':'2026-01-01T00:00:00Z','NextUpdate':'2100-01-01T00:00:00Z','sha256':'a'*64}) + return scans + def test_stage_covers_fixtures_and_metadata_and_rejects_reuse(self): + scans=self.prepared_scan();output=self.root/'dist' + with patch('evidence.source_identity',return_value={'commit':'a'*40,'version':'2.0.0','mode':'trial'}): + e.stage(self.root,scans,output,'a'*40,'2.0.0','trial') + paths={f['path'] for f in e.verify_manifest(output)['files']} + self.assertIn('artifacts/tiercache-core-2.0.0-test-fixtures.jar',paths) + self.assertIn('artifacts/tiercache-tck-2.0.0-tests.jar',paths) + self.assertIn('artifacts/tiercache-core-2.0.0.module',paths) + with self.assertRaises(FileExistsError):e.stage(self.root,scans,output,'a'*40,'2.0.0','trial') + def test_stage_rechecks_scanner_coverage_not_just_pass_label(self): + scans=self.prepared_scan();e.write(scans/'tiercache-core.json',report([SLF4J])) + with patch('evidence.source_identity',return_value={}): + with self.assertRaises(e.EvidenceError):e.stage(self.root,scans,self.root/'dist','a'*40,'2.0.0','trial') + def test_stage_rejects_changed_sbom_after_scan(self): + scans=self.prepared_scan();path=self.root/'aggregate.json';path.write_text(path.read_text()+' ') + with patch('evidence.source_identity',return_value={}): + with self.assertRaises(e.EvidenceError):e.stage(self.root,scans,self.root/'dist','a'*40,'2.0.0','trial') + def test_nested_manifest_is_not_exempt_from_checksums(self): + folder=self.root/'nested';folder.mkdir();(folder/'manifest.json').write_text('{}') + e.write(self.root/'manifest.json',{'schemaVersion':1,'files':[{'path':'nested/manifest.json','sha256':e.sha(folder/'manifest.json')}]}) + e.verify_manifest(self.root);(folder/'manifest.json').write_text('changed') + with self.assertRaises(e.EvidenceError):e.verify_manifest(self.root) + def test_snapshot_trial_allowed_but_final_rejected(self): + (self.root/'gradle.properties').write_text('version=1.5.0-SNAPSHOT\n') + with patch('evidence.subprocess.check_output',side_effect=['a'*40,'a'*40,'']): + self.assertEqual('trial',e.source_identity(self.root,'a'*40,'1.5.0-SNAPSHOT','trial')['mode']) + with patch('evidence.subprocess.check_output',side_effect=['a'*40,'a'*40,'']): + with self.assertRaises(e.EvidenceError):e.source_identity(self.root,'a'*40,'1.5.0-SNAPSHOT','final') + def test_final_version_tag_and_clean_tree_required(self): + (self.root/'gradle.properties').write_text('version=2.0.0\n') + for ref,dirty in [('main',''),('refs/tags/v2.0.0',' M file')]: + with patch('evidence.subprocess.check_output',side_effect=['a'*40,'a'*40,dirty]): + with self.assertRaises(e.EvidenceError):e.source_identity(self.root,ref,'2.0.0','final') + with patch('evidence.subprocess.check_output',side_effect=['a'*40,'a'*40,'']): + self.assertEqual('final',e.source_identity(self.root,'refs/tags/v2.0.0','2.0.0','final')['mode']) + +if __name__=='__main__': unittest.main() diff --git a/tiercache-core/src/main/java/io/tiercache/TierCacheFactory.java b/tiercache-core/src/main/java/io/tiercache/TierCacheFactory.java index c7925ef..74fdef2 100644 --- a/tiercache-core/src/main/java/io/tiercache/TierCacheFactory.java +++ b/tiercache-core/src/main/java/io/tiercache/TierCacheFactory.java @@ -61,6 +61,7 @@ public final class TierCacheFactory implements AutoCloseable { private final boolean ownsLockProvider; private final ScheduledExecutorService watchdog; private final VersionGenerator versionGenerator; + private final ScheduledExecutorService recoveryExecutor; private final InvalidationHandler invalidation; // null = single-node private final CircuitBreaker breaker; // null = unguarded L2 (opt-out) private final CacheMetricsListener metricsListener; @@ -78,7 +79,7 @@ public final class TierCacheFactory implements AutoCloseable { */ private final Object factoryLifecycleLock = new Object(); /** Set under {@link #factoryLifecycleLock} by {@link #close()}. */ - private boolean closed; + private volatile boolean closed; private TierCacheFactory(Builder builder) { this.defaults = builder.defaults; @@ -88,7 +89,13 @@ private TierCacheFactory(Builder builder) { this.singleflightEnabled = builder.singleflightEnabled; this.coordinationEnabled = builder.coordinationEnabled; this.caches = new LinkedHashMap<>(); - builder.overrides.forEach((name, override) -> caches.put(name, override.resolve(defaults))); + builder.overrides.forEach((name, override) -> { + try { + caches.put(name, override.resolve(defaults)); + } catch (IllegalArgumentException e) { + throw new CacheConfigurationException("Cache '" + name + "': " + e.getMessage()); + } + }); // Fail-fast validation of every configured cache. CacheConfigValidator.validate("", defaults); @@ -140,11 +147,17 @@ private TierCacheFactory(Builder builder) { new DaemonThreadFactory("tiercache-async"), new java.util.concurrent.ThreadPoolExecutor.AbortPolicy()); + java.util.concurrent.ScheduledThreadPoolExecutor recoveryPool = builder.invalidationFactory == null + ? null : new java.util.concurrent.ScheduledThreadPoolExecutor(2, + new DaemonThreadFactory("tiercache-recovery")); + if (recoveryPool != null) recoveryPool.setRemoveOnCancelPolicy(true); + this.recoveryExecutor = recoveryPool; this.versionGenerator = new VersionGenerator(); this.invalidation = builder.invalidationFactory != null ? builder.invalidationFactory.apply(versionGenerator) : null; if (this.invalidation != null) { + this.invalidation.configureRecoveryExecutor(recoveryExecutor); this.invalidation.setEventListener(builder.invalidationEventListener); } this.metricsListener = builder.metricsListener; @@ -163,15 +176,15 @@ public void onOpen() { @Override public void onClose() { - // Recovery: replay missed invalidations BEFORE we report - // recovery; L1 is never flushed here. - if (invalidation != null) { - invalidation.onL2Recovery(); - } - log.info("L2 circuit breaker CLOSED: L2 recovered, missed invalidations replayed."); - degradationListener.onRecovered(); + log.info("L2 circuit breaker CLOSED: data-path availability restored."); + if (!closed) degradationListener.onRecovered(); } }); + if (invalidation != null) { + this.breaker.configureRecovery(recoveryExecutor, () -> closed + ? java.util.concurrent.CompletableFuture.completedFuture(false) + : invalidation.recoverAsync(recoveryExecutor)); + } if (rawRemoteCache != null) { rawRemoteCache = new CircuitBreakerRemoteCache<>(rawRemoteCache, breaker); } @@ -218,8 +231,8 @@ public TierCache getCache(String name) { DefaultTierCache cache = new DefaultTierCache<>(n, l1, (RemoteCache) l2For(n), settings, singleflightEnabled, coordinationEnabled ? lockProvider : null, watchdog, versionGenerator, invalidation, - breaker, metricsListener, revalidationExecutor, jitter); - if (invalidation != null) { + breaker, metricsListener, revalidationExecutor, jitter, () -> !closed); + if (invalidation != null && !closed) { invalidation.registerTarget(n, cache); } return cache; @@ -323,9 +336,11 @@ public BreakerState breakerState() { * through {@link #asyncCache(String)} are failed with * {@link java.util.concurrent.CancellationException}, the revalidation * executor and the watchdog scheduler are stopped and the invalidation - * engine is closed. Caches already obtained remain usable but lose - * lease extension for in-flight coordination, and async operations - * submitted afterwards are rejected. + * engine is closed. Already-obtained synchronous caches remain usable + * while their supplied L2 is usable, with local coalescing but without + * coordination, renewal, refresh, publication or recovery. Continued + * cluster coherence is not promised. Async operations submitted afterwards + * are rejected. Caller-supplied resources retain caller ownership. * * @since 0.1.0 */ @@ -333,14 +348,18 @@ public BreakerState breakerState() { public void close() { java.util.List> views; synchronized (factoryLifecycleLock) { + if (closed) return; closed = true; views = new java.util.ArrayList<>(liveAsyncCaches.values()); } + if (breaker != null) breaker.detachRecovery(); // Draining happens outside the lock: completing stages may run // user callbacks. views.forEach(view -> ((io.tiercache.internal.DefaultAsyncTierCache) view).closeOutstanding()); - revalidationExecutor.shutdownNow(); + for (Runnable task : revalidationExecutor.shutdownNow()) { + if (task instanceof DefaultTierCache.DiscardableTask discarded) discarded.discard(); + } asyncExecutor.shutdownNow(); if (watchdog != null) { watchdog.shutdownNow(); @@ -352,8 +371,10 @@ public void close() { log.warn("Failed to close the derived rebuild-lock provider", e); } } - if (invalidation != null) { - invalidation.close(); + try { + if (invalidation != null) invalidation.close(); + } finally { + if (recoveryExecutor != null) recoveryExecutor.shutdownNow(); } } diff --git a/tiercache-core/src/main/java/io/tiercache/internal/BreakerLockProvider.java b/tiercache-core/src/main/java/io/tiercache/internal/BreakerLockProvider.java index fde99cf..8765bde 100644 --- a/tiercache-core/src/main/java/io/tiercache/internal/BreakerLockProvider.java +++ b/tiercache-core/src/main/java/io/tiercache/internal/BreakerLockProvider.java @@ -33,15 +33,19 @@ public BreakerLockProvider(DistributedLockProvider delegate, CircuitBreaker brea @Override public DistributedLock tryLock(String name, Duration lease) { - if (!breaker.tryAcquire()) { + CircuitBreaker.Permit permit = breaker.tryAcquirePermit(); + if (permit == null) { return null; } try { DistributedLock lock = delegate.tryLock(name, lease); - breaker.onSuccess(); + permit.success(); return lock; + } catch (LockProviderClosedException e) { + permit.cancel(); + throw e; } catch (RuntimeException e) { - breaker.onFailure(); + permit.failure(); return null; } } diff --git a/tiercache-core/src/main/java/io/tiercache/internal/CaffeineLocalCache.java b/tiercache-core/src/main/java/io/tiercache/internal/CaffeineLocalCache.java index 799627b..0e51c88 100644 --- a/tiercache-core/src/main/java/io/tiercache/internal/CaffeineLocalCache.java +++ b/tiercache-core/src/main/java/io/tiercache/internal/CaffeineLocalCache.java @@ -35,9 +35,15 @@ public final class CaffeineLocalCache implements LocalCache { * @since 0.1.0 */ public CaffeineLocalCache(CacheSettings settings) { + this(settings, System::nanoTime); + } + + /** Internal clock seam; the engine and L1 must use the same monotonic clock. */ + public CaffeineLocalCache(CacheSettings settings, java.util.function.LongSupplier clock) { @SuppressWarnings("unchecked") Caffeine> builder = (Caffeine>) (Caffeine) Caffeine.newBuilder(); this.cache = builder + .ticker(clock::getAsLong) .maximumSize(settings.l1MaxSize()) .expireAfter(new Expiry>() { @Override @@ -96,6 +102,18 @@ public boolean setIfAbsent(K key, StoredEntry entry, Duration ttl) { return cache.asMap().putIfAbsent(key, new Holder<>(entry, ttl.toNanos())) == null; } + @Override + public boolean supportsAtomicReplace() { return true; } + + @Override + public boolean replaceIfSame(K key, StoredEntry expected, StoredEntry replacement, Duration ttl) { + // A no-op computeIfPresent still invokes expireAfterUpdate and can reset + // another entry's TTL. Read quietly, then compare the exact holder at commit. + Holder current = cache.policy().getIfPresentQuietly(key); + if (current == null || current.entry != expected) return false; + return cache.asMap().replace(key, current, new Holder<>(replacement, ttl.toNanos())); + } + private static final class Holder { final StoredEntry entry; final long ttlNanos; diff --git a/tiercache-core/src/main/java/io/tiercache/internal/CircuitBreaker.java b/tiercache-core/src/main/java/io/tiercache/internal/CircuitBreaker.java index 3a7e36e..d0111fe 100644 --- a/tiercache-core/src/main/java/io/tiercache/internal/CircuitBreaker.java +++ b/tiercache-core/src/main/java/io/tiercache/internal/CircuitBreaker.java @@ -102,65 +102,64 @@ public interface Listener { private final Config config; private final Listener listener; + private final java.util.function.LongSupplier clock; private final boolean[] window; - private int windowPos; - private int windowCount; - private int windowFailures; + private int windowPos, windowCount, windowFailures; + private enum State { CLOSED, OPEN, HALF_OPEN } + private State state = State.CLOSED; + private long openedAtNanos, epoch, recoveryDelayNanos; + private int probesInFlight, probesSucceeded, recoveryFailures; + private boolean recoveryPending; + private java.util.concurrent.Executor recoveryExecutor; + private java.util.function.Supplier> recoveryHook; - private enum State { - CLOSED, OPEN, HALF_OPEN + /** Creates a breaker over count-based thresholds. */ + public CircuitBreaker(Config config, Listener listener) { + this(config, listener, System::nanoTime); } - private State state = State.CLOSED; - private long openedAtNanos; - private int probesInFlight; - private int probesSucceeded; + CircuitBreaker(Config config, Listener listener, java.util.function.LongSupplier clock) { + this.clock = clock; + this.config = Objects.requireNonNull(config); + this.listener = Objects.requireNonNull(listener); + this.window = new boolean[config.windowSize()]; + } - /** - * Creates a breaker. - * - * @param config the thresholds - * @param listener transition callback - * @since 0.1.0 - */ - public CircuitBreaker(Config config, Listener listener) { - this.config = config; - this.listener = listener; - this.window = new boolean[config.windowSize()]; // true = failure + /** Configures asynchronous coherence recovery; callers install this before use. */ + public synchronized void configureRecovery(java.util.concurrent.Executor executor, + java.util.function.Supplier> hook) { + recoveryExecutor = Objects.requireNonNull(executor); + recoveryHook = Objects.requireNonNull(hook); } - /** - * Whether the breaker is currently open (failing L2 calls fast). An open - * breaker whose wait has elapsed lazily transitions to half-open here. - * - * @return {@code true} while the breaker is open - * @since 0.1.0 - */ - public synchronized boolean isOpen() { - if (state == State.OPEN - && System.nanoTime() - openedAtNanos >= config.halfOpenAfter().toNanos()) { + /** Retires the coherence hook without manufacturing a successful data probe. */ + public synchronized void detachRecovery() { + recoveryHook = null; + recoveryExecutor = null; + recoveryDelayNanos = 0; + recoveryFailures = 0; + epoch++; + recoveryPending = false; + probesInFlight = 0; + probesSucceeded = 0; + // An actual OPEN episode keeps its original probe wait. + } + + private void advance() { + if (state == State.OPEN && clock.getAsLong() - openedAtNanos >= + Math.max(config.halfOpenAfter().toNanos(), recoveryDelayNanos)) { state = State.HALF_OPEN; probesInFlight = 0; probesSucceeded = 0; } - return state == State.OPEN; } - /** - * Current state of the breaker machine. An open breaker whose wait has - * elapsed is reported as {@link BreakerState#HALF_OPEN} even before the - * next probe call (the same lazy transition {@link #isOpen()} performs). - * - * @return the current state; never {@code null} - * @since 0.1.0 - */ + /** Whether the breaker is currently OPEN, including its configured wait. */ + public synchronized boolean isOpen() { advance(); return state == State.OPEN; } + + /** Current state; pending coherence recovery is HALF_OPEN, never CLOSED. */ public synchronized BreakerState state() { - if (state == State.OPEN - && System.nanoTime() - openedAtNanos >= config.halfOpenAfter().toNanos()) { - state = State.HALF_OPEN; - probesInFlight = 0; - probesSucceeded = 0; - } + advance(); return switch (state) { case CLOSED -> BreakerState.CLOSED; case OPEN -> BreakerState.OPEN; @@ -168,97 +167,148 @@ public synchronized BreakerState state() { }; } - /** - * Asks permission for an L2 call. When open (and not yet probe time) or - * when the probe budget is exhausted, returns {@code false}. - * - * @return {@code true} if the call may proceed - * @since 0.1.0 - */ + /** Admits one remote attempt without waiting for recovery. */ public synchronized boolean tryAcquire() { - switch (state) { - case CLOSED: - return true; - case OPEN: - if (System.nanoTime() - openedAtNanos < config.halfOpenAfter().toNanos()) { - return false; - } - state = State.HALF_OPEN; - probesInFlight = 0; - probesSucceeded = 0; - // fall through - case HALF_OPEN: - if (probesInFlight >= config.probesToClose()) { - return false; + advance(); + if (state == State.CLOSED) return true; + if (state == State.OPEN || recoveryPending || probesInFlight >= config.probesToClose()) return false; + probesInFlight++; + return true; + } + + /** Returns an epoch-bound, once-completable admission or null on rejection. */ + public synchronized Permit tryAcquirePermit() { return tryAcquire() ? new Permit(epoch) : null; } + + /** Once-only accounting, including neutral completion of unsupported calls. */ + public final class Permit { + private final long acquiredEpoch; + private boolean completed; + private Permit(long acquiredEpoch) { this.acquiredEpoch = acquiredEpoch; } + /** Records a successful remote execution. */ + public void success() { complete(1); } + /** Records a failed remote execution. */ + public void failure() { complete(-1); } + /** Returns admission without adding a remote outcome to the window. */ + public void cancel() { complete(0); } + private void complete(int outcome) { + Runnable action; + synchronized (CircuitBreaker.this) { + if (completed) return; + completed = true; + if (acquiredEpoch != epoch) return; + if (outcome == 0) { + if (state == State.HALF_OPEN && probesInFlight > 0) probesInFlight--; + return; } - probesInFlight++; - return true; - default: - throw new IllegalStateException("unknown state"); + action = outcome > 0 ? successLocked() : failureLocked(); + } + notifySafely(action); } } - /** - * Records a successful L2 call. - * - * @since 0.1.0 - */ - public synchronized void onSuccess() { + /** Compatibility accounting method; transition callbacks run outside the monitor. */ + public void onSuccess() { + Runnable action; + synchronized (this) { action = successLocked(); } + notifySafely(action); + } + + /** Compatibility accounting method; transition callbacks run outside the monitor. */ + public void onFailure() { + Runnable action; + synchronized (this) { action = failureLocked(); } + notifySafely(action); + } + + private Runnable successLocked() { + if (state == State.OPEN || recoveryPending) return null; if (state == State.HALF_OPEN) { - probesInFlight--; - if (++probesSucceeded >= config.probesToClose()) { - close(); - } - return; + if (probesInFlight > 0) probesInFlight--; + if (++probesSucceeded < config.probesToClose()) return null; + if (recoveryHook == null) return closeLocked(); + recoveryPending = true; + long expected = epoch; + var executor = recoveryExecutor; + var hook = recoveryHook; + return () -> { + try { executor.execute(() -> startRecovery(expected, hook)); } + catch (RuntimeException e) { finishRecovery(expected, false); } + }; } record(false); + return null; } - /** - * Records a failed L2 call. - * - * @since 0.1.0 - */ - public synchronized void onFailure() { - if (state == State.HALF_OPEN) { - open(); // probe failed: reopen, restart the wait - return; + private void startRecovery(long expected, + java.util.function.Supplier> hook) { + synchronized (this) { + if (epoch != expected || !recoveryPending || recoveryHook != hook) return; } + try { + hook.get().whenComplete((result, error) -> finishRecovery(expected, + error == null && Boolean.TRUE.equals(result))); + } catch (Throwable e) { finishRecovery(expected, false); } + } + + private void finishRecovery(long expected, boolean success) { + Runnable action; + synchronized (this) { + if (epoch != expected || !recoveryPending || recoveryHook == null) return; + recoveryPending = false; + if (success) { + recoveryFailures = 0; + recoveryDelayNanos = 0; + action = closeLocked(); + } else { + recoveryFailures = Math.min(6, recoveryFailures + 1); + recoveryDelayNanos = java.util.concurrent.TimeUnit.SECONDS.toNanos( + Math.min(30, 1L << (recoveryFailures - 1))); + action = openLocked(); + } + } + notifySafely(action); + } + + private Runnable failureLocked() { + if (state == State.HALF_OPEN) return openLocked(); record(true); if (windowCount >= config.minimumCalls() && windowFailures >= Math.ceil(config.failureRatio() * Math.min(windowCount, config.windowSize()))) { - open(); + return openLocked(); } + return null; } private void record(boolean failure) { - if (windowCount < config.windowSize()) { - windowCount++; - } else if (window[windowPos]) { - windowFailures--; // overwritten failure leaves the window - } - if (failure) { - windowFailures++; - } + if (windowCount < config.windowSize()) windowCount++; + else if (window[windowPos]) windowFailures--; + if (failure) windowFailures++; window[windowPos] = failure; windowPos = (windowPos + 1) % config.windowSize(); } - private void open() { - if (state != State.OPEN) { - state = State.OPEN; - openedAtNanos = System.nanoTime(); - listener.onOpen(); - } else { - openedAtNanos = System.nanoTime(); // re-opened: restart the wait - } + private Runnable openLocked() { + epoch++; + recoveryPending = false; + boolean changed = state != State.OPEN; + state = State.OPEN; + openedAtNanos = clock.getAsLong(); + return changed ? listener::onOpen : null; } - private void close() { + private Runnable closeLocked() { + epoch++; state = State.CLOSED; - windowPos = 0; - windowCount = 0; - windowFailures = 0; - listener.onClose(); + recoveryPending = false; + windowPos = windowCount = windowFailures = 0; + return listener::onClose; + } + + private static void notifySafely(Runnable action) { + if (action == null) return; + try { action.run(); } + catch (Throwable e) { + org.slf4j.LoggerFactory.getLogger(CircuitBreaker.class).warn("Circuit breaker observer failed", e); + } } } diff --git a/tiercache-core/src/main/java/io/tiercache/internal/CircuitBreakerRemoteCache.java b/tiercache-core/src/main/java/io/tiercache/internal/CircuitBreakerRemoteCache.java index 39f286b..7b7057c 100644 --- a/tiercache-core/src/main/java/io/tiercache/internal/CircuitBreakerRemoteCache.java +++ b/tiercache-core/src/main/java/io/tiercache/internal/CircuitBreakerRemoteCache.java @@ -3,6 +3,7 @@ import io.tiercache.Version; import io.tiercache.spi.RemoteCache; import io.tiercache.spi.StoredEntry; +import io.tiercache.spi.TaggedWriteOutcome; import java.time.Duration; @@ -127,23 +128,60 @@ public void putTagged(K key, StoredEntry entry, Duration ttl, String[] tags) }); } + @Override + public boolean supportsTaggedWriteOutcomes() { + return delegate.supportsTaggedWriteOutcomes(); + } + + @Override + public TaggedWriteOutcome putTaggedIfNewer(K key, StoredEntry entry, + Duration ttl, String[] tags) { + if (entry.version() != null && !supportsTaggedWriteOutcomes()) { + return TaggedWriteOutcome.UNSUPPORTED; + } + CircuitBreaker.Permit permit = breaker.tryAcquirePermit(); + if (permit == null) { + throw L2UnavailableException.OPEN; + } + try { + TaggedWriteOutcome result = delegate.putTaggedIfNewer(key, entry, ttl, tags); + if (result == TaggedWriteOutcome.UNSUPPORTED) { + permit.cancel(); + } else { + permit.success(); + } + return result; + } catch (L2UnavailableException e) { + permit.cancel(); + throw e; + } catch (Exception e) { + permit.failure(); + throw new L2UnavailableException("L2 call failed: " + e.getClass().getSimpleName(), e); + } catch (Error e) { + permit.cancel(); + throw e; + } + } + @Override public java.util.List keysByTag(String tag) { return guard(() -> delegate.keysByTag(tag)); } private T guard(java.util.concurrent.Callable call) { - if (!breaker.tryAcquire()) { + CircuitBreaker.Permit permit = breaker.tryAcquirePermit(); + if (permit == null) { throw L2UnavailableException.OPEN; } try { T result = call.call(); - breaker.onSuccess(); + permit.success(); return result; } catch (L2UnavailableException e) { + permit.cancel(); throw e; } catch (Exception e) { - breaker.onFailure(); + permit.failure(); throw new L2UnavailableException("L2 call failed: " + e.getClass().getSimpleName(), e); } } diff --git a/tiercache-core/src/main/java/io/tiercache/internal/DefaultTierCache.java b/tiercache-core/src/main/java/io/tiercache/internal/DefaultTierCache.java index 294a947..1d0dcb7 100644 --- a/tiercache-core/src/main/java/io/tiercache/internal/DefaultTierCache.java +++ b/tiercache-core/src/main/java/io/tiercache/internal/DefaultTierCache.java @@ -1,6 +1,7 @@ package io.tiercache.internal; import io.tiercache.CacheSettings; +import io.tiercache.CacheConfigurationException; import io.tiercache.InvalidationMode; import io.tiercache.InvalidationMessage; import io.tiercache.LookupResult; @@ -15,12 +16,12 @@ import io.tiercache.spi.LocalCache; import io.tiercache.spi.RemoteCache; import io.tiercache.spi.StoredEntry; +import io.tiercache.spi.TaggedWriteOutcome; import org.slf4j.Logger; import org.slf4j.LoggerFactory; import java.time.Duration; import java.util.Map; -import java.util.concurrent.CompletableFuture; import java.util.concurrent.ConcurrentHashMap; import java.util.concurrent.Executor; import java.util.concurrent.ScheduledExecutorService; @@ -49,9 +50,11 @@ * overwritten by the stale copy) and a single asynchronous revalidation per * key per instance * (claimed on the same in-flight map as singleflight) refreshes them through - * the coordinated load path; failures keep serving stale and never reach - * readers. XFetch (opt-in via {@code xfetchEnabled}) adds a probabilistic - * early refresh on fresh L2 hits, driven by entry age and a per-cache EMA of + * the coordinated load path. A skipped refresh is not a source miss: + * foreground demand promotes it to a bounded ordinary load, while stale + * readers keep their immediate result. Background failures do not replace + * an already served stale result. XFetch (opt-in via {@code xfetchEnabled}) + * adds a probabilistic early refresh on fresh L2 hits, driven by entry age and a per-cache EMA of * loader durations measured internally. * *

Hot-path discipline: a steady-state L1 hit performs exactly one @@ -94,6 +97,7 @@ public final class DefaultTierCache implements TierCache, Invalidati private final InvalidationHandler invalidation; // null = single-node private final CircuitBreaker breaker; // null = unguarded L2 (opt-out) private final CacheMetricsListener metrics; + private final java.util.function.BooleanSupplier auxiliaryOpen; private final Executor revalidationExecutor; // null = no async revalidation private final Duration staleTtl; private final boolean staleWindowEnabled; @@ -106,9 +110,10 @@ public final class DefaultTierCache implements TierCache, Invalidati private final long staleBoundaryMillis; // l2TtlMillis + staleTtl private final double xfetchBetaNanos; private final TtlJitter jitter; + private final java.util.function.LongSupplier localClock; /** EMA of loader durations in nanoseconds; updated on every load. */ private final AtomicLong loaderDurationEmaNanos = new AtomicLong(EMA_UNINITIALIZED); - private final Map>> inflight = new ConcurrentHashMap<>(); + private final Map> inflight = new ConcurrentHashMap<>(); /** * Per-key invalidation barrier: the highest version this instance has @@ -128,6 +133,7 @@ public final class DefaultTierCache implements TierCache, Invalidati private static final int L1_STRIPES = 64; private final L1BarrierMap l1Metas; + private final java.util.concurrent.atomic.AtomicLong recoveryGeneration = new java.util.concurrent.atomic.AtomicLong(); private final Object[] l1Locks; /** Bumped when protective L1 state is forgotten (barrier eviction, evictAll). */ private final AtomicLong l1Generation = new AtomicLong(); @@ -267,6 +273,37 @@ public DefaultTierCache(String cacheName, LocalCache l1, RemoteCache VersionGenerator versionGenerator, InvalidationHandler invalidation, CircuitBreaker breaker, CacheMetricsListener metrics, Executor revalidationExecutor, TtlJitter jitter) { + this(cacheName, l1, l2, settings, singleflightEnabled, lockProvider, watchdog, + versionGenerator, invalidation, breaker, metrics, revalidationExecutor, jitter, () -> true); + } + + /** Internal factory wiring with an auxiliary admission gate. */ + public DefaultTierCache(String cacheName, LocalCache l1, RemoteCache l2, + CacheSettings settings, boolean singleflightEnabled, + DistributedLockProvider lockProvider, ScheduledExecutorService watchdog, + VersionGenerator versionGenerator, InvalidationHandler invalidation, + CircuitBreaker breaker, CacheMetricsListener metrics, Executor revalidationExecutor, + TtlJitter jitter, java.util.function.BooleanSupplier auxiliaryOpen) { + this(cacheName, l1, l2, settings, singleflightEnabled, lockProvider, watchdog, + versionGenerator, invalidation, breaker, metrics, revalidationExecutor, jitter, + auxiliaryOpen, System::nanoTime); + } + + /** Internal clock seam for deterministic local-lifetime testing. */ + public DefaultTierCache(String cacheName, LocalCache l1, RemoteCache l2, + CacheSettings settings, boolean singleflightEnabled, + DistributedLockProvider lockProvider, ScheduledExecutorService watchdog, + VersionGenerator versionGenerator, InvalidationHandler invalidation, + CircuitBreaker breaker, CacheMetricsListener metrics, Executor revalidationExecutor, + TtlJitter jitter, java.util.function.BooleanSupplier auxiliaryOpen, + java.util.function.LongSupplier localClock) { + this.localClock = localClock; + if (!settings.degradationStaleTtl().isZero() && settings.l1ExpireAfterAccess() != null + && !l1.supportsAtomicReplace()) { + throw new IllegalArgumentException("Cache '" + cacheName + + "': degradationStaleTtl with l1ExpireAfterAccess requires LocalCache atomic replacement"); + } + this.auxiliaryOpen = auxiliaryOpen; this.cacheName = cacheName; this.l1 = l1; this.l2 = l2; @@ -295,7 +332,7 @@ public DefaultTierCache(String cacheName, LocalCache l1, RemoteCache cacheName); } this.l1Metas = new L1BarrierMap<>(L1_META_MAX, L1_META_EXPIRY, - l1Generation::incrementAndGet); + l1Generation::incrementAndGet, localClock); this.l1Locks = new Object[L1_STRIPES]; for (int i = 0; i < L1_STRIPES; i++) { l1Locks[i] = new Object(); @@ -459,23 +496,59 @@ public V getOrCompute(K key, Function loader) { ? CacheMetricsListener.Outcome.LOAD : CacheMetricsListener.Outcome.MISS); return unwrap(result); } - CompletableFuture> future = new CompletableFuture<>(); - CompletableFuture> existing = inflight.putIfAbsent(key, future); - if (existing != null) { - metrics.onRequest(cacheName, CacheMetricsListener.Outcome.COALESCED); - return unwrap(existing.join()); + LoadClaim.Demand demand = new LoadClaim.Demand<>(loader, + System.nanoTime() + OVERALL_BUDGET.toNanos()); + LoadClaim claim = new LoadClaim<>(demand); + LoadClaim existing = inflight.putIfAbsent(key, claim); + if (existing == null) { + return runForegroundClaim(key, claim, demand, true); } + existing.requireResult(demand); + metrics.onRequest(cacheName, CacheMetricsListener.Outcome.COALESCED); + LoadClaim.Outcome outcome = existing.result.join(); + if (!outcome.isSkipped()) { + return unwrap(outcome.resultEntry()); + } + return recoverSkippedRefresh(key, existing, demand); + } + + /** + * One map transition replaces a terminal skip or promotes its replacement. + * The selected claim cannot skip again: foreground demand is registered + * before the map transition ends. The old owner's finally uses identity + * removal and cannot delete this replacement. + */ + private V recoverSkippedRefresh(K key, LoadClaim skipped, + LoadClaim.Demand demand) { + LoadClaim replacement = new LoadClaim<>(demand); + LoadClaim selected = inflight.compute(key, (ignored, current) -> { + if (current == null || current == skipped || !current.requireResult(demand)) { + return replacement; + } + return current; + }); + if (selected == replacement) { + // The original request was already counted as coalesced. + return runForegroundClaim(key, replacement, demand, false); + } + return unwrap(selected.result.join().resultEntry()); + } + + private V runForegroundClaim(K key, LoadClaim claim, + LoadClaim.Demand demand, boolean recordOutcome) { try { - StoredEntry loaded = loadPath(key, loader); - metrics.onRequest(cacheName, loaded != null && !loaded.isNullMarker() - ? CacheMetricsListener.Outcome.LOAD : CacheMetricsListener.Outcome.MISS); - future.complete(loaded); + StoredEntry loaded = loadPath(key, demand.loader(), demand.deadlineNanos()); + if (recordOutcome) { + metrics.onRequest(cacheName, loaded != null && !loaded.isNullMarker() + ? CacheMetricsListener.Outcome.LOAD : CacheMetricsListener.Outcome.MISS); + } + claim.result.complete(LoadClaim.Outcome.result(loaded)); return unwrap(loaded); - } catch (RuntimeException e) { - future.completeExceptionally(e); + } catch (RuntimeException | Error e) { + claim.result.completeExceptionally(e); throw e; } finally { - inflight.remove(key, future); + inflight.remove(key, claim); } } @@ -577,6 +650,11 @@ public void put(K key, V value, String... tags) { put(key, value); return; } + // A missing SPI capability is a configuration error, including while + // OPEN: it must not silently become a successful local-only write. + if (versionGenerator != null && !l2.supportsTaggedWriteOutcomes()) { + throw unsupportedTaggedWrite(); + } long g0 = l1Generation.get(); Version version = nextVersion(); StoredEntry entry = StoredEntry.ofValue(value, version); @@ -584,13 +662,33 @@ public void put(K key, V value, String... tags) { warmL1(key, entry, g0); return; } + TaggedWriteOutcome outcome; try { - l2.putTagged(key, entry, settings.l2Ttl(), tags); - warmL1(key, entry, g0); - publishStore(key, entry, version); + outcome = l2.putTaggedIfNewer(key, entry, settings.l2Ttl(), tags); } catch (L2UnavailableException e) { warmL1(key, entry, g0); + return; + } + if (outcome == TaggedWriteOutcome.UNSUPPORTED) { + throw unsupportedTaggedWrite(); + } + if (outcome == TaggedWriteOutcome.LOST) { + StoredEntry current = l2Get(key); + if (current != null) { + warmL1(key, current, g0); + } else { + evictLocal(key); + } + return; } + warmL1(key, entry, g0); + publishStore(key, entry, version); + } + + private CacheConfigurationException unsupportedTaggedWrite() { + return new CacheConfigurationException("Cache '" + cacheName + + "' requires versioned tagged-write outcomes. Implement RemoteCache." + + "supportsTaggedWriteOutcomes() and putTaggedIfNewer() in the custom transport."); } @Override @@ -660,12 +758,36 @@ public void applyUpdateL1(Object key, Object value, Version eventVersion) { // One atomic step: lift the barrier AND install the payload (its // own version always passes — equality is not staleness). Duration ttl = jitter.apply(settings.l1ExpireAfterWrite(), settings.jitterAmplitude()); - long logicalDeadline = System.nanoTime() + ttl.toNanos(); - l1.put(typedKey, StoredEntry.ofValue((V) value, eventVersion), + Version barrier = maxVersion(eventVersion, highestSeen); + l1.put(typedKey, localCopy(StoredEntry.ofValue((V) value, eventVersion), ttl, barrier), degradationStaleEnabled ? ttl.plus(degradationStaleTtl) : ttl); - l1Metas.put(typedKey, new L1BarrierMap.L1Meta( - maxVersion(eventVersion, highestSeen), logicalDeadline, - logicalDeadline + degradationStaleTtl.toNanos())); + l1Metas.put(typedKey, barrier); + } + } + + @Override + public long recoveryGeneration() { return recoveryGeneration.get(); } + + @Override + public long resetRecovery(long expectedGeneration) { + if (!recoveryGeneration.compareAndSet(expectedGeneration, expectedGeneration + 1)) return -1; + clearL1Contents(); + return expectedGeneration + 1; + } + + @Override + @SuppressWarnings("unchecked") + public long applyRecovery(InvalidationMessage message, long expectedGeneration) { + if (message.type() == InvalidationMessage.Type.EVICT_ALL) return resetRecovery(expectedGeneration); + K key = (K) message.key(); + synchronized (l1LockFor(key)) { + if (recoveryGeneration.get() != expectedGeneration) return -1; + if (message.type() == InvalidationMessage.Type.UPDATE) { + applyUpdateL1(key, message.payload(), message.version()); + } else { + evictL1IfNewer(key, message.version()); + } + return expectedGeneration; } } @@ -705,22 +827,33 @@ private boolean commitL1(K key, StoredEntry entry, Duration ttl, long generat if (generationAtStart != l1Generation.get()) { return false; } - Version highestSeen = meta != null ? meta.highestSeen() : null; - if (entry.version() != null && highestSeen != null - && entry.version().compareTo(highestSeen) < 0) { + StoredEntry current = l1.get(key); + Version highestSeen = maxVersion(meta != null ? meta.highestSeen() : null, + current != null ? current.version() : null); + if (current != null && current.localFreshness() != null) { + highestSeen = maxVersion(highestSeen, current.localFreshness().highestSeen()); + } + if (generationAtStart != l1Generation.get()) return false; + if (highestSeen != null && (entry.version() == null + || entry.version().compareTo(highestSeen) < 0)) { return false; } - // Freshness deadlines are stamped with the ACTUAL jittered TTL; - // physical retention additionally covers the degradation window. - long logicalDeadline = System.nanoTime() + ttl.toNanos(); - long staleServeUntil = logicalDeadline + degradationStaleTtl.toNanos(); - l1.put(key, entry, degradationStaleEnabled ? ttl.plus(degradationStaleTtl) : ttl); - l1Metas.put(key, new L1BarrierMap.L1Meta( - maxVersion(entry.version(), highestSeen), logicalDeadline, staleServeUntil)); + Version barrier = maxVersion(entry.version(), highestSeen); + l1.put(key, localCopy(entry, ttl, barrier), + degradationStaleEnabled ? ttl.plus(degradationStaleTtl) : ttl); + l1Metas.put(key, barrier); return true; } } + private StoredEntry localCopy(StoredEntry entry, Duration ttl, Version barrier) { + if (!degradationStaleEnabled) return entry; + long logical = localClock.getAsLong() + ttl.toNanos(); + long retention = logical + degradationStaleTtl.toNanos(); + return entry.withLocalFreshness(new StoredEntry.LocalFreshness( + logical, retention, retention, retention, barrier)); + } + /** * Atomic invalidation: lifts the barrier (even for absent keys) and * removes the value when the event supersedes it. Local versioned @@ -768,6 +901,11 @@ private void evictLocal(K key) { /** Full local clear: generation bump first, then per-stripe ordering, then the clears. */ private void clearL1() { + recoveryGeneration.incrementAndGet(); + clearL1Contents(); + } + + private void clearL1Contents() { l1Generation.incrementAndGet(); // Ordering point with in-flight per-key commits: a commit that // passed its generation check before the bump completes its write @@ -783,14 +921,14 @@ private void clearL1() { } private void publish(Object key, Version version, InvalidationMessage.Type type) { - if (invalidation != null && version != null) { + if (auxiliaryOpen.getAsBoolean() && invalidation != null && version != null) { invalidation.onLocalWrite(cacheName, key, version, type); } } /** Publish for a stored entry: UPDATE (with payload) in update mode, else INVALIDATE. */ private void publishStore(K key, StoredEntry entry, Version version) { - if (invalidation == null || version == null) { + if (!auxiliaryOpen.getAsBoolean() || invalidation == null || version == null) { return; } if (settings.invalidationMode() == InvalidationMode.UPDATE && !entry.isNullMarker()) { @@ -865,22 +1003,21 @@ private void xfetchGate(K key, Function loader, */ private void triggerRevalidation(K key, Function loader, long servedWriteTimestamp) { - if (revalidationExecutor == null) { + if (!auxiliaryOpen.getAsBoolean() || revalidationExecutor == null) { return; // legacy wiring: stale keeps serving without revalidation } - CompletableFuture> claim = new CompletableFuture<>(); + LoadClaim claim = new LoadClaim<>(null); if (inflight.putIfAbsent(key, claim) != null) { return; // a load or revalidation for this key is already in flight } - metrics.onRevalidationTriggered(cacheName); try { - revalidationExecutor.execute( - () -> runRevalidation(key, loader, servedWriteTimestamp, claim)); + metrics.onRevalidationTriggered(cacheName); + revalidationExecutor.execute(new RevalidationTask(key, loader, servedWriteTimestamp, claim)); } catch (RuntimeException e) { // Executor rejected (saturated or shut down): complete the claim // first so waiters already joined on it fail fast instead of // hanging, then release the slot so a later read retries. - claim.completeExceptionally(e); + claim.result.completeExceptionally(e); inflight.remove(key, claim); metrics.onRevalidationFailed(cacheName); log.warn("Revalidation for key '{}' in cache '{}' could not be submitted " @@ -889,17 +1026,54 @@ private void triggerRevalidation(K key, Function loader, } } + /** A queued task must retire its claim when shutdown discards it. */ + public interface DiscardableTask extends Runnable { void discard(); } + + private final class RevalidationTask implements DiscardableTask { + private final K key; + private final Function loader; + private final long timestamp; + private final LoadClaim claim; + private final java.util.concurrent.atomic.AtomicBoolean claimed = new java.util.concurrent.atomic.AtomicBoolean(); + RevalidationTask(K key, Function loader, long timestamp, LoadClaim claim) { + this.key = key; this.loader = loader; this.timestamp = timestamp; this.claim = claim; + } + @Override public void run() { + if (!auxiliaryOpen.getAsBoolean()) { discard(); return; } + if (claimed.compareAndSet(false, true)) runRevalidation(key, loader, timestamp, claim); + } + @Override public void discard() { + if (claimed.compareAndSet(false, true)) { + inflight.remove(key, claim); + claim.result.complete(LoadClaim.Outcome.skippedRefresh()); + } + } + } + private void runRevalidation(K key, Function loader, - long servedWriteTimestamp, CompletableFuture> claim) { + long servedWriteTimestamp, LoadClaim claim) { try { - StoredEntry refreshed = revalidate(key, loader, servedWriteTimestamp); - claim.complete(refreshed); + LoadClaim.Outcome refreshed = revalidate(key, loader, servedWriteTimestamp); + if (refreshed.isSkipped()) { + LoadClaim.Demand demand = claim.skipOrForeground(); + if (demand != null) { + // Still the same local owner. A skipped acquisition used + // no loader budget; this one load path retains its normal + // two-execution bound and the original foreground deadline. + refreshed = LoadClaim.Outcome.result( + loadPath(key, demand.loader(), demand.deadlineNanos())); + } + } + claim.result.complete(refreshed); metrics.onRevalidationCompleted(cacheName); - } catch (RuntimeException e) { - claim.completeExceptionally(e); + } catch (RuntimeException | Error e) { + claim.result.completeExceptionally(e); metrics.onRevalidationFailed(cacheName); log.warn("Revalidation failed for key '{}' in cache '{}'; the stale entry " + "keeps serving until its window ends.", key, cacheName, e); + if (e instanceof Error error) { + throw error; + } } finally { inflight.remove(key, claim); } @@ -912,29 +1086,29 @@ private void runRevalidation(K key, Function loader, * served suppresses the reload. A lost lock race is not a failure: another * instance is refreshing, and the stale entry keeps serving. */ - private StoredEntry revalidate(K key, Function loader, + private LoadClaim.Outcome revalidate(K key, Function loader, long servedWriteTimestamp) { long g0 = l1Generation.get(); - if (lockProvider == null || watchdog == null || !l2Available()) { + if (!auxiliaryOpen.getAsBoolean() || lockProvider == null || watchdog == null || !l2Available()) { // No coordination possible: the in-flight claim already bounds // this to one load per key per instance. - return loadAndStore(key, loader); + return LoadClaim.Outcome.result(loadAndStore(key, loader)); } - DistributedLock lock = tryLockGuarded(cacheName + ":" + key); + DistributedLock lock; + try { lock = tryLockGuarded(cacheName + ":" + key); } + catch (LockProviderClosedException e) { return LoadClaim.Outcome.result(loadAndStore(key, loader)); } if (lock == null) { - return null; // another instance holds the rebuild lock + return LoadClaim.Outcome.skippedRefresh(); // no source absence was observed } - try { + try (LockScope scope = new LockScope(lock, key)) { StoredEntry current = l2Get(key); if (current != null && current.hasWriteTimestamp() && current.writeTimestampMillis() > servedWriteTimestamp) { // A newer write landed while we claimed the lock: converge, no load. warmL1(key, current, g0); - return current; + return LoadClaim.Outcome.result(current); } - return loadWithWatchdog(key, loader, lock); - } finally { - releaseGuarded(lock, key); + return LoadClaim.Outcome.result(loadWithWatchdog(key, loader, scope)); } } @@ -960,20 +1134,32 @@ public long loaderDurationEmaNanos() { // --- Load path selection: coordinated when possible --- private StoredEntry loadPath(K key, Function loader) { - if (lockProvider == null || watchdog == null || !l2Available()) { + return loadPath(key, loader, System.nanoTime() + OVERALL_BUDGET.toNanos()); + } + + private StoredEntry loadPath(K key, Function loader, + long overallDeadline) { + if (!auxiliaryOpen.getAsBoolean() || lockProvider == null || watchdog == null || !l2Available()) { // No coordination possible (or L2 down): per-instance load. return loadAndStore(key, loader); } - return coordinatedLoad(key, loader); + return coordinatedLoad(key, loader, overallDeadline); } - private StoredEntry coordinatedLoad(K key, Function loader) { + private StoredEntry coordinatedLoad(K key, Function loader, + long overallDeadline) { String lockName = cacheName + ":" + key; long g0 = l1Generation.get(); - long overallDeadline = System.nanoTime() + OVERALL_BUDGET.toNanos(); - long waitDeadline = System.nanoTime() + WAIT_SLICE.toNanos(); while (true) { - DistributedLock lock = tryLockGuarded(lockName); + if (System.nanoTime() >= overallDeadline) { + log.warn("Rebuild coordination budget exhausted for key '{}' in cache '{}'; " + + "loading without coordination (possible stampede after repeated " + + "winner failures).", key, cacheName); + return loadAndStore(key, loader); + } + DistributedLock lock; + try { lock = tryLockGuarded(lockName); } + catch (LockProviderClosedException e) { return loadAndStore(key, loader); } if (!l2Available()) { // L2 failed between the availability check and lock // acquisition: fall back to the per-instance load — after @@ -985,35 +1171,29 @@ private StoredEntry coordinatedLoad(K key, Function l return loadAndStore(key, loader); } if (lock != null) { - try { + try (LockScope scope = new LockScope(lock, key)) { // Mandatory double-check: the value may have // appeared while we were acquiring the lock. StoredEntry entry = readThrough(key, g0); if (entry != null) { return entry; } - return loadWithWatchdog(key, loader, lock); - } finally { - releaseGuarded(lock, key); + return loadWithWatchdog(key, loader, scope); } } + long waitDeadline = Math.min(overallDeadline, + System.nanoTime() + WAIT_SLICE.toNanos()); StoredEntry appeared = awaitValue(key, waitDeadline); if (appeared != null) { warmL1(key, appeared, g0); return appeared; } - if (System.nanoTime() > overallDeadline) { - log.warn("Rebuild coordination budget exhausted for key '{}' in cache '{}'; " - + "loading without coordination (possible stampede after repeated " - + "winner failures).", key, cacheName); - return loadAndStore(key, loader); - } - waitDeadline = System.nanoTime() + WAIT_SLICE.toNanos(); } } /** Lock acquisition through the breaker: fast-fail when open. */ private DistributedLock tryLockGuarded(String lockName) { + if (!auxiliaryOpen.getAsBoolean()) throw new LockProviderClosedException(); if (breaker != null && breaker.isOpen()) { return null; } @@ -1045,17 +1225,36 @@ private StoredEntry readThrough(K key, long generationAtStart) { return entry; } + private final class LockScope implements AutoCloseable { + final DistributedLock lock; + final K key; + ScheduledFuture extension; + boolean retired; + LockScope(DistributedLock lock, K key) { this.lock = lock; this.key = key; } + @Override public void close() { + if (retired) return; + retired = true; + if (extension != null) extension.cancel(false); + releaseGuarded(lock, key); + } + } + private StoredEntry loadWithWatchdog(K key, Function loader, - DistributedLock lock) { + LockScope scope) { + if (!auxiliaryOpen.getAsBoolean()) { + scope.close(); + return loadAndStore(key, loader); + } long periodMillis = LOCK_LEASE.toMillis() / 3; - ScheduledFuture extension = watchdog.scheduleAtFixedRate( - () -> lock.extend(LOCK_LEASE), - periodMillis, periodMillis, TimeUnit.MILLISECONDS); try { - return loadAndStore(key, loader); - } finally { - extension.cancel(false); + scope.extension = watchdog.scheduleAtFixedRate( + () -> { if (auxiliaryOpen.getAsBoolean()) scope.lock.extend(LOCK_LEASE); }, + periodMillis, periodMillis, TimeUnit.MILLISECONDS); + } catch (java.util.concurrent.RejectedExecutionException e) { + scope.close(); + if (auxiliaryOpen.getAsBoolean() && !watchdog.isShutdown()) throw e; } + return loadAndStore(key, loader); } /** Bounded wait for the winner's value in L2. Null on timeout. */ @@ -1286,37 +1485,30 @@ private FreshnessSnapshot freshnessOf(K key, StoredEntry entry) { synchronized (l1LockFor(key)) { // Coherent snapshot under the lock: the entry is re-read here, // so a completed concurrent write can never be overwritten by a - // stale caller-side read. The barrier metadata belongs to the - // same commit, so value and metadata always describe each other. + // stale caller-side read. Its immutable local descriptor belongs + // to this exact holder, independently of fencing-map eviction. entry = l1.get(key); - L1BarrierMap.L1Meta meta = l1Metas.get(key); - if (entry == null) { - // The value was evicted or removed concurrently: never hand - // out a null entry (a FRESH classification would NPE the - // caller). The protective barrier metadata stays untouched; - // the caller continues to the normal L2/loader path. - return new FreshnessSnapshot<>(null, L1Freshness.EXPIRED); - } - if (meta == null || meta.logicalDeadlineNanos() == 0L) { - return new FreshnessSnapshot<>(entry, L1Freshness.EXPIRED); - } - long now = System.nanoTime(); - if (now <= meta.logicalDeadlineNanos()) { - java.time.Duration accessTtl = settings.l1ExpireAfterAccess(); - if (accessTtl != null && entry != null && l1Metas.get(key) == meta) { - // Fresh access: slide freshness, the stale horizon AND - // the physical retention — identity-checked, so a - // concurrently replaced value keeps its own deadlines. + if (entry == null) return new FreshnessSnapshot<>(null, L1Freshness.EXPIRED); + StoredEntry.LocalFreshness meta = entry.localFreshness(); + if (meta == null) return new FreshnessSnapshot<>(entry, L1Freshness.EXPIRED); + long now = localClock.getAsLong(); + if (now - meta.logicalDeadlineNanos() < 0) { + Duration accessTtl = settings.l1ExpireAfterAccess(); + if (accessTtl != null) { long logical = now + accessTtl.toNanos(); - l1Metas.put(key, new L1BarrierMap.L1Meta(meta.highestSeen(), logical, - logical + degradationStaleTtl.toNanos())); - l1.put(key, entry, accessTtl.plus(degradationStaleTtl)); + long staleUntil = logical + degradationStaleTtl.toNanos(); + long retention = now + Math.max(meta.storeRetentionFloorNanos() - now, staleUntil - now); + StoredEntry refreshed = entry.withLocalFreshness(new StoredEntry.LocalFreshness( + logical, staleUntil, meta.storeRetentionFloorNanos(), retention, meta.highestSeen())); + if (!l1.replaceIfSame(key, entry, refreshed, Duration.ofNanos(retention - now))) { + return new FreshnessSnapshot<>(null, L1Freshness.EXPIRED); + } + entry = refreshed; } return new FreshnessSnapshot<>(entry, L1Freshness.FRESH); } - return new FreshnessSnapshot<>(entry, - now <= meta.staleServeUntilNanos() ? L1Freshness.STALE_ALLOWED - : L1Freshness.EXPIRED); + return new FreshnessSnapshot<>(entry, now - meta.staleServeUntilNanos() < 0 + ? L1Freshness.STALE_ALLOWED : L1Freshness.EXPIRED); } } diff --git a/tiercache-core/src/main/java/io/tiercache/internal/L1BarrierMap.java b/tiercache-core/src/main/java/io/tiercache/internal/L1BarrierMap.java index ecc94b7..eca31ec 100644 --- a/tiercache-core/src/main/java/io/tiercache/internal/L1BarrierMap.java +++ b/tiercache-core/src/main/java/io/tiercache/internal/L1BarrierMap.java @@ -6,11 +6,10 @@ import java.time.Duration; /** - * Bounded per-key L1 metadata for the engine, one entry per key: the - * invalidation barrier (highest version seen invalidated, recorded even - * for absent keys) plus the freshness deadlines of the value stored with - * it — all written and removed in the same atomic commit as the value, so - * metadata always describes the value it sits with. + * Bounded protective versions, including for keys absent from L1. + * Authoritative freshness lives on the actual L1 entry, independently of + * this map's eviction and expiry. Forgetting a fence bumps the generation + * to reject obsolete in-flight commits. * *

Every eviction (size cap or expiry) and every bulk invalidation * reports through the protection callback so the engine can bump its L1 @@ -28,18 +27,8 @@ */ final class L1BarrierMap { - /** - * Per-key metadata. - * - * @param highestSeen highest version invalidated/superseded - * locally (the barrier), or {@code null} - * @param logicalDeadlineNanos freshness deadline of the stored value - * (store time + actual jittered TTL) - * @param staleServeUntilNanos stale-serving horizon - * ({@code logicalDeadline + staleWindow}) - */ - record L1Meta(Version highestSeen, long logicalDeadlineNanos, long staleServeUntilNanos) { - } + /** Highest protective version seen for a key; freshness belongs to its L1 entry. */ + record L1Meta(Version highestSeen) { } private final com.github.benmanes.caffeine.cache.Cache barriers; @@ -50,7 +39,13 @@ record L1Meta(Version highestSeen, long logicalDeadlineNanos, long staleServeUnt * (entry eviction or bulk invalidation) */ L1BarrierMap(long maxSize, Duration expiry, Runnable onProtection) { + this(maxSize, expiry, onProtection, System::nanoTime); + } + + L1BarrierMap(long maxSize, Duration expiry, Runnable onProtection, + java.util.function.LongSupplier clock) { this.barriers = Caffeine.newBuilder() + .ticker(clock::getAsLong) .maximumSize(maxSize) .expireAfterWrite(expiry) // Removal callbacks must be synchronous with the evicting @@ -66,20 +61,12 @@ L1Meta get(K key) { return barriers.getIfPresent(key); } - /** Records the barrier version only (freshness deadlines unchanged). */ + /** Records the highest barrier; unversioned values need no separate fence. */ void put(K key, Version version) { - barriers.asMap().merge(key, new L1Meta(version, 0L, 0L), - (existing, barrierOnly) -> new L1Meta( - existing.highestSeen() != null - && (barrierOnly.highestSeen() == null - || existing.highestSeen().compareTo(barrierOnly.highestSeen()) >= 0) - ? existing.highestSeen() : barrierOnly.highestSeen(), - existing.logicalDeadlineNanos(), existing.staleServeUntilNanos())); - } - - /** Records full metadata (barrier + deadlines) for {@code key}. */ - void put(K key, L1Meta meta) { - barriers.put(key, meta); + if (version == null) return; + barriers.asMap().merge(key, new L1Meta(version), + (existing, incoming) -> existing.highestSeen().compareTo(incoming.highestSeen()) >= 0 + ? existing : incoming); } /** Drops the metadata for {@code key}. */ diff --git a/tiercache-core/src/main/java/io/tiercache/internal/LoadClaim.java b/tiercache-core/src/main/java/io/tiercache/internal/LoadClaim.java new file mode 100644 index 0000000..413ba2b --- /dev/null +++ b/tiercache-core/src/main/java/io/tiercache/internal/LoadClaim.java @@ -0,0 +1,69 @@ +package io.tiercache.internal; + +import io.tiercache.spi.StoredEntry; + +import java.util.concurrent.CompletableFuture; +import java.util.function.Function; + +/** + * One local execution owner, shared by foreground loads and background refreshes. + * Only demand/skip arbitration is synchronized; I/O, waiting, and future + * completion always happen outside that short transition. + */ +final class LoadClaim { + + record Demand(Function loader, long deadlineNanos) { + } + + enum Kind { + RESULT, SKIPPED_REFRESH + } + + record Outcome(Kind kind, StoredEntry entry) { + static Outcome result(StoredEntry entry) { + return new Outcome<>(Kind.RESULT, entry); + } + + static Outcome skippedRefresh() { + return new Outcome<>(Kind.SKIPPED_REFRESH, null); + } + + boolean isSkipped() { + return kind == Kind.SKIPPED_REFRESH; + } + + StoredEntry resultEntry() { + if (isSkipped()) { + throw new IllegalStateException("A skipped refresh is not a cache result"); + } + return entry; + } + } + + final CompletableFuture> result = new CompletableFuture<>(); + private Demand foreground; + private boolean skipped; + + LoadClaim(Demand foreground) { + this.foreground = foreground; + } + + /** False only when the refresh already committed its skipped outcome. */ + synchronized boolean requireResult(Demand demand) { + if (skipped) { + return false; + } + if (foreground == null) { + foreground = demand; + } + return true; + } + + /** Commit a skip, or return the first foreground demand that prevents it. */ + synchronized Demand skipOrForeground() { + if (foreground == null) { + skipped = true; + } + return foreground; + } +} diff --git a/tiercache-core/src/main/java/io/tiercache/internal/LockProviderClosedException.java b/tiercache-core/src/main/java/io/tiercache/internal/LockProviderClosedException.java new file mode 100644 index 0000000..5863b7f --- /dev/null +++ b/tiercache-core/src/main/java/io/tiercache/internal/LockProviderClosedException.java @@ -0,0 +1,6 @@ +package io.tiercache.internal; + +/** Internal lifecycle outcome: coordination is closed, not contended or failed remotely. */ +public final class LockProviderClosedException extends IllegalStateException { + public LockProviderClosedException() { super("Lock provider is closed"); } +} diff --git a/tiercache-core/src/main/java/io/tiercache/spi/CacheMetricsListener.java b/tiercache-core/src/main/java/io/tiercache/spi/CacheMetricsListener.java index 6fb009a..7c5bac2 100644 --- a/tiercache-core/src/main/java/io/tiercache/spi/CacheMetricsListener.java +++ b/tiercache-core/src/main/java/io/tiercache/spi/CacheMetricsListener.java @@ -63,6 +63,15 @@ enum Direction { DROPPED } + /** Batched terminal publication outcomes, dispatched outside transport I/O threads. */ + default void onPublication(String cache, PublicationOutcome outcome, long count) { } + + /** Streams failure category; labels must never contain row IDs, keys or exception text. */ + enum StreamResult { DECODE_FAILED, APPLY_FAILED, ACK_FAILED, RESYNC_FAILED } + + /** Reports a failed stream operation, independently of publication/delivery counters. */ + default void onStreamFailure(String cache, StreamResult result) { } + /** * A listener that ignores every event; costs nothing on the hot path. * @@ -175,7 +184,7 @@ default void onL2OperationEnd(String cache, String operation, boolean hit, Objec } /** - * Wraps inbound invalidation processing (tracing span in the binder). + * Observes committed invalidation outside state monitors (binder tracing span). * * @param cache the cache name * @return an opaque observation handle, or {@code null} @@ -195,4 +204,13 @@ default Object onInvalidationStart(String cache) { */ default void onInvalidationEnd(String cache, Object handle) { } + /** + * Registers the triggered-recovery pending gauge. The supplier is a + * nonblocking state read. Closing the returned handle unregisters this + * source without reporting a successful recovery. + */ + default AutoCloseable registerRecovery(String cache, java.util.function.BooleanSupplier pending) { + return () -> { }; + } + } diff --git a/tiercache-core/src/main/java/io/tiercache/spi/CheckedRange.java b/tiercache-core/src/main/java/io/tiercache/spi/CheckedRange.java index ab160a5..cdce057 100644 --- a/tiercache-core/src/main/java/io/tiercache/spi/CheckedRange.java +++ b/tiercache-core/src/main/java/io/tiercache/spi/CheckedRange.java @@ -10,11 +10,13 @@ * *

Semantics per cursor kind: *

    - *
  • Non-beginning cursor: {@code rows} starts AT the cursor row - * (inclusive read); {@code startIntact} is {@code true} iff the - * first returned row IS the cursor row. If it is not (the cursor - * row was trimmed), prefix integrity is unconfirmable and the - * receiver must take the flush path.
  • + *
  • Non-beginning cursor: the response atomically validates the raw cursor + * row ID and reads the following rows. The already-accounted anchor may + * be omitted without decoding its payload (including a corrupt anchor + * covered by a safe reset). Legacy providers may retain a valid typed + * anchor as the first row; consumers skip that row by cursor ID. + * A missing/trimmed anchor makes {@code startIntact} false, including + * when the returned event list is empty.
  • *
  • Beginning cursor (an end cursor recorded against an empty * journal): there is no cursor row; {@code rows} starts from the * journal's first row and {@code startIntact} reflects the atomic diff --git a/tiercache-core/src/main/java/io/tiercache/spi/InvalidationGapHandler.java b/tiercache-core/src/main/java/io/tiercache/spi/InvalidationGapHandler.java new file mode 100644 index 0000000..4703632 --- /dev/null +++ b/tiercache-core/src/main/java/io/tiercache/spi/InvalidationGapHandler.java @@ -0,0 +1,13 @@ +package io.tiercache.spi; + +import java.util.concurrent.CompletionStage; + +/** Internal handshake: covered rows require a committed, still-current per-cache reset proof. */ +public interface InvalidationGapHandler { + /** Request bounded baseline-before-clear recovery. Failed/closed results authorize no ACK. */ + CompletionStage reset(String cache); + /** Baseline established while registering an empty/reset target, or null when unavailable. */ + RecoveryResult registrationBaseline(String cache); + /** Side-effect-free validity check; rejects proofs from another cache or superseded lifecycle. */ + boolean isCurrent(String cache, RecoveryResult result); +} diff --git a/tiercache-core/src/main/java/io/tiercache/spi/InvalidationHandler.java b/tiercache-core/src/main/java/io/tiercache/spi/InvalidationHandler.java index 4aed201..3662368 100644 --- a/tiercache-core/src/main/java/io/tiercache/spi/InvalidationHandler.java +++ b/tiercache-core/src/main/java/io/tiercache/spi/InvalidationHandler.java @@ -63,14 +63,34 @@ default void setEventListener(InvalidationEventListener listener) { /** * Called when L2 recovers after a circuit-breaker episode: the engine - * replays the missed journal range for all registered caches. L1 is - * never flushed here (only journal-window overflow flushes). + * triggers recovery for registered caches. Built-in recovery is asynchronous; + * use recoverAsync for completion. Unconfirmable history may require a clear. * * @since 0.1.0 */ default void onL2Recovery() { } + /** Installs the factory-owned recovery workers before any target is registered. */ + default void configureRecoveryExecutor(java.util.concurrent.ScheduledExecutorService executor) { } + + /** + * Nonblocking recovery completion. The compatibility adapter runs the legacy + * hook on the supplied workers; legacy hooks must throw on failed recovery. + * Built-in engines compose cache passes without waiting on their own pool. + */ + default java.util.concurrent.CompletionStage recoverAsync(java.util.concurrent.Executor executor) { + return java.util.concurrent.CompletableFuture.supplyAsync(() -> { + onL2Recovery(); + return true; + }, executor); + } + + /** Requests a baseline-before-clear reset; unsupported legacy handlers fail explicitly. */ + default java.util.concurrent.CompletionStage resetAsync(String cache) { + return java.util.concurrent.CompletableFuture.completedFuture(RecoveryResult.failed()); + } + /** * Shuts the engine down: subscriptions are cancelled and resources * released. diff --git a/tiercache-core/src/main/java/io/tiercache/spi/InvalidationTarget.java b/tiercache-core/src/main/java/io/tiercache/spi/InvalidationTarget.java index 75a0a35..449c371 100644 --- a/tiercache-core/src/main/java/io/tiercache/spi/InvalidationTarget.java +++ b/tiercache-core/src/main/java/io/tiercache/spi/InvalidationTarget.java @@ -53,4 +53,29 @@ public interface InvalidationTarget { default void applyUpdateL1(Object key, Object value, Version eventVersion) { evictL1IfNewer(key, eventVersion); // default: no payload application } + /** Local clear epoch used to reject recovery responses obtained before a clear. */ + default long recoveryGeneration() { return 0; } + + /** + * Applies one recovered row if its local clear epoch is still valid. + * Returns the next epoch, or -1 when superseded. Targets with concurrent + * clears override this together with recoveryGeneration for atomic checks. + */ + default long applyRecovery(io.tiercache.InvalidationMessage message, long expectedGeneration) { + if (recoveryGeneration() != expectedGeneration) return -1; + switch (message.type()) { + case INVALIDATE -> evictL1IfNewer(message.key(), message.version()); + case UPDATE -> applyUpdateL1(message.key(), message.payload(), message.version()); + case EVICT_ALL -> evictAllL1(); + } + return recoveryGeneration(); + } + + /** Clears against a reserved epoch; -1 means a concurrent clear superseded the reservation. */ + default long resetRecovery(long expectedGeneration) { + if (recoveryGeneration() != expectedGeneration) return -1; + evictAllL1(); + return recoveryGeneration(); + } + } diff --git a/tiercache-core/src/main/java/io/tiercache/spi/InvalidationTransport.java b/tiercache-core/src/main/java/io/tiercache/spi/InvalidationTransport.java index 68edc2e..5e5f762 100644 --- a/tiercache-core/src/main/java/io/tiercache/spi/InvalidationTransport.java +++ b/tiercache-core/src/main/java/io/tiercache/spi/InvalidationTransport.java @@ -24,6 +24,21 @@ public interface InvalidationTransport extends AutoCloseable { */ void publish(InvalidationMessage message); + /** + * Observes submission without waiting for delivery. A normal legacy void return + * is unconfirmed; exceptional completion or cancellation is classified as failed. + * Implementations must not block waiting for command completion. + */ + default java.util.concurrent.CompletionStage publishAsync(InvalidationMessage message) { + try { + publish(message); + return java.util.concurrent.CompletableFuture.completedFuture(PublicationOutcome.UNCONFIRMED); + } catch (RuntimeException e) { + return java.util.concurrent.CompletableFuture.failedFuture(e); + } + } + + /** * Subscribes to a cache's invalidation channel. The handler is invoked * asynchronously; per-channel ordering is preserved. @@ -46,6 +61,15 @@ public interface InvalidationTransport extends AutoCloseable { default void setReconnectListener(Runnable listener) { } + /** Installs optional recovery authorization; legacy transports ignore this hook. */ + default void setGapHandler(InvalidationGapHandler handler) { } + + /** Installs optional Streams diagnostics; callbacks must run outside state monitors. */ + default void setMetricsListener(CacheMetricsListener metrics) { } + + /** Side-effect-free capability: registration must establish an empty L1 before serving traffic. */ + default boolean requiresRegistrationReset() { return false; } + /** * Shuts the transport down: subscriptions are cancelled and resources * released. diff --git a/tiercache-core/src/main/java/io/tiercache/spi/JournalCorruptionException.java b/tiercache-core/src/main/java/io/tiercache/spi/JournalCorruptionException.java new file mode 100644 index 0000000..37f0841 --- /dev/null +++ b/tiercache-core/src/main/java/io/tiercache/spi/JournalCorruptionException.java @@ -0,0 +1,13 @@ +package io.tiercache.spi; + +/** Sanitized journal integrity failure shared by transport decoding and recovery. */ +public class JournalCorruptionException extends IllegalStateException { + private final String cache; + private final String rowId; + public JournalCorruptionException(String cache, String rowId, String category) { + super("Invalid invalidation row: cache=" + cache + ", row=" + rowId + ", reason=" + category); + this.cache = cache; this.rowId = rowId; + } + public String cache() { return cache; } + public String rowId() { return rowId; } +} diff --git a/tiercache-core/src/main/java/io/tiercache/spi/LocalCache.java b/tiercache-core/src/main/java/io/tiercache/spi/LocalCache.java index b570c05..1adbc8e 100644 --- a/tiercache-core/src/main/java/io/tiercache/spi/LocalCache.java +++ b/tiercache-core/src/main/java/io/tiercache/spi/LocalCache.java @@ -8,7 +8,11 @@ *

    Implementations must be thread-safe. TTLs are per entry: each * {@link #put} carries the effective TTL computed by the core (base TTL with * jitter already applied). Entries are opaque {@link StoredEntry} holders — - * implementations must store null-markers like any other entry. + * implementations must store null-markers like any other entry and retain + * the exact opaque holder, including its immutable local freshness state. + * Reconstructing entries from only value/version loses that state. Combining + * degradation stale serving and access expiry requires atomic replacement; + * unsupported providers are rejected when that engine cache is created. * *

    Note: a two-level cache is eventually consistent by design. * Implementations must not claim or attempt to provide strong consistency. @@ -68,4 +72,16 @@ public interface LocalCache { * @since 0.1.0 */ boolean setIfAbsent(K key, StoredEntry entry, Duration ttl); + /** Whether this provider supports atomic identity-checked entry and TTL replacement. */ + default boolean supportsAtomicReplace() { return false; } + + /** + * Replaces value and TTL only while the exact expected holder is present and unexpired. + * Compare by reference identity, never value equality. Must not insert an absent key. + * Required only for degradation stale serving combined with expire-after-access. + */ + default boolean replaceIfSame(K key, StoredEntry expected, StoredEntry replacement, Duration ttl) { + throw new UnsupportedOperationException("LocalCache does not support atomic replacement"); + } + } diff --git a/tiercache-core/src/main/java/io/tiercache/spi/PublicationOutcome.java b/tiercache-core/src/main/java/io/tiercache/spi/PublicationOutcome.java new file mode 100644 index 0000000..b4e2990 --- /dev/null +++ b/tiercache-core/src/main/java/io/tiercache/spi/PublicationOutcome.java @@ -0,0 +1,6 @@ +package io.tiercache.spi; + +/** Terminal publication classification; acknowledgement is not receiver application. */ +public enum PublicationOutcome { + ACKNOWLEDGED, FAILED, UNCONFIRMED, NOT_REQUIRED +} diff --git a/tiercache-core/src/main/java/io/tiercache/spi/RecoveryResult.java b/tiercache-core/src/main/java/io/tiercache/spi/RecoveryResult.java new file mode 100644 index 0000000..16ad210 --- /dev/null +++ b/tiercache-core/src/main/java/io/tiercache/spi/RecoveryResult.java @@ -0,0 +1,15 @@ +package io.tiercache.spi; + +/** Completion of an internal recovery attempt, not a delivery/linearizability guarantee. */ +public record RecoveryResult(Status status, String baseline, long generation) { + /** Distinguishes verified replay/reset from incomplete recovery or shutdown. */ + public enum Status { CAUGHT_UP, RESET_SAFE, NO_JOURNAL, FAILED, CLOSED } + /** True only for a completed replay or a configured safe clear. */ + public boolean succeeded() { + return status == Status.CAUGHT_UP || status == Status.RESET_SAFE || status == Status.NO_JOURNAL; + } + /** Failed attempts do not authorize row skipping or closing the breaker. */ + public static RecoveryResult failed() { return new RecoveryResult(Status.FAILED, null, -1); } + /** Shutdown is cancellation, never recovered coherence. */ + public static RecoveryResult closed() { return new RecoveryResult(Status.CLOSED, null, -1); } +} diff --git a/tiercache-core/src/main/java/io/tiercache/spi/RemoteCache.java b/tiercache-core/src/main/java/io/tiercache/spi/RemoteCache.java index 2893cd9..bf18d87 100644 --- a/tiercache-core/src/main/java/io/tiercache/spi/RemoteCache.java +++ b/tiercache-core/src/main/java/io/tiercache/spi/RemoteCache.java @@ -179,6 +179,49 @@ default void putTagged(K key, StoredEntry entry, Duration ttl, String[] tags) put(key, entry, ttl); } + /** + * Whether versioned tagged stores expose their outcome without losing the + * relationship between the accepted value and its tag metadata. This + * capability check must have no I/O or mutation, including during outages. + * + * @return true if versioned tagged outcomes are implemented + * @since 1.5.0 + */ + default boolean supportsTaggedWriteOutcomes() { + return false; + } + + /** + * Stores a tagged candidate and reports its outcome. Implementations that + * support version comparisons must couple acceptance, data, tags and any + * journal append atomically. LOST must have no such side effects. + * Unversioned legacy stores retain their unconditional semantics; a + * versioned legacy store is unsupported and performs no work. Implementing + * this method also requires advertising the capability above. Existing + * providers still link, but versioned tagged calls through core fail with + * a configuration error until both methods are implemented, even during + * OPEN-breaker local fallback. + * + *

    A transport exception is not a confirmed loss: a timed-out command + * may have committed remotely. No rollback or exactly-once retry follows + * from this outcome contract. + * + * @param key the key to store + * @param entry the candidate entry + * @param ttl the entry lifetime + * @param tags the replacement tags + * @return the accepted, rejected, or unsupported outcome; never null + * @since 1.5.0 + */ + default TaggedWriteOutcome putTaggedIfNewer(K key, StoredEntry entry, + Duration ttl, String[] tags) { + if (entry.version() != null) { + return TaggedWriteOutcome.UNSUPPORTED; + } + putTagged(key, entry, ttl, tags); + return TaggedWriteOutcome.WON; + } + /** * Keys currently tagged with {@code tag} (deserialized). Default: none. * diff --git a/tiercache-core/src/main/java/io/tiercache/spi/StoredEntry.java b/tiercache-core/src/main/java/io/tiercache/spi/StoredEntry.java index 18a96f6..0ba8a5b 100644 --- a/tiercache-core/src/main/java/io/tiercache/spi/StoredEntry.java +++ b/tiercache-core/src/main/java/io/tiercache/spi/StoredEntry.java @@ -18,7 +18,9 @@ * the write timestamp (millis since epoch) of the write that produced them, * so readers can classify freshness without extra round trips. Entries from * legacy frames (or from stores that do not track write time) have no write - * timestamp. + * timestamp. Local L1 copies may additionally carry immutable monotonic + * freshness deadlines. They belong to that exact holder and are never part + * of the Redis value frame. Equality remains object identity. * *

    Internal — not part of the supported API. Exchanged between the * cache levels and the engine. @@ -32,18 +34,38 @@ public final class StoredEntry { private static final StoredEntry NULL_MARKER = new StoredEntry<>(null, true, null, NO_WRITE_TIMESTAMP); + /** Immutable engine-local deadlines; never encoded into Redis frames. */ + public record LocalFreshness(long logicalDeadlineNanos, long staleServeUntilNanos, + long storeRetentionFloorNanos, long retentionUntilNanos, Version highestSeen) { } + + private final LocalFreshness localFreshness; private final V value; private final boolean nullMarker; private final Version version; private final long writeTimestampMillis; private StoredEntry(V value, boolean nullMarker, Version version, long writeTimestampMillis) { + this(value, nullMarker, version, writeTimestampMillis, null); + } + + private StoredEntry(V value, boolean nullMarker, Version version, long writeTimestampMillis, + LocalFreshness localFreshness) { + this.localFreshness = localFreshness; this.value = value; this.nullMarker = nullMarker; this.version = version; this.writeTimestampMillis = writeTimestampMillis; } + /** Returns engine-local state, or null for undecorated/remote entries. */ + public LocalFreshness localFreshness() { return localFreshness; } + + /** Creates an independent local copy, including for the shared null marker. */ + public StoredEntry withLocalFreshness(LocalFreshness freshness) { + return new StoredEntry<>(value, nullMarker, version, writeTimestampMillis, + java.util.Objects.requireNonNull(freshness)); + } + /** * Creates an unversioned value entry. * diff --git a/tiercache-core/src/main/java/io/tiercache/spi/TaggedWriteOutcome.java b/tiercache-core/src/main/java/io/tiercache/spi/TaggedWriteOutcome.java new file mode 100644 index 0000000..37da098 --- /dev/null +++ b/tiercache-core/src/main/java/io/tiercache/spi/TaggedWriteOutcome.java @@ -0,0 +1,16 @@ +package io.tiercache.spi; + +/** + * Result of a tagged store, distinct from transport failure or rejected L2 admission. + * Internal SPI; not part of the supported application API. + * + * @since 1.5.0 + */ +public enum TaggedWriteOutcome { + /** The store was accepted under the transport's configured versioning contract. */ + WON, + /** A newer entry or tombstone rejected the candidate without side effects. */ + LOST, + /** This transport cannot report a versioned tagged-write outcome. */ + UNSUPPORTED +} diff --git a/tiercache-core/src/test/java/io/tiercache/BreakerCallbackMonitorTest.java b/tiercache-core/src/test/java/io/tiercache/BreakerCallbackMonitorTest.java new file mode 100644 index 0000000..4ac19f3 --- /dev/null +++ b/tiercache-core/src/test/java/io/tiercache/BreakerCallbackMonitorTest.java @@ -0,0 +1,22 @@ +package io.tiercache; + +import io.tiercache.internal.CircuitBreaker; +import org.junit.jupiter.api.Test; +import java.time.Duration; +import java.util.concurrent.atomic.AtomicReference; +import static org.junit.jupiter.api.Assertions.*; + +class BreakerCallbackMonitorTest { + @Test void transitionCallbacksRunOutsideBreakerMonitor() { + var owner = new AtomicReference(); + var held = new java.util.ArrayList(); + var breaker = new CircuitBreaker(new CircuitBreaker.Config(10, 0.5, 1, Duration.ZERO, 1), + new CircuitBreaker.Listener() { + public void onOpen() { held.add(Thread.holdsLock(owner.get())); } + public void onClose() { held.add(Thread.holdsLock(owner.get())); } + }); + owner.set(breaker); breaker.onFailure(); + assertTrue(breaker.tryAcquire()); breaker.onSuccess(); + assertEquals(java.util.List.of(false, false), held); + } +} diff --git a/tiercache-core/src/test/java/io/tiercache/ConsistencyRaceTest.java b/tiercache-core/src/test/java/io/tiercache/ConsistencyRaceTest.java index 72f499b..57bf7f3 100644 --- a/tiercache-core/src/test/java/io/tiercache/ConsistencyRaceTest.java +++ b/tiercache-core/src/test/java/io/tiercache/ConsistencyRaceTest.java @@ -6,6 +6,7 @@ import io.tiercache.spi.LocalCache; import io.tiercache.spi.RemoteCache; import io.tiercache.spi.StoredEntry; +import io.tiercache.spi.TaggedWriteOutcome; import io.tiercache.testkit.CountingLocalCache; import io.tiercache.testkit.InMemoryRemoteCache; import org.junit.jupiter.api.Test; @@ -52,6 +53,15 @@ public StoredEntry get(String key) { return entry; } + @Override + public boolean supportsAtomicReplace() { return true; } + + @Override + public boolean replaceIfSame(String key, StoredEntry expected, + StoredEntry replacement, Duration ttl) { + return delegate.replaceIfSame(key, expected, replacement, ttl); + } + @Override public void put(String key, StoredEntry entry, Duration ttl) { delegate.put(key, entry, ttl); @@ -439,11 +449,14 @@ public void put(String key, StoredEntry entry, Duration ttl) { } @Override - public void putTagged(String key, StoredEntry entry, Duration ttl, + public boolean supportsTaggedWriteOutcomes() { return true; } + + @Override + public TaggedWriteOutcome putTaggedIfNewer(String key, StoredEntry entry, Duration ttl, String[] tags) { taggedWritten.countDown(); awaitQuietly(releaseTagged); - delegate.putTagged(key, entry, ttl, tags); + return delegate.putTaggedIfNewer(key, entry, ttl, tags); } @Override diff --git a/tiercache-core/src/test/java/io/tiercache/DegradationTest.java b/tiercache-core/src/test/java/io/tiercache/DegradationTest.java index 542eb05..3616877 100644 --- a/tiercache-core/src/test/java/io/tiercache/DegradationTest.java +++ b/tiercache-core/src/test/java/io/tiercache/DegradationTest.java @@ -164,7 +164,7 @@ void recoveryKeepsL1() { @Test void recoveryTriggersInvalidationReplayHookBeforeRecovered() throws Exception { FailingRemoteCache l2 = new FailingRemoteCache<>(); - List events = new ArrayList<>(); + List events = new java.util.concurrent.CopyOnWriteArrayList<>(); InvalidationHandler handler = new InvalidationHandler() { @Override public void onLocalWrite(String cache, Object key, Version version, @@ -204,6 +204,8 @@ public void onRecovered() { Thread.sleep(100); cache.getOrCompute("k", key -> "v"); // probe -> close assertFalse(factory.isDegraded()); + long deadline = System.nanoTime() + java.util.concurrent.TimeUnit.SECONDS.toNanos(5); + while (events.size() < 2 && System.nanoTime() < deadline) Thread.sleep(5); assertEquals(List.of("replay", "recovered"), events, "journal replay runs before the recovered signal"); factory.close(); diff --git a/tiercache-core/src/test/java/io/tiercache/FreshnessLifetimeRegressionTest.java b/tiercache-core/src/test/java/io/tiercache/FreshnessLifetimeRegressionTest.java new file mode 100644 index 0000000..0d6abe7 --- /dev/null +++ b/tiercache-core/src/test/java/io/tiercache/FreshnessLifetimeRegressionTest.java @@ -0,0 +1,83 @@ +package io.tiercache; + +import io.tiercache.internal.CaffeineLocalCache; +import io.tiercache.internal.CircuitBreaker; +import io.tiercache.internal.DefaultTierCache; +import io.tiercache.spi.LocalCache; +import io.tiercache.spi.StoredEntry; +import io.tiercache.testkit.InMemoryRemoteCache; +import org.junit.jupiter.api.Test; +import java.time.Duration; +import java.util.concurrent.atomic.AtomicInteger; +import static org.junit.jupiter.api.Assertions.*; + +/** Reproductions for independent metadata expiry and L1 eviction during access refresh. */ +class FreshnessLifetimeRegressionTest { + private static CacheSettings settings(Duration access) { + return new CacheSettings(100, Duration.ofMinutes(30), access, Duration.ofHours(1), 0, + NullPolicy.deny(), InvalidationMode.INVALIDATE, 65536, + Duration.ZERO, false, Duration.ofSeconds(1), Duration.ofMinutes(30)); + } + + @Test + void expiryOfIndependentFenceCannotInvalidateRetainedFreshValue() throws Exception { + Duration previous = DefaultTierCache.L1_META_EXPIRY; + DefaultTierCache.L1_META_EXPIRY = Duration.ofMillis(20); + try { + var settings = settings(null); + var l1 = new CaffeineLocalCache(settings); + var breaker = new CircuitBreaker(new CircuitBreaker.Config(1, 1, 1, Duration.ofHours(1), 1), + new CircuitBreaker.Listener() { + public void onOpen() { } + public void onClose() { } + }); + var cache = new DefaultTierCache<>("c", l1, new InMemoryRemoteCache(), + settings, true, null, null, null, null, breaker, io.tiercache.spi.CacheMetricsListener.NOOP); + cache.put("hot", "retained"); + Thread.sleep(80); // Accelerate only the independent fence's expiry, not value TTL. + assertNotNull(l1.get("hot")); + breaker.onFailure(); + assertEquals("retained", cache.getOrCompute("hot", key -> { + fail("fresh retained value must not call the source after fence expiry"); + return null; + })); + } finally { + DefaultTierCache.L1_META_EXPIRY = previous; + } + } + + @Test + void accessRefreshCannotReinsertAnEntryRemovedAfterItsRead() { + var settings = settings(Duration.ofMinutes(20)); + var delegate = new CaffeineLocalCache(settings); + var reads = new AtomicInteger(); + LocalCache evicting = new LocalCache<>() { + public StoredEntry get(String key) { + var result = delegate.get(key); + // Evict after the engine's second (stripe-protected) read. This models + // removal by the L1 itself, which does not acquire the engine stripe. + if (reads.incrementAndGet() == 2) delegate.evict(key); + return result; + } + public boolean supportsAtomicReplace() { return true; } + public boolean replaceIfSame(String key, StoredEntry expected, StoredEntry replacement, Duration ttl) { + return delegate.replaceIfSame(key, expected, replacement, ttl); + } + public void put(String key, StoredEntry value, Duration ttl) { delegate.put(key, value, ttl); } + public boolean setIfAbsent(String key, StoredEntry value, Duration ttl) { + return delegate.setIfAbsent(key, value, ttl); + } + public void evict(String key) { delegate.evict(key); } + public void clear() { delegate.clear(); } + }; + var breaker = new CircuitBreaker(new CircuitBreaker.Config(1, 1, 1, Duration.ofHours(1), 1), + new CircuitBreaker.Listener() { public void onOpen() { } public void onClose() { } }); + var cache = new DefaultTierCache<>("c", evicting, new InMemoryRemoteCache(), + settings, true, null, null, null, null, breaker, io.tiercache.spi.CacheMetricsListener.NOOP); + cache.put("hot", "old"); + breaker.onFailure(); // Prevent a legitimate L2 re-warm from obscuring access refresh. + reads.set(0); + cache.get("hot"); + assertNull(delegate.get("hot"), "access refresh must not resurrect a removed L1 entry"); + } +} diff --git a/tiercache-core/src/test/java/io/tiercache/LoadStoreRaceTest.java b/tiercache-core/src/test/java/io/tiercache/LoadStoreRaceTest.java index 741e9a3..5bca072 100644 --- a/tiercache-core/src/test/java/io/tiercache/LoadStoreRaceTest.java +++ b/tiercache-core/src/test/java/io/tiercache/LoadStoreRaceTest.java @@ -4,6 +4,7 @@ import io.tiercache.spi.LocalCache; import io.tiercache.spi.RemoteCache; import io.tiercache.spi.StoredEntry; +import io.tiercache.spi.TaggedWriteOutcome; import io.tiercache.testkit.CountingLocalCache; import io.tiercache.testkit.InMemoryRemoteCache; import org.junit.jupiter.api.Test; @@ -90,8 +91,13 @@ public List keysByTag(String tag) { } @Override - public void putTagged(String key, StoredEntry entry, Duration ttl, String[] tags) { - delegate.putTagged(key, entry, ttl, tags); + public boolean supportsTaggedWriteOutcomes() { + return true; + } + + @Override + public TaggedWriteOutcome putTaggedIfNewer(String key, StoredEntry entry, Duration ttl, String[] tags) { + return delegate.putTaggedIfNewer(key, entry, ttl, tags); } } diff --git a/tiercache-core/src/test/java/io/tiercache/TaggedWriteOutcomeTest.java b/tiercache-core/src/test/java/io/tiercache/TaggedWriteOutcomeTest.java new file mode 100644 index 0000000..6046676 --- /dev/null +++ b/tiercache-core/src/test/java/io/tiercache/TaggedWriteOutcomeTest.java @@ -0,0 +1,248 @@ +package io.tiercache; + +import io.tiercache.internal.CircuitBreaker; +import io.tiercache.internal.CircuitBreakerRemoteCache; +import io.tiercache.internal.DefaultTierCache; +import io.tiercache.internal.L2UnavailableException; +import io.tiercache.spi.*; +import io.tiercache.testkit.CountingLocalCache; +import org.junit.jupiter.api.Test; + +import java.time.Duration; +import java.util.UUID; +import java.util.concurrent.atomic.AtomicInteger; + +import static io.tiercache.spi.TaggedWriteOutcome.*; +import static org.junit.jupiter.api.Assertions.*; + +class TaggedWriteOutcomeTest { + private static final Duration TTL = Duration.ofMinutes(1); + private static final Version NEWER = new Version(Long.MAX_VALUE, UUID.randomUUID()); + + private static class Legacy implements RemoteCache { + int writes; + @Override public StoredEntry get(String key) { return null; } + @Override public void put(String key, StoredEntry entry, Duration ttl) { writes++; } + @Override public void evict(String key) { } + @Override public void clear() { } + @Override public boolean setIfAbsent(String key, StoredEntry entry, Duration ttl) { return false; } + } + + private static final class Remote extends Legacy { + TaggedWriteOutcome outcome = WON; + StoredEntry current; + int reads; + Runnable duringWrite = () -> { }; + @Override public boolean supportsTaggedWriteOutcomes() { return true; } + @Override public StoredEntry get(String key) { reads++; return current; } + @Override public TaggedWriteOutcome putTaggedIfNewer(String key, StoredEntry entry, + Duration ttl, String[] tags) { + writes++; + duringWrite.run(); + return outcome; + } + } + + private static CircuitBreaker breaker(Duration delay, int probes) { + return new CircuitBreaker(new CircuitBreaker.Config(10, 0.5, 1, delay, probes), + new CircuitBreaker.Listener() { + @Override public void onOpen() { } + @Override public void onClose() { } + }); + } + + private static final class Harness { + final CountingLocalCache local = new CountingLocalCache<>(); + final CountingLocalCache receiverLocal = new CountingLocalCache<>(); + final AtomicInteger publications = new AtomicInteger(); + final DefaultTierCache cache; + Harness(RemoteCache remote, CircuitBreaker breaker, boolean versioned) { + var settings = new CacheSettings(100, Duration.ofSeconds(10), null, TTL, 0, + NullPolicy.deny(), InvalidationMode.UPDATE, 65536); + var receiver = new DefaultTierCache("c", receiverLocal, remote, + settings, true, null, null, new VersionGenerator(), null); + InvalidationHandler handler = new InvalidationHandler() { + @Override public void onLocalWrite(String c, Object k, Version v, InvalidationMessage.Type t) { + publications.incrementAndGet(); + } + @Override public void onLocalUpdate(String c, Object k, Object value, Version v) { + publications.incrementAndGet(); + receiver.applyUpdateL1(k, value, v); + } + @Override public void registerTarget(String c, InvalidationTarget t) { } + @Override public void close() { } + }; + cache = new DefaultTierCache<>("c", local, new CircuitBreakerRemoteCache<>(remote, breaker), + settings, true, null, null, versioned ? new VersionGenerator() : null, + handler, breaker, CacheMetricsListener.NOOP); + } + } + + @Test void lostWriteConvergesOnceWithoutPublishingToEmptyUpdateReceiverOrLoading() { + Remote remote = new Remote(); + remote.outcome = LOST; + remote.current = StoredEntry.ofValue("winner", NEWER); + Harness h = new Harness(remote, breaker(TTL, 1), true); + h.cache.put("k", "loser", "old"); + assertEquals("winner", h.local.get("k").value()); + assertEquals(1, remote.reads); + assertEquals(1, remote.writes); + assertEquals(0, h.publications.get()); + assertNull(h.receiverLocal.get("k")); + assertEquals("winner", h.cache.getOrCompute("k", key -> { throw new AssertionError("loader invoked"); })); + assertEquals(1, remote.reads); + } + + @Test void lostWriteWithAbsentWinnerEvictsLocalCandidate() { + Remote remote = new Remote(); remote.outcome = LOST; + Harness h = new Harness(remote, breaker(TTL, 1), true); + h.local.put("k", StoredEntry.ofValue("stale"), TTL); + h.cache.put("k", "loser", "tag"); + assertNull(h.local.get("k")); + assertNull(h.receiverLocal.get("k")); + assertEquals(1, remote.reads); + assertEquals(0, h.publications.get()); + } + + @Test void winningWriteWarmsAndPublishes() { + Remote remote = new Remote(); Harness h = new Harness(remote, breaker(TTL, 1), true); + h.cache.put("k", "winner", "tag"); + assertEquals("winner", h.local.get("k").value()); + assertEquals("winner", h.receiverLocal.get("k").value()); + assertEquals(1, h.publications.get()); + assertEquals(0, remote.reads); + } + + @Test void winningWriteCannotWarmAcrossGenerationChange() { + Remote remote = new Remote(); Harness h = new Harness(remote, breaker(TTL, 1), true); + remote.duringWrite = h.cache::evictAllL1; + h.cache.put("k", "winner", "tag"); + assertNull(h.local.get("k")); + assertEquals(1, h.publications.get(), "the remote win is still confirmed"); + } + + @Test void winningWriteCannotWarmAcrossNewerKeyBarrier() { + Remote remote = new Remote(); Harness h = new Harness(remote, breaker(TTL, 1), true); + remote.duringWrite = () -> h.cache.evictL1IfNewer("k", NEWER); + h.cache.put("k", "winner", "tag"); + assertNull(h.local.get("k")); + assertEquals(1, h.publications.get()); + } + + @Test void openBreakerFallsBackLocallyWithoutAttemptOrPublication() { + Remote remote = new Remote(); CircuitBreaker b = breaker(TTL, 1); b.onFailure(); + Harness h = new Harness(remote, b, true); + h.cache.put("k", "local", "tag"); + assertEquals("local", h.local.get("k").value()); + assertEquals(0, remote.writes); + assertEquals(0, h.publications.get()); + } + + @Test void exhaustedHalfOpenFallsBackLocally() { + Remote remote = new Remote(); CircuitBreaker b = breaker(Duration.ZERO, 1); b.onFailure(); + CircuitBreaker.Permit occupied = b.tryAcquirePermit(); assertNotNull(occupied); + Harness h = new Harness(remote, b, true); + h.cache.put("k", "local", "tag"); + assertEquals("local", h.local.get("k").value()); + assertEquals(0, remote.writes); + assertEquals(0, h.publications.get()); + assertEquals(BreakerState.HALF_OPEN, b.state()); occupied.cancel(); + } + + @Test void admittedFailureOpensBreakerAndFallsBackLocally() { + Remote remote = new Remote(); CircuitBreaker b = breaker(TTL, 1); + remote.duringWrite = () -> { throw new IllegalStateException("outage"); }; + Harness h = new Harness(remote, b, true); h.cache.put("k", "local", "tag"); + assertEquals(BreakerState.OPEN, b.state()); + assertEquals("local", h.local.get("k").value()); + assertEquals(1, remote.writes); + assertEquals(0, h.publications.get()); + } + + @Test void eachTaggedAttemptUsesExactlyOneProbe() { + Remote remote = new Remote(); CircuitBreaker b = breaker(Duration.ZERO, 2); b.onFailure(); + Harness h = new Harness(remote, b, true); + h.cache.put("k", "first", "tag"); + assertEquals(BreakerState.HALF_OPEN, b.state()); + h.cache.put("k", "second", "tag"); + assertEquals(BreakerState.CLOSED, b.state()); + assertEquals(2, remote.writes); + } + + @Test void losingAttemptAndConvergenceReadAreSeparateSuccessfulProbes() { + Remote remote = new Remote(); remote.outcome = LOST; + remote.current = StoredEntry.ofValue("winner", NEWER); + CircuitBreaker b = breaker(Duration.ZERO, 2); b.onFailure(); + Harness h = new Harness(remote, b, true); h.cache.put("k", "loser", "tag"); + assertEquals(BreakerState.CLOSED, b.state()); + assertEquals(1, remote.writes); assertEquals(1, remote.reads); + } + + @Test void legacyVersionedSpiIsUnsupportedEvenWhileOpen() { + for (boolean open : new boolean[]{false, true}) { + Legacy legacy = new Legacy(); CircuitBreaker b = breaker(TTL, 1); + if (open) b.onFailure(); + BreakerState before = b.state(); Harness h = new Harness(legacy, b, true); + assertThrows(CacheConfigurationException.class, () -> h.cache.put("k", "bad", "tag")); + assertEquals(before, b.state()); assertNull(h.local.get("k")); + assertEquals(0, legacy.writes); assertEquals(0, h.publications.get()); + assertEquals(UNSUPPORTED, new CircuitBreakerRemoteCache<>(legacy, b) + .putTaggedIfNewer("k", StoredEntry.ofValue("bad", NEWER), TTL, new String[]{"tag"})); + } + } + + @Test void unversionedLegacySpiDelegatesAndWarms() { + Legacy legacy = new Legacy(); Harness h = new Harness(legacy, breaker(TTL, 1), false); + h.cache.put("k", "ok", "tag"); + assertEquals(1, legacy.writes); assertEquals("ok", h.local.get("k").value()); + assertEquals(0, h.publications.get()); + assertEquals(UNSUPPORTED, legacy.putTaggedIfNewer("k", StoredEntry.ofValue("bad", NEWER), TTL, new String[]{"tag"})); + assertEquals(1, legacy.writes); + } + + @Test void unsupportedResultReturnsProbeWithoutSuccessOrFailure() { + Remote remote = new Remote(); remote.outcome = UNSUPPORTED; + CircuitBreaker b = breaker(Duration.ZERO, 1); b.onFailure(); Harness h = new Harness(remote, b, true); + for (int i = 0; i < 3; i++) { + assertThrows(CacheConfigurationException.class, () -> h.cache.put("k", "bad", "tag")); + assertEquals(BreakerState.HALF_OPEN, b.state()); + } + assertEquals(3, remote.writes); assertNull(h.local.get("k")); + remote.outcome = WON; h.cache.put("k", "ok", "tag"); + assertEquals(BreakerState.CLOSED, b.state()); + } + + @Test void neutralCompletionDoesNotPolluteClosedFailureWindow() { + Remote remote = new Remote(); remote.outcome = UNSUPPORTED; + CircuitBreaker b = new CircuitBreaker(new CircuitBreaker.Config(10, 0.5, 2, TTL, 1), + new CircuitBreaker.Listener() { public void onOpen() { } public void onClose() { } }); + Harness h = new Harness(remote, b, true); + assertThrows(CacheConfigurationException.class, () -> h.cache.put("k", "bad", "tag")); + b.onFailure(); assertEquals(BreakerState.CLOSED, b.state()); + b.onFailure(); assertEquals(BreakerState.OPEN, b.state()); + } + + @Test void permitIsOnceOnlyAndCannotRetireAnotherEpisode() { + CircuitBreaker b = breaker(Duration.ZERO, 1); b.onFailure(); + var old = b.tryAcquirePermit(); assertNotNull(old); b.onFailure(); + var current = b.tryAcquirePermit(); assertNotNull(current); + old.cancel(); assertNull(b.tryAcquirePermit()); + current.cancel(); current.success(); current.failure(); + assertEquals(BreakerState.HALF_OPEN, b.state()); + var next = b.tryAcquirePermit(); assertNotNull(next); next.success(); + assertEquals(BreakerState.CLOSED, b.state()); + } + + @Test void nestedRejectionAndErrorRetireAdmissionNeutrally() { + Remote remote = new Remote(); CircuitBreaker b = breaker(Duration.ZERO, 1); b.onFailure(); + Harness h = new Harness(remote, b, true); + remote.duringWrite = () -> { throw L2UnavailableException.OPEN; }; + h.cache.put("k", "local", "tag"); + assertEquals(BreakerState.HALF_OPEN, b.state()); + remote.duringWrite = () -> { throw new AssertionError("fatal"); }; + assertThrows(AssertionError.class, () -> h.cache.put("k", "bad", "tag")); + remote.duringWrite = () -> { }; h.cache.put("k", "ok", "tag"); + assertEquals(BreakerState.CLOSED, b.state()); + assertEquals(1, h.publications.get()); + } +} diff --git a/tiercache-core/src/test/java/io/tiercache/internal/AsyncBreakerRecoveryTest.java b/tiercache-core/src/test/java/io/tiercache/internal/AsyncBreakerRecoveryTest.java new file mode 100644 index 0000000..3f75c97 --- /dev/null +++ b/tiercache-core/src/test/java/io/tiercache/internal/AsyncBreakerRecoveryTest.java @@ -0,0 +1,160 @@ +package io.tiercache.internal; + +import io.tiercache.*; +import io.tiercache.spi.*; +import io.tiercache.testkit.*; +import org.junit.jupiter.api.Test; +import java.time.Duration; +import java.util.*; +import java.util.concurrent.*; +import java.util.concurrent.atomic.*; +import static org.junit.jupiter.api.Assertions.*; + +class AsyncBreakerRecoveryTest { + static final class Rig { + final AtomicLong now = new AtomicLong(1); + final List notifications = new CopyOnWriteArrayList<>(); + final Queue tasks = new ConcurrentLinkedQueue<>(); + final Queue> recoveries = new ConcurrentLinkedQueue<>(); + final CircuitBreaker breaker = new CircuitBreaker(new CircuitBreaker.Config(10, 0.5, 1, Duration.ZERO, 1), + new CircuitBreaker.Listener() { + public void onOpen() { notifications.add("open:" + Thread.holdsLock(breaker)); } + public void onClose() { notifications.add("close:" + Thread.holdsLock(breaker)); } + }, now::get); + Rig() { breaker.configureRecovery(tasks::add, () -> recoveries.remove()); } + CompletableFuture probe() { + CompletableFuture result = new CompletableFuture<>(); recoveries.add(result); + breaker.onFailure(); + var permit = breaker.tryAcquirePermit(); assertNotNull(permit); permit.success(); + assertEquals(BreakerState.HALF_OPEN, breaker.state()); + assertNull(breaker.tryAcquirePermit()); + tasks.remove().run(); + return result; + } + } + + @Test void successfulProbeReturnsBeforeRecoveryAndClosesOnlyOnCompletion() { + Rig r = new Rig(); var completion = r.probe(); + assertEquals(List.of("open:false"), r.notifications); + assertFalse(r.breaker.isOpen()); assertFalse(r.breaker.tryAcquire()); + completion.complete(true); + assertEquals(BreakerState.CLOSED, r.breaker.state()); + assertEquals(List.of("open:false", "close:false"), r.notifications); + assertTrue(r.breaker.tryAcquire()); + } + + @Test void failedRecoveryReopensWithExponentialMinimumDelay() { + Rig r = new Rig(); r.probe().complete(false); + assertEquals(BreakerState.OPEN, r.breaker.state()); + assertFalse(r.breaker.tryAcquire()); + r.now.addAndGet(Duration.ofMillis(999).toNanos()); assertTrue(r.breaker.isOpen()); + r.now.addAndGet(Duration.ofMillis(1).toNanos()); + var completion = new CompletableFuture(); r.recoveries.add(completion); + var permit = r.breaker.tryAcquirePermit(); assertNotNull(permit); permit.success(); r.tasks.remove().run(); + completion.completeExceptionally(new IllegalStateException("baseline unavailable")); + r.now.addAndGet(Duration.ofSeconds(1).toNanos()); assertFalse(r.breaker.tryAcquire()); + r.now.addAndGet(Duration.ofSeconds(1).toNanos()); assertTrue(r.breaker.tryAcquire()); + } + + @Test void lateRecoveryCannotCloseAReopenedEpisode() { + Rig r = new Rig(); var old = r.probe(); + r.breaker.onFailure(); old.complete(true); + assertFalse(r.notifications.stream().anyMatch(s -> s.startsWith("close"))); + assertEquals(BreakerState.HALF_OPEN, r.breaker.state()); + } + + @Test void detachReleasesPendingReservationWithoutRecoveredNotification() { + Rig r = new Rig(); var old = r.probe(); r.breaker.detachRecovery(); + old.complete(true); assertEquals(List.of("open:false"), r.notifications); + var realProbe = r.breaker.tryAcquirePermit(); assertNotNull(realProbe); realProbe.success(); + assertEquals(BreakerState.CLOSED, r.breaker.state()); + assertEquals(List.of("open:false", "close:false"), r.notifications); + } + + @Test void detachPreservesOpenWaitAndSkipsQueuedRetiredHook() { + var now = new AtomicLong(1); var calls = new AtomicInteger(); + var tasks = new ArrayDeque(); + var b = new CircuitBreaker(new CircuitBreaker.Config(2, 1, 1, Duration.ofSeconds(5), 1), + new CircuitBreaker.Listener() { public void onOpen() { } public void onClose() { } }, now::get); + b.configureRecovery(tasks::add, () -> { calls.incrementAndGet(); return CompletableFuture.completedFuture(true); }); + b.onFailure(); now.addAndGet(Duration.ofSeconds(2).toNanos()); b.detachRecovery(); + assertFalse(b.tryAcquire()); now.addAndGet(Duration.ofSeconds(3).toNanos()); assertTrue(b.tryAcquire()); + b.configureRecovery(tasks::add, () -> { calls.incrementAndGet(); return CompletableFuture.completedFuture(true); }); + b.onSuccess(); b.detachRecovery(); tasks.remove().run(); assertEquals(0, calls.get()); + } + + @Test void rejectedWorkerAndThrowingHookFailRecoveryInsteadOfClosing() { + Rig rejected = new Rig(); + rejected.breaker.configureRecovery(r -> { throw new RejectedExecutionException(); }, () -> CompletableFuture.completedFuture(true)); + rejected.breaker.onFailure(); assertTrue(rejected.breaker.tryAcquire()); rejected.breaker.onSuccess(); + assertEquals(BreakerState.OPEN, rejected.breaker.state()); + Rig throwing = new Rig(); + throwing.breaker.configureRecovery(throwing.tasks::add, () -> { throw new AssertionError("hook"); }); + throwing.breaker.onFailure(); assertTrue(throwing.breaker.tryAcquire()); throwing.breaker.onSuccess(); + throwing.tasks.remove().run(); assertEquals(BreakerState.OPEN, throwing.breaker.state()); + } + + @Test void observersCannotTurnSuccessfulCallsIntoInfrastructureFailures() { + var b = new CircuitBreaker(new CircuitBreaker.Config(2, 1, 1, Duration.ZERO, 1), + new CircuitBreaker.Listener() { + public void onOpen() { throw new IllegalStateException("observer"); } + public void onClose() { throw new AssertionError("observer"); } + }); + assertDoesNotThrow(b::onFailure); + var permit = b.tryAcquirePermit(); assertNotNull(permit); + assertDoesNotThrow(permit::success); assertEquals(BreakerState.CLOSED, b.state()); + } + + @Test void permitsIgnoreLateFailuresAndNeutralCompletionDoesNotConsumeProbes() { + Rig r = new Rig(); var closedCall = r.breaker.tryAcquirePermit(); + r.breaker.onFailure(); closedCall.failure(); + var neutral = r.breaker.tryAcquirePermit(); assertNotNull(neutral); neutral.cancel(); neutral.success(); + var real = r.breaker.tryAcquirePermit(); assertNotNull(real); + r.recoveries.add(CompletableFuture.completedFuture(true)); real.success(); real.failure(); + r.tasks.remove().run(); assertEquals(BreakerState.CLOSED, r.breaker.state()); + } + + @Test void recoveryHooksRunOffTheBusinessThreadAndSurvivingCacheProgressesAfterClose() throws Exception { + var started = new CountDownLatch(1); var release = new CountDownLatch(1); + var hookThread = new AtomicReference(); var recovered = new AtomicInteger(); + var remote = new FailingRemoteCache(); + InvalidationHandler handler = new InvalidationHandler() { + public void onLocalWrite(String c, Object k, Version v, InvalidationMessage.Type t) { } + public void registerTarget(String c, InvalidationTarget t) { } + public void onL2Recovery() { + hookThread.set(Thread.currentThread()); started.countDown(); + try { release.await(5, TimeUnit.SECONDS); } catch (InterruptedException e) { Thread.currentThread().interrupt(); } + } + public void close() { } + }; + var factory = TierCacheFactory.builder().remoteCache(remote).invalidation(v -> handler) + .circuitBreakerConfig(new CircuitBreaker.Config(2, 1, 1, Duration.ZERO, 1)) + .degradationListener(new DegradationListener() { public void onRecovered() { recovered.incrementAndGet(); } }).build(); + try { + var cache = factory.getCache("c"); remote.fail(); cache.get("outage"); remote.heal(); + assertNull(cache.get("probe")); assertTrue(started.await(5, TimeUnit.SECONDS)); + assertNotSame(Thread.currentThread(), hookThread.get()); + assertEquals(BreakerState.HALF_OPEN, factory.breakerState()); + factory.close(); + remote.put("after", StoredEntry.ofValue("available"), Duration.ofMinutes(1)); + assertEquals("available", cache.get("after")); + assertEquals(BreakerState.CLOSED, factory.breakerState()); + release.countDown(); assertEquals(0, recovered.get(), "closed coherence must not announce recovery"); + } finally { release.countDown(); factory.close(); } + } + + @Test void recoveryTargetRejectsUpdateAndResetFromAnOlderClearEpoch() { + var cache = new DefaultTierCache("c", new CountingLocalCache<>(), new CountingRemoteCache<>(), + CacheSettings.defaults(), true, null, null, new VersionGenerator(), null); + long old = cache.recoveryGeneration(); cache.evictAllL1(); + var v = new Version(10, UUID.randomUUID()); + var update = new InvalidationMessage("c", "k", v, v.instanceId(), InvalidationMessage.Type.UPDATE, "old"); + assertEquals(-1, cache.applyRecovery(update, old)); assertEquals(-1, cache.resetRecovery(old)); + long current = cache.recoveryGeneration(); assertEquals(current, cache.applyRecovery(update, current)); + assertEquals("old", cache.get("k")); + var invalidation = new InvalidationMessage("c", "k", new Version(11, v.instanceId()), v.instanceId(), InvalidationMessage.Type.INVALIDATE); + assertEquals(current, cache.applyRecovery(invalidation, current)); assertNull(cache.get("k")); + var clear = new InvalidationMessage("c", null, v, v.instanceId(), InvalidationMessage.Type.EVICT_ALL); + assertEquals(current + 1, cache.applyRecovery(clear, current)); + } +} diff --git a/tiercache-core/src/test/java/io/tiercache/internal/AuxiliaryLifecycleTest.java b/tiercache-core/src/test/java/io/tiercache/internal/AuxiliaryLifecycleTest.java new file mode 100644 index 0000000..8291e69 --- /dev/null +++ b/tiercache-core/src/test/java/io/tiercache/internal/AuxiliaryLifecycleTest.java @@ -0,0 +1,210 @@ +package io.tiercache.internal; + +import io.tiercache.*; +import io.tiercache.spi.*; +import io.tiercache.testkit.*; +import org.junit.jupiter.api.Test; +import java.time.Duration; +import java.util.*; +import java.util.concurrent.*; +import java.util.concurrent.atomic.*; +import java.util.function.Function; +import static org.junit.jupiter.api.Assertions.*; + +class AuxiliaryLifecycleTest { + static class RejectingScheduler extends ScheduledThreadPoolExecutor { + final boolean closing; + final List events; + final RejectedExecutionException failure = new RejectedExecutionException("test"); + RejectingScheduler(boolean closing, List events) { super(1); this.closing=closing; this.events=events; } + @Override public ScheduledFuture scheduleAtFixedRate(Runnable r,long a,long b,TimeUnit u) { + events.add("schedule"); if (closing) shutdown(); throw failure; + } + } + @Test void watchdogRejectionReleasesBeforeFallbackInBothPaths() throws Exception { + for (boolean refresh : new boolean[]{false,true}) for (boolean failLoader : new boolean[]{false,true}) { + var events = new ArrayList(); var failure = new IllegalArgumentException("loader"); + var scheduler = new RejectingScheduler(true,events); + var cache = cache(events,scheduler); + Function loader = k -> { events.add("load"); if(failLoader) throw failure; return "value"; }; + try { + if (refresh) { + var method=DefaultTierCache.class.getDeclaredMethod("revalidate",Object.class,Function.class,long.class); + method.setAccessible(true); + if (failLoader) assertSame(failure, assertThrows(java.lang.reflect.InvocationTargetException.class, + () -> method.invoke(cache,"x",loader,0L)).getCause()); + else assertEquals("value", ((LoadClaim.Outcome)method.invoke(cache,"x",loader,0L)).entry().value()); + } else if (failLoader) assertSame(failure,assertThrows(IllegalArgumentException.class,()->cache.getOrCompute("x",loader))); + else assertEquals("value",cache.getOrCompute("x",loader)); + assertEquals(List.of("acquire","schedule","release","load"),events); + } finally { scheduler.shutdownNow(); } + } + } + static DefaultTierCache cache(List events, ScheduledExecutorService scheduler) { + DistributedLockProvider provider=(n,d)->{ + events.add("acquire"); return new DistributedLock() { + public boolean extend(Duration lease) { return true; } + public void release() { events.add("release"); } + }; + }; + return new DefaultTierCache<>("c",new CountingLocalCache<>(),new InMemoryRemoteCache<>(), + CacheSettings.defaults(),true,provider,scheduler,null,null); + } + @Test void unrelatedLiveSchedulerRejectionRemainsVisible() { + var events=new ArrayList(); var scheduler=new RejectingScheduler(false,events); + try { + assertSame(scheduler.failure,assertThrows(RejectedExecutionException.class, + ()->cache(events,scheduler).getOrCompute("x",k->{ fail("must not load"); return null; }))); + assertEquals(List.of("acquire","schedule","release"),events); + } finally { scheduler.shutdownNow(); } + } + @Test void doubleCheckHitReleasesOnceWithoutScheduling() { + var l2=new InMemoryRemoteCache(); var releases=new AtomicInteger(); + var scheduler=new RejectingScheduler(false,new ArrayList<>()); + try { + DistributedLockProvider provider=(n,d)->{ + l2.put("x",StoredEntry.ofValue("winner"),Duration.ofMinutes(1)); + return new DistributedLock() { public boolean extend(Duration d){return true;} public void release(){releases.incrementAndGet();} }; + }; + var c=new DefaultTierCache<>("c",new CountingLocalCache(),l2,CacheSettings.defaults(),true,provider,scheduler,null,null); + assertEquals("winner",c.getOrCompute("x",k->{fail("no load");return null;})); assertEquals(1,releases.get()); + } finally { scheduler.shutdownNow(); } + } + @Test void factoryCloseDisablesAuxiliariesButPreservesBorrowedResourcesAndSyncView() { + var acquisitions=new AtomicInteger(); var publications=new AtomicInteger(); var closes=new AtomicInteger(); + class Provider implements DistributedLockProvider,AutoCloseable { + public DistributedLock tryLock(String n,Duration d){acquisitions.incrementAndGet();throw new AssertionError("post-close acquire");} + public void close(){closes.incrementAndGet();} + } + var l2=new InMemoryRemoteCache(); + var factory=TierCacheFactory.builder().remoteCache(l2).lockProvider(new Provider()).invalidation(v->new InvalidationHandler(){ + public void registerTarget(String c,InvalidationTarget t){} + public void onLocalWrite(String c,Object k,Version v,InvalidationMessage.Type t){publications.incrementAndGet();} + public void close(){closes.incrementAndGet();} + }).build(); + TierCache c=factory.getCache("c"); var async=factory.asyncCache("c"); factory.close(); factory.close(); + assertEquals("loaded",c.getOrCompute("x",k->"loaded")); c.put("y","written"); + assertEquals("written",l2.get("y").value()); c.evict("x"); c.evictAll(); + assertEquals(0,acquisitions.get()); assertEquals(0,publications.get()); assertEquals(1,closes.get()); + assertThrows(RuntimeException.class,()->async.getAsync("x").toCompletableFuture().join()); + } + @Test void discardedRefreshClaimDoesNotStrandForegroundReader() throws Exception { + var queue=new ArrayDeque(); var l2=new InMemoryRemoteCache(); + var settings=new CacheSettings(100,Duration.ofMinutes(1),null,Duration.ofMinutes(1),0,NullPolicy.deny(), + InvalidationMode.INVALIDATE,65536,Duration.ofMinutes(5),false,Duration.ofSeconds(1)); + var c=new DefaultTierCache<>("c",new CountingLocalCache(),l2,settings,true,null,null,null,null, + null,CacheMetricsListener.NOOP,queue::add); + l2.put("x",StoredEntry.ofValue("stale",new Version(1,UUID.randomUUID()),System.currentTimeMillis()-120000),Duration.ofMinutes(10)); + assertEquals("stale",c.getOrCompute("x",k->"refresh")); assertEquals(1,queue.size()); + l2.evict("x"); + var pool=Executors.newSingleThreadExecutor(); + try { + var foreground=pool.submit(()->c.getOrCompute("x",k->"foreground")); + ((DefaultTierCache.DiscardableTask)queue.remove()).discard(); + assertEquals("foreground",foreground.get(2,TimeUnit.SECONDS)); + } finally {pool.shutdownNow();} + } + @Test void closedProviderIsNeutralInClosedAndHalfOpenAndLateEpoch() throws Exception { + var b=new CircuitBreaker(new CircuitBreaker.Config(2,1,1,Duration.ZERO,1),new CircuitBreaker.Listener(){public void onOpen(){}public void onClose(){}}); + var wrapped=new BreakerLockProvider((n,d)->{throw new LockProviderClosedException();},b); + assertThrows(LockProviderClosedException.class,()->wrapped.tryLock("x",Duration.ofSeconds(1))); + assertEquals(BreakerState.CLOSED,b.state()); b.onFailure(); + assertThrows(LockProviderClosedException.class,()->wrapped.tryLock("x",Duration.ofSeconds(1))); + assertEquals(BreakerState.HALF_OPEN,b.state()); + var old=b.tryAcquirePermit(); assertNotNull(old); b.onFailure(); + var current=b.tryAcquirePermit(); assertNotNull(current); old.cancel(); old.cancel(); + assertNull(b.tryAcquirePermit()); current.success(); assertEquals(BreakerState.CLOSED,b.state()); + } + @Test void closedProviderFallbackDoesNotWaitForContention() { + var scheduler=Executors.newSingleThreadScheduledExecutor(); + try { + var c=new DefaultTierCache<>("c",new CountingLocalCache(),new InMemoryRemoteCache(), + CacheSettings.defaults(),true,(n,d)->{throw new LockProviderClosedException();},scheduler,null,null); + assertTimeoutPreemptively(Duration.ofSeconds(1),()->assertEquals("v",c.getOrCompute("x",k->"v"))); + } finally {scheduler.shutdownNow();} + } + @Test void factoryCloseRetiresPendingRecoveryWithoutFalseRecoveryNotification() throws Exception { + var pending=new CompletableFuture(); var entered=new CountDownLatch(1); + var recovered=new AtomicInteger(); var l2=new InMemoryRemoteCache(); + var factory=TierCacheFactory.builder().remoteCache(l2) + .circuitBreakerConfig(new CircuitBreaker.Config(2,1,1,Duration.ZERO,1)) + .degradationListener(new DegradationListener(){public void onDegraded(){}public void onRecovered(){recovered.incrementAndGet();}}) + .invalidation(v->new InvalidationHandler(){ + public void registerTarget(String c,InvalidationTarget t){} + public void onLocalWrite(String c,Object k,Version v,InvalidationMessage.Type t){} + public CompletionStage recoverAsync(Executor e){entered.countDown();return pending;} + public void close(){} + }).build(); + try { + TierCache cache=factory.getCache("c"); + var field=TierCacheFactory.class.getDeclaredField("breaker");field.setAccessible(true); + var breaker=(CircuitBreaker)field.get(factory);breaker.onFailure();cache.get("missing"); + assertTrue(entered.await(5,TimeUnit.SECONDS));assertNull(breaker.tryAcquirePermit()); + factory.close();pending.complete(true);assertEquals(0,recovered.get()); + l2.put("after",StoredEntry.ofValue("healthy"),Duration.ofMinutes(1)); + assertEquals("healthy",cache.get("after"));assertEquals(BreakerState.CLOSED,breaker.state()); + assertEquals(0,recovered.get()); + } finally {factory.close();} + } + @Test void factoryShutdownDiscardsActualQueuedRefresh() throws Exception { + var l2=new InMemoryRemoteCache(); + var settings=new CacheSettings(100,Duration.ofMinutes(1),null,Duration.ofMinutes(1),0,NullPolicy.deny(), + InvalidationMode.INVALIDATE,65536,Duration.ofMinutes(5),false,Duration.ofSeconds(1)); + var factory=TierCacheFactory.builder().remoteCache(l2).defaults(settings).build(); + var gate=new CountDownLatch(1); + try { + var f=TierCacheFactory.class.getDeclaredField("revalidationExecutor");f.setAccessible(true); + var executor=(ThreadPoolExecutor)f.get(factory);var started=new CountDownLatch(executor.getCorePoolSize()); + for(int i=0;i{started.countDown();try{gate.await();}catch(InterruptedException ignored){}}); + assertTrue(started.await(5,TimeUnit.SECONDS)); + TierCache cache=factory.getCache("c"); + l2.put("x",StoredEntry.ofValue("stale",new Version(1,UUID.randomUUID()),System.currentTimeMillis()-120000),Duration.ofMinutes(10)); + assertEquals("stale",cache.getOrCompute("x",k->{fail("discarded refresh ran");return null;})); + assertEquals(1,executor.getQueue().size());factory.close();l2.evict("x"); + assertTimeoutPreemptively(Duration.ofSeconds(2),()->assertEquals("new",cache.getOrCompute("x",k->"new"))); + } finally {gate.countDown();factory.close();} + } + + @Test void closeAfterAcquisitionReleasesBeforeLoaderWithoutScheduling() { + var open=new AtomicBoolean(true); var events=new ArrayList(); + var scheduler=new RejectingScheduler(false,events); + DistributedLockProvider provider=(n,d)->{ + open.set(false);events.add("acquire"); + return new DistributedLock(){public boolean extend(Duration d){fail("renewal after close");return false;}public void release(){events.add("release");}}; + }; + try { + var cache=new DefaultTierCache<>("c",new CountingLocalCache(),new InMemoryRemoteCache(), + CacheSettings.defaults(),true,provider,scheduler,null,null,null,CacheMetricsListener.NOOP,null,new TtlJitter(),open::get); + assertEquals("v",cache.getOrCompute("x",k->{events.add("load");return "v";})); + assertEquals(List.of("acquire","release","load"),events); + } finally {scheduler.shutdownNow();} + } + @Test void admittedRenewalStopsCallingProviderAfterAuxiliaryClose() { + var open=new AtomicBoolean(true); var renewals=new AtomicInteger();var releases=new AtomicInteger(); + var callback=new AtomicReference(); + var scheduler=new ScheduledThreadPoolExecutor(1){ + @Override public ScheduledFuture scheduleAtFixedRate(Runnable r,long a,long b,TimeUnit u){ + callback.set(r);return super.scheduleAtFixedRate(()->{},1,1,TimeUnit.DAYS); + } + }; + DistributedLockProvider provider=(n,d)->new DistributedLock(){public boolean extend(Duration d){renewals.incrementAndGet();return true;}public void release(){releases.incrementAndGet();}}; + try { + var cache=new DefaultTierCache<>("c",new CountingLocalCache(),new InMemoryRemoteCache(), + CacheSettings.defaults(),true,provider,scheduler,null,null,null,CacheMetricsListener.NOOP,null,new TtlJitter(),open::get); + assertEquals("v",cache.getOrCompute("x",k->{callback.get().run();open.set(false);callback.get().run();return "v";})); + assertEquals(1,renewals.get());assertEquals(1,releases.get()); + } finally {scheduler.shutdownNow();} + } + @Test void dequeuedRefreshObservesCloseAndDiscardIsIdempotent() { + var queue=new ArrayDeque();var open=new AtomicBoolean(true);var l2=new InMemoryRemoteCache(); + var settings=new CacheSettings(100,Duration.ofMinutes(1),null,Duration.ofMinutes(1),0,NullPolicy.deny(), + InvalidationMode.INVALIDATE,65536,Duration.ofMinutes(5),false,Duration.ofSeconds(1)); + var cache=new DefaultTierCache<>("c",new CountingLocalCache(),l2,settings,true,null,null,null,null, + null,CacheMetricsListener.NOOP,queue::add,new TtlJitter(),open::get); + l2.put("x",StoredEntry.ofValue("stale",new Version(1,UUID.randomUUID()),System.currentTimeMillis()-120000),Duration.ofMinutes(10)); + assertEquals("stale",cache.getOrCompute("x",k->{fail("closed refresh must not load");return null;})); + var task=(DefaultTierCache.DiscardableTask)queue.remove();open.set(false);task.run();task.discard(); + l2.evict("x");assertEquals("v",cache.getOrCompute("x",k->"v")); + } + +} diff --git a/tiercache-core/src/test/java/io/tiercache/internal/LocalFreshnessTest.java b/tiercache-core/src/test/java/io/tiercache/internal/LocalFreshnessTest.java new file mode 100644 index 0000000..908ad15 --- /dev/null +++ b/tiercache-core/src/test/java/io/tiercache/internal/LocalFreshnessTest.java @@ -0,0 +1,150 @@ +package io.tiercache.internal; + +import io.tiercache.*; +import io.tiercache.spi.*; +import io.tiercache.testkit.*; +import org.junit.jupiter.api.Test; +import java.time.Duration; +import java.util.UUID; +import java.util.concurrent.atomic.*; +import static org.junit.jupiter.api.Assertions.*; + +class LocalFreshnessTest { + static CacheSettings settings(long ttl, Long access, long window, NullPolicy policy) { + return new CacheSettings(100, Duration.ofMinutes(ttl), access == null ? null : Duration.ofMinutes(access), + Duration.ofHours(4), 0, policy, InvalidationMode.INVALIDATE, 65536, + Duration.ZERO, false, Duration.ofSeconds(1), Duration.ofMinutes(window)); + } + static final class Rig { + final AtomicLong now = new AtomicLong(1); + final AtomicInteger stale = new AtomicInteger(); + final CaffeineLocalCache l1; + final InMemoryRemoteCache l2 = new InMemoryRemoteCache<>(); + final CircuitBreaker breaker = new CircuitBreaker(new CircuitBreaker.Config(1,1,1,Duration.ofDays(1),1), + new CircuitBreaker.Listener(){public void onOpen(){} public void onClose(){}}); + final DefaultTierCache cache; + Rig(CacheSettings settings) { + l1 = new CaffeineLocalCache<>(settings, now::get); + cache = new DefaultTierCache<>("c",l1,new CircuitBreakerRemoteCache<>(l2,breaker),settings,true, + null,null,new VersionGenerator(),null,breaker,new CacheMetricsListener(){ + public void onRequest(String c,Outcome outcome){if(outcome==Outcome.STALE_DEGRADED)stale.incrementAndGet();} + },null,new TtlJitter(),()->true,now::get); + } + void minute(long value){now.set(1+Duration.ofMinutes(value).toNanos());} + @SuppressWarnings("unchecked") L1BarrierMap fences() throws Exception { + var f=DefaultTierCache.class.getDeclaredField("l1Metas");f.setAccessible(true);return (L1BarrierMap)f.get(cache); + } + } + static String noLoad(String key){throw new AssertionError("unexpected source call: "+key);} + + @Test void thirtyMinuteValueRemainsFreshAfterIndependentFenceExpires() throws Exception { + var r=new Rig(settings(30,null,30,NullPolicy.deny()));r.cache.put("x","value");r.breaker.onFailure(); + r.minute(11);assertNull(r.fences().get("x")); + r.minute(29);assertEquals("value",r.cache.getOrCompute("x",LocalFreshnessTest::noLoad));assertEquals(0,r.stale.get()); + r.minute(30);assertEquals("value",r.cache.getOrCompute("x",LocalFreshnessTest::noLoad));assertEquals(1,r.stale.get()); + r.minute(60);assertEquals("new",r.cache.getOrCompute("x",k->"new")); + } + @Test void freshAccessSlidesBeyondOldHorizonWhileHotStaleReadsNeverSlide() { + var r=new Rig(settings(20,20L,30,NullPolicy.deny()));r.cache.put("x","value");r.breaker.onFailure(); + r.minute(15);assertEquals("value",r.cache.get("x")); + r.minute(30);assertEquals("value",r.cache.get("x")); + var snapshot=r.l1.get("x").localFreshness(); + assertEquals(1+Duration.ofMinutes(50).toNanos(),snapshot.logicalDeadlineNanos()); + for(int minute=50;minute<80;minute++) {r.minute(minute);assertEquals("value",r.cache.getOrCompute("x",LocalFreshnessTest::noLoad));assertSame(snapshot,r.l1.get("x").localFreshness());} + r.minute(80);assertNull(r.cache.get("x"));assertEquals(30,r.stale.get()); + } + @Test void shortAccessDoesNotShortenStoreFloorOrExtendStaleCutoff() { + var r=new Rig(settings(30,2L,10,NullPolicy.deny()));r.cache.put("x","value");r.breaker.onFailure(); + r.minute(1);assertEquals("value",r.cache.get("x")); + var state=r.l1.get("x").localFreshness(); + assertEquals(1+Duration.ofMinutes(40).toNanos(),state.retentionUntilNanos()); + r.minute(12);assertEquals("value",r.cache.get("x")); + r.minute(13);assertNotNull(r.l1.get("x"));assertNull(r.cache.get("x")); + r.minute(39);assertNotNull(r.l1.get("x"));r.minute(40);assertNull(r.l1.get("x")); + } + @Test void fencesStayBoundedWithoutErasingFreshnessOrCurrentVersion() throws Exception { + var r=new Rig(settings(30,null,10,NullPolicy.deny()));r.cache.put("x","value"); + var old=r.l1.get("x"); + var field=DefaultTierCache.class.getDeclaredField("l1Generation");field.setAccessible(true);var generation=(AtomicLong)field.get(r.cache); + var fences=new L1BarrierMap(2,Duration.ofMinutes(10),generation::incrementAndGet,r.now::get); + field=DefaultTierCache.class.getDeclaredField("l1Metas");field.setAccessible(true);field.set(r.cache,fences); + long before=generation.get(); + UUID origin=UUID.randomUUID(); + for(int i=0;i<1000;i++) r.cache.evictL1IfNewer("absent-"+i,new Version(i+1,origin)); + var rawField=L1BarrierMap.class.getDeclaredField("barriers");rawField.setAccessible(true); + // PIT may load the shaded core artifact; use its declared cache interface. + var raw=rawField.get(fences);var cacheType=rawField.getType(); + cacheType.getMethod("cleanUp").invoke(raw); + assertTrue((Long)cacheType.getMethod("estimatedSize").invoke(raw)<=2);assertTrue(generation.get()>before); + r.breaker.onFailure();assertEquals("value",r.cache.getOrCompute("x",LocalFreshnessTest::noLoad)); + var commit=DefaultTierCache.class.getDeclaredMethod("commitL1",Object.class,StoredEntry.class,Duration.class,long.class);commit.setAccessible(true); + assertEquals(false,commit.invoke(r.cache,"x",StoredEntry.ofValue("obsolete",new Version(1,origin)),Duration.ofMinutes(30),generation.get())); + assertEquals(false,commit.invoke(r.cache,"absent-old",old,Duration.ofMinutes(30),before)); + assertSame(old,r.l1.get("x")); + } + @Test void markerOwnsItsTtlAndReplacementNeverInheritsPreviousDeadlines() { + var r=new Rig(settings(30,null,10,NullPolicy.allow(Duration.ofMinutes(2))));r.cache.put("x","value"); + var value=r.l1.get("x");r.minute(1);r.cache.putNull("x");var marker=r.l1.get("x"); + assertNotSame(value.localFreshness(),marker.localFreshness());assertTrue(marker.isNullMarker()); + r.breaker.onFailure();r.minute(4);assertNull(r.cache.getOrCompute("x",LocalFreshnessTest::noLoad)); + r.minute(13);assertEquals("new",r.cache.getOrCompute("x",k->"new")); + r.cache.evictAll();assertNull(r.l1.get("x"));assertEquals("after-clear",r.cache.getOrCompute("x",k->"after-clear")); + r.l1.evict("x");assertNull(r.cache.get("x")); + } + @Test void losingPutIfAbsentAndDuplicateUpdateDoNotExtendDeadline() { + var r=new Rig(settings(30,null,10,NullPolicy.deny()));r.cache.put("x","value");var entry=r.l1.get("x"); + r.minute(5);assertFalse(r.cache.putIfAbsent("x","loser"));assertSame(entry,r.l1.get("x")); + r.cache.applyUpdateL1("x","duplicate",entry.version());assertSame(entry,r.l1.get("x")); + var next=new Version(entry.version().sequence()+1,entry.version().instanceId()); + r.cache.applyUpdateL1("x","update",next);assertEquals("update",r.l1.get("x").value());assertNotSame(entry.localFreshness(),r.l1.get("x").localFreshness()); + } + @Test void sharedMarkersAndRemoteEntriesReceiveIndependentLocalCopies() { + var a=new Rig(settings(30,null,10,NullPolicy.allow(Duration.ofMinutes(2)))); + var b=new Rig(settings(30,null,20,NullPolicy.allow(Duration.ofMinutes(3)))); + var shared=StoredEntry.nullMarker();a.l2.put("x",shared,Duration.ofHours(1));b.l2.put("x",shared,Duration.ofHours(1)); + a.cache.get("x");b.cache.get("x"); + assertNull(shared.localFreshness());assertNotSame(shared,a.l1.get("x"));assertNotSame(a.l1.get("x"),b.l1.get("x")); + assertNotEquals(a.l1.get("x").localFreshness(),b.l1.get("x").localFreshness()); + assertNotEquals(StoredEntry.ofValue("same"),StoredEntry.ofValue("same")); + } + @Test void defaultOffKeepsOpaqueEntryAndLegacyAccessExpiry() { + var r=new Rig(settings(30,20L,0,NullPolicy.deny()));r.cache.put("x","value");r.breaker.onFailure(); + assertNull(r.l1.get("x").localFreshness());r.minute(15);assertEquals("value",r.cache.get("x")); + r.minute(30);assertEquals("value",r.cache.get("x"));r.minute(51);assertNull(r.cache.get("x"));assertEquals(0,r.stale.get()); + } + @Test void failedAtomicReplacementDoesNotRefreshAnotherValue() { + var now=new AtomicLong(1);var config=settings(30,20L,10,NullPolicy.deny()); + var l1=new CaffeineLocalCache(config,now::get); + var old=StoredEntry.ofValue("old");var current=StoredEntry.ofValue("current"); + l1.put("x",current,Duration.ofSeconds(5));now.addAndGet(Duration.ofSeconds(4).toNanos()); + assertFalse(l1.replaceIfSame("x",old,StoredEntry.ofValue("candidate"),Duration.ofMinutes(30))); + assertSame(current,l1.get("x"));now.addAndGet(Duration.ofSeconds(1).toNanos());assertNull(l1.get("x")); + assertFalse(l1.replaceIfSame("x",current,old,Duration.ofMinutes(30)));assertNull(l1.get("x")); + } + @Test void customOpaqueProviderWorksExceptUnsupportedSlidingCombination() { + var retained=new CountingLocalCache(); + LocalCache legacy=new LocalCache<>() { + public StoredEntry get(String k){return retained.get(k);} + public void put(String k,StoredEntry v,Duration ttl){retained.put(k,v,ttl);} + public boolean setIfAbsent(String k,StoredEntry v,Duration ttl){return retained.setIfAbsent(k,v,ttl);} + public void evict(String k){retained.evict(k);} + public void clear(){retained.clear();} + }; + assertFalse(legacy.supportsAtomicReplace()); + assertThrows(UnsupportedOperationException.class,()->legacy.replaceIfSame("x",StoredEntry.nullMarker(),StoredEntry.nullMarker(),Duration.ofSeconds(1))); + var error=assertThrows(IllegalArgumentException.class,()->new DefaultTierCache<>("custom-l1",legacy, + new InMemoryRemoteCache(),settings(30,20L,10,NullPolicy.deny()),true,null,null,null,null)); + assertTrue(error.getMessage().contains("custom-l1"));assertTrue(error.getMessage().contains("atomic replacement")); + var cache=new DefaultTierCache<>("custom-l1",legacy,new InMemoryRemoteCache(),settings(30,null,10,NullPolicy.deny()),true,null,null,null,null); + cache.put("x","value");assertNotNull(retained.get("x").localFreshness());assertEquals("value",cache.get("x")); + assertDoesNotThrow(()->new DefaultTierCache<>("off",legacy,new InMemoryRemoteCache(),settings(30,20L,0,NullPolicy.deny()),true,null,null,null,null)); + } + @Test void oldRemoteTimestampDoesNotShortenNewLocalFreshness() { + var r=new Rig(settings(30,null,10,NullPolicy.deny())); + r.l2.put("x",StoredEntry.ofValue("remote",new Version(1,UUID.randomUUID()),0),Duration.ofHours(1)); + r.minute(30);assertEquals("remote",r.cache.get("x"));r.breaker.onFailure(); + r.minute(59);assertEquals("remote",r.cache.getOrCompute("x",LocalFreshnessTest::noLoad));assertEquals(0,r.stale.get()); + r.minute(60);assertEquals("remote",r.cache.getOrCompute("x",LocalFreshnessTest::noLoad));assertEquals(1,r.stale.get()); + } + +} diff --git a/tiercache-core/src/test/java/io/tiercache/internal/SkippedRefreshTest.java b/tiercache-core/src/test/java/io/tiercache/internal/SkippedRefreshTest.java new file mode 100644 index 0000000..26a917d --- /dev/null +++ b/tiercache-core/src/test/java/io/tiercache/internal/SkippedRefreshTest.java @@ -0,0 +1,612 @@ +package io.tiercache.internal; + +import io.tiercache.CacheSettings; +import io.tiercache.InvalidationMode; +import io.tiercache.NullPolicy; +import io.tiercache.VersionGenerator; +import io.tiercache.spi.CacheMetricsListener; +import io.tiercache.spi.DistributedLock; +import io.tiercache.spi.DistributedLockProvider; +import io.tiercache.spi.RemoteCache; +import io.tiercache.spi.StoredEntry; +import io.tiercache.testkit.CountingLocalCache; +import io.tiercache.testkit.InMemoryRemoteCache; +import org.junit.jupiter.api.Test; + +import java.lang.reflect.Field; +import java.lang.reflect.Method; +import java.time.Duration; +import java.util.ArrayList; +import java.util.List; +import java.util.concurrent.BlockingQueue; +import java.util.concurrent.CompletionException; +import java.util.concurrent.ConcurrentHashMap; +import java.util.concurrent.CountDownLatch; +import java.util.concurrent.ExecutionException; +import java.util.concurrent.Executor; +import java.util.concurrent.ExecutorService; +import java.util.concurrent.Executors; +import java.util.concurrent.Future; +import java.util.concurrent.LinkedBlockingQueue; +import java.util.concurrent.RejectedExecutionException; +import java.util.concurrent.ScheduledExecutorService; +import java.util.concurrent.TimeUnit; +import java.util.concurrent.atomic.AtomicBoolean; +import java.util.concurrent.atomic.AtomicInteger; +import java.util.concurrent.atomic.AtomicLong; +import java.util.function.BiFunction; +import java.util.function.Function; + +import static org.junit.jupiter.api.Assertions.*; + +/** A skipped refresh is coordination control flow, not evidence of a missing value. */ +class SkippedRefreshTest { + private static final String KEY = "key"; + private static final Duration L2_TTL = Duration.ofMinutes(10); + + @Test + void swrForegroundJoinDoesNotReturnSkippedRefreshAsNull() throws Exception { + assertForegroundJoinObtainsValue(false); + } + + @Test + void xfetchForegroundJoinDoesNotReturnSkippedRefreshAsNull() throws Exception { + assertForegroundJoinObtainsValue(true); + } + + private void assertForegroundJoinObtainsValue(boolean xfetch) throws Exception { + try (Rig rig = new Rig(xfetch, NullPolicy.deny(), 1)) { + AtomicInteger refreshLoads = new AtomicInteger(); + rig.queueRefresh(key -> { + refreshLoads.incrementAndGet(); + return "refresh-value"; + }); + Runnable background = rig.refresh.take(); + rig.removeCachedValue(); + AtomicInteger foregroundLoads = new AtomicInteger(); + rig.metrics.joined = new CountDownLatch(1); + Future caller = rig.workers.submit(() -> rig.cache.getOrCompute(KEY, key -> { + foregroundLoads.incrementAndGet(); + return "current"; + })); + await(rig.metrics.joined); + background.run(); + assertEquals("current", caller.get(5, TimeUnit.SECONDS)); + assertEquals(1, foregroundLoads.get()); + assertEquals(0, refreshLoads.get()); + assertEquals(2, rig.acquisitions.get()); + assertEquals(1, rig.releases.get()); + assertTrue(rig.claims.isEmpty()); + } + } + + @Test + void terminalSkipWithEmptySlotCreatesForegroundOwner() throws Exception { + assertLateJoin(Replacement.ABSENT); + } + + @Test + void oldRefreshCleanupCannotRemoveForegroundReplacement() throws Exception { + assertLateJoin(Replacement.SAME_SKIPPED); + } + + @Test + void terminalSkipPromotesCompetingActiveRefresh() throws Exception { + assertLateJoin(Replacement.ACTIVE_REFRESH); + } + + @Test + void terminalSkipReplacesAnotherAlreadySkippedRefresh() throws Exception { + assertLateJoin(Replacement.SKIPPED_REFRESH); + } + + @Test + void terminalSkipJoinsCompetingForegroundOwner() throws Exception { + assertLateJoin(Replacement.FOREGROUND); + } + + private enum Replacement { ABSENT, SAME_SKIPPED, ACTIVE_REFRESH, SKIPPED_REFRESH, FOREGROUND } + + /** + * Park the first foreground reader after it obtained the old claim but + * before it can promote it. Then change the map behind that exact reference. + * The map seam controls scheduling; assertions concern real readers/loaders. + */ + private void assertLateJoin(Replacement replacement) throws Exception { + int skips = replacement == Replacement.ACTIVE_REFRESH + || replacement == Replacement.SKIPPED_REFRESH ? 2 : 1; + CountDownLatch releaseCompletion = new CountDownLatch(1); + CountDownLatch releaseLoad = new CountDownLatch(1); + try (Rig rig = new Rig(false, NullPolicy.deny(), skips)) { + AtomicInteger loads = new AtomicInteger(); + CountDownLatch loading = new CountDownLatch(1); + Function loader = key -> { + loads.incrementAndGet(); + loading.countDown(); + await(releaseLoad); + return "current"; + }; + rig.queueRefresh(key -> fail("a skipped refresh must not invoke its loader")); + Runnable oldRefresh = rig.refresh.take(); + LoadClaim old = rig.claims.get(KEY); + rig.removeCachedValue(); + rig.claims.parkNextJoin.set(true); + Future caller = rig.workers.submit(() -> rig.cache.getOrCompute(KEY, loader)); + await(rig.claims.captured); + + CountDownLatch completed = new CountDownLatch(1); + Runnable completionGate = () -> { + completed.countDown(); + await(releaseCompletion); + }; + Future parkedRefresh = null; + if (replacement == Replacement.SAME_SKIPPED) { + rig.metrics.onCompleted = completionGate; + parkedRefresh = rig.workers.submit(oldRefresh); + await(completed); + } else { + oldRefresh.run(); + } + + Runnable competingRefresh = null; + Future competingForeground = null; + if (replacement == Replacement.ACTIVE_REFRESH + || replacement == Replacement.SKIPPED_REFRESH) { + rig.queueRefresh(key -> fail("promoted skip must use foreground demand")); + competingRefresh = rig.refresh.take(); + rig.removeCachedValue(); + if (replacement == Replacement.SKIPPED_REFRESH) { + rig.metrics.onCompleted = completionGate; + parkedRefresh = rig.workers.submit(competingRefresh); + await(completed); + } + } else if (replacement == Replacement.FOREGROUND) { + competingForeground = rig.workers.submit(() -> rig.cache.getOrCompute(KEY, loader)); + await(loading); + } + + rig.claims.releaseJoin.countDown(); + await(rig.claims.recovered); + Future activeRefresh = replacement == Replacement.ACTIVE_REFRESH + ? rig.workers.submit(competingRefresh) : null; + await(loading); + LoadClaim selected = rig.claims.get(KEY); + assertNotNull(selected); + assertNotSame(old, selected); + releaseCompletion.countDown(); + if (parkedRefresh != null) { + parkedRefresh.get(5, TimeUnit.SECONDS); + } + assertSame(selected, rig.claims.get(KEY), "old cleanup must retain the replacement"); + + rig.metrics.joined = new CountDownLatch(1); + Future follower = rig.workers.submit(() -> rig.cache.getOrCompute(KEY, loader)); + await(rig.metrics.joined); + assertEquals(1, loads.get(), "replacement keeps one local owner"); + releaseLoad.countDown(); + assertEquals("current", caller.get(5, TimeUnit.SECONDS)); + assertEquals("current", follower.get(5, TimeUnit.SECONDS)); + if (competingForeground != null) { + assertEquals("current", competingForeground.get(5, TimeUnit.SECONDS)); + } + if (activeRefresh != null) { + activeRefresh.get(5, TimeUnit.SECONDS); + } + assertEquals(1, loads.get()); + assertTrue(rig.claims.isEmpty()); + } finally { + releaseLoad.countDown(); + releaseCompletion.countDown(); + } + } + + @Test + void manyForegroundWaitersShareOnePromotedRefresh() throws Exception { + try (Rig rig = new Rig(false, NullPolicy.deny(), 1)) { + rig.queueRefresh(key -> fail("refresh must not fabricate a result")); + Runnable refresh = rig.refresh.take(); + rig.removeCachedValue(); + int count = 6; + rig.metrics.joined = new CountDownLatch(count); + AtomicInteger loads = new AtomicInteger(); + List> callers = new ArrayList<>(); + for (int i = 0; i < count; i++) { + callers.add(rig.workers.submit(() -> rig.cache.getOrCompute(KEY, key -> { + loads.incrementAndGet(); + return "current"; + }))); + } + await(rig.metrics.joined); + refresh.run(); + for (Future caller : callers) { + assertEquals("current", caller.get(5, TimeUnit.SECONDS)); + } + assertEquals(1, loads.get()); + } + } + + @Test + void realRefreshNullUnderDenyIsNotRetried() throws Exception { + assertRealNullIsShared(NullPolicy.deny()); + } + + @Test + void realRefreshNullUnderAllowIsCachedWithoutRetry() throws Exception { + assertRealNullIsShared(NullPolicy.allow(Duration.ofMinutes(1))); + } + + private void assertRealNullIsShared(NullPolicy policy) throws Exception { + try (Rig rig = new Rig(false, policy, 0)) { + AtomicInteger refreshLoads = new AtomicInteger(); + rig.queueRefresh(key -> { + refreshLoads.incrementAndGet(); + return null; + }); + Runnable refresh = rig.refresh.take(); + rig.removeCachedValue(); + rig.metrics.joined = new CountDownLatch(1); + Future caller = rig.workers.submit(() -> rig.cache.getOrCompute(KEY, + key -> fail("genuine null is a result, not skipped coordination"))); + await(rig.metrics.joined); + refresh.run(); + assertNull(caller.get(5, TimeUnit.SECONDS)); + assertEquals(1, refreshLoads.get()); + if (policy.markerTtl() == null) { + assertNull(rig.l2.get(KEY)); + } else { + assertTrue(rig.l2.get(KEY).isNullMarker()); + assertNull(rig.cache.getOrCompute(KEY, key -> fail("cached marker"))); + } + assertTrue(rig.claims.isEmpty()); + } + } + + @Test + void newerL2ValueCompletesJoinedRefreshWithoutLoader() throws Exception { + try (Rig rig = new Rig(false, NullPolicy.deny(), 0)) { + rig.queueRefresh(key -> fail("newer L2 suppresses refresh loader")); + Runnable refresh = rig.refresh.take(); + rig.removeCachedValue(); + rig.metrics.joined = new CountDownLatch(1); + Future caller = rig.workers.submit(() -> rig.cache.getOrCompute(KEY, + key -> fail("double-check supplies foreground result"))); + await(rig.metrics.joined); + rig.l2.put(KEY, StoredEntry.ofValue("newer", null, System.currentTimeMillis()), + Duration.ofHours(1)); + refresh.run(); + assertEquals("newer", caller.get(5, TimeUnit.SECONDS)); + assertEquals("newer", rig.l1.get(KEY).value()); + } + } + + @Test + void loaderFailureUnblocksForegroundAndReleasesClaim() throws Exception { + assertFailureShared(new IllegalArgumentException("loader failed")); + } + + @Test + void loaderErrorAlsoSettlesItsSharedClaim() throws Exception { + assertFailureShared(new AssertionError("loader error")); + } + + private void assertFailureShared(Throwable failure) throws Exception { + try (Rig rig = new Rig(false, NullPolicy.deny(), 0)) { + rig.queueRefresh(key -> { + if (failure instanceof Error error) { + throw error; + } + throw (RuntimeException) failure; + }); + Runnable refresh = rig.refresh.take(); + rig.removeCachedValue(); + rig.metrics.joined = new CountDownLatch(1); + Future caller = rig.workers.submit(() -> rig.cache.getOrCompute(KEY, + key -> fail("joined failure must not start a second loader"))); + await(rig.metrics.joined); + if (failure instanceof Error) { + assertSame(failure, assertThrows(Error.class, refresh::run)); + } else { + refresh.run(); + } + ExecutionException thrown = assertThrows(ExecutionException.class, + () -> caller.get(5, TimeUnit.SECONDS)); + assertSame(failure, assertInstanceOf(CompletionException.class, thrown.getCause()).getCause()); + assertTrue(rig.claims.isEmpty()); + assertEquals("retry", rig.cache.getOrCompute(KEY, key -> "retry")); + } + } + + @Test + void rejectedSchedulingFailsJoinedReaderButKeepsStaleCallerResult() throws Exception { + CountDownLatch submitted = new CountDownLatch(1); + CountDownLatch reject = new CountDownLatch(1); + RejectedExecutionException failure = new RejectedExecutionException("queue full"); + Executor rejected = task -> { + submitted.countDown(); + await(reject); + throw failure; + }; + try (Rig rig = new Rig(false, NullPolicy.deny(), 0, rejected)) { + rig.seed(); + Future stale = rig.workers.submit(() -> rig.cache.getOrCompute(KEY, + key -> fail("rejected task cannot load"))); + await(submitted); + rig.removeCachedValue(); + rig.metrics.joined = new CountDownLatch(1); + Future caller = rig.workers.submit(() -> rig.cache.getOrCompute(KEY, + key -> fail("a joined rejected claim must fail"))); + await(rig.metrics.joined); + reject.countDown(); + assertEquals("cached", stale.get(5, TimeUnit.SECONDS)); + ExecutionException thrown = assertThrows(ExecutionException.class, + () -> caller.get(5, TimeUnit.SECONDS)); + assertSame(failure, assertInstanceOf(CompletionException.class, thrown.getCause()).getCause()); + assertTrue(rig.claims.isEmpty()); + assertEquals("retry", rig.cache.getOrCompute(KEY, key -> "retry")); + } finally { + reject.countDown(); + } + } + + @Test + void promotedRefreshKeepsTwoLoadBudgetAcrossTombstonesAndGenerationChanges() throws Exception { + try (Rig rig = new Rig(false, NullPolicy.deny(), 1)) { + rig.queueRefresh(key -> fail("skip has consumed no loader execution")); + Runnable refresh = rig.refresh.take(); + rig.removeCachedValue(); + rig.l2.rejectWrites = true; + rig.metrics.joined = new CountDownLatch(1); + AtomicInteger loads = new AtomicInteger(); + Future caller = rig.workers.submit(() -> rig.cache.getOrCompute(KEY, key -> { + int attempt = loads.incrementAndGet(); + rig.cache.evictAllL1(); + rig.cache.evictAllL1(); + return "value-" + attempt; + })); + await(rig.metrics.joined); + refresh.run(); + assertEquals("value-2", caller.get(5, TimeUnit.SECONDS)); + assertEquals(2, loads.get()); + assertEquals(2, rig.l2.conditionalWrites.get()); + assertNull(rig.l1.get(KEY)); + assertNull(rig.l2.get(KEY)); + assertTrue(rig.claims.isEmpty()); + } + } + + @Test + void promotionKeepsTheFirstForegroundDeadline() throws Exception { + try (Rig rig = new Rig(false, NullPolicy.deny(), Integer.MAX_VALUE)) { + rig.queueRefresh(key -> fail("no refresh loader after lock loss")); + Runnable refresh = rig.refresh.take(); + rig.removeCachedValue(); + LoadClaim claim = rig.claims.get(KEY); + AtomicInteger loads = new AtomicInteger(); + // Simulate the original demand expiring while refresh was queued, + // without sleeping through a real 30-second coordination budget. + assertTrue(claim.requireResult(new LoadClaim.Demand<>(key -> { + loads.incrementAndGet(); + return "expired-budget-result"; + }, System.nanoTime() - 1))); + rig.metrics.joined = new CountDownLatch(1); + Future caller = rig.workers.submit(() -> rig.cache.getOrCompute(KEY, + key -> fail("a later waiter must not replace the first demand"))); + await(rig.metrics.joined); + Future background = rig.workers.submit(refresh); + assertEquals("expired-budget-result", caller.get(5, TimeUnit.SECONDS)); + background.get(5, TimeUnit.SECONDS); + assertEquals(1, rig.acquisitions.get(), "only the original refresh tries its lock"); + assertEquals(1, loads.get()); + } + } + + @Test + void terminalSkipRecoveryKeepsItsDeadlineAndTwoLoadBudget() throws Exception { + try (Rig rig = new Rig(false, NullPolicy.deny(), Integer.MAX_VALUE)) { + rig.queueRefresh(key -> fail("refresh is skipped")); + Runnable refresh = rig.refresh.take(); + LoadClaim skipped = rig.claims.get(KEY); + rig.removeCachedValue(); + refresh.run(); + assertTrue(skipped.result.join().isSkipped()); + assertThrows(IllegalStateException.class, () -> skipped.result.join().resultEntry()); + rig.claims.put(KEY, skipped); + rig.l2.rejectWrites = true; + AtomicInteger loads = new AtomicInteger(); + LoadClaim.Demand demand = new LoadClaim.Demand<>(key -> { + rig.cache.evictAllL1(); + return "bounded-" + loads.incrementAndGet(); + }, System.nanoTime() - 1); + Method recover = DefaultTierCache.class.getDeclaredMethod("recoverSkippedRefresh", + Object.class, LoadClaim.class, LoadClaim.Demand.class); + recover.setAccessible(true); + Future result = rig.workers.submit(() -> recover.invoke(rig.cache, KEY, skipped, demand)); + assertEquals("bounded-2", result.get(5, TimeUnit.SECONDS)); + assertEquals(1, rig.acquisitions.get(), "recovery must not restart coordination"); + assertEquals(2, loads.get()); + assertTrue(rig.claims.isEmpty()); + } + } + + private static void await(CountDownLatch latch) { + try { + assertTrue(latch.await(5, TimeUnit.SECONDS), "controlled step did not complete"); + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + throw new AssertionError("interrupted controlled step", e); + } + } + + private static final class Rig implements AutoCloseable { + final CountingLocalCache l1 = new CountingLocalCache<>(); + final TestRemote l2 = new TestRemote(); + final QueuedExecutor refresh = new QueuedExecutor(); + final Metrics metrics = new Metrics(); + final ClaimMap claims = new ClaimMap(); + final AtomicInteger acquisitions = new AtomicInteger(); + final AtomicInteger releases = new AtomicInteger(); + final ExecutorService workers = Executors.newFixedThreadPool(10); + final ScheduledExecutorService watchdog = Executors.newSingleThreadScheduledExecutor(); + final DefaultTierCache cache; + final boolean xfetch; + + Rig(boolean xfetch, NullPolicy policy, int refusedLocks) throws Exception { + this(xfetch, policy, refusedLocks, null); + } + + Rig(boolean xfetch, NullPolicy policy, int refusedLocks, Executor executor) throws Exception { + this.xfetch = xfetch; + DistributedLockProvider provider = (name, lease) -> + acquisitions.incrementAndGet() <= refusedLocks ? null : lock(releases); + CacheSettings settings = new CacheSettings(10_000, Duration.ofMinutes(1), null, + L2_TTL, 0.0, policy, InvalidationMode.INVALIDATE, 64 * 1024, + Duration.ofMinutes(5), xfetch, Duration.ofNanos(1)); + cache = new DefaultTierCache<>("cache", l1, l2, settings, true, provider, watchdog, + new VersionGenerator(), null, null, metrics, executor == null ? refresh : executor); + Field field = DefaultTierCache.class.getDeclaredField("inflight"); + field.setAccessible(true); + field.set(cache, claims); + if (xfetch) { + field = DefaultTierCache.class.getDeclaredField("loaderDurationEmaNanos"); + field.setAccessible(true); + ((AtomicLong) field.get(cache)).set(TimeUnit.SECONDS.toNanos(1)); + } + } + + void seed() { + long age = xfetch ? L2_TTL.toMillis() / 2 : L2_TTL.toMillis() + 60_000; + l2.put(KEY, StoredEntry.ofValue("cached", null, System.currentTimeMillis() - age), + Duration.ofHours(1)); + } + + void queueRefresh(Function loader) { + seed(); + assertEquals("cached", cache.getOrCompute(KEY, loader)); + } + + void removeCachedValue() { + l1.evict(KEY); + l2.evict(KEY); + } + + @Override + public void close() { + claims.releaseJoin.countDown(); + workers.shutdownNow(); + watchdog.shutdownNow(); + } + } + + private static DistributedLock lock(AtomicInteger releases) { + return new DistributedLock() { + @Override + public boolean extend(Duration lease) { + return true; + } + + @Override + public void release() { + releases.incrementAndGet(); + } + }; + } + + private static final class QueuedExecutor implements Executor { + private final BlockingQueue tasks = new LinkedBlockingQueue<>(); + + @Override + public void execute(Runnable command) { + tasks.add(command); + } + + Runnable take() throws InterruptedException { + Runnable task = tasks.poll(5, TimeUnit.SECONDS); + assertNotNull(task, "refresh must be queued"); + return task; + } + } + + private static final class Metrics implements CacheMetricsListener { + volatile CountDownLatch joined = new CountDownLatch(0); + volatile Runnable onCompleted = () -> { }; + + @Override + public void onRequest(String cache, Outcome outcome) { + if (outcome == Outcome.COALESCED) { + joined.countDown(); + } + } + + @Override + public void onRevalidationCompleted(String cache) { + onCompleted.run(); + } + } + + private static final class ClaimMap extends ConcurrentHashMap> { + final AtomicBoolean parkNextJoin = new AtomicBoolean(); + final CountDownLatch captured = new CountDownLatch(1); + final CountDownLatch releaseJoin = new CountDownLatch(1); + final CountDownLatch recovered = new CountDownLatch(1); + + @Override + public LoadClaim putIfAbsent(String key, LoadClaim value) { + LoadClaim existing = super.putIfAbsent(key, value); + if (existing != null && parkNextJoin.compareAndSet(true, false)) { + captured.countDown(); + await(releaseJoin); + } + return existing; + } + + @Override + public LoadClaim compute(String key, + BiFunction, + ? extends LoadClaim> remappingFunction) { + LoadClaim selected = super.compute(key, remappingFunction); + recovered.countDown(); + return selected; + } + } + + private static final class TestRemote implements RemoteCache { + final InMemoryRemoteCache delegate = new InMemoryRemoteCache<>(); + final AtomicInteger conditionalWrites = new AtomicInteger(); + volatile boolean rejectWrites; + + @Override + public StoredEntry get(String key) { + return delegate.get(key); + } + + @Override + public void put(String key, StoredEntry value, Duration ttl) { + delegate.put(key, value, ttl); + } + + @Override + public boolean putIfNewer(String key, StoredEntry value, Duration ttl, Duration staleTtl) { + conditionalWrites.incrementAndGet(); + if (rejectWrites) { + return false; + } + delegate.put(key, value, ttl, staleTtl); + return true; + } + + @Override + public void evict(String key) { + delegate.evict(key); + } + + @Override + public void clear() { + delegate.clear(); + } + + @Override + public boolean setIfAbsent(String key, StoredEntry value, Duration ttl) { + return delegate.setIfAbsent(key, value, ttl); + } + } +} diff --git a/tiercache-core/src/testFixtures/java/io/tiercache/testkit/CountingLocalCache.java b/tiercache-core/src/testFixtures/java/io/tiercache/testkit/CountingLocalCache.java index 31eb251..38cce85 100644 --- a/tiercache-core/src/testFixtures/java/io/tiercache/testkit/CountingLocalCache.java +++ b/tiercache-core/src/testFixtures/java/io/tiercache/testkit/CountingLocalCache.java @@ -46,4 +46,12 @@ public void clear() { public boolean setIfAbsent(K key, StoredEntry entry, Duration ttl) { return store.putIfAbsent(key, entry) == null; } + @Override + public boolean supportsAtomicReplace() { return true; } + + @Override + public boolean replaceIfSame(K key, StoredEntry expected, StoredEntry replacement, Duration ttl) { + return store.replace(key, expected, replacement); + } + } diff --git a/tiercache-core/src/testFixtures/java/io/tiercache/testkit/CountingRemoteCache.java b/tiercache-core/src/testFixtures/java/io/tiercache/testkit/CountingRemoteCache.java index a5fcca0..45d220d 100644 --- a/tiercache-core/src/testFixtures/java/io/tiercache/testkit/CountingRemoteCache.java +++ b/tiercache-core/src/testFixtures/java/io/tiercache/testkit/CountingRemoteCache.java @@ -2,6 +2,7 @@ import io.tiercache.spi.RemoteCache; import io.tiercache.spi.StoredEntry; +import io.tiercache.spi.TaggedWriteOutcome; import java.time.Duration; import java.util.concurrent.atomic.AtomicInteger; @@ -50,4 +51,15 @@ public void clear() { public boolean setIfAbsent(K key, StoredEntry entry, Duration ttl) { return delegate.setIfAbsent(key, entry, ttl); } + @Override + public boolean supportsTaggedWriteOutcomes() { + return delegate.supportsTaggedWriteOutcomes(); + } + + @Override + public TaggedWriteOutcome putTaggedIfNewer(K key, StoredEntry entry, Duration ttl, String[] tags) { + puts.incrementAndGet(); + return delegate.putTaggedIfNewer(key, entry, ttl, tags); + } + } diff --git a/tiercache-core/src/testFixtures/java/io/tiercache/testkit/FailingRemoteCache.java b/tiercache-core/src/testFixtures/java/io/tiercache/testkit/FailingRemoteCache.java index 9d30f91..a10a06b 100644 --- a/tiercache-core/src/testFixtures/java/io/tiercache/testkit/FailingRemoteCache.java +++ b/tiercache-core/src/testFixtures/java/io/tiercache/testkit/FailingRemoteCache.java @@ -3,6 +3,7 @@ import io.tiercache.Version; import io.tiercache.spi.RemoteCache; import io.tiercache.spi.StoredEntry; +import io.tiercache.spi.TaggedWriteOutcome; import java.time.Duration; import java.util.concurrent.atomic.AtomicBoolean; @@ -71,4 +72,15 @@ public boolean setIfAbsent(K key, StoredEntry entry, Duration ttl) { maybeFail(); return delegate.setIfAbsent(key, entry, ttl); } + @Override + public boolean supportsTaggedWriteOutcomes() { + return delegate.supportsTaggedWriteOutcomes(); + } + + @Override + public TaggedWriteOutcome putTaggedIfNewer(K key, StoredEntry entry, Duration ttl, String[] tags) { + maybeFail(); + return delegate.putTaggedIfNewer(key, entry, ttl, tags); + } + } diff --git a/tiercache-core/src/testFixtures/java/io/tiercache/testkit/InMemoryRemoteCache.java b/tiercache-core/src/testFixtures/java/io/tiercache/testkit/InMemoryRemoteCache.java index ec45b54..ecea5fa 100644 --- a/tiercache-core/src/testFixtures/java/io/tiercache/testkit/InMemoryRemoteCache.java +++ b/tiercache-core/src/testFixtures/java/io/tiercache/testkit/InMemoryRemoteCache.java @@ -2,6 +2,7 @@ import io.tiercache.spi.RemoteCache; import io.tiercache.spi.StoredEntry; +import io.tiercache.spi.TaggedWriteOutcome; import java.time.Duration; import java.util.Map; @@ -27,7 +28,7 @@ public static Function> perName() { } @Override - public StoredEntry get(K key) { + public synchronized StoredEntry get(K key) { Entry entry = store.get(key); if (entry == null) { return null; @@ -40,7 +41,7 @@ public StoredEntry get(K key) { } @Override - public void put(K key, StoredEntry value, Duration ttl) { + public synchronized void put(K key, StoredEntry value, Duration ttl) { store.put(key, new Entry<>(value, System.nanoTime() + ttl.toNanos())); } @@ -50,7 +51,7 @@ public void put(K key, StoredEntry value, Duration ttl) { * the current write time, so reads can classify freshness by age. */ @Override - public void put(K key, StoredEntry entry, Duration ttl, Duration staleTtl) { + public synchronized void put(K key, StoredEntry entry, Duration ttl, Duration staleTtl) { if (staleTtl == null || staleTtl.isZero() || staleTtl.isNegative()) { put(key, entry, ttl); return; @@ -70,7 +71,7 @@ private static StoredEntry stampedWithWriteTime(StoredEntry entry) { } @Override - public void evict(K key) { + public synchronized void evict(K key) { store.remove(key); java.util.Set tags = keyTags.remove(key); if (tags != null) { @@ -84,12 +85,12 @@ public void evict(K key) { } @Override - public void clear() { + public synchronized void clear() { store.clear(); } @Override - public boolean setIfAbsent(K key, StoredEntry entry, Duration ttl) { + public synchronized boolean setIfAbsent(K key, StoredEntry entry, Duration ttl) { long now = System.nanoTime(); Entry candidate = new Entry<>(entry, now + ttl.toNanos()); Entry result = store.merge(key, candidate, @@ -101,16 +102,34 @@ public boolean setIfAbsent(K key, StoredEntry entry, Duration ttl) { private final Map> keyTags = new ConcurrentHashMap<>(); @Override - public void putTagged(K key, StoredEntry entry, Duration ttl, String[] tags) { + public synchronized void putTagged(K key, StoredEntry entry, Duration ttl, String[] tags) { + putTaggedIfNewer(key, entry, ttl, tags); + } + + @Override + public boolean supportsTaggedWriteOutcomes() { + return true; + } + + @Override + public synchronized TaggedWriteOutcome putTaggedIfNewer( + K key, StoredEntry entry, Duration ttl, String[] tags) { + StoredEntry current = get(key); + if (entry.version() != null && current != null && current.version() != null + && entry.version().compareTo(current.version()) < 0) { + return TaggedWriteOutcome.LOST; + } + evict(key); put(key, entry, ttl); for (String tag : tags) { tagIndex.computeIfAbsent(tag, t -> ConcurrentHashMap.newKeySet()).add(key); keyTags.computeIfAbsent(key, k -> ConcurrentHashMap.newKeySet()).add(tag); } + return TaggedWriteOutcome.WON; } @Override - public java.util.List keysByTag(String tag) { + public synchronized java.util.List keysByTag(String tag) { return java.util.List.copyOf(tagIndex.getOrDefault(tag, java.util.Set.of())); } diff --git a/tiercache-invalidation/src/main/java/io/tiercache/invalidation/InvalidationService.java b/tiercache-invalidation/src/main/java/io/tiercache/invalidation/InvalidationService.java index 4e98c8b..1a861c4 100644 --- a/tiercache-invalidation/src/main/java/io/tiercache/invalidation/InvalidationService.java +++ b/tiercache-invalidation/src/main/java/io/tiercache/invalidation/InvalidationService.java @@ -2,506 +2,622 @@ import io.tiercache.InvalidationMessage; import io.tiercache.Version; -import io.tiercache.spi.CacheMetricsListener; -import io.tiercache.spi.CheckedRange; -import io.tiercache.spi.InvalidationHandler; -import io.tiercache.spi.InvalidationEventListener; -import io.tiercache.spi.InvalidationJournal; -import io.tiercache.spi.InvalidationListener; -import io.tiercache.spi.InvalidationTarget; -import io.tiercache.spi.InvalidationTransport; -import io.tiercache.spi.JournalRow; +import io.tiercache.spi.*; import org.slf4j.Logger; import org.slf4j.LoggerFactory; -import java.util.HashSet; -import java.util.LinkedHashSet; -import java.util.List; -import java.util.Map; -import java.util.Set; -import java.util.UUID; -import java.util.concurrent.ConcurrentHashMap; +import java.util.*; +import java.util.concurrent.*; /** - * The invalidation engine: publishes local writes, applies inbound events - * to registered caches (last-write-wins), and heals missed events after a - * transport reconnect by replaying the journal — with a controlled full L1 - * flush when the journal window was exceeded or a read's integrity cannot - * be confirmed. - * - *

    The replay cursor always marks the end of the contiguous applied - * prefix of the journal: live deliveries are tracked in a bounded - * applied-version window, the cursor advances on a bounded cadence over - * exactly the contiguous rows already applied (never past an unconsumed - * row, never to the stream end), and every cursor read is one atomic - * {@link InvalidationJournal#checkedRead} — integrity proof and range from - * the same response. A window overflow behind a delayed row falls back to - * journal-driven catch-up (the journal is the source of truth). - * - *

    Created per factory via {@code TierCacheFactory.Builder.invalidation(...)}: - *

    {@code
    - * .invalidation(versions -> new InvalidationService(transport, journal,
    - *         versions.instanceId(), listener))
    - * }
    - * - *

    Internal — not part of the supported API. The default invalidation - * engine behind the {@code io.tiercache.spi} interfaces; may change in any - * release without notice. - * - * @since 0.1.0 + * Versioned invalidation and bounded, triggered journal recovery. State gates + * cover only in-memory target/cursor commits. Journal I/O and observers never + * execute under those gates. Internal API, not an application entry point. */ public final class InvalidationService implements InvalidationHandler { - private static final Logger log = LoggerFactory.getLogger(InvalidationService.class); - - /** Live deliveries between cursor ticks (bounded-cadence tracking). */ - private static final long CURSOR_TICK_EVERY = 64L; - /** Versions tracked per cache since the confirmed cursor (nominal trigger). */ - private static final int APPLIED_WINDOW_CAPACITY = 128; - /** Hard cap for the applied window; a breach after an unfinished catch-up forces resync-required. */ - private static final int APPLIED_WINDOW_HARD_CAP = 512; - /** Recently confirmed versions per cache (duplicates of them are not re-tracked). */ - private static final int CONFIRMED_SET_CAPACITY = 256; - /** Minimum interval between resync attempts for one cache in resync-required state. */ - private static final long RESYNC_MIN_INTERVAL_NANOS = 1_000_000_000L; - /** Rows read per checked read on the tick and catch-up paths. */ private static final int READ_BATCH = 256; - /** Catch-up batches per trigger (progress is guaranteed; the rest continues on the next trigger). */ - private static final int MAX_CATCHUP_BATCHES = 16; + private static final int MAX_BATCHES = 16; + private static final int WINDOW_CAP = 128; + private static final int WINDOW_HARD_CAP = 512; + private static final int CONFIRMED_CAP = 256; private final InvalidationTransport transport; - private final InvalidationJournal journal; // null = no replay capability + private final InvalidationJournal journal; private final UUID originInstanceId; private final InvalidationListener listener; - private final Map targets = new ConcurrentHashMap<>(); - private final Map subscriptions = new ConcurrentHashMap<>(); - private final Map cursors = new ConcurrentHashMap<>(); - private final Map> appliedWindows = new ConcurrentHashMap<>(); - /** Recently confirmed versions per cache (bounded; duplicates are not re-tracked). */ - private final Map> confirmedVersions = new ConcurrentHashMap<>(); - /** Caches whose version tracking is unconfirmable; resync retries are throttled and lazy. */ - private final Map resyncRequired = new ConcurrentHashMap<>(); - private final Map lastResyncAttemptNanos = new ConcurrentHashMap<>(); - private final Map deliveriesSinceCursorTick = new ConcurrentHashMap<>(); - private final Map cacheLocks = new ConcurrentHashMap<>(); - private final CacheMetricsListener metrics; + private final PublicationObserver publications; + private final java.util.function.LongSupplier clock; private volatile InvalidationEventListener eventListener = InvalidationEventListener.NOOP; + private final Map states = new ConcurrentHashMap<>(); + private final Object lifecycle = new Object(); + private final Object executorGate = new Object(); + private ScheduledExecutorService executor; + private boolean ownsExecutor; + private volatile boolean closed; + private CompletableFuture aggregate; + private Set aggregateTargets = Set.of(); + + private static final class CacheState { + final String cache; + InvalidationTarget target; + String cursor; + long generation, registration, token, deliveries; + final Map applied = new HashMap<>(); + final LinkedHashSet confirmed = new LinkedHashSet<>(); + ScheduledFuture scheduled; + boolean running, ready, tick, replay, reset, resyncRequired, retired, resetOnFailure; + boolean metricClaimed; + AutoCloseable subscription, gauge; + volatile boolean pending; + int failures; + long retryAt, nextCorruptionLog; + RecoveryResult lastResult, safeResult; + long safeTargetGeneration; + CompletableFuture resetCompletion; + CacheState(String cache) { this.cache = cache; } + } + + private record Snapshot(long generation, long token, String cursor, long targetGeneration, + InvalidationTarget target, long deliverySequence) { } + private enum Kind { TICK_DONE, CAUGHT_UP, RESET_SAFE, NO_JOURNAL, FAILED, OBSOLETE } + private record Pass(Kind kind, String baseline, long generation, long targetGeneration) { + Pass(Kind kind, String baseline, long generation) { this(kind, baseline, generation, -1); } + } - /** - * Creates the service without metrics reporting. - * - * @param transport the invalidation transport; its reconnect - * listener is taken over by this service - * @param journal the invalidation journal used to heal missed - * events after a reconnect, or {@code null} to - * disable replay (reconnect then flushes L1 - * entirely) - * @param originInstanceId ID of this instance; own writes are skipped on - * receipt - * @param listener lifecycle listener, or {@code null} for no - * callbacks - * @since 0.1.0 - */ + /** Creates an engine with no metrics binder. */ public InvalidationService(InvalidationTransport transport, InvalidationJournal journal, UUID originInstanceId, InvalidationListener listener) { this(transport, journal, originInstanceId, listener, CacheMetricsListener.NOOP); } - /** - * Creates the service. - * - * @param transport the invalidation transport; its reconnect - * listener is taken over by this service - * @param journal the invalidation journal used to heal missed - * events after a reconnect, or {@code null} to - * disable replay (reconnect then flushes L1 - * entirely) - * @param originInstanceId ID of this instance; own writes are skipped on - * receipt - * @param listener lifecycle listener, or {@code null} for no - * callbacks - * @param metrics metrics listener for invalidation traffic - * @since 0.1.0 - */ + /** Creates an engine; standalone engines lazily own two workers unless configured by a factory. */ public InvalidationService(InvalidationTransport transport, InvalidationJournal journal, UUID originInstanceId, InvalidationListener listener, CacheMetricsListener metrics) { - this.transport = transport; + this(transport, journal, originInstanceId, listener, metrics, System::nanoTime); + } + + InvalidationService(InvalidationTransport transport, InvalidationJournal journal, + UUID originInstanceId, InvalidationListener listener, CacheMetricsListener metrics, + java.util.function.LongSupplier clock) { + this.clock = clock; + this.transport = Objects.requireNonNull(transport); this.journal = journal; - this.originInstanceId = originInstanceId; - this.listener = listener != null ? listener : InvalidationListener.NOOP; - this.metrics = metrics; + this.originInstanceId = Objects.requireNonNull(originInstanceId); + this.listener = listener == null ? InvalidationListener.NOOP : listener; + this.metrics = metrics == null ? CacheMetricsListener.NOOP : metrics; + this.publications = new PublicationObserver(this.metrics, clock); + transport.setMetricsListener(this.metrics); + transport.setGapHandler(new InvalidationGapHandler() { + public CompletionStage reset(String cache) { return resetAsync(cache); } + public RecoveryResult registrationBaseline(String cache) { + CacheState state = states.get(cache); + if (state == null) return null; + synchronized (state) { return currentProof(state, state.safeResult) ? state.safeResult : null; } + } + public boolean isCurrent(String cache, RecoveryResult result) { + CacheState state = states.get(cache); + if (state == null) return false; + synchronized (state) { return currentProof(state, result); } + } + }); transport.setReconnectListener(this::onReconnect); } - @Override - public void onLocalWrite(String cache, Object key, io.tiercache.Version version, - InvalidationMessage.Type type) { - transport.publish(new InvalidationMessage(cache, key, version, originInstanceId, type)); - metrics.onInvalidation(cache, CacheMetricsListener.Direction.SENT); + private boolean currentProof(CacheState state, RecoveryResult result) { + return !closed && !state.retired && state.ready && result != null + && result == state.safeResult && result.status() == RecoveryResult.Status.RESET_SAFE + && state.generation == result.generation() + && state.target.recoveryGeneration() == state.safeTargetGeneration; } @Override - public void onLocalUpdate(String cache, Object key, Object value, io.tiercache.Version version) { - transport.publish(new InvalidationMessage(cache, key, version, originInstanceId, - InvalidationMessage.Type.UPDATE, value)); - metrics.onInvalidation(cache, CacheMetricsListener.Direction.SENT); + public void configureRecoveryExecutor(ScheduledExecutorService workers) { + synchronized (executorGate) { + if (executor != null || closed) throw new IllegalStateException("Recovery executor must be configured before use"); + executor = Objects.requireNonNull(workers); + } + } + + private ScheduledExecutorService workers() { + synchronized (executorGate) { + if (closed) throw new RejectedExecutionException("Invalidation service is closed"); + if (executor == null) { + ScheduledThreadPoolExecutor pool = new ScheduledThreadPoolExecutor(2, runnable -> { + Thread thread = new Thread(runnable, "tiercache-recovery"); + thread.setDaemon(true); + return thread; + }); + pool.setRemoveOnCancelPolicy(true); + executor = pool; + ownsExecutor = true; + } + return executor; + } } @Override public void registerTarget(String cache, InvalidationTarget target) { - targets.put(cache, target); - subscriptions.computeIfAbsent(cache, c -> { - // First subscribe: start from the journal's end — a fresh target - // has an empty L1, so replaying history would only evict fresh - // entries by stale version order. - if (journal != null) { - cursors.put(c, journal.endCursor(c)); + CacheState state; + long registration; + boolean resetRegistration = transport.requiresRegistrationReset(); + AutoCloseable retiredSubscription; + synchronized (lifecycle) { + if (closed) return; + state = states.computeIfAbsent(cache, CacheState::new); + publications.register(cache); + synchronized (state) { + if (state.ready && state.subscription != null && !resetRegistration) { + state.target = target; + state.generation++; + state.lastResult = null; + if (state.pending) { state.replay = true; schedule(state); } + return; + } + retiredSubscription = state.subscription; + state.subscription = null; + state.target = target; + state.generation++; + registration = ++state.registration; + state.ready = false; + state.lastResult = null; } - return transport.subscribe(c, this::onMessage); - }); + } + // Initial baseline and subscription are outside service/state monitors. + closeQuietly(retiredSubscription); + String baseline = journal == null ? "0-0" : journal.endCursor(cache); + synchronized (state) { + if (closed || state.registration != registration) return; + if (resetRegistration) { + long next = target.resetRecovery(target.recoveryGeneration()); + if (next < 0 || target.recoveryGeneration() != next) throw new IllegalStateException("Registration reset was superseded"); + } + state.cursor = baseline; + state.safeTargetGeneration = target.recoveryGeneration(); + state.safeResult = journal == null ? null : new RecoveryResult(RecoveryResult.Status.RESET_SAFE, baseline, state.generation); + state.applied.clear(); + state.confirmed.clear(); + state.ready = true; + } + AutoCloseable subscription = transport.subscribe(cache, message -> onMessage(message, registration)); + AutoCloseable previous; + boolean registerMetric = false; + synchronized (state) { + if (closed || state.registration != registration) previous = subscription; + else { + previous = state.subscription; + state.subscription = subscription; + if (!state.metricClaimed) { state.metricClaimed = true; registerMetric = true; } + } + } + closeQuietly(previous); + if (registerMetric) { + AutoCloseable gauge = null; + try { gauge = metrics.registerRecovery(cache, () -> state.pending); } + catch (Throwable e) { log.warn("Recovery metric registration failed ({})", e.getClass().getSimpleName()); } + boolean discard; + synchronized (state) { discard = closed; if (!discard) state.gauge = gauge; } + if (discard) closeQuietly(gauge); + } + synchronized (state) { if (state.pending) schedule(state); } } - private void onMessage(InvalidationMessage message) { - if (message.originInstanceId().equals(originInstanceId)) { - return; // own write: our L1 is already correct - } - InvalidationTarget target = targets.get(message.cache()); - if (target == null) { - return; + @Override + public void onLocalWrite(String cache, Object key, Version version, InvalidationMessage.Type type) { + if (closed) return; + InvalidationMessage message = new InvalidationMessage(cache, key, version, originInstanceId, type); + PublicationObserver.Attempt attempt = publications.admit(cache); + if (attempt == null) return; + if (type == InvalidationMessage.Type.EVICT_ALL) { + CacheState state = states.get(cache); + if (state != null) synchronized (state) { + state.generation++; + if (state.pending) { state.replay = true; state.lastResult = null; schedule(state); } + } } - applyInbound(target, message); - recordDelivery(message.cache(), target, message.version()); + publish(message, attempt); + } + + @Override + public void onLocalUpdate(String cache, Object key, Object value, Version version) { + if (closed) return; + InvalidationMessage message = new InvalidationMessage(cache, key, version, originInstanceId, + InvalidationMessage.Type.UPDATE, value); + PublicationObserver.Attempt attempt = publications.admit(cache); + if (attempt == null) return; + publish(message, attempt); } - /** Applies an inbound event to the target (live, replayed or caught-up). */ - private void applyInbound(InvalidationTarget target, InvalidationMessage message) { - Object span = metrics.onInvalidationStart(message.cache()); - metrics.onInvalidation(message.cache(), CacheMetricsListener.Direction.RECEIVED); + private void publish(InvalidationMessage message, PublicationObserver.Attempt attempt) { + observe(() -> metrics.onInvalidation(message.cache(), CacheMetricsListener.Direction.SENT)); try { - switch (message.type()) { - case INVALIDATE -> target.evictL1IfNewer(message.key(), message.version()); - case UPDATE -> target.applyUpdateL1(message.key(), message.payload(), - message.version()); - case EVICT_ALL -> target.evictAllL1(); - } - eventListener.onEvent(message.cache(), message); - } finally { - metrics.onInvalidationEnd(message.cache(), span); + transport.publishAsync(message).whenComplete(attempt::complete); + } catch (Throwable failure) { + // Publication follows the data commit; a submission/serialization error + // must not turn that successful write into a reported business failure. + attempt.complete(null, failure); } } - /** - * Tracks a live-applied event for cursor advancement. A version the - * cursor has already passed (bounded confirmed-set) is a late duplicate: - * applied to L1 but NOT re-tracked. Window overflow falls back to - * journal-driven catch-up; a catch-up that reached the journal end - * accounts for everything left in the window (a live event's row is - * journaled before its publish, so unseen versions sit at or before the - * cursor). In resync-required state tracking is skipped entirely and - * resync retries on a bounded time throttle. - */ - private void recordDelivery(String cache, InvalidationTarget target, Version version) { - if (journal == null) { + private void onMessage(InvalidationMessage message, long registration) { + if (closed) { + if (transport.requiresRegistrationReset()) throw new IllegalStateException("Invalidation service is closed"); return; } - Object lock = cacheLocks.computeIfAbsent(cache, c -> new Object()); - synchronized (lock) { - if (resyncRequired.containsKey(cache)) { - long now = System.nanoTime(); - Long last = lastResyncAttemptNanos.get(cache); - if (last == null || now - last >= RESYNC_MIN_INTERVAL_NANOS) { - lastResyncAttemptNanos.put(cache, now); - if (catchUpFromJournal(cache, target) == CatchUpOutcome.CAUGHT_UP) { - resyncRequired.remove(cache); - } - } - return; // degraded: no tracking until the resync completes + if (message.originInstanceId().equals(originInstanceId)) return; + CacheState state = states.get(message.cache()); + if (state == null) return; + synchronized (state) { + if (closed || !state.ready || state.registration != registration) { + throw new IllegalStateException("Invalidation registration was closed or superseded"); } - LinkedHashSet confirmed = confirmedVersions.get(cache); - if (confirmed != null && confirmed.contains(version)) { - return; // late duplicate of an already-accounted row + applyLive(state.target, message); + if (message.type() == InvalidationMessage.Type.EVICT_ALL) { + state.generation++; + if (state.running) state.replay = true; } - Set window = appliedWindows.computeIfAbsent(cache, c -> new HashSet<>()); - window.add(version); - if (window.size() > APPLIED_WINDOW_CAPACITY) { - CatchUpOutcome outcome = catchUpFromJournal(cache, target); - if (outcome == CatchUpOutcome.CAUGHT_UP) { - window.clear(); - } else if (outcome == CatchUpOutcome.FAILED - || window.size() > APPLIED_WINDOW_HARD_CAP) { - enterResync(cache); + if (journal != null) { + if (!state.resyncRequired && !state.confirmed.contains(message.version())) { + if (state.applied.size() >= WINDOW_HARD_CAP && !state.applied.containsKey(message.version())) { + state.applied.clear(); + state.resyncRequired = true; + state.generation++; + } else state.applied.putIfAbsent(message.version(), state.deliveries + 1); } - } - long deliveries = deliveriesSinceCursorTick.merge(cache, 1L, Long::sum); - if (deliveries % CURSOR_TICK_EVERY == 0) { - advanceCursor(cache, target); + if (state.resyncRequired || state.applied.size() > WINDOW_CAP) state.replay = true; + if (++state.deliveries % JournalProtocol.CURSOR_CADENCE == 0) state.tick = true; + if (state.replay || state.tick) { state.pending = true; schedule(state); } } } + notifyEvent(message, false); } - /** Marks a version as cursor-confirmed (bounded per cache; insertion-ordered eviction). */ - private void confirm(String cache, Version version) { - LinkedHashSet confirmed = - confirmedVersions.computeIfAbsent(cache, c -> new LinkedHashSet<>()); - confirmed.add(version); - while (confirmed.size() > CONFIRMED_SET_CAPACITY) { - confirmed.remove(confirmed.iterator().next()); + private static void applyLive(InvalidationTarget target, InvalidationMessage message) { + switch (message.type()) { + case INVALIDATE -> target.evictL1IfNewer(message.key(), message.version()); + case UPDATE -> target.applyUpdateL1(message.key(), message.payload(), message.version()); + case EVICT_ALL -> target.evictAllL1(); } } - /** - * Enters the resync-required state: tracking memory is freed and the - * cursor stays where it is. While set, recovery is lazy and throttled; - * missed rows are applied at the next recovery trigger (a delivery - * after reads heal, or a reconnect replay) — until then L1 may serve - * stale data. - */ - private void enterResync(String cache) { - Set window = appliedWindows.get(cache); - if (window != null) { - window.clear(); - } - resyncRequired.put(cache, Boolean.TRUE); - lastResyncAttemptNanos.put(cache, System.nanoTime()); - log.warn("Version tracking for cache '{}' is unconfirmable (catch-up failed or the " - + "hard cap was exceeded); entering resync-required state. Missed rows are " - + "applied at the next recovery trigger; L1 may serve stale data until then.", - cache); + private void notifyEvent(InvalidationMessage message, boolean replayed) { + Object span = null; + try { span = metrics.onInvalidationStart(message.cache()); } + catch (Throwable e) { log.warn("Invalidation observer failed ({})", e.getClass().getSimpleName()); } + observe(() -> metrics.onInvalidation(message.cache(), CacheMetricsListener.Direction.RECEIVED)); + observe(() -> eventListener.onEvent(message.cache(), message)); + Object handle = span; + observe(() -> metrics.onInvalidationEnd(message.cache(), handle)); + if (replayed) observe(() -> metrics.onInvalidation(message.cache(), CacheMetricsListener.Direction.REPLAYED)); } - /** The outcome of one catch-up pass. */ - private enum CatchUpOutcome { - /** The read reached the current journal end: everything left in the window is accounted. */ - CAUGHT_UP, - /** The batch budget was exhausted before the journal end; progress is kept. */ - MORE_WORK, - /** A read failed before the journal end. */ - FAILED + @Override public void setEventListener(InvalidationEventListener listener) { + eventListener = listener == null ? InvalidationEventListener.NOOP : listener; } + @Override public void onL2Recovery() { onReconnect(); } + private void onReconnect() { requestAll(false); } - /** - * Cadence tick: advances the cursor over exactly the contiguous rows - * from one checked read whose versions are already accounted for (in - * the applied window, or own writes — our L1 holds them by definition), - * stopping at the first unconsumed row. Never advances to the stream - * end and never past an unconsumed row. - */ - private void advanceCursor(String cache, InvalidationTarget target) { - String cursor = cursors.get(cache); - if (cursor == null) { - return; - } - CheckedRange range; - try { - range = journal.checkedRead(cache, cursor, READ_BATCH); - } catch (RuntimeException e) { - // A failed tick read proves nothing either way; the cursor stays - // put and the next tick retries (a later replay may widen). - log.debug("Cursor tick read failed for cache '{}'; will retry on a later tick.", - cache, e); - return; - } - if (!range.startIntact()) { - flushL1(cache, target, - "the replay cursor row was trimmed; prefix integrity is unconfirmable", null); - return; - } - Set window = appliedWindows.get(cache); - String confirmed = cursor; - for (JournalRow row : rowsAfterCursor(range, cursor)) { - if (row.message().originInstanceId().equals(originInstanceId) - || (window != null && window.remove(row.message().version()))) { - confirm(cache, row.message().version()); - confirmed = row.cursor(); - } else { - break; // the first unconsumed row: never advance past it + @Override + public CompletionStage recoverAsync(Executor ignored) { return requestAll(true); } + + private CompletionStage requestAll(boolean awaitResult) { + CompletableFuture result; + synchronized (lifecycle) { + if (closed) return CompletableFuture.completedFuture(false); + if (awaitResult && (aggregate == null || aggregate.isDone())) aggregate = new CompletableFuture<>(); + Set selected = new HashSet<>(); + for (CacheState state : states.values()) synchronized (state) { + if (!state.ready) continue; + selected.add(state); + state.resetOnFailure = true; + state.generation++; + state.lastResult = null; + state.replay = true; + state.pending = true; + schedule(state); } + if (aggregate != null && !aggregate.isDone()) aggregateTargets = selected; + result = aggregate; } - cursors.put(cache, confirmed); + checkAggregate(); + return result == null ? CompletableFuture.completedFuture(false) : result; } - /** - * Window overflow behind a delayed row: catches up directly from the - * journal (the source of truth), applying rows in order and advancing - * the cursor over every row read — applied or stale-dropped, both are - * accounted. Returns the tri-state outcome; only {@link - * CatchUpOutcome#CAUGHT_UP} proves the window fully accounted. - */ - private CatchUpOutcome catchUpFromJournal(String cache, InvalidationTarget target) { - String cursor = cursors.get(cache); - if (cursor == null) { - return CatchUpOutcome.CAUGHT_UP; // no cursor: nothing to account against + @Override + public CompletionStage resetAsync(String cache) { + CacheState state = states.get(cache); + if (state == null || closed) return CompletableFuture.completedFuture(RecoveryResult.closed()); + synchronized (state) { + if (closed || !state.ready) return CompletableFuture.completedFuture(RecoveryResult.closed()); + if (state.resetCompletion == null || state.resetCompletion.isDone()) state.resetCompletion = new CompletableFuture<>(); + state.generation++; + state.lastResult = null; + state.reset = true; + state.pending = true; + schedule(state); + return state.resetCompletion; } - Set window = appliedWindows.get(cache); - for (int batch = 0; batch < MAX_CATCHUP_BATCHES; batch++) { - CheckedRange range; - try { - range = journal.checkedRead(cache, cursor, READ_BATCH); - } catch (RuntimeException e) { - log.debug("Catch-up read failed for cache '{}'; will retry on the next trigger.", - cache, e); - return CatchUpOutcome.FAILED; - } - if (!range.startIntact()) { - flushL1(cache, target, - "the replay cursor row was trimmed; prefix integrity is unconfirmable", null); - return CatchUpOutcome.CAUGHT_UP; // the flush settles the state completely - } - List rows = rowsAfterCursor(range, cursor); - if (rows.isEmpty()) { - return CatchUpOutcome.CAUGHT_UP; + } + + // Caller owns the state gate. A running pass absorbs triggers; it alone + // schedules its continuation. Thus at most one token exists per cache. + private void schedule(CacheState state) { + if (closed || state.retired || !state.ready || state.running || state.scheduled != null) return; + long delay = state.retryAt == 0 ? 0 : Math.max(0, state.retryAt - clock.getAsLong()); + try { state.scheduled = workers().schedule(() -> run(state), delay, TimeUnit.NANOSECONDS); } + catch (RejectedExecutionException e) { if (!closed) throw e; } + } + + private void run(CacheState state) { + boolean replay, reset, resetOnFailure; + long generation; + synchronized (state) { + state.scheduled = null; + if (closed || state.retired || !state.ready || state.running) return; + state.running = true; + state.token++; + generation = state.generation; + replay = state.replay || state.resyncRequired; + reset = state.reset; + resetOnFailure = state.resetOnFailure; + state.replay = state.tick = state.reset = false; + } + Pass result; + try { result = pass(state, replay, reset, resetOnFailure, generation); } + catch (Throwable e) { + log.warn("Recovery pass failed for cache '{}' ({})", state.cache, e.getClass().getSimpleName()); + result = new Pass(Kind.FAILED, null, generation); + } + CompletableFuture resetWaiter = null; + RecoveryResult completion = null; + synchronized (state) { + state.running = false; + if (closed || state.retired) return; + if (result.kind != Kind.OBSOLETE && (state.generation != result.generation + || (result.kind == Kind.RESET_SAFE && state.target.recoveryGeneration() != result.targetGeneration))) { + result = obsolete(state); } - for (JournalRow row : rows) { - if (!row.message().originInstanceId().equals(originInstanceId)) { - applyInbound(target, row.message()); + if (result.kind == Kind.FAILED) { + state.resyncRequired = true; + state.applied.clear(); + state.replay = true; + if (reset || result.targetGeneration >= 0) state.reset = true; + state.failures = Math.min(6, state.failures + 1); + long seconds = Math.min(30, 1L << (state.failures - 1)); + state.retryAt = clock.getAsLong() + TimeUnit.SECONDS.toNanos(seconds); + completion = RecoveryResult.failed(); + } else if (result.kind != Kind.OBSOLETE) { + if (result.kind == Kind.RESET_SAFE) { + // The baseline makes the clear safe, but a repeatedly + // unreadable history must not cause an immediate flush loop. + state.failures = Math.min(6, state.failures + 1); + state.retryAt = clock.getAsLong() + TimeUnit.SECONDS.toNanos( + Math.min(30, 1L << (state.failures - 1))); + } else { + state.failures = 0; + state.retryAt = 0; } - if (window != null) { - window.remove(row.message().version()); + if (result.kind != Kind.TICK_DONE) { + completion = new RecoveryResult(switch (result.kind) { + case RESET_SAFE -> RecoveryResult.Status.RESET_SAFE; + case NO_JOURNAL -> RecoveryResult.Status.NO_JOURNAL; + default -> RecoveryResult.Status.CAUGHT_UP; + }, result.baseline, result.generation); } - confirm(cache, row.message().version()); - cursor = row.cursor(); } - cursors.put(cache, cursor); + // A newer request may have arrived after the final batch committed. + // Do not settle its completion using this obsolete pass's result. + if (completion != null && state.generation == result.generation) { + state.lastResult = completion; + if (completion.status() == RecoveryResult.Status.RESET_SAFE) { + state.safeResult = completion; + state.safeTargetGeneration = result.targetGeneration; + } + resetWaiter = state.resetCompletion; + state.resetCompletion = null; + } else completion = null; + state.pending = state.replay || state.tick || state.reset || state.resyncRequired; + if (state.pending) schedule(state); } - return CatchUpOutcome.MORE_WORK; + if (resetWaiter != null) resetWaiter.complete(completion); + checkAggregate(); } - @Override - public void setEventListener(InvalidationEventListener listener) { - this.eventListener = listener != null ? listener : InvalidationEventListener.NOOP; + private Snapshot snapshot(CacheState state) { + return new Snapshot(state.generation, state.token, state.cursor, + state.target.recoveryGeneration(), state.target, state.deliveries); } - - @Override - public void onL2Recovery() { - onReconnect(); + private boolean valid(CacheState state, Snapshot snapshot) { + return !closed && !state.retired && state.generation == snapshot.generation + && state.token == snapshot.token && Objects.equals(state.cursor, snapshot.cursor) + && state.target == snapshot.target && state.target.recoveryGeneration() == snapshot.targetGeneration; + } + private Pass obsolete(CacheState state) { + state.replay = true; + if (state.resetCompletion != null) state.reset = true; + return new Pass(Kind.OBSOLETE, null, -1); } - private void onReconnect() { - if (journal == null) { - // No replay capability: the honest fallback is a full flush. - log.warn("Invalidation transport reconnected without a journal; " - + "flushing L1 for all caches to avoid stale entries."); - targets.forEach((cache, target) -> target.evictAllL1()); - return; - } - targets.forEach((cache, target) -> { - Object lock = cacheLocks.computeIfAbsent(cache, c -> new Object()); - synchronized (lock) { - String cursor = cursors.get(cache); - if (cursor == null) { - return; + private Pass pass(CacheState state, boolean replay, boolean reset, boolean resetOnFailure, long expectedGeneration) { + if (reset || journal == null) return reset(state, expectedGeneration); + for (int batch = 0; batch < (replay ? MAX_BATCHES : 1); batch++) { + Snapshot snapshot; + synchronized (state) { + if (closed || state.retired) return new Pass(Kind.OBSOLETE, null, -1); + if (state.generation != expectedGeneration) return obsolete(state); + snapshot = snapshot(state); + } + CheckedRange range; + try { range = journal.checkedRead(state.cache, snapshot.cursor, READ_BATCH); } + catch (RuntimeException e) { + synchronized (state) { if (!valid(state, snapshot)) return obsolete(state); } + if (e instanceof JournalCorruptionException corrupt) { + boolean report; + synchronized (state) { + long now = clock.getAsLong(); + report = state.nextCorruptionLog == 0 || now - state.nextCorruptionLog >= 0; + if (report) state.nextCorruptionLog = now + TimeUnit.SECONDS.toNanos(30); + } + observe(() -> metrics.onStreamFailure(state.cache, CacheMetricsListener.StreamResult.DECODE_FAILED)); + if (report) observe(() -> log.warn("Journal corruption: cache={}, row={}, failure={}", + state.cache, corrupt.rowId(), corrupt.getClass().getSimpleName())); + return reset(state, expectedGeneration); } - try { - while (true) { - CheckedRange range = journal.checkedRead(cache, cursor, READ_BATCH); - if (!range.startIntact()) { - flushL1(cache, target, - "the journal window was exceeded during the disconnect", null); - return; - } - List rows = rowsAfterCursor(range, cursor); - if (rows.isEmpty()) { - // Replay reached the journal end: this is a full - // journal-driven recovery — normal tracking resumes. - resyncRequired.remove(cache); - return; // nothing missed - } - Set window = appliedWindows.get(cache); - for (JournalRow row : rows) { - if (!row.message().originInstanceId().equals(originInstanceId)) { - applyInbound(target, row.message()); - metrics.onInvalidation(cache, - CacheMetricsListener.Direction.REPLAYED); - } - if (window != null) { - window.remove(row.message().version()); - } - confirm(cache, row.message().version()); - // The cursor records the ID of the last row - // actually read — never the stream end: a row - // written after this read sits past the cursor - // and is picked up next time. - cursor = row.cursor(); + return resetOnFailure ? reset(state, expectedGeneration) + : new Pass(Kind.FAILED, null, expectedGeneration); + } + List notifications = new ArrayList<>(); + Pass done = null; + synchronized (state) { + if (!valid(state, snapshot)) return obsolete(state); + if (!range.startIntact()) { /* baseline acquisition below, outside the gate */ } + else { + List rows = rowsAfterCursor(range, snapshot.cursor); + long targetGeneration = snapshot.targetGeneration; + for (JournalRow row : rows) { + InvalidationMessage message = row.message(); + boolean own = message.originInstanceId().equals(originInstanceId); + if (!replay && !own && !state.applied.containsKey(message.version())) break; + if (replay && (!own || message.type() == InvalidationMessage.Type.EVICT_ALL)) { + long next = state.target.applyRecovery(message, targetGeneration); + if (next < 0 || state.target.recoveryGeneration() != next) return obsolete(state); + targetGeneration = next; + if (!own) notifications.add(message); + if (message.type() == InvalidationMessage.Type.EVICT_ALL) expectedGeneration = ++state.generation; } - cursors.put(cache, cursor); + state.applied.remove(message.version()); + state.confirmed.add(message.version()); + while (state.confirmed.size() > CONFIRMED_CAP) state.confirmed.remove(state.confirmed.iterator().next()); + state.cursor = row.cursor(); } - } catch (RuntimeException e) { - // A failed replay read (e.g. the trim counter is - // unreadable): integrity is unconfirmable — take the - // flush path, never assume "no loss". - flushL1(cache, target, "the journal replay read failed", e); + if (rows.isEmpty()) { + state.applied.values().removeIf(sequence -> sequence <= snapshot.deliverySequence); + state.resyncRequired = false; + state.resetOnFailure = false; + done = new Pass(replay ? Kind.CAUGHT_UP : Kind.TICK_DONE, state.cursor, state.generation); + } else if (!replay) done = new Pass(Kind.TICK_DONE, state.cursor, state.generation); } } - }); + if (!range.startIntact()) return reset(state, expectedGeneration); + for (InvalidationMessage message : notifications) notifyEvent(message, true); + if (done != null) return done; + } + synchronized (state) { return obsolete(state); } // bounded continuation, retaining cursor progress } - /** - * The flush path: L1 is dropped for the cache, the flush is signaled - * (log + listener + dropped metric), and the cursor re-baselines. - * - *

    Order matters: the baseline is captured BEFORE L1 is cleared — - * rows journaled up to it are covered by the clear (a re-warm reads - * current L2, which includes them), while rows journaled after the - * clear stay ahead of the stored cursor and are applied by the next - * replay or catch-up. If the baseline read fails, the previous - * confirmed cursor is kept (never advance past unread rows). Observers - * (log, listener, metrics) fire last, after the state is settled, so an - * observer failure cannot cancel the clear. - */ - private void flushL1(String cache, InvalidationTarget target, String reason, Throwable cause) { - String baseline; - try { - baseline = journal.endCursor(cache); - } catch (RuntimeException e) { - baseline = cursors.get(cache); - log.warn("Journal baseline read failed for cache '{}'; keeping the previous " - + "confirmed cursor for the flush.", cache, e); + private Pass reset(CacheState state, long expectedGeneration) { + Snapshot snapshot; + synchronized (state) { + if (closed || state.retired) return new Pass(Kind.OBSOLETE, null, -1); + if (state.generation != expectedGeneration) return obsolete(state); + state.generation++; + snapshot = snapshot(state); + } + String baseline = null; + boolean safe = journal == null; + if (journal != null) { + try { baseline = journal.endCursor(state.cache); safe = baseline != null; } + catch (RuntimeException e) { log.debug("Recovery baseline unavailable for '{}' ({})", state.cache, e.getClass().getSimpleName()); } } - target.evictAllL1(); - cursors.put(cache, baseline); - resyncRequired.remove(cache); // the flush settles the state completely - lastResyncAttemptNanos.remove(cache); - Set window = appliedWindows.get(cache); - if (window != null) { - window.clear(); + long generation; + long targetGeneration; + synchronized (state) { + if (!valid(state, snapshot)) return obsolete(state); + long next; + try { next = state.target.resetRecovery(snapshot.targetGeneration); } + catch (Throwable error) { + // The reservation already advanced generation. Report failure + // for that reservation, not the pass's obsolete starting epoch. + return new Pass(Kind.FAILED, null, state.generation, snapshot.targetGeneration); + } + if (next < 0 || state.target.recoveryGeneration() != next) return obsolete(state); + targetGeneration = next; + generation = ++state.generation; + if (safe && journal != null) state.cursor = baseline; + state.applied.clear(); + state.confirmed.clear(); + state.resyncRequired = !safe; + // A safe reset covers only the baseline. Follow with a bounded + // continuation; rows appended after it must still be consumed. + state.replay = journal != null; } - if (cause == null) { - log.warn("Invalidation journal cannot confirm contiguous history for cache '{}' " - + "({}); flushing L1 entirely.", cache, reason); + boolean baselineEstablished = safe; + if (journal == null) { + observe(() -> log.warn("Recovery without a journal for cache '{}'; clearing L1 without claiming replay", state.cache)); } else { - log.warn("Invalidation journal cannot confirm contiguous history for cache '{}' " - + "({}); flushing L1 entirely.", cache, reason, cause); + observe(() -> log.warn("Conservative L1 reset for cache '{}'; baseline established: {}", state.cache, baselineEstablished)); + observe(() -> listener.onJournalOverflow(state.cache)); + observe(() -> metrics.onInvalidation(state.cache, CacheMetricsListener.Direction.DROPPED)); } - listener.onJournalOverflow(cache); - metrics.onInvalidation(cache, CacheMetricsListener.Direction.DROPPED); + return new Pass(!safe ? Kind.FAILED : journal == null ? Kind.NO_JOURNAL : Kind.RESET_SAFE, baseline, generation, targetGeneration); + } + + private void checkAggregate() { + CompletableFuture completion = null; + Boolean result = null; + synchronized (lifecycle) { + if (aggregate == null || aggregate.isDone()) return; + boolean all = true; + for (CacheState state : aggregateTargets) synchronized (state) { + if (state.lastResult != null && !state.lastResult.succeeded()) { result = false; break; } + if (state.lastResult == null) all = false; + } + if (result != null || all) { + completion = aggregate; + aggregate = null; + aggregateTargets = Set.of(); + if (result == null) result = true; + } + } + if (completion != null) completion.complete(result); } - /** - * The rows to process from a checked read: all of them for a beginning - * cursor, everything after the cursor's own row otherwise (an intact - * non-beginning read starts AT the cursor row, which is already - * accounted for). - */ private static List rowsAfterCursor(CheckedRange range, String cursor) { List rows = range.rows(); - if (!rows.isEmpty() && rows.get(0).cursor().equals(cursor)) { - return rows.subList(1, rows.size()); - } - return rows; + return !rows.isEmpty() && rows.get(0).cursor().equals(cursor) ? rows.subList(1, rows.size()) : rows; + } + + private static void observe(Runnable callback) { + try { callback.run(); } + catch (Throwable e) { log.warn("Invalidation observer failed ({})", e.getClass().getSimpleName()); } + } + private static void closeQuietly(AutoCloseable resource) { + if (resource != null) observe(() -> { + try { resource.close(); } catch (Exception e) { throw new IllegalStateException(e); } + }); } @Override public void close() { - subscriptions.values().forEach(subscription -> { - try { - subscription.close(); - } catch (Exception e) { - log.warn("Failed to close invalidation subscription", e); + List resources = new ArrayList<>(); + List> completions = new ArrayList<>(); + CompletableFuture recovery; + synchronized (lifecycle) { + if (closed) return; + closed = true; + publications.stopAdmission(); + recovery = aggregate; + aggregate = null; + aggregateTargets = Set.of(); + for (CacheState state : states.values()) synchronized (state) { + state.retired = true; + state.generation++; + if (state.scheduled != null) state.scheduled.cancel(false); + state.scheduled = null; + if (state.resetCompletion != null) completions.add(state.resetCompletion); + resources.add(state.subscription); + resources.add(state.gauge); } - }); - subscriptions.clear(); - targets.clear(); - transport.close(); + states.clear(); + } + if (recovery != null) recovery.complete(false); + for (CompletableFuture completion : completions) completion.complete(RecoveryResult.closed()); + resources.forEach(InvalidationService::closeQuietly); + ScheduledExecutorService owned; + synchronized (executorGate) { owned = ownsExecutor ? executor : null; } + if (owned != null) owned.shutdownNow(); + publications.close(); + closeQuietly(transport); } } diff --git a/tiercache-invalidation/src/main/java/io/tiercache/invalidation/JournalProtocol.java b/tiercache-invalidation/src/main/java/io/tiercache/invalidation/JournalProtocol.java new file mode 100644 index 0000000..9b149d4 --- /dev/null +++ b/tiercache-invalidation/src/main/java/io/tiercache/invalidation/JournalProtocol.java @@ -0,0 +1,20 @@ +package io.tiercache.invalidation; + +/** Internal protocol limits shared by receivers, Redis journals and framework wiring. */ +public final class JournalProtocol { + public static final int CURSOR_CADENCE = 64; + public static final int MIN_CAPACITY = CURSOR_CADENCE + 1; + public static final int DEFAULT_CAPACITY = 10_000; + + private JournalProtocol() { } + + /** Validates the configured row target, not a guarantee of sufficient recovery retention. */ + public static int requireCapacity(int capacity) { + if (capacity < MIN_CAPACITY) { + throw new IllegalArgumentException("tiercache.invalidation.journal-capacity=" + capacity + + " must be at least " + MIN_CAPACITY + ": the " + CURSOR_CADENCE + + "-event cursor cadence also requires the previous confirmed cursor row"); + } + return capacity; + } +} diff --git a/tiercache-invalidation/src/main/java/io/tiercache/invalidation/PublicationObserver.java b/tiercache-invalidation/src/main/java/io/tiercache/invalidation/PublicationObserver.java new file mode 100644 index 0000000..895eaff --- /dev/null +++ b/tiercache-invalidation/src/main/java/io/tiercache/invalidation/PublicationObserver.java @@ -0,0 +1,176 @@ +package io.tiercache.invalidation; + +import io.tiercache.spi.CacheMetricsListener; +import io.tiercache.spi.PublicationOutcome; +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; + +import java.util.ArrayDeque; +import java.util.HashMap; +import java.util.Map; +import java.util.concurrent.*; +import java.util.concurrent.atomic.AtomicBoolean; +import java.util.concurrent.atomic.AtomicLongArray; +import java.util.concurrent.atomic.AtomicReference; +import java.util.concurrent.locks.ReentrantLock; +import java.util.function.LongSupplier; + +/** Bounded terminal accounting and coalesced publication diagnostics. */ +final class PublicationObserver implements AutoCloseable { + static final String OVERFLOW_CACHE = "__tiercache_unregistered__"; + private static final Logger log = LoggerFactory.getLogger(PublicationObserver.class); + private static final long WARNING_INTERVAL = TimeUnit.SECONDS.toNanos(30); + private static final PublicationOutcome[] OUTCOMES = PublicationOutcome.values(); + private final CacheMetricsListener metrics; + private final LongSupplier clock; + private final ReentrantLock gate = new ReentrantLock(); + private final Map caches = new HashMap<>(); + private final Bucket overflow = new Bucket(OVERFLOW_CACHE); + private final ArrayDeque ready = new ArrayDeque<>(); + private final ThreadPoolExecutor worker; + private volatile Thread workerThread; + private boolean accepting = true, scheduling = true, notifying = true, scheduled, closing; + + private static final class Bucket { + final String cache; + final AtomicLongArray totals = new AtomicLongArray(OUTCOMES.length); + final long[] exported = new long[OUTCOMES.length]; // worker-owned observation cursors + final AtomicReference failure = new AtomicReference<>(); + boolean queued; + volatile long nextWarning; + Bucket(String cache) { this.cache = cache; } + } + + static final class Attempt { + private final PublicationObserver owner; + private final Bucket bucket; + private final AtomicBoolean completed = new AtomicBoolean(); + Attempt(PublicationObserver owner, Bucket bucket) { this.owner = owner; this.bucket = bucket; } + void complete(PublicationOutcome outcome, Throwable error) { + if (!completed.compareAndSet(false, true)) return; + PublicationOutcome terminal = error != null || outcome == null ? PublicationOutcome.FAILED : outcome; + bucket.totals.incrementAndGet(terminal.ordinal()); + if (terminal == PublicationOutcome.FAILED) { + Throwable cause = error; + for (int depth = 0; depth < 8 && (cause instanceof CompletionException || cause instanceof ExecutionException) + && cause.getCause() != null; depth++) cause = cause.getCause(); + String failure = cause == null ? "PublicationFailure" : cause.getClass().getSimpleName(); + bucket.failure.set(failure.substring(0, Math.min(128, failure.length()))); + } + owner.enqueue(bucket); + } + long count(PublicationOutcome outcome) { return bucket.totals.get(outcome.ordinal()); } + } + + PublicationObserver(CacheMetricsListener metrics, LongSupplier clock) { + this.metrics = metrics; + this.clock = clock; + worker = new ThreadPoolExecutor(1, 1, 0, TimeUnit.MILLISECONDS, + new ArrayBlockingQueue<>(1), runnable -> { + Thread thread = new Thread(runnable, "tiercache-publication-observer"); + thread.setDaemon(true); + workerThread = thread; + return thread; + }, new ThreadPoolExecutor.AbortPolicy()); + } + + void register(String cache) { + gate.lock(); + try { if (accepting) caches.computeIfAbsent(cache, Bucket::new); } + finally { gate.unlock(); } + } + + Attempt admit(String cache) { + gate.lock(); + try { return accepting ? new Attempt(this, caches.getOrDefault(cache, overflow)) : null; } + finally { gate.unlock(); } + } + + void stopAdmission() { + gate.lock(); + try { accepting = false; } + finally { gate.unlock(); } + } + + private void enqueue(Bucket bucket) { + gate.lock(); + try { + if (!scheduling) return; + if (!bucket.queued) { bucket.queued = true; ready.add(bucket); } + if (!scheduled) { + scheduled = true; + worker.execute(this::drain); + } + } finally { gate.unlock(); } + } + + private void drain() { + while (true) { + Bucket bucket; + gate.lock(); + try { + bucket = ready.poll(); + if (bucket == null || !notifying) { scheduled = false; return; } + // Completion after this dequeue can enqueue exactly one next token. + bucket.queued = false; + } finally { gate.unlock(); } + String failure = bucket.failure.getAndSet(null); + if (failure != null && notificationAdmitted()) warn(bucket, failure); + for (PublicationOutcome outcome : OUTCOMES) { + int index = outcome.ordinal(); + long total = bucket.totals.get(index); + long count = total - bucket.exported[index]; + bucket.exported[index] = total; + if (count != 0 && notificationAdmitted()) { + try { metrics.onPublication(bucket.cache, outcome, count); } + catch (Throwable error) { warn(bucket, "Observer:" + error.getClass().getSimpleName()); } + } + } + } + } + + private boolean notificationAdmitted() { + gate.lock(); + try { return notifying; } + finally { gate.unlock(); } + } + + private void warn(Bucket bucket, String failure) { + long now = clock.getAsLong(); + if (bucket.nextWarning != 0 && now - bucket.nextWarning < 0) return; + bucket.nextWarning = now + WARNING_INTERVAL; + try { log.warn("Invalidation publication diagnostic: cache={}, failure={}", bucket.cache, failure); } + catch (Throwable ignored) { /* Diagnostics cannot terminate the worker. */ } + } + + /** Diagnostic seam: pending tokens cannot exceed registered caches plus one. */ + int pendingTokens() { + gate.lock(); + try { return ready.size(); } + finally { gate.unlock(); } + } + int bucketCount() { + gate.lock(); + try { return caches.size() + 1; } + finally { gate.unlock(); } + } + + @Override public void close() { + gate.lock(); + try { + if (closing) return; + closing = true; + accepting = false; + scheduling = false; + worker.shutdown(); + } finally { gate.unlock(); } + if (Thread.currentThread() != workerThread) { + try { worker.awaitTermination(1, TimeUnit.SECONDS); } + catch (InterruptedException e) { Thread.currentThread().interrupt(); } + } + gate.lock(); + try { notifying = false; ready.clear(); } + finally { gate.unlock(); } + worker.shutdownNow(); + } +} diff --git a/tiercache-invalidation/src/test/java/io/tiercache/invalidation/GapAuthorizationTest.java b/tiercache-invalidation/src/test/java/io/tiercache/invalidation/GapAuthorizationTest.java new file mode 100644 index 0000000..dd37132 --- /dev/null +++ b/tiercache-invalidation/src/test/java/io/tiercache/invalidation/GapAuthorizationTest.java @@ -0,0 +1,49 @@ +package io.tiercache.invalidation; + +import io.tiercache.*; +import io.tiercache.spi.*; +import io.tiercache.testkit.InMemoryJournal; +import org.junit.jupiter.api.Test; +import java.util.*; +import java.util.concurrent.*; +import java.util.concurrent.atomic.*; +import static org.junit.jupiter.api.Assertions.*; + +class GapAuthorizationTest { + static class Transport implements InvalidationTransport { + InvalidationGapHandler gaps; int subscriptions; + public void publish(InvalidationMessage message) { } + public AutoCloseable subscribe(String cache, java.util.function.Consumer handler) { subscriptions++; return () -> { }; } + public void setGapHandler(InvalidationGapHandler handler) { gaps = handler; } + public boolean requiresRegistrationReset() { return true; } + public void close() { } + } + @Test void registrationAndResetProofsAreCacheAndGenerationScoped() throws Exception { + var transport = new Transport(); var target = new RecoveryProtocolTest.Target(); + try (var service = new InvalidationService(transport, new InMemoryJournal(100), UUID.randomUUID(), InvalidationListener.NOOP)) { + service.registerTarget("c", target); assertEquals(1, target.clears.get()); + var first = transport.gaps.registrationBaseline("c"); assertNotNull(first); + assertTrue(transport.gaps.isCurrent("c", first)); assertFalse(transport.gaps.isCurrent("other", first)); + assertFalse(transport.gaps.isCurrent("c", new RecoveryResult(first.status(), first.baseline(), first.generation()))); + var reset = transport.gaps.reset("c").toCompletableFuture().get(5, TimeUnit.SECONDS); + assertTrue(transport.gaps.isCurrent("c", reset)); assertFalse(transport.gaps.isCurrent("c", first)); + target.evictAllL1(); assertFalse(transport.gaps.isCurrent("c", reset)); + var again = transport.gaps.reset("c").toCompletableFuture().get(5, TimeUnit.SECONDS); + assertTrue(transport.gaps.isCurrent("c", again)); + service.registerTarget("c", new RecoveryProtocolTest.Target()); + assertEquals(2, transport.subscriptions); assertFalse(transport.gaps.isCurrent("c", again)); + service.close(); assertFalse(transport.gaps.isCurrent("c", again)); + } + } + @Test void retargetBeforeResetCompletionCannotIssueTheOldProof() throws Exception { + var transport = new Transport(); var first = new RecoveryProtocolTest.Target(); var second = new RecoveryProtocolTest.Target(); + var owner = new AtomicReference(); var once = new AtomicBoolean(); + try (var service = new InvalidationService(transport, new InMemoryJournal(100), UUID.randomUUID(), cache -> { + if (once.compareAndSet(false, true)) owner.get().registerTarget(cache, second); + })) { + owner.set(service); service.registerTarget("c", first); + var result = transport.gaps.reset("c").toCompletableFuture().get(5, TimeUnit.SECONDS); + assertTrue(transport.gaps.isCurrent("c", result)); assertEquals(2, second.clears.get()); + } + } +} diff --git a/tiercache-invalidation/src/test/java/io/tiercache/invalidation/InvalidationServiceTest.java b/tiercache-invalidation/src/test/java/io/tiercache/invalidation/InvalidationServiceTest.java index ad0b335..fd7997f 100644 --- a/tiercache-invalidation/src/test/java/io/tiercache/invalidation/InvalidationServiceTest.java +++ b/tiercache-invalidation/src/test/java/io/tiercache/invalidation/InvalidationServiceTest.java @@ -169,6 +169,7 @@ void shortDisconnectHealsByReplay() { assertEquals(new Version(1, idA), targetB.entries.get("k"), "lost event: still stale"); b.transport().reconnect(); + awaitFirstPass(b.service()); assertNull(targetB.entries.get("k"), "replay must heal the missed invalidation"); a.service().close(); b.service().close(); @@ -194,6 +195,7 @@ void overflowFlushesWholeL1WithListenerEvent() { idA, InvalidationMessage.Type.INVALIDATE)); } b.transport().reconnect(); + awaitFirstPass(b.service()); assertEquals(0, targetB.entries.size(), "overflow must flush L1"); assertEquals(1, overflows.get(), "listener must be notified"); @@ -384,10 +386,12 @@ void replayMetricsAreEmittedAndTheCursorAdvances() { a.onLocalWrite("c", "k", v2, InvalidationMessage.Type.INVALIDATE); transportB.reconnect(); + awaitFirstPass(b); assertEquals(1, metrics.count(Direction.REPLAYED), "the missed event is replayed"); transportB.disconnect(); transportB.reconnect(); + awaitFirstPass(b); assertEquals(1, metrics.count(Direction.REPLAYED), "the cursor advanced: nothing is replayed twice"); a.close(); @@ -414,6 +418,7 @@ void journalOverflowIsCountedAsDropped() { idA, InvalidationMessage.Type.INVALIDATE)); } transportB.reconnect(); + awaitFirstPass(b); assertEquals(1, metrics.count(Direction.DROPPED), "the overflow is surfaced"); assertEquals(0, targetB.entries.size(), "overflow still flushes L1"); @@ -430,6 +435,7 @@ void l2RecoveryWithoutJournalFlushesL1() { a.registerTarget("c", targetA); a.onL2Recovery(); + awaitFirstPass(a); assertEquals(1, targetA.flushCount.get(), "no journal: the honest fallback is a full flush"); assertTrue(targetA.entries.isEmpty()); @@ -473,6 +479,7 @@ void cursorNeverAdvancesPastAnUnconsumedRow() { b.transport().disconnect(); b.transport().reconnect(); + awaitFirstPass(b.service()); assertNull(targetB.entries.get("k"), "the cursor must not have advanced past the unconsumed row: replay applies it"); assertEquals(0, targetB.flushCount.get(), "nothing was trimmed: no flush expected"); @@ -516,6 +523,7 @@ public void evictL1IfNewer(Object key, Version eventVersion) { journal.append("c", new InvalidationMessage("c", "k3", new Version(3, idA), idA, InvalidationMessage.Type.INVALIDATE)); b.transport().reconnect(); + awaitFirstPass(b.service()); assertNull(targetB.entries.get("k2")); assertNull(targetB.entries.get("k3")); @@ -562,6 +570,7 @@ void trimmedCursorRowForcesTheFlushPath() { InvalidationMessage.Type.INVALIDATE)); } transportB.reconnect(); + awaitFirstPass(b); assertEquals(1, targetB.flushCount.get(), "unconfirmable cursor integrity must take the flush path"); @@ -602,6 +611,7 @@ void windowOverflowFallsBackToJournalCatchUp() { a.service().onLocalWrite("c", "x" + i, v, InvalidationMessage.Type.INVALIDATE); } + awaitCondition(() -> targetB.entries.get("kd") == null); assertNull(targetB.entries.get("kd"), "the delayed row must be applied by journal catch-up, without a reconnect"); assertEquals(0, targetB.flushCount.get(), "catch-up is not a flush: nothing was trimmed"); @@ -646,6 +656,7 @@ void firstTickWithBeginningCursorIsNotAFalseLoss() { journal.append("c", new InvalidationMessage("c", "k", v66, idA, InvalidationMessage.Type.INVALIDATE)); transportB.reconnect(); + awaitFirstPass(b); assertNull(targetB.entries.get("k"), "the missed row is replayed"); assertEquals(0, targetB.flushCount.get(), "still no flush"); a.service().close(); @@ -699,6 +710,7 @@ public boolean isTrimmed(String cache, String cursor) { transportB.disconnect(); transportB.reconnect(); + awaitFirstPass(b); assertEquals(1, targetB.flushCount.get(), "a failed read is unconfirmable integrity: flush, never \"no loss\""); @@ -749,12 +761,14 @@ public void evictAllL1() { InvalidationMessage.Type.INVALIDATE)); } b.transport().reconnect(); + awaitFirstPass(b.service()); assertEquals(1, targetB.flushCount.get(), "trimmed cursor: the flush fired"); assertEquals(new Version(1, idA), targetB.entries.get("k"), "the stale re-warm is in place after the clear"); b.transport().disconnect(); b.transport().reconnect(); + awaitFirstPass(b.service()); assertNull(targetB.entries.get("k"), "the row journaled during the flush must be replayed, not baselined away"); a.service().close(); @@ -822,10 +836,12 @@ public boolean isTrimmed(String cache, String cursor) { } failBaseline.set(true); // the flush's baseline read will fail once transportB.reconnect(); + awaitFirstPass(b); assertEquals(1, targetB.flushCount.get()); transportB.disconnect(); transportB.reconnect(); + awaitFirstPass(b); assertEquals(2, targetB.flushCount.get(), "the kept cursor is still trimmed: the flush path repeats honestly"); assertEquals("3", checkedReadCursors.get(checkedReadCursors.size() - 1), @@ -879,28 +895,39 @@ public boolean isTrimmed(String cache, String cursor) { } } - @SuppressWarnings("unchecked") - private static int windowSize(InvalidationService service, String cache) { + private static Object stateField(InvalidationService service, String cache, String name) { try { - var field = InvalidationService.class.getDeclaredField("appliedWindows"); - field.setAccessible(true); - Set window = ((Map>) field.get(service)).get(cache); - return window == null ? 0 : window.size(); - } catch (ReflectiveOperationException e) { - throw new AssertionError(e); - } + var states = InvalidationService.class.getDeclaredField("states"); states.setAccessible(true); + Object state = ((Map) states.get(service)).get(cache); + if (state == null) return null; + synchronized (state) { + var field = state.getClass().getDeclaredField(name); field.setAccessible(true); + Object value = field.get(state); + return value instanceof Map map ? Map.copyOf(map) : value; + } + } catch (ReflectiveOperationException e) { throw new AssertionError(e); } } + private static int windowSize(InvalidationService service, String cache) { + Object value = stateField(service, cache, "applied"); + return value == null ? 0 : ((Map) value).size(); + } private static boolean resyncRequired(InvalidationService service, String cache) { - try { - var field = InvalidationService.class.getDeclaredField("resyncRequired"); - field.setAccessible(true); - return ((Map) field.get(service)).containsKey(cache); - } catch (NoSuchFieldException e) { - return false; // the state does not exist before the fix - } catch (ReflectiveOperationException e) { - throw new AssertionError(e); + return Boolean.TRUE.equals(stateField(service, cache, "resyncRequired")); + } + private static void awaitCondition(java.util.function.BooleanSupplier condition) { + long deadline = System.nanoTime() + java.util.concurrent.TimeUnit.SECONDS.toNanos(5); + while (!condition.getAsBoolean() && System.nanoTime() < deadline) { + try { Thread.sleep(5); } catch (InterruptedException e) { throw new AssertionError(e); } } + assertTrue(condition.getAsBoolean(), "asynchronous recovery did not settle"); + } + private static void awaitFirstPass(InvalidationService service) { + awaitCondition(() -> stateField(service, "c", "lastResult") != null + && !Boolean.TRUE.equals(stateField(service, "c", "running"))); + } + private static void awaitIdle(InvalidationService service) { + awaitCondition(() -> !Boolean.TRUE.equals(stateField(service, "c", "pending"))); } /** @@ -931,7 +958,8 @@ void duplicateDeliveryAfterConfirmationStaysBounded() { messages.add(message); journal.append("c", message); } - b.transport().reconnect(); // replay applies all 1,000; cursor at the end + b.transport().reconnect(); + awaitFirstPass(b.service()); // replay applies all 1,000; cursor at the end journal.readCalls.set(0); int peakWindow = 0; @@ -941,7 +969,8 @@ void duplicateDeliveryAfterConfirmationStaysBounded() { peakWindow = Math.max(peakWindow, windowSize(b.service(), "c")); } - assertTrue(peakWindow <= 200, + awaitIdle(b.service()); + assertTrue(peakWindow <= 512, "duplicates must not accumulate without bound: peak window " + peakWindow); assertTrue(windowSize(b.service(), "c") <= APPLIED_WINDOW_NOMINAL, "re-tracked duplicates beyond the confirmed horizon stay within the nominal " @@ -980,6 +1009,7 @@ void readFailuresKeepTrackingBoundedAndHeal() throws Exception { a.service().onLocalWrite("c", "k" + i, v, InvalidationMessage.Type.INVALIDATE); } + awaitFirstPass(b.service()); assertTrue(resyncRequired(b.service(), "c"), "failed catch-up must enter the resync-required state"); assertEquals(0, windowSize(b.service(), "c"), @@ -1006,6 +1036,7 @@ void readFailuresKeepTrackingBoundedAndHeal() throws Exception { InvalidationMessage.Type.INVALIDATE)); a.service().onLocalWrite("c", "k1002", v1002, InvalidationMessage.Type.INVALIDATE); + awaitCondition(() -> !resyncRequired(b.service(), "c")); assertFalse(resyncRequired(b.service(), "c"), "a successful resync to the journal end clears the state"); assertEquals(0, windowSize(b.service(), "c")); @@ -1014,14 +1045,9 @@ void readFailuresKeepTrackingBoundedAndHeal() throws Exception { b.service().close(); } - /** - * The lazy-recovery contract: an invalidation missed before the - * failures stays unapplied while no delivery or reconnect occurs — - * L1 may serve stale data — and is applied at the next recovery - * trigger, never silently dropped. - */ + /** Triggered failed recovery retries on its own; healthy caches are not polled. */ @Test - void lazyRecoveryStaysIncompleteUntilTheNextTrigger() throws Exception { + void triggeredRecoveryRetriesAfterReadsHealWithoutAnotherDelivery() throws Exception { var hub = new InMemoryInvalidationTransport.Hub(); StubJournal journal = new StubJournal(new InMemoryJournal(2000)); UUID idA = UUID.randomUUID(); @@ -1045,26 +1071,15 @@ void lazyRecoveryStaysIncompleteUntilTheNextTrigger() throws Exception { InvalidationMessage.Type.INVALIDATE)); a.service().onLocalWrite("c", "x" + i, v, InvalidationMessage.Type.INVALIDATE); } + awaitFirstPass(b.service()); assertTrue(resyncRequired(b.service(), "c")); assertEquals(new Version(1, idA), targetB.entries.get("k"), "the missed invalidation is not yet applied"); - - // Reads heal, but no delivery or reconnect occurs: recovery stays - // incomplete — the stale entry is still served. journal.failReads.set(false); - assertEquals(new Version(1, idA), targetB.entries.get("k"), - "lazy recovery: L1 may serve stale data until the next trigger"); - - // The next delivery (throttle interval elapsed) triggers the resync. - Thread.sleep(1_100); - Version trigger = new Version(1_000, idA); - journal.append("c", new InvalidationMessage("c", "xT", trigger, idA, - InvalidationMessage.Type.INVALIDATE)); - a.service().onLocalWrite("c", "xT", trigger, InvalidationMessage.Type.INVALIDATE); - - assertNull(targetB.entries.get("k"), - "the missed invalidation is applied at the next recovery trigger"); - assertFalse(resyncRequired(b.service(), "c")); + // No new live event or reconnect: the already-triggered retry owns progress. + awaitCondition(() -> targetB.entries.get("k") == null); + awaitCondition(() -> !resyncRequired(b.service(), "c")); + assertEquals(0, windowSize(b.service(), "c")); a.service().close(); b.service().close(); } diff --git a/tiercache-invalidation/src/test/java/io/tiercache/invalidation/JournalCapacityProtocolTest.java b/tiercache-invalidation/src/test/java/io/tiercache/invalidation/JournalCapacityProtocolTest.java new file mode 100644 index 0000000..ed72d16 --- /dev/null +++ b/tiercache-invalidation/src/test/java/io/tiercache/invalidation/JournalCapacityProtocolTest.java @@ -0,0 +1,53 @@ +package io.tiercache.invalidation; + +import io.tiercache.*; +import io.tiercache.spi.*; +import io.tiercache.testkit.*; +import org.junit.jupiter.api.Test; +import java.util.*; +import static org.junit.jupiter.api.Assertions.*; + +class JournalCapacityProtocolTest { + private int exercise(int capacity, boolean delayed, boolean failReads) throws Exception { + var data=new InMemoryJournal(capacity); + InvalidationJournal journal=new InvalidationJournal(){ + public String append(String c,InvalidationMessage m){return data.append(c,m);} + public List readRange(String c,String p){return data.readRange(c,p);} + public String endCursor(String c){return data.endCursor(c);} + public boolean isTrimmed(String c,String p){return data.isTrimmed(c,p);} + public CheckedRange checkedRead(String c,String p,int n){ + if(failReads)throw new IllegalStateException("read unavailable");return data.checkedRead(c,p,n); + } + }; + var target=new RecoveryProtocolTest.Target(); + var transport=new InMemoryInvalidationTransport(new InMemoryInvalidationTransport.Hub()); + try(var scheduler=new RecoveryProtocolTest.Manual();var service=new InvalidationService(transport,journal,UUID.randomUUID(),InvalidationListener.NOOP,CacheMetricsListener.NOOP,scheduler.now::get)) { + service.configureRecoveryExecutor(scheduler);service.registerTarget("c",target); + for(int i=1;i<=128;i++) { + var message=RecoveryProtocolTest.update("c","k"+i,i);journal.append("c",message);transport.publish(message); + if(!delayed)scheduler.drain(); + if(i==64 && !delayed && !failReads)assertEquals("64",RecoveryProtocolTest.field(RecoveryProtocolTest.state(service,"c"),"cursor")); + } + scheduler.drain(); + if (failReads) { + // A failed live tick retains the cursor. Recovery then uses its + // existing conservative reset path, with the normal retry delay. + assertEquals("0",RecoveryProtocolTest.field(RecoveryProtocolTest.state(service,"c"),"cursor")); + service.recoverAsync(Runnable::run); + scheduler.advance(1); scheduler.drain(); + } + assertEquals("128",RecoveryProtocolTest.field(RecoveryProtocolTest.state(service,"c"),"cursor")); + return target.clears.get(); + } + } + @Test void exactRetentionNeedsCursorPlusSixtyFourLaterEvents() throws Exception { + assertEquals(64,JournalProtocol.CURSOR_CADENCE);assertEquals(65,JournalProtocol.MIN_CAPACITY); + assertEquals(1,exercise(64,false,false));assertEquals(0,exercise(65,false,false)); + } + @Test void validCapacityDoesNotGuaranteeRetentionUnderDelayedRecovery() throws Exception { + assertTrue(exercise(65,true,false)>0); + } + @Test void validCapacityDoesNotBypassConservativeHandlingOfReadFailure() throws Exception { + assertTrue(exercise(65,false,true)>0); + } +} diff --git a/tiercache-invalidation/src/test/java/io/tiercache/invalidation/PublicationObserverTest.java b/tiercache-invalidation/src/test/java/io/tiercache/invalidation/PublicationObserverTest.java new file mode 100644 index 0000000..fdda710 --- /dev/null +++ b/tiercache-invalidation/src/test/java/io/tiercache/invalidation/PublicationObserverTest.java @@ -0,0 +1,159 @@ +package io.tiercache.invalidation; + +import io.tiercache.*; +import io.tiercache.spi.*; +import io.tiercache.testkit.*; +import org.junit.jupiter.api.Test; +import java.util.*; +import java.util.concurrent.*; +import java.util.concurrent.atomic.*; +import java.util.function.*; +import static org.junit.jupiter.api.Assertions.*; + +class PublicationObserverTest { + static class Metrics implements CacheMetricsListener { + final AtomicLongArray outcomes=new AtomicLongArray(4); + final AtomicInteger sent=new AtomicInteger(); + volatile Thread callbackThread; + public void onInvalidation(String c,Direction d){if(d==Direction.SENT)sent.incrementAndGet();} + public void onPublication(String c,PublicationOutcome outcome,long count){callbackThread=Thread.currentThread();outcomes.addAndGet(outcome.ordinal(),count);} + long count(PublicationOutcome outcome){return outcomes.get(outcome.ordinal());} + } + static class Transport implements InvalidationTransport { + final AtomicInteger calls=new AtomicInteger(); + volatile Supplier> action=()->CompletableFuture.completedFuture(PublicationOutcome.ACKNOWLEDGED); + public void publish(InvalidationMessage message){throw new AssertionError("async override must be used");} + public CompletionStage publishAsync(InvalidationMessage m){calls.incrementAndGet();return action.get();} + public AutoCloseable subscribe(String c,Consumer h){return ()->{};} + public void close(){} + } + static void await(BooleanSupplier condition)throws Exception { + long end=System.nanoTime()+TimeUnit.SECONDS.toNanos(5); + while(!condition.getAsBoolean()&&System.nanoTime()();transport.action=()->future; + var metrics=new Metrics();var remote=new InMemoryRemoteCache<>(); + try(var factory=TierCacheFactory.builder().remoteCache(remote).invalidation(v->new InvalidationService(transport,null,v.instanceId(),InvalidationListener.NOOP,metrics)).build()) { + var cache=factory.getCache("c");cache.put("x","committed");assertFalse(future.isDone());assertEquals("committed",remote.get("x").value()); + assertEquals(1,metrics.sent.get());assertEquals(0,metrics.count(PublicationOutcome.ACKNOWLEDGED)); + Thread io=new Thread(()->future.completeExceptionally(new IllegalStateException("secret-payload")),"simulated-lettuce-io");io.start();io.join(); + await(()->metrics.count(PublicationOutcome.FAILED)==1);assertNotSame(io,metrics.callbackThread);assertEquals(1,transport.calls.get()); + assertEquals(0,metrics.count(PublicationOutcome.ACKNOWLEDGED));factory.close();cache.put("after","still-usable"); + assertEquals(1,transport.calls.get());assertEquals(1,metrics.sent.get()); + } + } + @Test void cancellationSynchronousThrowAndNullStageBecomeOneFailureEach() throws Exception { + var transport=new Transport();var metrics=new Metrics(); + try(var service=new InvalidationService(transport,null,UUID.randomUUID(),InvalidationListener.NOOP,metrics)) { + transport.action=()->{throw new IllegalArgumentException("secret");};service.onLocalWrite("c","x",version(),InvalidationMessage.Type.INVALIDATE); + transport.action=()->{var f=new CompletableFuture();f.cancel(false);return f;};service.onLocalWrite("c","x",version(),InvalidationMessage.Type.INVALIDATE); + transport.action=()->null;service.onLocalUpdate("c","x","payload",version()); + await(()->metrics.count(PublicationOutcome.FAILED)==3);assertEquals(3,metrics.sent.get());assertEquals(3,transport.calls.get()); + } + } + @Test void duplicateCompletionCannotDoubleCount() throws Exception { + var metrics=new Metrics(); + try(var observer=new PublicationObserver(metrics,System::nanoTime)) { + observer.register("c");var attempt=observer.admit("c"); + attempt.complete(PublicationOutcome.ACKNOWLEDGED,null);attempt.complete(null,new IllegalStateException()); + await(()->metrics.count(PublicationOutcome.ACKNOWLEDGED)==1); + assertEquals(0,attempt.count(PublicationOutcome.FAILED));assertEquals(1,attempt.count(PublicationOutcome.ACKNOWLEDGED)); + } + } + @Test void legacyVoidTransportIsUnconfirmedAndLegacyListenerStillWorks() throws Exception { + var calls=new AtomicInteger();InvalidationTransport legacy=new InvalidationTransport(){ + public void publish(InvalidationMessage m){calls.incrementAndGet();} + public AutoCloseable subscribe(String c,Consumer h){return ()->{};} + public void close(){} + }; + assertEquals(PublicationOutcome.UNCONFIRMED,legacy.publishAsync(message()).toCompletableFuture().get());assertEquals(1,calls.get()); + new CacheMetricsListener(){}.onPublication("c",PublicationOutcome.UNCONFIRMED,1); + var metrics=new Metrics();try(var service=new InvalidationService(legacy,null,UUID.randomUUID(),InvalidationListener.NOOP,metrics)) { + service.onLocalWrite("c","x",version(),InvalidationMessage.Type.INVALIDATE); + await(()->metrics.count(PublicationOutcome.UNCONFIRMED)==1);assertEquals(2,calls.get()); + } + } + @Test void burstAndConcurrentDrainKeepBoundedTokensAndExactTotals() throws Exception { + var entered=new CountDownLatch(1);var release=new CountDownLatch(1);var block=new AtomicBoolean(true);var metrics=new Metrics(){ + public void onPublication(String c,PublicationOutcome o,long n){if(block.compareAndSet(true,false)){entered.countDown();gate(release);}super.onPublication(c,o,n);} + }; + var observer=new PublicationObserver(metrics,System::nanoTime);var producers=Executors.newFixedThreadPool(4); + try { + observer.register("a");observer.register("b");observer.admit("a").complete(PublicationOutcome.ACKNOWLEDGED,null);gate(entered); + var jobs=new ArrayList>(); + for(int thread=0;thread<4;thread++)jobs.add(producers.submit(()->{ + for(int i=0;i<10000;i++)observer.admit(i%2==0?"b":"unknown-"+i).complete(PublicationOutcome.ACKNOWLEDGED,null); + })); + for(var job:jobs)job.get(5,TimeUnit.SECONDS); + assertEquals(3,observer.bucketCount());assertTrue(observer.pendingTokens()<=3); + release.countDown();await(()->metrics.count(PublicationOutcome.ACKNOWLEDGED)==40001); + jobs.clear();for(int t=0;t<4;t++)jobs.add(producers.submit(()->{for(int i=0;i<1000;i++)observer.admit("a").complete(PublicationOutcome.NOT_REQUIRED,null);})); + for(var job:jobs)job.get(5,TimeUnit.SECONDS);await(()->metrics.count(PublicationOutcome.NOT_REQUIRED)==4000); + } finally {release.countDown();observer.close();producers.shutdownNow();} + } + @Test void throwingObserverIsNotRetriedAndDoesNotChangeInternalOutcome() throws Exception { + var calls=new AtomicInteger(); + try(var observer=new PublicationObserver(new CacheMetricsListener(){public void onPublication(String c,PublicationOutcome o,long n){calls.incrementAndGet();throw new IllegalStateException("observer secret");}},System::nanoTime)) { + var attempt=observer.admit("c");attempt.complete(PublicationOutcome.FAILED,null);await(()->calls.get()==1); + assertEquals(1,attempt.count(PublicationOutcome.FAILED));observer.close();assertEquals(1,calls.get()); + } + } + @Test void closeDrainsForAtMostOneSecondAndLateCompletionDoesNotResurrectWorker() throws Exception { + var entered=new CountDownLatch(1);var release=new CountDownLatch(1);var calls=new AtomicInteger(); + var observer=new PublicationObserver(new CacheMetricsListener(){public void onPublication(String c,PublicationOutcome o,long n){ + calls.incrementAndGet();entered.countDown();boolean done=false;while(!done)try{done=release.await(5,TimeUnit.SECONDS);}catch(InterruptedException ignored){} + }},System::nanoTime); + try { + var late=observer.admit("late");observer.admit("c").complete(PublicationOutcome.ACKNOWLEDGED,null);gate(entered); + assertTimeoutPreemptively(java.time.Duration.ofSeconds(2),observer::close); + assertNull(observer.admit("after"));late.complete(PublicationOutcome.ACKNOWLEDGED,null);assertEquals(2,late.count(PublicationOutcome.ACKNOWLEDGED)); + assertEquals(0,observer.pendingTokens());assertEquals(1,calls.get()); + release.countDown();observer.close(); + } finally {release.countDown();observer.close();} + } + @Test void observerCanCloseItselfWithoutWaitingOnItsOwnThread() throws Exception { + var reference=new AtomicReference();var completed=new CountDownLatch(1); + var observer=new PublicationObserver(new CacheMetricsListener(){public void onPublication(String c,PublicationOutcome o,long n){reference.get().close();completed.countDown();}},System::nanoTime);reference.set(observer); + observer.admit("c").complete(PublicationOutcome.ACKNOWLEDGED,null);assertTrue(completed.await(2,TimeUnit.SECONDS));observer.close(); + } + static long warningDeadline(Object bucket) { + try{var f=bucket.getClass().getDeclaredField("nextWarning");f.setAccessible(true);return f.getLong(bucket);} + catch(ReflectiveOperationException e){throw new AssertionError(e);} + } + @Test void warningRateLimitWorksWithoutMetricsAndRetainsNoThrowable() throws Exception { + var now=new AtomicLong(1); + try(var observer=new PublicationObserver(CacheMetricsListener.NOOP,now::get)) { + observer.register("c");var first=observer.admit("c");var field=first.getClass().getDeclaredField("bucket");field.setAccessible(true);var bucket=field.get(first); + first.complete(null,new CompletionException(new IllegalArgumentException("sensitive payload"))); + await(()->warningDeadline(bucket)!=0);long deadline=warningDeadline(bucket); + for(int i=0;i<10000;i++)observer.admit("c").complete(null,new IllegalArgumentException("sensitive payload")); + observer.close();assertEquals(deadline,warningDeadline(bucket));assertEquals(10001,first.count(PublicationOutcome.FAILED)); + for(var member:bucket.getClass().getDeclaredFields())assertFalse(Throwable.class.isAssignableFrom(member.getType())); + } + try(var observer=new PublicationObserver(CacheMetricsListener.NOOP,now::get)) { + var first=observer.admit("unknown");var field=first.getClass().getDeclaredField("bucket");field.setAccessible(true);var bucket=field.get(first); + first.complete(PublicationOutcome.FAILED,null);await(()->warningDeadline(bucket)!=0); + long deadline=warningDeadline(bucket);now.addAndGet(TimeUnit.SECONDS.toNanos(31)); + observer.admit("different-unknown").complete(PublicationOutcome.FAILED,null);await(()->warningDeadline(bucket)>deadline); + assertEquals(1,observer.bucketCount()); + } + } + + @Test void firstCompletionAfterCloseDoesNotStartAWorker() throws Exception { + var metrics=new Metrics();var observer=new PublicationObserver(metrics,System::nanoTime); + var attempt=observer.admit("not-registered");observer.close(); + attempt.complete(PublicationOutcome.ACKNOWLEDGED,null); + var field=PublicationObserver.class.getDeclaredField("workerThread");field.setAccessible(true); + assertNull(field.get(observer));assertEquals(1,attempt.count(PublicationOutcome.ACKNOWLEDGED)); + assertEquals(0,metrics.count(PublicationOutcome.ACKNOWLEDGED));assertEquals(0,observer.pendingTokens()); + observer.register("late-registration");assertEquals(1,observer.bucketCount()); + } + +} diff --git a/tiercache-invalidation/src/test/java/io/tiercache/invalidation/PublishFailureRegressionTest.java b/tiercache-invalidation/src/test/java/io/tiercache/invalidation/PublishFailureRegressionTest.java new file mode 100644 index 0000000..551f806 --- /dev/null +++ b/tiercache-invalidation/src/test/java/io/tiercache/invalidation/PublishFailureRegressionTest.java @@ -0,0 +1,22 @@ +package io.tiercache.invalidation; +import io.tiercache.*; +import io.tiercache.spi.*; +import io.tiercache.testkit.*; +import org.junit.jupiter.api.Test; +import static org.junit.jupiter.api.Assertions.*; + +class PublishFailureRegressionTest { + @Test void synchronousPublicationFailureCannotReplaceCommittedWriteResult() { + var remote=new InMemoryRemoteCache(); + InvalidationTransport transport=new InvalidationTransport(){ + public void publish(InvalidationMessage m){throw new IllegalStateException("publication failed");} + public AutoCloseable subscribe(String c,java.util.function.Consumer h){return ()->{};} + public void close(){} + }; + try(var factory=TierCacheFactory.builder().remoteCache(remote) + .invalidation(v->new InvalidationService(transport,null,v.instanceId(),InvalidationListener.NOOP)).build()) { + var cache=factory.getCache("c");assertDoesNotThrow(()->cache.put("x","committed")); + assertEquals("committed",remote.get("x").value());assertEquals("committed",cache.get("x")); + } + } +} diff --git a/tiercache-invalidation/src/test/java/io/tiercache/invalidation/RecoveryMonitorTest.java b/tiercache-invalidation/src/test/java/io/tiercache/invalidation/RecoveryMonitorTest.java new file mode 100644 index 0000000..c6807c4 --- /dev/null +++ b/tiercache-invalidation/src/test/java/io/tiercache/invalidation/RecoveryMonitorTest.java @@ -0,0 +1,51 @@ +package io.tiercache.invalidation; + +import io.tiercache.*; +import io.tiercache.spi.*; +import io.tiercache.testkit.*; +import org.junit.jupiter.api.Test; +import java.util.*; +import java.util.concurrent.*; +import java.util.concurrent.atomic.*; +import static org.junit.jupiter.api.Assertions.*; + +class RecoveryMonitorTest { + @Test void parkedJournalDoesNotHoldCacheMonitorOrCallbackThread() throws Exception { + var entered = new CountDownLatch(1); var release = new CountDownLatch(1); + var returned = new CountDownLatch(1); var held = new AtomicBoolean(); + var owner = new AtomicReference(); + InvalidationJournal journal = new InvalidationJournal() { + public String append(String c, InvalidationMessage m) { return "0"; } + public List readRange(String c, String cursor) { return List.of(); } + public String endCursor(String c) { return "0"; } + public boolean isTrimmed(String c, String cursor) { return false; } + public CheckedRange checkedRead(String c, String cursor, int count) { + try { + java.lang.reflect.Field field; + try { field = InvalidationService.class.getDeclaredField("states"); } + catch (NoSuchFieldException e) { field = InvalidationService.class.getDeclaredField("cacheLocks"); } + field.setAccessible(true); + Object gate = ((Map) field.get(owner.get())).get(c); + held.set(Thread.holdsLock(gate)); entered.countDown(); + if (!release.await(5, TimeUnit.SECONDS)) throw new AssertionError("journal gate timeout"); + return new CheckedRange(true, List.of()); + } catch (ReflectiveOperationException | InterruptedException e) { throw new AssertionError(e); } + } + }; + var transport = new InMemoryInvalidationTransport(new InMemoryInvalidationTransport.Hub()); + var service = new InvalidationService(transport, journal, UUID.randomUUID(), InvalidationListener.NOOP); + owner.set(service); + service.registerTarget("c", new InvalidationTarget() { + public Version versionOfL1Entry(Object key) { return null; } + public void evictL1IfNewer(Object key, Version version) { } + public void evictAllL1() { } + }); + Thread callback = new Thread(() -> { service.onL2Recovery(); returned.countDown(); }, "transport-callback"); + callback.start(); + try { + assertTrue(entered.await(5, TimeUnit.SECONDS)); + assertFalse(held.get(), "journal I/O must not own the cache state monitor"); + assertTrue(returned.await(1, TimeUnit.SECONDS), "callback must return before journal I/O completes"); + } finally { release.countDown(); callback.join(5000); service.close(); } + } +} diff --git a/tiercache-invalidation/src/test/java/io/tiercache/invalidation/RecoveryProtocolTest.java b/tiercache-invalidation/src/test/java/io/tiercache/invalidation/RecoveryProtocolTest.java new file mode 100644 index 0000000..03a64c0 --- /dev/null +++ b/tiercache-invalidation/src/test/java/io/tiercache/invalidation/RecoveryProtocolTest.java @@ -0,0 +1,341 @@ +package io.tiercache.invalidation; + +import io.tiercache.*; +import io.tiercache.internal.CircuitBreaker; +import io.tiercache.spi.*; +import io.tiercache.testkit.*; +import org.junit.jupiter.api.Test; +import java.time.Duration; +import java.util.*; +import java.util.concurrent.*; +import java.util.concurrent.atomic.*; +import java.util.function.*; +import static org.junit.jupiter.api.Assertions.*; + +class RecoveryProtocolTest { + static final UUID WRITER = UUID.randomUUID(); + static InvalidationMessage update(String cache, String key, long version) { + return new InvalidationMessage(cache, key, new Version(version, WRITER), WRITER, + InvalidationMessage.Type.UPDATE, "v" + version); + } + static class Journal implements InvalidationJournal { + final InMemoryJournal data = new InMemoryJournal(20000); + final AtomicInteger reads = new AtomicInteger(); + final AtomicInteger ends = new AtomicInteger(); + volatile TriRead read = (c, p, n) -> data.checkedRead(c, p, n); + volatile Function end = data::endCursor; + public String append(String c, InvalidationMessage m) { return data.append(c, m); } + public List readRange(String c, String p) { return data.readRange(c, p); } + public CheckedRange checkedRead(String c, String p, int n) { reads.incrementAndGet(); return read.read(c, p, n); } + public String endCursor(String c) { ends.incrementAndGet(); return end.apply(c); } + public boolean isTrimmed(String c, String p) { return data.isTrimmed(c, p); } + } + interface TriRead { CheckedRange read(String cache, String cursor, int count); } + static class Target implements InvalidationTarget { + final Map values = new ConcurrentHashMap<>(); + final AtomicLong generation = new AtomicLong(); + final AtomicInteger clears = new AtomicInteger(); + public Version versionOfL1Entry(Object k) { var v = values.get(k); return v == null ? null : v.version(); } + public void evictL1IfNewer(Object k, Version v) { values.computeIfPresent(k, (key, old) -> v.compareTo(old.version()) > 0 ? null : old); } + public void evictAllL1() { generation.incrementAndGet(); values.clear(); clears.incrementAndGet(); } + public long recoveryGeneration() { return generation.get(); } + public void applyUpdateL1(Object k, Object value, Version v) { + values.compute(k, (key, old) -> old == null || old.version().compareTo(v) < 0 + ? new InvalidationMessage("c", k, v, v.instanceId(), InvalidationMessage.Type.UPDATE, value) : old); + } + } + static class Harness implements AutoCloseable { + final Journal journal = new Journal(); + final Target target = new Target(); + final InMemoryInvalidationTransport.Hub hub = new InMemoryInvalidationTransport.Hub(); + final InMemoryInvalidationTransport transport = new InMemoryInvalidationTransport(hub); + final AtomicReference pending = new AtomicReference<>(); + final AtomicInteger removed = new AtomicInteger(); + final InvalidationService service; + Harness() { this(null); } + Harness(Manual scheduler) { + CacheMetricsListener metrics = new CacheMetricsListener() { + public AutoCloseable registerRecovery(String cache, BooleanSupplier state) { + if (cache.equals("c")) pending.set(state); + return removed::incrementAndGet; + } + }; + service = scheduler == null + ? new InvalidationService(transport, journal, UUID.randomUUID(), InvalidationListener.NOOP, metrics) + : new InvalidationService(transport, journal, UUID.randomUUID(), InvalidationListener.NOOP, metrics, scheduler.now::get); + if (scheduler != null) service.configureRecoveryExecutor(scheduler); + service.registerTarget("c", target); + } + CompletableFuture recover() { return service.recoverAsync(Runnable::run).toCompletableFuture(); } + public void close() { service.close(); } + } + static void await(BooleanSupplier condition) throws Exception { + long until = System.nanoTime() + TimeUnit.SECONDS.toNanos(5); + while (!condition.getAsBoolean() && System.nanoTime() < until) Thread.sleep(5); + assertTrue(condition.getAsBoolean(), "condition did not settle"); + } + static void gate(CountDownLatch release) { + try { if (!release.await(5, TimeUnit.SECONDS)) throw new AssertionError("gate timeout"); } + catch (InterruptedException e) { Thread.currentThread().interrupt(); throw new AssertionError(e); } + } + static Object state(InvalidationService service, String cache) throws Exception { + var field = InvalidationService.class.getDeclaredField("states"); field.setAccessible(true); + return ((Map) field.get(service)).get(cache); + } + static Object field(Object state, String name) throws Exception { + synchronized (state) { var field = state.getClass().getDeclaredField(name); field.setAccessible(true); return field.get(state); } + } + + @Test void liveDeliveryCanProgressWhileJournalIsParked() throws Exception { + try (Harness h = new Harness()) { + h.journal.append("c", update("c", "k", 1)); + var entered = new CountDownLatch(1); var release = new CountDownLatch(1); var once = new AtomicBoolean(); + h.journal.read = (c, p, n) -> { + var rows = h.journal.data.checkedRead(c, p, n); + if (once.compareAndSet(false, true)) { entered.countDown(); gate(release); } + return rows; + }; + var recovery = h.recover(); + try { + assertTrue(entered.await(5, TimeUnit.SECONDS)); + var newer = update("c", "k", 2); h.journal.append("c", newer); + try (var sender = new InMemoryInvalidationTransport(h.hub)) { sender.publish(newer); } + assertEquals(newer.version(), h.target.versionOfL1Entry("k")); + assertFalse(recovery.isDone()); + } finally { release.countDown(); } + assertTrue(recovery.get(5, TimeUnit.SECONDS)); + assertEquals(new Version(2, WRITER), h.target.versionOfL1Entry("k")); + assertEquals("2", field(state(h.service, "c"), "cursor")); + } + } + + @Test void secondRecoveryDiscardsTheOlderRead() throws Exception { + staleReadIsDiscarded(false); + } + @Test void localClearDiscardsTheOlderRead() throws Exception { + staleReadIsDiscarded(true); + } + private void staleReadIsDiscarded(boolean localClear) throws Exception { + try (Harness h = new Harness()) { + h.journal.append("c", update("c", "current", 1)); + var entered = new CountDownLatch(1); var release = new CountDownLatch(1); var once = new AtomicBoolean(); + h.journal.read = (c, p, n) -> { + if (once.compareAndSet(false, true)) { + entered.countDown(); gate(release); + return new CheckedRange(true, List.of(new JournalRow("999", update("c", "obsolete", 999)))); + } + return h.journal.data.checkedRead(c, p, n); + }; + var completion = h.recover(); + try { + assertTrue(entered.await(5, TimeUnit.SECONDS)); + if (localClear) h.target.evictAllL1(); else assertSame(completion, h.recover()); + } finally { release.countDown(); } + assertTrue(completion.get(5, TimeUnit.SECONDS)); + assertNull(h.target.versionOfL1Entry("obsolete")); + assertNotNull(h.target.versionOfL1Entry("current")); + assertEquals("1", field(state(h.service, "c"), "cursor")); + } + } + + @Test void closeCancelsCompletionAndRejectsLateJournalResponse() throws Exception { + Harness h = new Harness(); + var entered = new CountDownLatch(1); var release = new CountDownLatch(1); + h.journal.read = (c, p, n) -> { + entered.countDown(); + // Model a transport that finishes after cancellation, ignoring interruption. + while (release.getCount() != 0) { try { release.await(); } catch (InterruptedException ignored) { } } + return new CheckedRange(true, List.of(new JournalRow("1", update("c", "late", 1)))); + }; + var completion = h.recover(); assertTrue(entered.await(5, TimeUnit.SECONDS)); + h.close(); assertFalse(completion.get(1, TimeUnit.SECONDS)); release.countDown(); + assertEquals(1, h.removed.get()); + assertEquals(RecoveryResult.Status.CLOSED, h.service.resetAsync("c").toCompletableFuture().get().status()); + assertFalse(h.recover().get()); + await(() -> Thread.getAllStackTraces().keySet().stream().noneMatch(t -> t.getName().equals("tiercache-recovery") && t.isAlive())); + assertNull(h.target.versionOfL1Entry("late")); + h.close(); assertEquals(1, h.removed.get()); + } + + @Test void resetCapturesBaselineBeforeClearAndReplaysRowsAfterIt() throws Exception { + try (Harness h = new Harness()) { + h.target.values.put("old", update("c", "old", 1)); + var captured = new CountDownLatch(1); var release = new CountDownLatch(1); + h.journal.end = c -> { + try { assertFalse(Thread.holdsLock(state(h.service, c)), "baseline I/O must be outside the state gate"); } + catch (Exception e) { throw new AssertionError(e); } + String baseline = h.journal.data.endCursor(c); captured.countDown(); gate(release); return baseline; + }; + var reset = h.service.resetAsync("c").toCompletableFuture(); + try { + assertTrue(captured.await(5, TimeUnit.SECONDS)); + assertEquals(0, h.target.clears.get()); + h.journal.append("c", update("c", "after", 2)); + } finally { release.countDown(); } + RecoveryResult result = reset.get(5, TimeUnit.SECONDS); + assertEquals(RecoveryResult.Status.RESET_SAFE, result.status()); + assertEquals("0", result.baseline()); assertTrue(result.generation() > 0); + assertNull(h.target.versionOfL1Entry("old")); + await(() -> h.target.versionOfL1Entry("after") != null); + await(() -> !h.pending.get().getAsBoolean()); + assertEquals("1", field(state(h.service, "c"), "cursor")); + } + } + + @Test void failedBaselineClearsButRetainsCursorAndDoesNotCloseBreaker() throws Exception { + try (Harness h = new Harness()) { + h.target.values.put("old", update("c", "old", 1)); + h.journal.read = (c, p, n) -> { throw new IllegalStateException("journal offline"); }; + h.journal.end = c -> { throw new IllegalStateException("baseline offline"); }; + var b = new CircuitBreaker(new CircuitBreaker.Config(2, 1, 1, Duration.ZERO, 1), + new CircuitBreaker.Listener() { public void onOpen() { } public void onClose() { fail("unsafe recovery"); } }); + var executor = Executors.newSingleThreadExecutor(); + try { + b.configureRecovery(executor, h::recover); b.onFailure(); + var probe = b.tryAcquirePermit(); assertNotNull(probe); probe.success(); + await(() -> h.target.clears.get() == 1); + await(() -> b.state() == BreakerState.OPEN); + assertEquals("0", field(state(h.service, "c"), "cursor")); + assertTrue(h.pending.get().getAsBoolean()); + assertNull(h.target.versionOfL1Entry("old")); + var result = (RecoveryResult) field(state(h.service, "c"), "lastResult"); + assertEquals(RecoveryResult.Status.FAILED, result.status()); assertNull(result.baseline()); + } finally { b.detachRecovery(); executor.shutdownNow(); } + } + } + + @Test void observerFailuresCannotCancelApplicationCursorOrSafeReset() throws Exception { + var hub = new InMemoryInvalidationTransport.Hub(); var journal = new Journal(); var target = new Target(); + CacheMetricsListener metrics = new CacheMetricsListener() { + public void onInvalidation(String c, Direction d) { throw new IllegalStateException("metric"); } + public Object onInvalidationStart(String c) { throw new AssertionError("span"); } + public void onInvalidationEnd(String c, Object span) { throw new IllegalStateException("end"); } + }; + try (var service = new InvalidationService(new InMemoryInvalidationTransport(hub), journal, UUID.randomUUID(), + c -> { throw new IllegalStateException("listener"); }, metrics)) { + service.registerTarget("c", target); + service.setEventListener((c, e) -> { throw new IllegalStateException("event listener"); }); + journal.append("c", update("c", "k", 1)); + assertTrue(service.recoverAsync(Runnable::run).toCompletableFuture().get(5, TimeUnit.SECONDS)); + assertNotNull(target.versionOfL1Entry("k")); assertEquals("1", field(state(service, "c"), "cursor")); + var reset = service.resetAsync("c").toCompletableFuture().get(5, TimeUnit.SECONDS); + assertEquals(RecoveryResult.Status.RESET_SAFE, reset.status()); assertEquals(1, target.clears.get()); + } + } + + @Test void mailboxesCoalesceAndLargePassYieldsToAnotherCache() throws Exception { + try (Manual scheduler = new Manual(); Harness h = new Harness(scheduler)) { + Target other = new Target(); h.service.registerTarget("z", other); + for (int i = 1; i <= 5000; i++) h.journal.append("c", update("c", "k" + i, i)); + h.journal.append("z", update("z", "small", 1)); + var completion = h.recover(); + for (int i = 0; i < 1000; i++) assertSame(completion, h.recover()); + assertEquals(2, scheduler.queue.size()); + scheduler.next(); + assertTrue(h.journal.reads.get() <= 16); + assertTrue(scheduler.queue.size() <= 2); + scheduler.next(); + assertNotNull(other.versionOfL1Entry("small"), "the first continuation goes behind other ready caches"); + scheduler.drain(); + assertTrue(completion.get()); assertEquals(5000, h.target.values.size()); + assertTrue(scheduler.queue.isEmpty()); assertFalse(h.pending.get().getAsBoolean()); + int calls = h.journal.reads.get(); scheduler.advance(60); scheduler.drain(); assertEquals(calls, h.journal.reads.get()); + } + } + + @Test void failedRetriesUseOneTokenAndOneTwoFourEightSixteenThirtySecondBackoff() throws Exception { + try (Manual scheduler = new Manual(); Harness h = new Harness(scheduler)) { + h.journal.read = (c, p, n) -> { throw new IllegalStateException("read"); }; + h.journal.end = c -> { throw new IllegalStateException("baseline"); }; + var initial = h.recover(); scheduler.next(); assertFalse(initial.get()); + for (int delay : new int[]{1, 2, 4, 8, 16, 30, 30}) { + assertEquals(1, scheduler.queue.size()); + assertEquals(TimeUnit.SECONDS.toNanos(delay), scheduler.queue.peek().due - scheduler.now.get()); + int before = h.journal.ends.get(); + for (int i = 0; i < 100; i++) h.service.onL2Recovery(); + assertEquals(1, scheduler.queue.size()); assertEquals(before, h.journal.ends.get()); + scheduler.advance(delay); scheduler.next(); + assertEquals(before + 1, h.journal.ends.get()); assertTrue(h.pending.get().getAsBoolean()); + } + h.journal.read = (c, p, n) -> h.journal.data.checkedRead(c, p, n); + h.journal.end = h.journal.data::endCursor; + scheduler.advance(30); scheduler.drain(); // safe reset; its follow-up is also bounded + scheduler.advance(30); scheduler.drain(); assertFalse(h.pending.get().getAsBoolean()); + } + } + + @Test void actualWorkersBoundConcurrentReadsAndCoalesceRunningCaches() throws Exception { + var release = new CountDownLatch(1); + try (Harness h = new Harness()) { + h.service.registerTarget("d", new Target()); h.service.registerTarget("e", new Target()); + var entered = new CountDownLatch(2); var active = new AtomicInteger(); var maximum = new AtomicInteger(); + h.journal.read = (c, p, n) -> { + int running = active.incrementAndGet(); maximum.accumulateAndGet(running, Math::max); + try { entered.countDown(); gate(release); return h.journal.data.checkedRead(c, p, n); } + finally { active.decrementAndGet(); } + }; + var completion = h.recover(); + try { + assertTrue(entered.await(5, TimeUnit.SECONDS)); + for (int i = 0; i < 100; i++) h.service.onL2Recovery(); + assertEquals(2, h.journal.reads.get()); + assertEquals(2, maximum.get()); + } finally { release.countDown(); } + assertTrue(completion.get(5, TimeUnit.SECONDS)); assertTrue(maximum.get() <= 2); + } finally { release.countDown(); } + } + + @Test void closingCancelsQueuedTokensAndUnregistersGaugeWithoutFakeSuccess() throws Exception { + try (Manual scheduler = new Manual()) { + Harness h = new Harness(scheduler); var completion = h.recover(); + assertEquals(1, scheduler.queue.size()); assertTrue(h.pending.get().getAsBoolean()); + h.close(); assertEquals(0, scheduler.queue.size()); assertFalse(completion.get()); + assertEquals(1, h.removed.get()); + assertTrue(h.pending.get().getAsBoolean(), "unregister rather than rewriting pending as successful"); + scheduler.drain(); assertEquals(0, h.journal.reads.get()); + } + } + + @Test void noJournalClearDoesNotPretendToBeJournalOverflowOrReplay() throws Exception { + var overflows = new AtomicInteger(); var signals = new AtomicInteger(); var target = new Target(); + var metrics = new CacheMetricsListener() { + public void onInvalidation(String cache, Direction direction) { signals.incrementAndGet(); } + }; + try (var service = new InvalidationService(new InMemoryInvalidationTransport(new InMemoryInvalidationTransport.Hub()), + null, UUID.randomUUID(), cache -> overflows.incrementAndGet(), metrics)) { + service.registerTarget("c", target); + var result = service.resetAsync("c").toCompletableFuture().get(5, TimeUnit.SECONDS); + assertEquals(RecoveryResult.Status.NO_JOURNAL, result.status()); assertNull(result.baseline()); + assertEquals(1, target.clears.get()); assertEquals(0, overflows.get()); assertEquals(0, signals.get()); + } + } + + /** Deterministic two-worker scheduler seam: no task runs inline on submission. */ + static final class Manual extends ScheduledThreadPoolExecutor implements AutoCloseable { + final AtomicLong now = new AtomicLong(1); + final PriorityQueue queue = new PriorityQueue<>(); long sequence; + Manual() { super(2); } + class Task implements ScheduledFuture { + final Runnable runnable; final long due, order = sequence++; boolean cancelled, done; + Task(Runnable runnable, long due) { this.runnable = runnable; this.due = due; } + public long getDelay(TimeUnit unit) { return unit.convert(due - now.get(), TimeUnit.NANOSECONDS); } + public int compareTo(Delayed other) { Task t = (Task) other; int c = Long.compare(due, t.due); return c == 0 ? Long.compare(order, t.order) : c; } + public boolean cancel(boolean interrupt) { cancelled = true; queue.remove(this); return true; } + public boolean isCancelled() { return cancelled; } + public boolean isDone() { return done || cancelled; } + public Object get() { throw new UnsupportedOperationException(); } + public Object get(long timeout, TimeUnit unit) { throw new UnsupportedOperationException(); } + } + @Override public ScheduledFuture schedule(Runnable runnable, long delay, TimeUnit unit) { + if (isShutdown()) throw new RejectedExecutionException(); + Task task = new Task(runnable, now.get() + unit.toNanos(delay)); queue.add(task); return task; + } + void next() { + Task task = queue.remove(); assertTrue(task.due <= now.get()); + if (!task.cancelled) task.runnable.run(); task.done = true; + } + void drain() { int count = 0; while (!queue.isEmpty() && queue.peek().due <= now.get()) { assertTrue(count++ < 100); next(); } } + void advance(long seconds) { now.addAndGet(TimeUnit.SECONDS.toNanos(seconds)); } + public void close() { shutdownNow(); } + } +} diff --git a/tiercache-micrometer/src/main/java/io/tiercache/micrometer/MicrometerCacheMetrics.java b/tiercache-micrometer/src/main/java/io/tiercache/micrometer/MicrometerCacheMetrics.java index 2d584bc..3af36ad 100644 --- a/tiercache-micrometer/src/main/java/io/tiercache/micrometer/MicrometerCacheMetrics.java +++ b/tiercache-micrometer/src/main/java/io/tiercache/micrometer/MicrometerCacheMetrics.java @@ -57,12 +57,38 @@ public MicrometerCacheMetrics(MeterRegistry registry) { this.registry = registry; } + private final Map recoveryGauges = new ConcurrentHashMap<>(); + + private static final class RecoveryGauge { + final Map sources = new ConcurrentHashMap<>(); + Gauge meter; + double pending() { return sources.values().stream().anyMatch(java.util.function.BooleanSupplier::getAsBoolean) ? 1 : 0; } + } + + @Override + public AutoCloseable registerRecovery(String cache, java.util.function.BooleanSupplier pending) { + Object registration = new Object(); + recoveryGauges.compute(cache, (name, existing) -> { + RecoveryGauge state = existing == null ? new RecoveryGauge() : existing; + state.sources.put(registration, pending); + if (state.meter == null) state.meter = Gauge.builder("tiercache.invalidation.recovery.pending", state, RecoveryGauge::pending) + .tag("cache", cache).register(registry); + return state; + }); + return () -> recoveryGauges.computeIfPresent(cache, (name, state) -> { + state.sources.remove(registration); + if (!state.sources.isEmpty()) return state; + registry.remove(state.meter); + return null; + }); + } + // --- CacheMetricsListener --- @Override public void onRequest(String cache, Outcome outcome) { counter(requestCounters, cache, "requests", "result", - outcome.name().toLowerCase()).increment(); + outcome.name().toLowerCase(java.util.Locale.ROOT)).increment(); if (outcome == Outcome.LOAD) { lastStoreNanos.put(cache, System.nanoTime()); } @@ -71,7 +97,7 @@ public void onRequest(String cache, Outcome outcome) { @Override public void onLatency(String cache, Level level, long nanos) { latencyTimers.computeIfAbsent(cache + ":" + level, k -> Timer.builder("tiercache.latency") - .tags("cache", cache, "level", level.name().toLowerCase()) + .tags("cache", cache, "level", level.name().toLowerCase(java.util.Locale.ROOT)) .register(registry)) .record(nanos, TimeUnit.NANOSECONDS); } @@ -79,7 +105,23 @@ public void onLatency(String cache, Level level, long nanos) { @Override public void onInvalidation(String cache, Direction direction) { counter(invalidationCounters, cache, "invalidation", "direction", - direction.name().toLowerCase()).increment(); + direction.name().toLowerCase(java.util.Locale.ROOT)).increment(); + } + + private final Map publicationCounters = new ConcurrentHashMap<>(); + + @Override + public void onPublication(String cache, io.tiercache.spi.PublicationOutcome outcome, long count) { + counter(publicationCounters, cache, "invalidation.publish", "outcome", + outcome.name().toLowerCase(java.util.Locale.ROOT)).increment(count); + } + + private final Map streamFailures = new ConcurrentHashMap<>(); + + @Override + public void onStreamFailure(String cache, StreamResult result) { + counter(streamFailures, cache, "invalidation.stream", "result", + result.name().toLowerCase(java.util.Locale.ROOT)).increment(); } @Override diff --git a/tiercache-micrometer/src/main/java/io/tiercache/micrometer/TiercacheInspection.java b/tiercache-micrometer/src/main/java/io/tiercache/micrometer/TiercacheInspection.java index b420eac..fa5aa9a 100644 --- a/tiercache-micrometer/src/main/java/io/tiercache/micrometer/TiercacheInspection.java +++ b/tiercache-micrometer/src/main/java/io/tiercache/micrometer/TiercacheInspection.java @@ -7,10 +7,16 @@ import io.tiercache.spi.InvalidationJournal; import javax.management.MBeanServer; +import javax.management.InstanceAlreadyExistsException; +import javax.management.InstanceNotFoundException; import javax.management.ObjectName; import java.lang.management.ManagementFactory; import java.util.List; import java.util.Locale; +import java.util.concurrent.locks.ReentrantLock; +import java.util.function.Supplier; +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; /** * JMX registration for {@link TiercacheInspectionMXBean}. One instance per @@ -25,7 +31,11 @@ public final class TiercacheInspection implements TiercacheInspectionMXBean, Aut private final TierCacheFactory factory; private final InvalidationJournal journal; private final List cacheNames; + private static final Logger log = LoggerFactory.getLogger(TiercacheInspection.class); + private final ReentrantLock lifecycle = new ReentrantLock(); + private final Supplier serverSupplier; private ObjectName objectName; + private MBeanServer ownerServer; /** * Creates an inspection MBean reading counters from the given registry @@ -42,6 +52,13 @@ public final class TiercacheInspection implements TiercacheInspectionMXBean, Aut */ public TiercacheInspection(MeterRegistry registry, TierCacheFactory factory, InvalidationJournal journal, List cacheNames) { + this(registry, factory, journal, cacheNames, ManagementFactory::getPlatformMBeanServer); + } + + /** Package-local seam for deterministic registration failure and lifecycle races. */ + TiercacheInspection(MeterRegistry registry, TierCacheFactory factory, + InvalidationJournal journal, List cacheNames, Supplier serverSupplier) { + this.serverSupplier = serverSupplier; this.registry = registry; this.factory = factory; this.journal = journal; @@ -57,14 +74,22 @@ public TiercacheInspection(MeterRegistry registry, TierCacheFactory factory, * @since 0.1.0 */ public void register() { + lifecycle.lock(); try { - objectName = new ObjectName("io.tiercache:type=Inspection"); - MBeanServer server = ManagementFactory.getPlatformMBeanServer(); - if (!server.isRegistered(objectName)) { - server.registerMBean(this, objectName); + if (objectName != null) return; // Preserve this ownership period on repeated registration. + ObjectName name = new ObjectName("io.tiercache:type=Inspection"); + MBeanServer server = serverSupplier.get(); + try { + server.registerMBean(this, name); + } catch (InstanceAlreadyExistsException occupied) { + return; // A foreign or competing registration grants no ownership. } + ownerServer = server; + objectName = name; } catch (Exception e) { throw new IllegalStateException("JMX registration failed", e); + } finally { + lifecycle.unlock(); } } @@ -111,14 +136,28 @@ private double counter(String cache, String result) { } } + /** + * Relinquishes this successful registration once. Skipped registrations and + * repeated closes are harmless; a later register may start a new ownership period. + */ @Override public void close() { + lifecycle.lock(); try { - if (objectName != null) { - ManagementFactory.getPlatformMBeanServer().unregisterMBean(objectName); + ObjectName owned = objectName; + MBeanServer server = ownerServer; + objectName = null; + ownerServer = null; + if (owned == null) return; + try { + server.unregisterMBean(owned); + } catch (InstanceNotFoundException absent) { + // External removal is benign; this ownership period is already consumed. + } catch (Exception failure) { + log.warn("JMX inspection unregister failed ({})", failure.getClass().getSimpleName()); } - } catch (Exception e) { - // best effort + } finally { + lifecycle.unlock(); } } } diff --git a/tiercache-micrometer/src/test/java/io/tiercache/micrometer/DashboardCoverageTest.java b/tiercache-micrometer/src/test/java/io/tiercache/micrometer/DashboardCoverageTest.java index 464f781..d369366 100644 --- a/tiercache-micrometer/src/test/java/io/tiercache/micrometer/DashboardCoverageTest.java +++ b/tiercache-micrometer/src/test/java/io/tiercache/micrometer/DashboardCoverageTest.java @@ -23,7 +23,7 @@ void dashboardCoversEveryMetric() throws Exception { // Every exported metric appears (Prometheus naming: dots -> underscores, // counters gain _total). for (String metric : new String[]{"tiercache_requests_total", "tiercache_latency", - "tiercache_invalidation_total", "tiercache_degraded", "tiercache_breaker_state", + "tiercache_invalidation_total", "tiercache_invalidation_publish_total", "tiercache_degraded", "tiercache_breaker_state", "tiercache_journal_size", "tiercache_last_load_age", "tiercache_null_entries_total", "tiercache_l2_stale_hits_total", "tiercache_l2_revalidation_triggers_total", "tiercache_l2_revalidation_completions_total", "tiercache_l2_revalidation_failures_total"}) { @@ -36,6 +36,7 @@ void alertRulesDefined() throws Exception { String yaml = Files.readString(DOCS.resolve("alerts.yml")); assertTrue(yaml.contains("TiercacheMissGrowth"), "miss growth alert"); assertTrue(yaml.contains("TiercacheDroppedInvalidations"), "dropped invalidations alert"); + assertTrue(yaml.contains("TiercachePublicationFailures"), "publication failures alert"); assertTrue(yaml.contains("TiercacheDegraded"), "degraded alert"); assertTrue(yaml.contains("TiercacheRevalidationFailures"), "revalidation failures alert"); assertTrue(yaml.contains("tiercache_requests_total")); diff --git a/tiercache-micrometer/src/test/java/io/tiercache/micrometer/InspectionLifecycleTest.java b/tiercache-micrometer/src/test/java/io/tiercache/micrometer/InspectionLifecycleTest.java new file mode 100644 index 0000000..d62fd33 --- /dev/null +++ b/tiercache-micrometer/src/test/java/io/tiercache/micrometer/InspectionLifecycleTest.java @@ -0,0 +1,110 @@ +package io.tiercache.micrometer; + +import io.micrometer.core.instrument.simple.SimpleMeterRegistry; +import io.tiercache.TierCacheFactory; +import io.tiercache.testkit.InMemoryRemoteCache; +import org.junit.jupiter.api.Test; +import javax.management.*; +import java.lang.reflect.*; +import java.util.List; +import java.util.concurrent.*; +import java.util.concurrent.atomic.*; +import static org.junit.jupiter.api.Assertions.*; + +class InspectionLifecycleTest { + static final ObjectName NAME=InspectionRegressionTest.NAME; + static class Rig implements AutoCloseable { + final SimpleMeterRegistry registry=new SimpleMeterRegistry(); + final TierCacheFactory factory=TierCacheFactory.builder().remoteCache(new InMemoryRemoteCache<>()).build(); + final TiercacheInspection inspection; + Rig(String name,MBeanServer server){inspection=new TiercacheInspection(registry,factory,null,List.of(name),()->server);} + public void close(){inspection.close();factory.close();registry.close();} + } + static MBeanServer proxy(InvocationHandler handler) { + return (MBeanServer)Proxy.newProxyInstance(InspectionLifecycleTest.class.getClassLoader(),new Class[]{MBeanServer.class},handler); + } + static Object invoke(MBeanServer server,Method method,Object[] args)throws Throwable { + try{return method.invoke(server,args);}catch(InvocationTargetException e){throw e.getCause();} + } + static void await(CountDownLatch latch) { + try{assertTrue(latch.await(5,TimeUnit.SECONDS));}catch(InterruptedException e){throw new AssertionError(e);} + } + static String owner(MBeanServer server)throws Exception{return ((String[])server.getAttribute(NAME,"CacheNames"))[0];} + + @Test void failedRegistrationNeverGrantsCleanupAuthorityAndCanBeRetried() throws Exception { + var server=MBeanServerFactory.newMBeanServer();var fail=new AtomicBoolean(true);var unregisters=new AtomicInteger(); + var failure=new MBeanRegistrationException(new Exception("registration failed")); + var instrumented=proxy((p,m,a)->{ + if(m.getName().equals("registerMBean")&&fail.getAndSet(false))throw failure; + if(m.getName().equals("unregisterMBean"))unregisters.incrementAndGet(); + return invoke(server,m,a); + }); + try(var a=new Rig("a",instrumented);var b=new Rig("b",server)) { + assertSame(failure,assertThrows(IllegalStateException.class,a.inspection::register).getCause()); + b.inspection.register();a.inspection.close();assertEquals(0,unregisters.get());assertEquals("b",owner(server)); + b.inspection.close();a.inspection.register();assertEquals("a",owner(server));a.inspection.close();assertEquals(1,unregisters.get()); + } + } + @Test void unregisterFailureConsumesOwnershipBeforeAReplacementAppears() throws Exception { + var server=MBeanServerFactory.newMBeanServer();var unregisters=new AtomicInteger(); + var instrumented=proxy((p,m,a)->{ + if(m.getName().equals("unregisterMBean")){unregisters.incrementAndGet();throw new SecurityException("denied");} + return invoke(server,m,a); + }); + try(var a=new Rig("a",instrumented);var b=new Rig("b",server)) { + a.inspection.register();assertDoesNotThrow(a.inspection::close);assertEquals(1,unregisters.get()); + server.unregisterMBean(NAME);b.inspection.register();a.inspection.close();assertEquals(1,unregisters.get());assertEquals("b",owner(server)); + } + } + @Test void externallyAbsentOwnedNameIsBenignAndAllowsNewOwnershipPeriod() throws Exception { + var server=MBeanServerFactory.newMBeanServer(); + try(var rig=new Rig("a",server)) { + rig.inspection.register();server.unregisterMBean(NAME);assertDoesNotThrow(rig.inspection::close); + rig.inspection.register();assertEquals("a",owner(server));rig.inspection.close();assertFalse(server.isRegistered(NAME)); + } + } + @Test void concurrentRegistrantsHaveOneOwnerAndLoserCannotRemoveWinner() throws Exception { + var server=MBeanServerFactory.newMBeanServer();var meet=new CyclicBarrier(2);var occupied=new AtomicInteger(); + var instrumented=proxy((p,m,a)->{ + if(m.getName().equals("registerMBean"))meet.await(5,TimeUnit.SECONDS); + try{return invoke(server,m,a);}catch(InstanceAlreadyExistsException e){occupied.incrementAndGet();throw e;} + }); + var pool=Executors.newFixedThreadPool(2); + try(var a=new Rig("a",instrumented);var b=new Rig("b",instrumented)) { + var first=pool.submit(a.inspection::register);var second=pool.submit(b.inspection::register); + first.get(5,TimeUnit.SECONDS);second.get(5,TimeUnit.SECONDS);assertEquals(1,occupied.get()); + String name=owner(server);var winner=name.equals("a")?a:b;var loser=name.equals("a")?b:a; + loser.inspection.close();assertEquals(name,owner(server));winner.inspection.register();assertEquals(name,owner(server)); + winner.inspection.close();assertFalse(server.isRegistered(NAME)); + } finally {pool.shutdownNow();} + } + @Test void closeWaitsForRegistrationPublicationAndSkippedCompetitorRemainsUnowned() throws Exception { + var server=MBeanServerFactory.newMBeanServer();var registered=new CountDownLatch(1);var release=new CountDownLatch(1);var closing=new CountDownLatch(1); + var instrumented=proxy((p,m,a)->{ + Object result=invoke(server,m,a); + if(m.getName().equals("registerMBean")){registered.countDown();await(release);} + return result; + }); + var pool=Executors.newFixedThreadPool(2); + try(var a=new Rig("a",instrumented);var b=new Rig("b",server)) { + var registration=pool.submit(a.inspection::register);await(registered); + var close=pool.submit(()->{closing.countDown();a.inspection.close();});await(closing); + b.inspection.register();b.inspection.close();assertEquals("a",owner(server)); + release.countDown();registration.get(5,TimeUnit.SECONDS);close.get(5,TimeUnit.SECONDS);assertFalse(server.isRegistered(NAME)); + b.inspection.register();a.inspection.close();assertEquals("b",owner(server)); + } finally {release.countDown();pool.shutdownNow();} + } + @Test void registrationFollowingAnInFlightCloseStartsANewOwnershipPeriod() throws Exception { + var server=MBeanServerFactory.newMBeanServer();var entered=new CountDownLatch(1);var release=new CountDownLatch(1);var gate=new AtomicBoolean(true); + var instrumented=proxy((p,m,a)->{ + if(m.getName().equals("unregisterMBean")&&gate.getAndSet(false)){entered.countDown();await(release);} + return invoke(server,m,a); + }); + var pool=Executors.newFixedThreadPool(2); + try(var rig=new Rig("a",instrumented)) { + rig.inspection.register();var close=pool.submit(rig.inspection::close);await(entered); + var register=pool.submit(rig.inspection::register);release.countDown();close.get(5,TimeUnit.SECONDS);register.get(5,TimeUnit.SECONDS); + assertEquals("a",owner(server));rig.inspection.close();assertFalse(server.isRegistered(NAME)); + } finally {release.countDown();pool.shutdownNow();} + } +} diff --git a/tiercache-micrometer/src/test/java/io/tiercache/micrometer/InspectionRegressionTest.java b/tiercache-micrometer/src/test/java/io/tiercache/micrometer/InspectionRegressionTest.java new file mode 100644 index 0000000..62dc69d --- /dev/null +++ b/tiercache-micrometer/src/test/java/io/tiercache/micrometer/InspectionRegressionTest.java @@ -0,0 +1,111 @@ +package io.tiercache.micrometer; + +import io.micrometer.core.instrument.simple.SimpleMeterRegistry; +import io.tiercache.*; +import io.tiercache.spi.CacheMetricsListener; +import io.tiercache.testkit.InMemoryRemoteCache; +import org.junit.jupiter.api.Test; +import org.junit.jupiter.api.parallel.ResourceLock; +import javax.management.*; +import java.lang.management.ManagementFactory; +import java.util.*; +import static org.junit.jupiter.api.Assertions.*; + +@ResourceLock("tiercache-platform-inspection") +@ResourceLock("java.util.Locale.default") +class InspectionRegressionTest { + static final ObjectName NAME; + static {try{NAME=new ObjectName("io.tiercache:type=Inspection");}catch(Exception e){throw new ExceptionInInitializerError(e);}} + static final MBeanServer SERVER=ManagementFactory.getPlatformMBeanServer(); + static class Rig implements AutoCloseable { + final SimpleMeterRegistry registry=new SimpleMeterRegistry(); + final TierCacheFactory factory=TierCacheFactory.builder().remoteCache(new InMemoryRemoteCache<>()).build(); + final TiercacheInspection inspection; + Rig(String name){inspection=new TiercacheInspection(registry,factory,null,List.of(name));} + public void close(){inspection.close();factory.close();registry.close();} + } + static void clean() throws Exception {if(SERVER.isRegistered(NAME))SERVER.unregisterMBean(NAME);} + static void owner(String name) throws Exception {assertArrayEquals(new String[]{name},(String[])SERVER.getAttribute(NAME,"CacheNames"));} + @Test void closingSkippedRegistrantPreservesTheOwner() throws Exception { + assertFalse(SERVER.isRegistered(NAME)); + try(var a=new Rig("a");var b=new Rig("b")) { + a.inspection.register();b.inspection.register();b.inspection.close();owner("a"); + } finally {clean();} + } + @Test void repeatedCloseCannotRemoveReplacement() throws Exception { + assertFalse(SERVER.isRegistered(NAME)); + try(var a=new Rig("a");var b=new Rig("b")) { + a.inspection.register();a.inspection.close();b.inspection.register();a.inspection.close();owner("b"); + } finally {clean();} + } + public interface ForeignMBean {String getMarker();} + public static class Foreign implements ForeignMBean {public String getMarker(){return "foreign";}} + @Test void skippedInspectionCannotRemoveForeignRegistration() throws Exception { + assertFalse(SERVER.isRegistered(NAME)); + try(var rig=new Rig("a")) { + SERVER.registerMBean(new Foreign(),NAME);rig.inspection.register();rig.inspection.close(); + assertEquals("foreign",SERVER.getAttribute(NAME,"Marker")); + } finally {clean();} + } + @Test void repeatedRegisterPreservesOwnershipAndExplicitReregisterWorks() throws Exception { + assertFalse(SERVER.isRegistered(NAME)); + try(var rig=new Rig("a")) { + rig.inspection.register();rig.inspection.register();owner("a");rig.inspection.close();assertFalse(SERVER.isRegistered(NAME)); + rig.inspection.close();rig.inspection.register();owner("a");rig.inspection.close();assertFalse(SERVER.isRegistered(NAME)); + } finally {clean();} + } + @Test void turkishLocaleKeepsAsciiTagsAndQueryableJmxRatiosAcrossLocaleChanges() throws Exception { + assertFalse(SERVER.isRegistered(NAME));Locale previous=Locale.getDefault(); + try(var rig=new Rig("IstanbulCache")) { + var metrics=new MicrometerCacheMetrics(rig.registry);Locale.setDefault(Locale.forLanguageTag("tr-TR")); + metrics.onRequest("IstanbulCache",CacheMetricsListener.Outcome.L1_HIT); + metrics.onRequest("IstanbulCache",CacheMetricsListener.Outcome.L2_HIT); + metrics.onRequest("IstanbulCache",CacheMetricsListener.Outcome.MISS); + metrics.onInvalidation("IstanbulCache",CacheMetricsListener.Direction.RECEIVED); + rig.inspection.register(); + for(String tag:new String[]{"l1_hit","l2_hit","miss"})assertNotNull(rig.registry.find("tiercache.requests").tags("cache","IstanbulCache","result",tag).counter()); + assertNotNull(rig.registry.find("tiercache.invalidation").tags("cache","IstanbulCache","direction","received").counter()); + assertEquals(1.0/3,(double)SERVER.invoke(NAME,"getL1HitRatio",new Object[]{"IstanbulCache"},new String[]{String.class.getName()}),0.00001); + assertEquals(1.0/3,(double)SERVER.invoke(NAME,"getL2HitRatio",new Object[]{"IstanbulCache"},new String[]{String.class.getName()}),0.00001); + int meters=rig.registry.getMeters().size();Locale.setDefault(Locale.ROOT); + metrics.onRequest("IstanbulCache",CacheMetricsListener.Outcome.L1_HIT); + metrics.onRequest("IstanbulCache",CacheMetricsListener.Outcome.L2_HIT); + metrics.onRequest("IstanbulCache",CacheMetricsListener.Outcome.MISS); + metrics.onInvalidation("IstanbulCache",CacheMetricsListener.Direction.RECEIVED); + assertEquals(meters,rig.registry.getMeters().size()); + assertEquals(2,rig.registry.get("tiercache.requests").tags("cache","IstanbulCache","result","l1_hit").counter().count()); + assertEquals(1.0/3,(double)SERVER.invoke(NAME,"getL1HitRatio",new Object[]{"IstanbulCache"},new String[]{String.class.getName()}),0.00001); + owner("IstanbulCache"); + } finally {Locale.setDefault(previous);clean();} + } + @Test void allEnumLabelsStayStableAcrossDefaultLocaleChanges() { + Locale previous=Locale.getDefault();var registry=new SimpleMeterRegistry(); + try { + var metrics=new MicrometerCacheMetrics(registry);String cache="IstanbulCache"; + for(var locale:List.of(Locale.forLanguageTag("tr-TR"),Locale.ROOT)) { + Locale.setDefault(locale); + for(var outcome:CacheMetricsListener.Outcome.values())metrics.onRequest(cache,outcome); + for(var direction:CacheMetricsListener.Direction.values())metrics.onInvalidation(cache,direction); + for(var level:CacheMetricsListener.Level.values())metrics.onLatency(cache,level,1000); + } + for(var outcome:CacheMetricsListener.Outcome.values())assertEquals(2,registry.get("tiercache.requests") + .tags("cache",cache,"result",outcome.name().toLowerCase(Locale.ROOT)).counter().count()); + for(var direction:CacheMetricsListener.Direction.values())assertEquals(2,registry.get("tiercache.invalidation") + .tags("cache",cache,"direction",direction.name().toLowerCase(Locale.ROOT)).counter().count()); + for(var level:CacheMetricsListener.Level.values())assertEquals(2,registry.get("tiercache.latency") + .tags("cache",cache,"level",level.name().toLowerCase(Locale.ROOT)).timer().count()); + assertEquals(CacheMetricsListener.Outcome.values().length+CacheMetricsListener.Direction.values().length + +CacheMetricsListener.Level.values().length,registry.getMeters().size()); + } finally {Locale.setDefault(previous);registry.close();} + } + + @Test void publicJmxNameAttributesAndOperationsStayUnchanged() throws Exception { + assertFalse(SERVER.isRegistered(NAME)); + try(var rig=new Rig("a")) { + rig.inspection.register();var info=SERVER.getMBeanInfo(NAME); + assertEquals(Set.of("CacheNames","BreakerState"),Arrays.stream(info.getAttributes()).map(MBeanAttributeInfo::getName).collect(java.util.stream.Collectors.toSet())); + assertEquals(Set.of("getL1HitRatio","getL2HitRatio","getJournalSize"),Arrays.stream(info.getOperations()).map(MBeanOperationInfo::getName).collect(java.util.stream.Collectors.toSet())); + } finally {clean();} + } + +} diff --git a/tiercache-micrometer/src/test/java/io/tiercache/micrometer/PublicationMetricsTest.java b/tiercache-micrometer/src/test/java/io/tiercache/micrometer/PublicationMetricsTest.java new file mode 100644 index 0000000..6998afe --- /dev/null +++ b/tiercache-micrometer/src/test/java/io/tiercache/micrometer/PublicationMetricsTest.java @@ -0,0 +1,23 @@ +package io.tiercache.micrometer; +import io.micrometer.core.instrument.simple.SimpleMeterRegistry; +import io.tiercache.spi.PublicationOutcome; +import org.junit.jupiter.api.Test; +import java.util.Locale; +import static org.junit.jupiter.api.Assertions.*; + +@org.junit.jupiter.api.parallel.ResourceLock("java.util.Locale.default") +class PublicationMetricsTest { + @Test void batchesHaveFixedAsciiLabelsEvenUnderTurkishLocale() { + Locale previous=Locale.getDefault();Locale.setDefault(Locale.forLanguageTag("tr-TR")); + var registry=new SimpleMeterRegistry(); + try { + var metrics=new MicrometerCacheMetrics(registry); + for(var outcome:PublicationOutcome.values()) { + metrics.onPublication("c",outcome,7);metrics.onPublication("c",outcome,3); + var meter=registry.find("tiercache.invalidation.publish").tags("cache","c","outcome",outcome.name().toLowerCase(Locale.ROOT)).counter(); + assertNotNull(meter);assertEquals(10,meter.count());assertEquals(2,meter.getId().getTags().size()); + } + assertEquals(4,registry.getMeters().size()); + } finally {registry.close();Locale.setDefault(previous);} + } +} diff --git a/tiercache-micrometer/src/test/java/io/tiercache/micrometer/RecoveryGaugeTest.java b/tiercache-micrometer/src/test/java/io/tiercache/micrometer/RecoveryGaugeTest.java new file mode 100644 index 0000000..166b3e5 --- /dev/null +++ b/tiercache-micrometer/src/test/java/io/tiercache/micrometer/RecoveryGaugeTest.java @@ -0,0 +1,26 @@ +package io.tiercache.micrometer; + +import io.micrometer.core.instrument.simple.SimpleMeterRegistry; +import org.junit.jupiter.api.Test; +import java.util.concurrent.atomic.AtomicBoolean; +import static org.junit.jupiter.api.Assertions.*; + +class RecoveryGaugeTest { + @Test void pendingTracksSourcesAndCloseDoesNotRemoveAnotherOwner() throws Exception { + var registry = new SimpleMeterRegistry(); + try { + var metrics = new MicrometerCacheMetrics(registry); + var first = new AtomicBoolean(); var second = new AtomicBoolean(true); + var one = metrics.registerRecovery("c", first::get); + var two = metrics.registerRecovery("c", second::get); + var gauge = registry.get("tiercache.invalidation.recovery.pending").tag("cache", "c").gauge(); + assertEquals(1, gauge.value()); second.set(false); assertEquals(0, gauge.value()); + first.set(true); assertEquals(1, gauge.value()); one.close(); assertEquals(0, gauge.value()); + two.close(); assertNull(registry.find("tiercache.invalidation.recovery.pending").gauge()); + var replacement = metrics.registerRecovery("c", first::get); + one.close(); two.close(); + assertEquals(1, registry.get("tiercache.invalidation.recovery.pending").gauge().value()); + replacement.close(); assertNull(registry.find("tiercache.invalidation.recovery.pending").gauge()); + } finally { registry.close(); } + } +} diff --git a/tiercache-micrometer/src/test/java/io/tiercache/micrometer/StreamFailureMetricsTest.java b/tiercache-micrometer/src/test/java/io/tiercache/micrometer/StreamFailureMetricsTest.java new file mode 100644 index 0000000..b233861 --- /dev/null +++ b/tiercache-micrometer/src/test/java/io/tiercache/micrometer/StreamFailureMetricsTest.java @@ -0,0 +1,20 @@ +package io.tiercache.micrometer; +import io.micrometer.core.instrument.simple.SimpleMeterRegistry; +import io.tiercache.spi.CacheMetricsListener.StreamResult; +import org.junit.jupiter.api.Test; +import java.util.*; +import static org.junit.jupiter.api.Assertions.*; +@org.junit.jupiter.api.parallel.ResourceLock("java.util.Locale.default") +class StreamFailureMetricsTest { + @Test void streamFailuresUseFixedLocaleIndependentLowCardinalityLabels() { + Locale previous = Locale.getDefault(); var registry = new SimpleMeterRegistry(); + try { + Locale.setDefault(Locale.forLanguageTag("tr-TR")); var metrics = new MicrometerCacheMetrics(registry); + for (var result : StreamResult.values()) { + metrics.onStreamFailure("c", result); + var counter = registry.get("tiercache.invalidation.stream").tags("cache","c","result",result.name().toLowerCase(Locale.ROOT)).counter(); + assertEquals(1,counter.count()); assertEquals(2,counter.getId().getTags().size()); + } + } finally { Locale.setDefault(previous); registry.close(); } + } +} diff --git a/tiercache-micrometer/src/test/java/io/tiercache/micrometer/TiercacheInspectionTest.java b/tiercache-micrometer/src/test/java/io/tiercache/micrometer/TiercacheInspectionTest.java index 0df408a..80b9f2b 100644 --- a/tiercache-micrometer/src/test/java/io/tiercache/micrometer/TiercacheInspectionTest.java +++ b/tiercache-micrometer/src/test/java/io/tiercache/micrometer/TiercacheInspectionTest.java @@ -15,6 +15,7 @@ import static org.junit.jupiter.api.Assertions.assertEquals; /** JMX registration and state exposure. */ +@org.junit.jupiter.api.parallel.ResourceLock("tiercache-platform-inspection") class TiercacheInspectionTest { @Test diff --git a/tiercache-micronaut/src/main/java/io/tiercache/micronaut/TiercacheMicronautConfiguration.java b/tiercache-micronaut/src/main/java/io/tiercache/micronaut/TiercacheMicronautConfiguration.java index 166a1e9..156babc 100644 --- a/tiercache-micronaut/src/main/java/io/tiercache/micronaut/TiercacheMicronautConfiguration.java +++ b/tiercache-micronaut/src/main/java/io/tiercache/micronaut/TiercacheMicronautConfiguration.java @@ -1,5 +1,6 @@ package io.tiercache.micronaut; +import io.tiercache.invalidation.JournalProtocol; import io.lettuce.core.ClientOptions; import io.lettuce.core.RedisClient; import io.lettuce.core.SocketOptions; @@ -77,6 +78,9 @@ public TiercacheMicronautConfiguration() { @Bean(preDestroy = "shutdown") @Requires(missingBeans = RemoteCache.class) RedisClient tiercacheRedisClient(TiercacheProperties properties) { + if (properties.getInvalidation().isEnabled()) { + JournalProtocol.requireCapacity(properties.getInvalidation().getJournalCapacity()); + } if (properties.getRedisUri() == null || properties.getRedisUri().isBlank()) { throw new IllegalStateException( "tiercache.redis-uri is required when tiercache.enabled=true " @@ -97,6 +101,7 @@ RedisClient tiercacheRedisClient(TiercacheProperties properties) { @Requires(property = "tiercache.invalidation.enabled", notEquals = "false") RedisStreamJournal tiercacheInvalidationJournal(RedisClient tiercacheRedisClient, TiercacheProperties properties) { + JournalProtocol.requireCapacity(properties.getInvalidation().getJournalCapacity()); return new RedisStreamJournal(tiercacheRedisClient.connect(ByteArrayCodec.INSTANCE), properties.getInvalidation().getJournalCapacity(), new JdkCacheSerializer<>()); } @@ -150,8 +155,13 @@ TierCacheFactory tierCacheFactory(TiercacheProperties properties, BeanProvider metrics, BeanProvider lockProvider) { Map overridesByName = overridesByName(caches); - TierCacheFactory.Builder builder = TierCacheFactory.builder() - .defaults(properties.getDefaults().toSettings(io.tiercache.CacheSettings.defaults())); + io.tiercache.CacheSettings defaults; + try { + defaults = properties.getDefaults().toSettings(io.tiercache.CacheSettings.defaults()); + } catch (IllegalArgumentException e) { + throw new io.tiercache.CacheConfigurationException("Cache '': " + e.getMessage()); + } + TierCacheFactory.Builder builder = TierCacheFactory.builder().defaults(defaults); overridesByName.forEach((name, props) -> builder.cache(name, props.toOverride())); if (properties.getAsyncExecutorThreads() > 0) { builder.asyncExecutorThreads(properties.getAsyncExecutorThreads()); diff --git a/tiercache-micronaut/src/main/java/io/tiercache/micronaut/TiercacheProperties.java b/tiercache-micronaut/src/main/java/io/tiercache/micronaut/TiercacheProperties.java index bb3ede8..5c5ee73 100644 --- a/tiercache-micronaut/src/main/java/io/tiercache/micronaut/TiercacheProperties.java +++ b/tiercache-micronaut/src/main/java/io/tiercache/micronaut/TiercacheProperties.java @@ -1,5 +1,6 @@ package io.tiercache.micronaut; +import io.tiercache.invalidation.JournalProtocol; import io.micronaut.context.annotation.ConfigurationInject; import io.micronaut.context.annotation.ConfigurationProperties; import io.micronaut.core.annotation.Introspected; @@ -191,7 +192,7 @@ public String getProfile() { * @since 1.1.0 */ public int getJournalCapacity() { - return journalCapacity != null ? journalCapacity : 10_000; + return journalCapacity != null ? journalCapacity : JournalProtocol.DEFAULT_CAPACITY; } } diff --git a/tiercache-micronaut/src/test/java/io/tiercache/micronaut/CacheableAnnotationEndToEndTest.java b/tiercache-micronaut/src/test/java/io/tiercache/micronaut/CacheableAnnotationEndToEndTest.java index 459654e..56053f0 100644 --- a/tiercache-micronaut/src/test/java/io/tiercache/micronaut/CacheableAnnotationEndToEndTest.java +++ b/tiercache-micronaut/src/test/java/io/tiercache/micronaut/CacheableAnnotationEndToEndTest.java @@ -94,7 +94,7 @@ void cacheableSkipsTheMethodBodyOnSecondCall() { // The entry lives in L2 under the cache's namespace. try (RedisClient probe = RedisClient.create(redisUri()); io.lettuce.core.api.StatefulRedisConnection conn = probe.connect()) { - assertThat(conn.sync().keys("micronaut:numbers:*")).isNotEmpty(); + assertThat(conn.sync().keys(new String(io.tiercache.redis.RedisKeyspace.dataPrefix("micronaut:numbers"), java.nio.charset.StandardCharsets.US_ASCII) + "*")).isNotEmpty(); } } diff --git a/tiercache-micronaut/src/test/java/io/tiercache/micronaut/DegradationWindowBindingTest.java b/tiercache-micronaut/src/test/java/io/tiercache/micronaut/DegradationWindowBindingTest.java new file mode 100644 index 0000000..47706f5 --- /dev/null +++ b/tiercache-micronaut/src/test/java/io/tiercache/micronaut/DegradationWindowBindingTest.java @@ -0,0 +1,46 @@ +package io.tiercache.micronaut; + +import io.micronaut.context.ApplicationContext; +import io.tiercache.*; +import org.junit.jupiter.api.Test; +import java.time.Duration; +import java.util.Map; +import static org.assertj.core.api.Assertions.assertThat; +import static org.junit.jupiter.api.Assertions.assertThrows; + +class DegradationWindowBindingTest { + private static CacheSettings actual(TierCacheFactory factory,String name)throws Exception { + var cache=factory.getCache(name);var field=cache.getClass().getDeclaredField("settings"); + field.setAccessible(true);return (CacheSettings)field.get(cache); + } + @Test void omittedWindowIsOffInActualEngine() throws Exception { + try(var context=ApplicationContext.run(Map.of("tiercache.enabled",true),"tiercache-inmemory-l2")) { + assertThat(actual(context.getBean(TierCacheFactory.class),"orders").degradationStaleTtl()).isEqualTo(Duration.ZERO); + } + } + @Test void inheritanceOverrideAndExplicitZeroReachActualEngine() throws Exception { + try(var context=ApplicationContext.run(Map.of("tiercache.enabled",true, + "tiercache.defaults.degradation-stale-ttl","10m", + "tiercache.caches.orders.degradation-stale-ttl","20m", + "tiercache.caches.off.degradation-stale-ttl","0s"),"tiercache-inmemory-l2")) { + var factory=context.getBean(TierCacheFactory.class); + assertThat(actual(factory,"inherit").degradationStaleTtl()).isEqualTo(Duration.ofMinutes(10)); + assertThat(actual(factory,"orders").degradationStaleTtl()).isEqualTo(Duration.ofMinutes(20)); + assertThat(actual(factory,"off").degradationStaleTtl()).isEqualTo(Duration.ZERO); + assertThat(actual(factory,"off").l2Ttl()).isEqualTo(actual(factory,"inherit").l2Ttl()); + } + } + @Test void negativeGlobalAndNamedWindowsFailWithContext() { + for(String name:new String[]{"defaults","caches.orders"}) { + var error=assertThrows(RuntimeException.class,()->{ + try(var context=ApplicationContext.run(Map.of("tiercache.enabled",true, + "tiercache."+name+".degradation-stale-ttl","-1s"),"tiercache-inmemory-l2")) { + context.getBean(TierCacheFactory.class); + } + }); + StringBuilder messages=new StringBuilder();Throwable cause=error; + while(cause!=null){messages.append(cause.getMessage());cause=cause.getCause();} + assertThat(messages.toString()).contains("degradationStaleTtl",name.equals("defaults")?"global defaults":"orders"); + } + } +} diff --git a/tiercache-micronaut/src/test/java/io/tiercache/micronaut/JournalCapacityConfigurationTest.java b/tiercache-micronaut/src/test/java/io/tiercache/micronaut/JournalCapacityConfigurationTest.java new file mode 100644 index 0000000..43f83db --- /dev/null +++ b/tiercache-micronaut/src/test/java/io/tiercache/micronaut/JournalCapacityConfigurationTest.java @@ -0,0 +1,62 @@ +package io.tiercache.micronaut; +import io.tiercache.TierCacheFactory; +import io.tiercache.redis.RedisStreamJournal; +import io.micronaut.context.ApplicationContext; +import org.junit.jupiter.api.Test; +import org.testcontainers.containers.GenericContainer; +import org.testcontainers.utility.DockerImageName; +import java.util.*; +import static org.junit.jupiter.api.Assertions.*; + +class JournalCapacityConfigurationTest { + @Test void invalidEnabledCapacitiesFailBeforeConnectionAndTraffic() { + for(int value:new int[]{-1,0,1,32,63,64}) { + var failure=assertThrows(RuntimeException.class,()->{ + try(var context=ApplicationContext.run(Map.of("tiercache.enabled",true,"tiercache.redis-uri","redis://127.0.0.1:1", + "tiercache.invalidation.journal-capacity",value))) {context.getBean(TierCacheFactory.class);} + });check(failure,value); + } + } + @Test void boundaryDefaultAndDisabledSettingsStartAndServeTraffic() { + try(var redis=new GenericContainer<>(DockerImageName.parse("redis:6.2-alpine")).withExposedPorts(6379)) { + redis.start();String uri="redis://"+redis.getHost()+":"+redis.getMappedPort(6379); + for(String value:new String[]{"65","10000","omitted"}) { + Map props=new HashMap<>();props.put("tiercache.enabled",true);props.put("tiercache.redis-uri",uri); + if(!value.equals("omitted"))props.put("tiercache.invalidation.journal-capacity",value); + try(var context=ApplicationContext.run(props)) { + assertEquals(value.equals("65")?65:10000,context.getBean(RedisStreamJournal.class).capacity()); + var cache=context.getBean(TierCacheFactory.class).getCache("c");cache.put("x","v");assertEquals("v",cache.get("x")); + } + } + try(var context=ApplicationContext.run(Map.of("tiercache.enabled",true,"tiercache.redis-uri",uri, + "tiercache.invalidation.enabled",false,"tiercache.invalidation.journal-capacity",1))) { + assertFalse(context.containsBean(RedisStreamJournal.class)); + var cache=context.getBean(TierCacheFactory.class).getCache("c");cache.put("x","v");assertEquals("v",cache.get("x")); + } + } + } + + static class NeverConnect extends io.lettuce.core.RedisClient { + int connects; + @Override public io.lettuce.core.api.StatefulRedisConnection connect(io.lettuce.core.codec.RedisCodec codec) { + connects++;throw new AssertionError("invalid configuration attempted a connection"); + } + } + @Test void suppliedClientIsNotConnectedBeforeValidation() { + try(var client=new NeverConnect()) { + for(int value:new int[]{-1,0,1,32,63,64}) { + var properties=new TiercacheProperties(true,null,null,new TiercacheProperties.InvalidationProps(true,null,value),null); + var failure=assertThrows(IllegalArgumentException.class, + ()->new TiercacheMicronautConfiguration().tiercacheInvalidationJournal(client,properties)); + check(failure,value);assertEquals(0,client.connects); + } + } + } + static String messages(Throwable cause) { + StringBuilder out=new StringBuilder();while(cause!=null){out.append(cause.getMessage()).append("\n");cause=cause.getCause();}return out.toString(); + } + static void check(Throwable failure,int capacity) { + String text=messages(failure);assertTrue(text.contains("tiercache.invalidation.journal-capacity="+capacity),text); + assertTrue(text.contains("65"),text);assertTrue(text.contains("64-event"),text); + } +} diff --git a/tiercache-micronaut/src/test/java/io/tiercache/micronaut/RedisNamespaceWiringTest.java b/tiercache-micronaut/src/test/java/io/tiercache/micronaut/RedisNamespaceWiringTest.java new file mode 100644 index 0000000..6a36528 --- /dev/null +++ b/tiercache-micronaut/src/test/java/io/tiercache/micronaut/RedisNamespaceWiringTest.java @@ -0,0 +1,103 @@ +package io.tiercache.micronaut; + +import io.lettuce.core.codec.ByteArrayCodec; +import io.lettuce.core.pubsub.StatefulRedisPubSubConnection; +import io.tiercache.TierCacheFactory; +import io.tiercache.TierCache; +import io.tiercache.InvalidationMessage; +import io.tiercache.redis.JdkCacheSerializer; +import io.tiercache.redis.RedisKeyspace; +import io.tiercache.redis.RedisStreamJournal; +import io.tiercache.spi.InvalidationHandler; +import org.junit.jupiter.api.Test; +import org.testcontainers.containers.GenericContainer; +import org.testcontainers.utility.DockerImageName; +import java.time.Duration; +import java.util.concurrent.CountDownLatch; +import java.util.concurrent.TimeUnit; +import static org.junit.jupiter.api.Assertions.*; + +import io.micronaut.context.ApplicationContext; +import java.util.Map; + +class RedisNamespaceWiringTest { + @Test void v2NamespacesAndReplayWorkThroughTwoMicronautContexts() throws Exception { + for (String image : new String[]{"redis:6.2-alpine", "valkey/valkey:8.0-alpine"}) { + try (var server = new GenericContainer<>(DockerImageName.parse(image)).withExposedPorts(6379)) { + server.start(); + String uri = "redis://" + server.getHost() + ":" + server.getMappedPort(6379); + Map config = Map.of("tiercache.enabled", "true", "tiercache.redis-uri", uri); + try (var a = ApplicationContext.run(config); var b = ApplicationContext.run(config)) { + verify(a.getBean(TierCacheFactory.class), b.getBean(TierCacheFactory.class), + a.getBean(RedisStreamJournal.class), a.getBean(io.lettuce.core.RedisClient.class), "micronaut:"); + } + } + } + } + + private static Object field(Object object, String name) throws Exception { + var field = object.getClass().getDeclaredField(name); + field.setAccessible(true); + return field.get(object); + } + + @SuppressWarnings("unchecked") + private static void verify(TierCacheFactory a, TierCacheFactory b, RedisStreamJournal journal, + io.lettuce.core.RedisClient client, String prefix) throws Exception { + TierCache writer = a.getCache("user"); + TierCache reader = b.getCache("user"); + TierCache rolesA = a.getCache("user:roles"); + TierCache rolesB = b.getCache("user:roles"); + InvalidationHandler handler = (InvalidationHandler) field(b, "invalidation"); + Object transport = field(handler, "transport"); + var subscriber = (StatefulRedisPubSubConnection) field(transport, "connection"); + Runnable reconnect = (Runnable) field(transport, "reconnectListener"); + var initial = new CountDownLatch(1); + handler.setEventListener((cache, event) -> { if ("user".equals(cache) && "k".equals(event.key())) initial.countDown(); }); + writer.put("k", "old"); + assertTrue(initial.await(5, TimeUnit.SECONDS)); + assertEquals("old", reader.get("k")); + try (var probe = client.connect(ByteArrayCodec.INSTANCE)) { + var serializer = new JdkCacheSerializer(); + assertNotNull(probe.sync().get(RedisKeyspace.dataKey(prefix + "user", serializer.toBytes("k")))); + assertTrue(journal.size("user") > 0); + assertEquals(0, journal.size(prefix + "user"), "physical prefix must not replace logical journal identity"); + + // Deterministic Pub/Sub gap. Invoke the very callback registered by production wiring. + subscriber.sync().unsubscribe(); + writer.put("k", "new"); + assertEquals("old", reader.get("k")); + subscriber.sync().subscribe(RedisKeyspace.channel("user"), RedisKeyspace.channel("user:roles")); + reconnect.run(); + assertTrue(handler.recoverAsync(Runnable::run).toCompletableFuture().get(5, TimeUnit.SECONDS)); + assertEquals("new", reader.get("k"), "reconnect replay must use the writer's logical journal"); + + subscriber.sync().unsubscribe(); + writer.evict("k"); + writer.put("clear", "cached"); + assertEquals("cached", reader.get("clear")); + rolesA.put("keep", "roles"); + assertEquals("roles", rolesB.get("keep")); + writer.evictAll(); + assertEquals("cached", reader.get("clear"), "the test must have a real missed invalidation"); + assertNotNull(probe.sync().get(RedisKeyspace.dataKey(prefix + "user:roles", serializer.toBytes("keep")))); + assertTrue(journal.readRange("user", "0-0").stream() + .anyMatch(row -> row.message().type() == InvalidationMessage.Type.EVICT_ALL)); + subscriber.sync().subscribe(RedisKeyspace.channel("user"), RedisKeyspace.channel("user:roles")); + reconnect.run(); + assertTrue(handler.recoverAsync(Runnable::run).toCompletableFuture().get(5, TimeUnit.SECONDS)); + assertNull(reader.get("clear")); assertNull(reader.get("k")); + assertEquals("roles", rolesB.get("keep")); + + var tagged = new CountDownLatch(1); + handler.setEventListener((cache, event) -> { if ("tagged".equals(event.key())) tagged.countDown(); }); + writer.put("tagged", "v", "A:B"); + assertTrue(tagged.await(5, TimeUnit.SECONDS)); + assertEquals("v", reader.get("tagged")); + writer.evictByTag("A:B"); + long deadline = System.nanoTime() + Duration.ofSeconds(5).toNanos(); + while (reader.get("tagged") != null && System.nanoTime() < deadline) Thread.sleep(10); + assertNull(reader.get("tagged")); + } + } +} diff --git a/tiercache-micronaut/src/test/java/io/tiercache/micronaut/TiercacheMicronautConfigurationTest.java b/tiercache-micronaut/src/test/java/io/tiercache/micronaut/TiercacheMicronautConfigurationTest.java index 572636e..764f83c 100644 --- a/tiercache-micronaut/src/test/java/io/tiercache/micronaut/TiercacheMicronautConfigurationTest.java +++ b/tiercache-micronaut/src/test/java/io/tiercache/micronaut/TiercacheMicronautConfigurationTest.java @@ -180,8 +180,8 @@ void defaultWiringOverRealRedis() { try (RedisClient probe = RedisClient.create(uri); io.lettuce.core.api.StatefulRedisConnection conn = probe.connect()) { - assertThat(conn.sync().keys("micronaut:demo:*")).isNotEmpty(); - assertThat(conn.sync().keys("micronaut:greetings:*")).isEmpty(); + assertThat(conn.sync().keys(new String(io.tiercache.redis.RedisKeyspace.dataPrefix("micronaut:demo"), java.nio.charset.StandardCharsets.US_ASCII) + "*")).isNotEmpty(); + assertThat(conn.sync().keys(new String(io.tiercache.redis.RedisKeyspace.dataPrefix("micronaut:greetings"), java.nio.charset.StandardCharsets.US_ASCII) + "*")).isEmpty(); } } diff --git a/tiercache-micronaut/src/test/java/io/tiercache/micronaut/TiercacheMicronautInvalidationMetricsTest.java b/tiercache-micronaut/src/test/java/io/tiercache/micronaut/TiercacheMicronautInvalidationMetricsTest.java index 448b29c..87f63be 100644 --- a/tiercache-micronaut/src/test/java/io/tiercache/micronaut/TiercacheMicronautInvalidationMetricsTest.java +++ b/tiercache-micronaut/src/test/java/io/tiercache/micronaut/TiercacheMicronautInvalidationMetricsTest.java @@ -69,6 +69,11 @@ public void onInvalidation(String cache, Direction direction) { counts.computeIfAbsent(direction, d -> new AtomicInteger()).incrementAndGet(); } + final java.util.concurrent.atomic.AtomicLong acknowledged = new java.util.concurrent.atomic.AtomicLong(); + @Override public void onPublication(String cache, io.tiercache.spi.PublicationOutcome outcome, long count) { + if (outcome == io.tiercache.spi.PublicationOutcome.ACKNOWLEDGED) acknowledged.addAndGet(count); + } + int count(Direction direction) { var counter = counts.get(direction); return counter == null ? 0 : counter.get(); @@ -84,18 +89,19 @@ private Map config() { } @Test - void customListenerReceivesInvalidationMetrics() { + void customListenerReceivesInvalidationMetrics() throws Exception { try (ApplicationContext context = ApplicationContext.run(config(), "custom-listener-test")) { RecordingListener listener = context.getBean(RecordingListener.class); context.getBean(TierCacheFactory.class).getCache("m").put("k", "v"); assertThat(listener.count(CacheMetricsListener.Direction.SENT)) .as("the custom listener must count invalidation SENT (pre-fix: NOOP)") .isGreaterThanOrEqualTo(1); + awaitPublication(() -> listener.acknowledged.get() == 1); } } @Test - void autoCreatedListenerCountsInvalidations() { + void autoCreatedListenerCountsInvalidations() throws Exception { try (ApplicationContext context = ApplicationContext.run(config(), "registry-listener-test")) { SimpleMeterRegistry registry = context.getBean(SimpleMeterRegistry.class); context.getBean(TierCacheFactory.class).getCache("m").put("k", "v"); @@ -103,6 +109,11 @@ void autoCreatedListenerCountsInvalidations() { assertThat(counter).as("invalidation SENT must reach the auto-created listener") .isNotNull(); assertThat(counter.count()).isGreaterThanOrEqualTo(1.0); + awaitPublication(() -> { + var acknowledged = registry.find("tiercache.invalidation.publish") + .tags("cache", "m", "outcome", "acknowledged").counter(); + return acknowledged != null && acknowledged.count() == 1; + }); } } @@ -115,4 +126,10 @@ void wiringWorksWithoutAnyListener() { assertThat(context.containsBean(TierCacheFactory.class)).isTrue(); } } + private static void awaitPublication(java.util.function.BooleanSupplier condition) throws Exception { + long until = System.nanoTime() + java.util.concurrent.TimeUnit.SECONDS.toNanos(5); + while (!condition.getAsBoolean() && System.nanoTime() < until) Thread.sleep(5); + org.junit.jupiter.api.Assertions.assertTrue(condition.getAsBoolean()); + } + } diff --git a/tiercache-micronaut/src/test/java/io/tiercache/micronaut/TiercacheMicronautLockProviderShutdownTest.java b/tiercache-micronaut/src/test/java/io/tiercache/micronaut/TiercacheMicronautLockProviderShutdownTest.java index 649caad..f576ebe 100644 --- a/tiercache-micronaut/src/test/java/io/tiercache/micronaut/TiercacheMicronautLockProviderShutdownTest.java +++ b/tiercache-micronaut/src/test/java/io/tiercache/micronaut/TiercacheMicronautLockProviderShutdownTest.java @@ -39,6 +39,10 @@ void contextCloseTerminatesTheCompensationScheduler() throws Exception { assertThat(context.containsBean(LettuceLockProvider.class)).isTrue(); LettuceLockProvider provider = context.getBean(LettuceLockProvider.class); + provider.tryLock("preflight", Duration.ofSeconds(5)).release(); + var field = LettuceLockProvider.class.getDeclaredField("ownedConnection"); + field.setAccessible(true); + var owned = (io.lettuce.core.api.StatefulRedisConnection) field.get(provider); redis.getDockerClient().pauseContainerCmd(redis.getContainerId()).exec(); try { org.junit.jupiter.api.Assertions.assertThrows(Exception.class, @@ -55,6 +59,10 @@ void contextCloseTerminatesTheCompensationScheduler() throws Exception { } context.close(); + assertThat(owned.isOpen()).isFalse(); + provider.close(); + org.junit.jupiter.api.Assertions.assertThrows(io.tiercache.internal.LockProviderClosedException.class, + () -> provider.tryLock("closed", Duration.ofSeconds(5))); long deadline = System.nanoTime() + Duration.ofSeconds(5).toNanos(); while (compensationThreads() > 0 && System.nanoTime() < deadline) { Thread.sleep(20); diff --git a/tiercache-spring-boot-starter/build.gradle.kts b/tiercache-spring-boot-starter/build.gradle.kts index 0a22117..78fb835 100644 --- a/tiercache-spring-boot-starter/build.gradle.kts +++ b/tiercache-spring-boot-starter/build.gradle.kts @@ -47,3 +47,12 @@ mavenPublishing { ) } } + +// A JVM inside the fixture network reaches the exact addresses advertised by Sentinel. +tasks.register("sentinelRuntime") { + dependsOn(tasks.testClasses) + into(layout.buildDirectory.dir("sentinel-runtime")) + from(sourceSets.test.get().output) { into("classes") } + from(sourceSets.main.get().output) { into("classes") } + from(configurations.testRuntimeClasspath) { into("lib") } +} diff --git a/tiercache-spring-boot-starter/src/main/java/io/tiercache/spring/TierCacheManager.java b/tiercache-spring-boot-starter/src/main/java/io/tiercache/spring/TierCacheManager.java index 5ef0711..e77daee 100644 --- a/tiercache-spring-boot-starter/src/main/java/io/tiercache/spring/TierCacheManager.java +++ b/tiercache-spring-boot-starter/src/main/java/io/tiercache/spring/TierCacheManager.java @@ -53,6 +53,10 @@ protected Collection loadCaches() { */ @Override protected Cache getMissingCache(String name) { - return new TierCacheSpringCache(name, factory.getCache(name)); + // Initialize the synchronous cache before async-view acquisition enters + // the factory lifecycle gate; no new monitor wraps lazy transport I/O. + var sync = factory.getCache(name); + var async = factory.asyncCache(name); + return new TierCacheSpringCache(name, sync, async); } } diff --git a/tiercache-spring-boot-starter/src/main/java/io/tiercache/spring/TierCacheSpringCache.java b/tiercache-spring-boot-starter/src/main/java/io/tiercache/spring/TierCacheSpringCache.java index 3d457df..76e6b55 100644 --- a/tiercache-spring-boot-starter/src/main/java/io/tiercache/spring/TierCacheSpringCache.java +++ b/tiercache-spring-boot-starter/src/main/java/io/tiercache/spring/TierCacheSpringCache.java @@ -1,12 +1,15 @@ package io.tiercache.spring; +import io.tiercache.AsyncTierCache; import io.tiercache.LookupResult; import io.tiercache.TierCache; import org.springframework.cache.support.AbstractValueAdaptingCache; import org.springframework.cache.support.NullValue; +import org.springframework.cache.support.SimpleValueWrapper; import java.util.concurrent.Callable; import java.util.concurrent.CompletableFuture; +import java.util.function.Supplier; /** * Spring Cache adapter over a core {@link TierCache}, built on @@ -16,9 +19,10 @@ * *

    Load-bearing mappings: {@link #get(Object, Callable)} delegates to * {@code getOrCompute} — singleflight and cluster-wide rebuild - * coordination apply under annotations by construction. {@link #retrieve(Object)} - * implements the honest multilevel cascade. Null store values map onto the - * null-marker policy: stored as a marker under {@code allow}, + * coordination apply to synchronous {@code @Cacheable(sync = true)} calls. + * Both retrieve overloads use the factory-managed async view. Ordinary + * {@code sync = false} annotations keep Spring's separate read/invoke/write path. + * Null store values map onto the null-marker policy: stored as a marker under {@code allow}, * skipped under {@code deny}. * *

    Internal: not part of the supported public API. @@ -34,17 +38,29 @@ public class TierCacheSpringCache extends AbstractValueAdaptingCache { private final String name; private final TierCache delegate; + private final AsyncTierCache async; /** - * Creates an adapter over the given core cache. + * Creates a synchronous-only adapter. Both retrieve overloads return failed + * futures without a managed async view; use TierCacheManager or the constructor + * accepting both views for asynchronous retrieval. * * @param name the cache name exposed to Spring's cache abstraction * @param delegate the core two-level cache backing this adapter */ public TierCacheSpringCache(String name, TierCache delegate) { + this(name, delegate, null); + } + + /** + * Creates a fully wired adapter with both views of the same factory cache. + * The adapter does not own or close either view or their executor. + */ + public TierCacheSpringCache(String name, TierCache delegate, AsyncTierCache async) { super(true); // we convert nulls ourselves (NullValue <-> null-marker) this.name = name; this.delegate = (TierCache) delegate; + this.async = (AsyncTierCache) async; } /** @@ -118,21 +134,36 @@ public T get(Object key, Callable valueLoader) { } /** - * Implements the multilevel {@code retrieve} contract: an already - * completed future carrying the honest cascade result (L1, then L2 with - * L1 warm-up). - * - * @param key the key to look up - * @return a completed future holding the value wrapper, or a completed - * {@code null} future on a miss + * Asynchronous cascade lookup. A miss completes with null, a value hit + * with a wrapper, and a cached null with a non-null wrapper holding null. + * The factory's bounded async view performs all cache I/O. */ @Override public CompletableFuture retrieve(Object key) { - // Sync core for now; the future completes immediately with the - // honest cascade result (L1 -> L2 with warm-up -> empty). - Object storeValue = lookup(key); - return CompletableFuture.completedFuture( - storeValue != null ? toValueWrapper(storeValue) : null); + if (async == null) return missingAsyncView(); + return async.lookupAsync(key).thenApply(result -> { + if (result instanceof LookupResult.Hit hit) { + return (ValueWrapper) new SimpleValueWrapper(hit.value()); + } + if (result instanceof LookupResult.CachedNull) { + return (ValueWrapper) new SimpleValueWrapper(null); + } + return null; + }).toCompletableFuture(); + } + + /** Loads through the same engine claim as other async and synchronous callers. */ + @Override + public CompletableFuture retrieve(Object key, Supplier> valueLoader) { + if (async == null) return missingAsyncView(); + return async.getOrComputeAsyncStage(key, ignored -> valueLoader.get()) + .thenApply(value -> (T) value).toCompletableFuture(); + } + + private static CompletableFuture missingAsyncView() { + return CompletableFuture.failedFuture(new UnsupportedOperationException( + "Async retrieval requires a factory-managed AsyncTierCache; use TierCacheManager " + + "or the TierCacheSpringCache constructor accepting both cache views")); } /** diff --git a/tiercache-spring-boot-starter/src/main/java/io/tiercache/spring/TiercacheAutoConfiguration.java b/tiercache-spring-boot-starter/src/main/java/io/tiercache/spring/TiercacheAutoConfiguration.java index 0421156..c391559 100644 --- a/tiercache-spring-boot-starter/src/main/java/io/tiercache/spring/TiercacheAutoConfiguration.java +++ b/tiercache-spring-boot-starter/src/main/java/io/tiercache/spring/TiercacheAutoConfiguration.java @@ -1,5 +1,6 @@ package io.tiercache.spring; +import io.tiercache.invalidation.JournalProtocol; import io.lettuce.core.ClientOptions; import io.lettuce.core.RedisClient; import io.lettuce.core.SocketOptions; @@ -63,6 +64,9 @@ public class TiercacheAutoConfiguration { @Bean(destroyMethod = "shutdown") @ConditionalOnMissingBean(RemoteCache.class) RedisClient tiercacheRedisClient(TiercacheProperties properties) { + if (properties.getInvalidation().isEnabled()) { + JournalProtocol.requireCapacity(properties.getInvalidation().getJournalCapacity()); + } if (properties.getRedisUri() == null || properties.getRedisUri().isBlank()) { throw new IllegalStateException( "tiercache.redis-uri is required when tiercache.enabled=true " @@ -89,6 +93,7 @@ RedisClient tiercacheRedisClient(TiercacheProperties properties) { havingValue = "true", matchIfMissing = true) RedisStreamJournal tiercacheInvalidationJournal(RedisClient tiercacheRedisClient, TiercacheProperties properties) { + JournalProtocol.requireCapacity(properties.getInvalidation().getJournalCapacity()); return new RedisStreamJournal(tiercacheRedisClient.connect(ByteArrayCodec.INSTANCE), properties.getInvalidation().getJournalCapacity(), new JdkCacheSerializer<>()); } @@ -135,8 +140,13 @@ TierCacheFactory tierCacheFactory(TiercacheProperties properties, ObjectProvider> invalidation, ObjectProvider metrics, ObjectProvider lockProvider) { - TierCacheFactory.Builder builder = TierCacheFactory.builder() - .defaults(properties.getDefaults().toSettings(io.tiercache.CacheSettings.defaults())); + io.tiercache.CacheSettings defaults; + try { + defaults = properties.getDefaults().toSettings(io.tiercache.CacheSettings.defaults()); + } catch (IllegalArgumentException e) { + throw new io.tiercache.CacheConfigurationException("Cache '': " + e.getMessage()); + } + TierCacheFactory.Builder builder = TierCacheFactory.builder().defaults(defaults); properties.getCaches().forEach((name, props) -> builder.cache(name, props.toOverride())); if (properties.getAsyncExecutorThreads() > 0) { builder.asyncExecutorThreads(properties.getAsyncExecutorThreads()); diff --git a/tiercache-spring-boot-starter/src/main/java/io/tiercache/spring/TiercacheProperties.java b/tiercache-spring-boot-starter/src/main/java/io/tiercache/spring/TiercacheProperties.java index 426b2c1..6ff3ee9 100644 --- a/tiercache-spring-boot-starter/src/main/java/io/tiercache/spring/TiercacheProperties.java +++ b/tiercache-spring-boot-starter/src/main/java/io/tiercache/spring/TiercacheProperties.java @@ -1,5 +1,6 @@ package io.tiercache.spring; +import io.tiercache.invalidation.JournalProtocol; import io.tiercache.CacheOverride; import io.tiercache.CacheSettings; import io.tiercache.NullPolicy; @@ -179,7 +180,7 @@ public static class InvalidationProps { private String profile = "pubsub"; /** Max journal entries kept per cache stream. */ - private int journalCapacity = 10_000; + private int journalCapacity = JournalProtocol.DEFAULT_CAPACITY; /** * Returns whether cross-instance invalidation is active. diff --git a/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/AsyncCacheInterceptionTest.java b/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/AsyncCacheInterceptionTest.java new file mode 100644 index 0000000..5ca58a3 --- /dev/null +++ b/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/AsyncCacheInterceptionTest.java @@ -0,0 +1,93 @@ +package io.tiercache.spring; + +import io.tiercache.*; +import io.tiercache.spi.CacheMetricsListener; +import io.tiercache.testkit.InMemoryRemoteCache; +import org.junit.jupiter.api.Test; +import org.springframework.cache.CacheManager; +import org.springframework.cache.annotation.*; +import org.springframework.context.annotation.*; +import java.util.concurrent.*; +import java.util.concurrent.atomic.*; +import reactor.core.publisher.Mono; +import static org.junit.jupiter.api.Assertions.*; + +class AsyncCacheInterceptionTest { + static class State { + final AtomicInteger calls=new AtomicInteger(); + final CompletableFuture result=new CompletableFuture<>(); + final CountDownLatch entered=new CountDownLatch(1),joined=new CountDownLatch(1); + volatile Thread loaderThread; + } + public static class Service { + final State state; + Service(State state){this.state=state;} + @Cacheable(cacheNames="future-values",sync=true) + public CompletableFuture future(String key){ + state.calls.incrementAndGet();state.loaderThread=Thread.currentThread();state.entered.countDown();return state.result; + } + @Cacheable(cacheNames="mono-values",sync=true) + public Mono mono(String key){ + state.calls.incrementAndGet();state.loaderThread=Thread.currentThread();state.entered.countDown();return Mono.fromFuture(state.result); + } + @Cacheable(cacheNames="sync-values",sync=true) + public String sync(String key){ + state.calls.incrementAndGet();state.entered.countDown();return state.result.join(); + } + } + @Configuration(proxyBeanMethods=false) @EnableCaching + static class Config { + @Bean State state(){return new State();} + @Bean(destroyMethod="close") TierCacheFactory factory(State state){ + return TierCacheFactory.builder().remoteCache(new InMemoryRemoteCache<>()).asyncExecutorThreads(4) + .metricsListener(new CacheMetricsListener(){public void onRequest(String c,Outcome o){if(o==Outcome.COALESCED)state.joined.countDown();}}).build(); + } + @Bean CacheManager cacheManager(TierCacheFactory factory){return new TierCacheManager(factory);} + @Bean Service service(State state){return new Service(state);} + } + @Test void synchronizedFutureAnnotationUsesNonBlockingCoalescedRetrieval() throws Exception { + try(var context=new AnnotationConfigApplicationContext(Config.class)) { + var state=context.getBean(State.class);var service=context.getBean(Service.class);Thread caller=Thread.currentThread(); + try { + var first=service.future("x");AsyncRetrievalTest.await(state.entered); + var second=service.future("x");AsyncRetrievalTest.await(state.joined); + assertFalse(first.isDone());assertNotSame(caller,state.loaderThread);assertEquals(1,state.calls.get()); + state.result.complete("value");assertEquals("value",first.get(2,TimeUnit.SECONDS));assertEquals("value",second.get(2,TimeUnit.SECONDS)); + assertEquals("value",service.future("x").get(2,TimeUnit.SECONDS));assertEquals(1,state.calls.get()); + } finally {state.result.complete("cleanup");} + } + } + @Test void synchronizedMonoAnnotationUsesSameCoalescedRetrieval() throws Exception { + try(var context=new AnnotationConfigApplicationContext(Config.class)) { + var state=context.getBean(State.class);var service=context.getBean(Service.class);Thread caller=Thread.currentThread(); + try { + var first=service.mono("x").toFuture();AsyncRetrievalTest.await(state.entered); + var second=service.mono("x").toFuture();AsyncRetrievalTest.await(state.joined); + assertFalse(first.isDone());assertNotSame(caller,state.loaderThread);assertEquals(1,state.calls.get()); + state.result.complete("value");assertEquals("value",first.get(2,TimeUnit.SECONDS));assertEquals("value",second.get(2,TimeUnit.SECONDS)); + assertEquals("value",service.mono("x").toFuture().get(2,TimeUnit.SECONDS));assertEquals(1,state.calls.get()); + } finally {state.result.complete("cleanup");} + } + } + @Test void synchronousSyncTrueStillCoalescesThroughCallablePath() throws Exception { + var callers=Executors.newFixedThreadPool(2); + try(var context=new AnnotationConfigApplicationContext(Config.class)) { + var state=context.getBean(State.class);var service=context.getBean(Service.class); + try { + var first=callers.submit(()->service.sync("x"));AsyncRetrievalTest.await(state.entered); + var second=callers.submit(()->service.sync("x"));AsyncRetrievalTest.await(state.joined); + state.result.complete("value");assertEquals("value",first.get(2,TimeUnit.SECONDS));assertEquals("value",second.get(2,TimeUnit.SECONDS)); + assertEquals("value",service.sync("x"));assertEquals(1,state.calls.get()); + } finally {state.result.complete("cleanup");} + } finally {callers.shutdownNow();} + } + @Test void annotatedFuturePreservesExceptionalResultCause() { + try(var context=new AnnotationConfigApplicationContext(Config.class)) { + var state=context.getBean(State.class);var service=context.getBean(Service.class); + var result=service.future("failed");AsyncRetrievalTest.await(state.entered); + var error=new IllegalArgumentException("application failure");state.result.completeExceptionally(error); + assertSame(error,AsyncRetrievalTest.failure(result)); + } + } + +} diff --git a/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/AsyncRetrievalTest.java b/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/AsyncRetrievalTest.java new file mode 100644 index 0000000..007d9bc --- /dev/null +++ b/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/AsyncRetrievalTest.java @@ -0,0 +1,154 @@ +package io.tiercache.spring; + +import io.tiercache.*; +import io.tiercache.spi.*; +import io.tiercache.testkit.*; +import org.junit.jupiter.api.Test; +import org.springframework.cache.Cache; +import java.time.Duration; +import java.util.concurrent.*; +import java.util.concurrent.atomic.*; +import static org.junit.jupiter.api.Assertions.*; + +class AsyncRetrievalTest { + static class GatedRemote implements RemoteCache { + final CountingRemoteCache delegate=new CountingRemoteCache<>(); + final CountDownLatch entered=new CountDownLatch(1),release=new CountDownLatch(1); + volatile boolean gated=true; + volatile Thread ioThread; + public StoredEntry get(Object key){ + ioThread=Thread.currentThread(); + if(gated){entered.countDown();await(release);}return delegate.get(key); + } + public void put(Object k,StoredEntry v,Duration ttl){delegate.put(k,v,ttl);} + public boolean setIfAbsent(Object k,StoredEntry v,Duration ttl){return delegate.setIfAbsent(k,v,ttl);} + public void evict(Object k){delegate.evict(k);} + public void clear(){delegate.clear();} + } + static void await(CountDownLatch latch){ + try{assertTrue(latch.await(5,TimeUnit.SECONDS));}catch(InterruptedException e){Thread.currentThread().interrupt();throw new AssertionError(e);} + } + static Cache managed(TierCacheFactory factory){return new TierCacheManager(factory).getCache("c");} + @Test void retrieveReturnsBeforeGatedL2AndWarmsL1() throws Exception { + var remote=new GatedRemote();remote.delegate.put("x",StoredEntry.ofValue("value"),Duration.ofMinutes(1)); + var caller=Executors.newSingleThreadExecutor();var callerThread=new AtomicReference(); + try(var factory=TierCacheFactory.builder().remoteCache(remote).build()) { + var cache=managed(factory); + try { + var invocation=caller.submit(()->{callerThread.set(Thread.currentThread());return cache.retrieve("x");}); + await(remote.entered); + var future=invocation.get(1,TimeUnit.SECONDS); + assertNotNull(future);assertFalse(future.isDone());assertNotSame(callerThread.get(),remote.ioThread); + remote.gated=false;remote.release.countDown(); + assertEquals("value",((Cache.ValueWrapper)future.get(2,TimeUnit.SECONDS)).get()); + int reads=remote.delegate.gets.get(); + assertEquals("value",((Cache.ValueWrapper)cache.retrieve("x").get(2,TimeUnit.SECONDS)).get()); + assertEquals(reads,remote.delegate.gets.get()); + } finally {remote.release.countDown();caller.shutdownNow();} + } + } + static Throwable root(Throwable error) { + while((error instanceof CompletionException || error instanceof ExecutionException) && error.getCause()!=null) error=error.getCause(); + return error; + } + static Throwable failure(CompletableFuture future) { + try {future.get(3,TimeUnit.SECONDS);throw new AssertionError("expected failure");} + catch(ExecutionException | CancellationException e){return root(e);} + catch(Exception e){throw new AssertionError(e);} + } + static TierCacheFactory factory(NullPolicy policy) { + return TierCacheFactory.builder().remoteCache(new InMemoryRemoteCache<>()) + .cache("c",new CacheOverride().nullPolicy(policy)).asyncExecutorThreads(4).build(); + } + @Test void singleArgumentDistinguishesHitCachedNullAndMissAndSuppressesHitSuppliers() throws Exception { + try(var factory=factory(NullPolicy.allow(Duration.ofMinutes(1)))) { + var cache=managed(factory);cache.put("value","v");cache.put("null",null); + assertEquals("v",((Cache.ValueWrapper)cache.retrieve("value").get(2,TimeUnit.SECONDS)).get()); + var marker=cache.retrieve("null").get(2,TimeUnit.SECONDS);assertInstanceOf(Cache.ValueWrapper.class,marker); + assertNull(((Cache.ValueWrapper)marker).get());assertNull(cache.retrieve("absent").get(2,TimeUnit.SECONDS)); + assertEquals("v",cache.retrieve("value",()->{fail("supplier on hit");return null;}).get(2,TimeUnit.SECONDS)); + assertNull(cache.retrieve("null",()->{fail("supplier on cached null");return null;}).get(2,TimeUnit.SECONDS)); + } + } + @Test void supplierNullFollowsAllowAndDenyWithoutExposingSpringWrappers() throws Exception { + for(boolean allow:new boolean[]{false,true})try(var factory=factory(allow?NullPolicy.allow(Duration.ofMinutes(1)):NullPolicy.deny())) { + var cache=managed(factory);var calls=new AtomicInteger(); + java.util.function.Supplier> supplier=()->{calls.incrementAndGet();return CompletableFuture.completedFuture(null);}; + assertNull(cache.retrieve("x",supplier).get(2,TimeUnit.SECONDS));assertNull(cache.retrieve("x",supplier).get(2,TimeUnit.SECONDS)); + assertEquals(allow?1:2,calls.get());var result=cache.retrieve("x").get(2,TimeUnit.SECONDS); + if(allow){assertInstanceOf(Cache.ValueWrapper.class,result);assertNull(((Cache.ValueWrapper)result).get());}else assertNull(result); + } + } + @Test void loaderThrowsExceptionalStageAndCancellationPreserveCauses() { + try(var factory=factory(NullPolicy.deny())) { + var cache=managed(factory);var thrown=new IllegalArgumentException("supplier"); + assertSame(thrown,failure(cache.retrieve("throws",()->{throw thrown;}))); + var failed=new java.io.IOException("stage"); + assertSame(failed,failure(cache.retrieve("failed",()->CompletableFuture.failedFuture(failed)))); + var cancelled=new CompletableFuture();cancelled.cancel(false); + assertInstanceOf(CancellationException.class,failure(cache.retrieve("cancelled",()->cancelled))); + } + } + @Test void cancelledWaiterDoesNotCancelSharedSupplierOrOtherCaller() throws Exception { + var entered=new CountDownLatch(1);var joined=new CountDownLatch(1);var source=new CompletableFuture(); + var calls=new AtomicInteger();var loaderThread=new AtomicReference();Thread caller=Thread.currentThread(); + try(var factory=TierCacheFactory.builder().remoteCache(new InMemoryRemoteCache<>()).asyncExecutorThreads(2) + .metricsListener(new CacheMetricsListener(){public void onRequest(String c,Outcome o){if(o==Outcome.COALESCED)joined.countDown();}}).build()) { + var cache=managed(factory); + try { + var first=cache.retrieve("shared",()->{calls.incrementAndGet();loaderThread.set(Thread.currentThread());entered.countDown();return source;}); + await(entered);var second=cache.retrieve("shared",()->{calls.incrementAndGet();return CompletableFuture.completedFuture("wrong");}); + await(joined);assertTrue(first.cancel(true));assertFalse(source.isCancelled());assertFalse(second.isDone()); + source.complete("value");assertEquals("value",second.get(2,TimeUnit.SECONDS));assertEquals(1,calls.get());assertNotSame(caller,loaderThread.get()); + assertTrue(first.isCancelled()); + } finally {source.complete("cleanup");} + } + } + @Test void closeSettlesRunningAndQueuedLookupsAndRejectsLaterRetrieval() throws Exception { + var remote=new GatedRemote();var factory=TierCacheFactory.builder().remoteCache(remote).asyncExecutorThreads(1).build(); + var manager=new TierCacheManager(factory);var cache=manager.getCache("c"); + try { + cache.put("done","v");var completed=cache.retrieve("done");assertEquals("v",((Cache.ValueWrapper)completed.get(2,TimeUnit.SECONDS)).get()); + var running=cache.retrieve("running");await(remote.entered);var queued=cache.retrieve("queued"); + var supplierCalls=new AtomicInteger();var queuedSupplier=cache.retrieve("queued-loader",()->{supplierCalls.incrementAndGet();return CompletableFuture.completedFuture("bad");}); + factory.close(); + assertInstanceOf(CancellationException.class,failure(running));assertInstanceOf(CancellationException.class,failure(queued)); + assertInstanceOf(CancellationException.class,failure(queuedSupplier));assertEquals(0,supplierCalls.get()); + assertInstanceOf(CancellationException.class,failure(cache.retrieve("later"))); + assertInstanceOf(CancellationException.class,failure(cache.retrieve("later-loader",()->{fail("loader after close");return null;}))); + assertEquals("v",((Cache.ValueWrapper)completed.get()).get()); + assertThrows(IllegalStateException.class,()->manager.getCache("never-created")); + assertFalse(manager.getCacheNames().contains("never-created")); + } finally {remote.release.countDown();factory.close();} + } + @Test void closeSettlesRunningSupplierWithoutCancellingApplicationStage() { + var source=new CompletableFuture();var entered=new CountDownLatch(1); + var factory=factory(NullPolicy.deny());var cache=managed(factory); + try { + var result=cache.retrieve("x",()->{entered.countDown();return source;});await(entered);factory.close(); + assertInstanceOf(CancellationException.class,failure(result));assertFalse(source.isCancelled()); + } finally {source.complete("late-value");factory.close();} + } + @Test void saturationFailsBothOverloadsWithoutCallerRunsOrSupplierExecution() throws Exception { + var source=new CompletableFuture();var entered=new CountDownLatch(1);var calls=new AtomicInteger(); + var factory=TierCacheFactory.builder().remoteCache(new InMemoryRemoteCache<>()).asyncExecutorThreads(1).build(); + try { + var cache=managed(factory);cache.retrieve("busy",()->{entered.countDown();return source;});await(entered); + var field=TierCacheFactory.class.getDeclaredField("asyncExecutor");field.setAccessible(true);var executor=(ThreadPoolExecutor)field.get(factory); + int capacity=executor.getQueue().remainingCapacity();assertTrue(capacity>0); + for(int i=0;i{}); + assertInstanceOf(RejectedExecutionException.class,failure(cache.retrieve("overflow"))); + assertInstanceOf(RejectedExecutionException.class,failure(cache.retrieve("overflow-loader",()->{calls.incrementAndGet();return CompletableFuture.completedFuture("bad");}))); + assertEquals(0,calls.get());assertEquals(1,executor.getPoolSize()); + } finally {factory.close();source.complete("cleanup");} + } + @Test void legacyConstructorRetainsSyncMethodsButRejectsBothAsyncOverloads() { + try(var factory=factory(NullPolicy.deny())) { + var cache=new TierCacheSpringCache("c",factory.getCache("c"));cache.put("x","v");assertEquals("v",cache.get("x").get()); + var error=failure(cache.retrieve("x"));assertInstanceOf(UnsupportedOperationException.class,error);assertTrue(error.getMessage().contains("TierCacheManager")); + assertInstanceOf(UnsupportedOperationException.class,failure(cache.retrieve("x",()->{fail("legacy supplier");return null;}))); + cache.evict("x");assertNull(cache.get("x"));assertEquals("loaded",cache.get("x",()->"loaded")); + } + } + +} diff --git a/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/CustomRedisClientDocumentationTest.java b/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/CustomRedisClientDocumentationTest.java new file mode 100644 index 0000000..71eddca --- /dev/null +++ b/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/CustomRedisClientDocumentationTest.java @@ -0,0 +1,63 @@ +package io.tiercache.spring; + +import io.lettuce.core.*; +import io.tiercache.TierCacheFactory; +import io.tiercache.TierCache; +import io.tiercache.redis.RedisStreamJournal; +import org.junit.jupiter.api.Test; +import org.springframework.beans.factory.annotation.Value; +import org.springframework.boot.autoconfigure.AutoConfigurations; +import org.springframework.boot.test.context.runner.ApplicationContextRunner; +import org.springframework.context.annotation.*; +import org.testcontainers.containers.GenericContainer; +import org.testcontainers.utility.DockerImageName; +import java.time.Duration; +import static org.assertj.core.api.Assertions.assertThat; + +class CustomRedisClientDocumentationTest { + @Configuration(proxyBeanMethods = false) + static class CacheClientConfiguration { + @Bean(destroyMethod = "shutdown") + @Primary + RedisClient applicationRedisClient(@Value("${tiercache.redis-uri}") String uri) { + RedisClient client = RedisClient.create(uri); + client.setOptions(ClientOptions.builder() + .socketOptions(SocketOptions.builder() + .connectTimeout(Duration.ofMillis(150)).build()) + .timeoutOptions(TimeoutOptions.enabled(Duration.ofMillis(350))) + .build()); + return client; + } + } + @Test void primaryClientPreservesJournalAndCrossInstanceInvalidation() { + try (var redis = new GenericContainer<>(DockerImageName.parse("redis:6.2.24-alpine")).withExposedPorts(6379)) { + redis.start(); + var runner = new ApplicationContextRunner() + .withConfiguration(AutoConfigurations.of(TiercacheAutoConfiguration.class)) + .withUserConfiguration(CacheClientConfiguration.class) + .withPropertyValues("tiercache.enabled=true", + "tiercache.redis-uri=redis://"+redis.getHost()+":"+redis.getMappedPort(6379)); + runner.run(a -> runner.run(b -> { + assertThat(a).hasNotFailed(); assertThat(b).hasNotFailed(); + assertThat(a.getBeansOfType(RedisClient.class)).hasSize(2); + assertThat(a.getBean(RedisClient.class)).isSameAs(a.getBean("applicationRedisClient")); + assertThat(a.getBean(RedisClient.class).getOptions().getSocketOptions().getConnectTimeout()) + .isEqualTo(Duration.ofMillis(150)); + var timeout = a.getBean(RedisClient.class).getOptions().getTimeoutOptions().getSource(); + assertThat(timeout.getTimeUnit().toMillis(timeout.getTimeout(null))).isEqualTo(350); + TierCache left=a.getBean(TierCacheFactory.class).getCache("client-doc"); + TierCache right=b.getBean(TierCacheFactory.class).getCache("client-doc"); + left.put("key","old"); assertThat(right.get("key")).isEqualTo("old"); + left.put("key","new"); + long deadline=System.nanoTime()+Duration.ofSeconds(5).toNanos(); + while (!"new".equals(right.get("key")) && System.nanoTime(){assertThat(context).hasNotFailed(); + assertThat(actual(context.getBean(TierCacheFactory.class),"orders").degradationStaleTtl()).isEqualTo(Duration.ZERO);}); + } + @Test void inheritanceOverrideAndExplicitZeroReachActualEngine() { + runner.withPropertyValues("tiercache.defaults.degradation-stale-ttl=10m", + "tiercache.caches.orders.degradation-stale-ttl=20m", + "tiercache.caches.off.degradation-stale-ttl=0s").run(context->{ + assertThat(context).hasNotFailed();var factory=context.getBean(TierCacheFactory.class); + assertThat(actual(factory,"inherit").degradationStaleTtl()).isEqualTo(Duration.ofMinutes(10)); + assertThat(actual(factory,"orders").degradationStaleTtl()).isEqualTo(Duration.ofMinutes(20)); + assertThat(actual(factory,"off").degradationStaleTtl()).isEqualTo(Duration.ZERO); + assertThat(actual(factory,"off").l2Ttl()).isEqualTo(actual(factory,"inherit").l2Ttl()); + }); + } + @Test void negativeGlobalAndNamedWindowsFailWithContext() { + for(String name:new String[]{"defaults","caches.orders"}) { + runner.withPropertyValues("tiercache."+name+".degradation-stale-ttl=-1s").run(context->{ + assertThat(context).hasFailed();var error=context.getStartupFailure(); + StringBuilder messages=new StringBuilder();while(error!=null){messages.append(error.getMessage());error=error.getCause();} + assertThat(messages.toString()).contains("degradationStaleTtl",name.equals("defaults")?"global defaults":"orders"); + }); + } + } +} diff --git a/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/InvalidationAutoConfigurationTest.java b/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/InvalidationAutoConfigurationTest.java index 4ae6193..d222df6 100644 --- a/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/InvalidationAutoConfigurationTest.java +++ b/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/InvalidationAutoConfigurationTest.java @@ -79,6 +79,11 @@ public void onInvalidation(String cache, Direction direction) { .incrementAndGet(); } + final java.util.concurrent.atomic.AtomicLong acknowledged = new java.util.concurrent.atomic.AtomicLong(); + @Override public void onPublication(String cache, io.tiercache.spi.PublicationOutcome outcome, long count) { + if (outcome == io.tiercache.spi.PublicationOutcome.ACKNOWLEDGED) acknowledged.addAndGet(count); + } + int count(Direction direction) { var counter = counts.get(direction); return counter == null ? 0 : counter.get(); @@ -129,6 +134,7 @@ void invalidationEventsReachTheCustomMetricsListener() throws Exception { awaitTrue(() -> listenerA.count( io.tiercache.spi.CacheMetricsListener.Direction.SENT) >= 1, "A's listener must count SENT"); + awaitTrue(() -> listenerA.acknowledged.get() == 1, "A's listener must count actual publication acknowledgement"); awaitTrue(() -> listenerB.count( io.tiercache.spi.CacheMetricsListener.Direction.RECEIVED) >= 1, "B's listener must count RECEIVED"); @@ -162,6 +168,11 @@ void autoConfiguredMicrometerListenerCountsInvalidations() { assertThat(counter).as("invalidation SENT must reach the auto-configured " + "Micrometer listener").isNotNull(); assertThat(counter.count()).isGreaterThanOrEqualTo(1.0); + awaitTrue(() -> { + var acknowledged = registry.find("tiercache.invalidation.publish") + .tags("cache", "m", "outcome", "acknowledged").counter(); + return acknowledged != null && acknowledged.count() == 1; + }, "publication completion must reach the configured registry"); }); } } diff --git a/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/JournalCapacityConfigurationTest.java b/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/JournalCapacityConfigurationTest.java new file mode 100644 index 0000000..3272f7f --- /dev/null +++ b/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/JournalCapacityConfigurationTest.java @@ -0,0 +1,61 @@ +package io.tiercache.spring; +import io.tiercache.TierCacheFactory; +import io.tiercache.redis.RedisStreamJournal; +import org.junit.jupiter.api.Test; +import org.springframework.boot.autoconfigure.AutoConfigurations; +import org.springframework.boot.test.context.runner.ApplicationContextRunner; +import org.testcontainers.containers.GenericContainer; +import org.testcontainers.utility.DockerImageName; +import static org.junit.jupiter.api.Assertions.*; + +class JournalCapacityConfigurationTest { + final ApplicationContextRunner runner=new ApplicationContextRunner().withConfiguration(AutoConfigurations.of(TiercacheAutoConfiguration.class)) + .withPropertyValues("tiercache.enabled=true"); + @Test void invalidEnabledCapacitiesFailBeforeConnectionAndTraffic() { + for(int value:new int[]{-1,0,1,32,63,64}) runner.withPropertyValues("tiercache.redis-uri=redis://127.0.0.1:1", + "tiercache.invalidation.journal-capacity="+value).run(context->{ + assertNotNull(context.getStartupFailure());check(context.getStartupFailure(),value); + assertThrows(IllegalStateException.class,()->context.getBean(TierCacheFactory.class)); + }); + } + @Test void boundaryDefaultAndDisabledSettingsStartAndServeTraffic() { + try(var redis=new GenericContainer<>(DockerImageName.parse("redis:6.2-alpine")).withExposedPorts(6379)) { + redis.start();String uri="redis://"+redis.getHost()+":"+redis.getMappedPort(6379); + for(String value:new String[]{"65","10000","omitted"}) { + var configured=runner.withPropertyValues("tiercache.redis-uri="+uri); + if(!value.equals("omitted"))configured=configured.withPropertyValues("tiercache.invalidation.journal-capacity="+value); + configured.run(context->{assertNull(context.getStartupFailure()); + assertEquals(value.equals("65")?65:10000,context.getBean(RedisStreamJournal.class).capacity()); + var cache=context.getBean(TierCacheFactory.class).getCache("c");cache.put("x","v");assertEquals("v",cache.get("x")); + }); + } + runner.withPropertyValues("tiercache.redis-uri="+uri,"tiercache.invalidation.enabled=false","tiercache.invalidation.journal-capacity=1") + .run(context->{assertNull(context.getStartupFailure());assertTrue(context.getBeansOfType(RedisStreamJournal.class).isEmpty()); + var cache=context.getBean(TierCacheFactory.class).getCache("c");cache.put("x","v");assertEquals("v",cache.get("x"));}); + } + } + + static class NeverConnect extends io.lettuce.core.RedisClient { + int connects; + @Override public io.lettuce.core.api.StatefulRedisConnection connect(io.lettuce.core.codec.RedisCodec codec) { + connects++;throw new AssertionError("invalid configuration attempted a connection"); + } + } + @Test void suppliedClientIsNotConnectedBeforeValidation() { + try(var client=new NeverConnect()) { + for(int value:new int[]{-1,0,1,32,63,64}) { + var properties=new TiercacheProperties(); properties.getInvalidation().setJournalCapacity(value); + var failure=assertThrows(IllegalArgumentException.class, + ()->new TiercacheAutoConfiguration().tiercacheInvalidationJournal(client,properties)); + check(failure,value);assertEquals(0,client.connects); + } + } + } + static String messages(Throwable cause) { + StringBuilder out=new StringBuilder();while(cause!=null){out.append(cause.getMessage()).append("\n");cause=cause.getCause();}return out.toString(); + } + static void check(Throwable failure,int capacity) { + String text=messages(failure);assertTrue(text.contains("tiercache.invalidation.journal-capacity="+capacity),text); + assertTrue(text.contains("65"),text);assertTrue(text.contains("64-event"),text); + } +} diff --git a/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/LockProviderShutdownTest.java b/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/LockProviderShutdownTest.java index dbe86ef..018de2d 100644 --- a/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/LockProviderShutdownTest.java +++ b/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/LockProviderShutdownTest.java @@ -34,6 +34,8 @@ void contextCloseTerminatesTheCompensationScheduler() throws Exception { DockerImageName.parse("redis:6.2-alpine")).withExposedPorts(6379)) { redis.start(); String uri = "redis://" + redis.getHost() + ":" + redis.getMappedPort(6379); + var owned = new java.util.concurrent.atomic.AtomicReference>(); + var managed = new java.util.concurrent.atomic.AtomicReference(); runner.withPropertyValues("tiercache.enabled=true", "tiercache.redis-uri=" + uri) .run(context -> { assertThat(context).hasNotFailed(); @@ -41,6 +43,11 @@ void contextCloseTerminatesTheCompensationScheduler() throws Exception { LettuceLockProvider provider = context.getBean(LettuceLockProvider.class); // Force an ambiguous acquire so the scheduler spins up. + provider.tryLock("preflight", Duration.ofSeconds(5)).release(); + var field = LettuceLockProvider.class.getDeclaredField("ownedConnection"); + field.setAccessible(true); + owned.set((io.lettuce.core.api.StatefulRedisConnection) field.get(provider)); + managed.set(provider); redis.getDockerClient().pauseContainerCmd(redis.getContainerId()).exec(); try { org.junit.jupiter.api.Assertions.assertThrows(Exception.class, @@ -57,6 +64,10 @@ void contextCloseTerminatesTheCompensationScheduler() throws Exception { .unpauseContainerCmd(redis.getContainerId()).exec(); } }); + assertThat(owned.get().isOpen()).isFalse(); + managed.get().close(); // Repeated framework/provider cleanup is harmless. + org.junit.jupiter.api.Assertions.assertThrows(io.tiercache.internal.LockProviderClosedException.class, + () -> managed.get().tryLock("closed", Duration.ofSeconds(5))); // After the context is closed, the scheduler must be gone. long deadline = System.nanoTime() + Duration.ofSeconds(5).toNanos(); while (compensationThreads() > 0 && System.nanoTime() < deadline) { diff --git a/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/RedisNamespaceWiringTest.java b/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/RedisNamespaceWiringTest.java new file mode 100644 index 0000000..8efcefb --- /dev/null +++ b/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/RedisNamespaceWiringTest.java @@ -0,0 +1,105 @@ +package io.tiercache.spring; + +import io.lettuce.core.codec.ByteArrayCodec; +import io.lettuce.core.pubsub.StatefulRedisPubSubConnection; +import io.tiercache.TierCacheFactory; +import io.tiercache.TierCache; +import io.tiercache.InvalidationMessage; +import io.tiercache.redis.JdkCacheSerializer; +import io.tiercache.redis.RedisKeyspace; +import io.tiercache.redis.RedisStreamJournal; +import io.tiercache.spi.InvalidationHandler; +import org.junit.jupiter.api.Test; +import org.testcontainers.containers.GenericContainer; +import org.testcontainers.utility.DockerImageName; +import java.time.Duration; +import java.util.concurrent.CountDownLatch; +import java.util.concurrent.TimeUnit; +import static org.junit.jupiter.api.Assertions.*; + +import org.springframework.boot.autoconfigure.AutoConfigurations; +import org.springframework.boot.test.context.runner.ApplicationContextRunner; + +class RedisNamespaceWiringTest { + @Test void v2NamespacesAndReplayWorkThroughTwoSpringContexts() { + for (String image : new String[]{"redis:6.2-alpine", "valkey/valkey:8.0-alpine"}) { + try (var server = new GenericContainer<>(DockerImageName.parse(image)).withExposedPorts(6379)) { + server.start(); + String uri = "redis://" + server.getHost() + ":" + server.getMappedPort(6379); + var runner = new ApplicationContextRunner().withConfiguration(AutoConfigurations.of(TiercacheAutoConfiguration.class)) + .withPropertyValues("tiercache.enabled=true", "tiercache.redis-uri=" + uri); + runner.run(a -> runner.run(b -> { + assertNull(a.getStartupFailure()); assertNull(b.getStartupFailure()); + verify(a.getBean(TierCacheFactory.class), b.getBean(TierCacheFactory.class), + a.getBean(RedisStreamJournal.class), a.getBean(io.lettuce.core.RedisClient.class), "spring:"); + })); + } + } + } + + private static Object field(Object object, String name) throws Exception { + var field = object.getClass().getDeclaredField(name); + field.setAccessible(true); + return field.get(object); + } + + @SuppressWarnings("unchecked") + private static void verify(TierCacheFactory a, TierCacheFactory b, RedisStreamJournal journal, + io.lettuce.core.RedisClient client, String prefix) throws Exception { + TierCache writer = a.getCache("user"); + TierCache reader = b.getCache("user"); + TierCache rolesA = a.getCache("user:roles"); + TierCache rolesB = b.getCache("user:roles"); + InvalidationHandler handler = (InvalidationHandler) field(b, "invalidation"); + Object transport = field(handler, "transport"); + var subscriber = (StatefulRedisPubSubConnection) field(transport, "connection"); + Runnable reconnect = (Runnable) field(transport, "reconnectListener"); + var initial = new CountDownLatch(1); + handler.setEventListener((cache, event) -> { if ("user".equals(cache) && "k".equals(event.key())) initial.countDown(); }); + writer.put("k", "old"); + assertTrue(initial.await(5, TimeUnit.SECONDS)); + assertEquals("old", reader.get("k")); + try (var probe = client.connect(ByteArrayCodec.INSTANCE)) { + var serializer = new JdkCacheSerializer(); + assertNotNull(probe.sync().get(RedisKeyspace.dataKey(prefix + "user", serializer.toBytes("k")))); + assertTrue(journal.size("user") > 0); + assertEquals(0, journal.size(prefix + "user"), "physical prefix must not replace logical journal identity"); + + // Deterministic Pub/Sub gap. Invoke the very callback registered by production wiring. + subscriber.sync().unsubscribe(); + writer.put("k", "new"); + assertEquals("old", reader.get("k")); + subscriber.sync().subscribe(RedisKeyspace.channel("user"), RedisKeyspace.channel("user:roles")); + reconnect.run(); + assertTrue(handler.recoverAsync(Runnable::run).toCompletableFuture().get(5, TimeUnit.SECONDS)); + assertEquals("new", reader.get("k"), "reconnect replay must use the writer's logical journal"); + + subscriber.sync().unsubscribe(); + writer.evict("k"); + writer.put("clear", "cached"); + assertEquals("cached", reader.get("clear")); + rolesA.put("keep", "roles"); + assertEquals("roles", rolesB.get("keep")); + writer.evictAll(); + assertEquals("cached", reader.get("clear"), "the test must have a real missed invalidation"); + assertNotNull(probe.sync().get(RedisKeyspace.dataKey(prefix + "user:roles", serializer.toBytes("keep")))); + assertTrue(journal.readRange("user", "0-0").stream() + .anyMatch(row -> row.message().type() == InvalidationMessage.Type.EVICT_ALL)); + subscriber.sync().subscribe(RedisKeyspace.channel("user"), RedisKeyspace.channel("user:roles")); + reconnect.run(); + assertTrue(handler.recoverAsync(Runnable::run).toCompletableFuture().get(5, TimeUnit.SECONDS)); + assertNull(reader.get("clear")); assertNull(reader.get("k")); + assertEquals("roles", rolesB.get("keep")); + + var tagged = new CountDownLatch(1); + handler.setEventListener((cache, event) -> { if ("tagged".equals(event.key())) tagged.countDown(); }); + writer.put("tagged", "v", "A:B"); + assertTrue(tagged.await(5, TimeUnit.SECONDS)); + assertEquals("v", reader.get("tagged")); + writer.evictByTag("A:B"); + long deadline = System.nanoTime() + Duration.ofSeconds(5).toNanos(); + while (reader.get("tagged") != null && System.nanoTime() < deadline) Thread.sleep(10); + assertNull(reader.get("tagged")); + } + } +} diff --git a/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/SentinelProbe.java b/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/SentinelProbe.java new file mode 100644 index 0000000..99ed32c --- /dev/null +++ b/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/SentinelProbe.java @@ -0,0 +1,156 @@ +package io.tiercache.spring; + +import io.lettuce.core.RedisClient; +import io.lettuce.core.RedisURI; +import io.tiercache.*; +import io.tiercache.redis.*; +import io.tiercache.spi.*; +import org.springframework.context.annotation.AnnotationConfigApplicationContext; +import org.springframework.core.env.MapPropertySource; +import java.nio.file.*; +import java.time.Duration; +import java.util.*; +import java.util.concurrent.*; +import java.util.concurrent.atomic.AtomicInteger; +import static org.junit.jupiter.api.Assertions.*; + +/** Invoked in the isolated Docker network by compatibility/run-sentinel.py. */ +public final class SentinelProbe { + static final Path CONTROL = Path.of("/control"); + static final class Metrics implements CacheMetricsListener { + final AtomicInteger replayed = new AtomicInteger(), l1Hits = new AtomicInteger(); + public void onRequest(String cache, Outcome outcome) { if (outcome == Outcome.L1_HIT) l1Hits.incrementAndGet(); } + public void onInvalidation(String cache, Direction direction) { + if (direction == Direction.REPLAYED) replayed.incrementAndGet(); + } + } + static AnnotationConfigApplicationContext application(String id, Metrics metrics) { + var uri = RedisURI.Builder.sentinel("sentinel0",26379,"mymaster") + .withSentinel("sentinel1",26379).withSentinel("sentinel2",26379).build(); + var context = new AnnotationConfigApplicationContext(); + context.getEnvironment().getPropertySources().addFirst(new MapPropertySource("fixture", Map.of( + "tiercache.enabled", "true", "tiercache.redis-uri", uri.toURI().toString(), + "tiercache.defaults.l1-expire-after-write", "10m", "tiercache.defaults.l2-ttl", "20m"))); + context.registerBean("fixtureMetrics", CacheMetricsListener.class, () -> metrics); + context.register(TiercacheAutoConfiguration.class); + context.refresh(); + System.out.println("APPLICATION_READY " + id); + return context; + } + static void signal(String phase) throws Exception { + Files.writeString(CONTROL.resolve(phase), "ready"); + System.out.println("PHASE " + phase); + } + static void await(String phase) throws Exception { + long deadline = System.nanoTime()+TimeUnit.SECONDS.toNanos(90); + while (!Files.exists(CONTROL.resolve(phase))) { + if (System.nanoTime()>deadline) throw new AssertionError("Timeout waiting for " + phase); + Thread.sleep(50); + } + } + public static void main(String[] args) throws Exception { + String mode = args[0]; + System.out.println("RUNTIME jdk="+System.getProperty("java.runtime.version") + +" lettuce="+RedisClient.class.getPackage().getImplementationVersion() + +" boot="+org.springframework.boot.SpringBootVersion.getVersion()); + Metrics ma = new Metrics(), mb = new Metrics(); + try (var a = application("a", ma); var b = application("b", mb)) { + var fa = a.getBean(TierCacheFactory.class); var fb = b.getBean(TierCacheFactory.class); + TierCache ca = fa.getCache("sentinel"), cb = fb.getCache("sentinel"); + ca.put("changed", "v1"); ca.put("unaffected", "stable"); + assertEquals("v1", cb.get("changed")); assertEquals("stable", cb.get("unaffected")); + if (!mode.equals("outage")) { + // Persist a versioned update and journal row without publishing it: deterministic + // missed delivery, recovered through the actual starter's reconnect callback. + try (var writer = LettuceRemoteCache.builder("redis://unused") + .client(a.getBean(RedisClient.class)).cacheName("spring:sentinel") + .journalName("sentinel").journal(a.getBean(RedisStreamJournal.class)).build()) { + writer.put("changed", StoredEntry.ofValue("v2", new Version( + System.currentTimeMillis()*1000+1_000_000, UUID.randomUUID())), Duration.ofMinutes(20)); + } + assertEquals("v1", cb.get("changed"), "L1 deliberately missed this update"); + } + var journal = a.getBean(RedisStreamJournal.class); + long changedRowsBefore = journal.readRange("sentinel", "0-0").stream() + .filter(row -> "changed".equals(row.message().key())).count(); + var source = new java.util.concurrent.atomic.AtomicReference<>("v1"); + signal("ready"); + var traffic = Executors.newSingleThreadExecutor(); + try { + var work = traffic.submit(() -> { + int count=0; long max=0; + while (!Files.exists(CONTROL.resolve("topology-ready"))) { + long start=System.nanoTime(); + String value=ca.getOrCompute("traffic-"+count, k -> "source-v1"); + assertEquals("source-v1", value); + max=Math.max(max,System.nanoTime()-start); count++; + assertTrue(max "probe"); + cb.getOrCompute("recover-b-"+System.nanoTime(), k -> "probe"); + if (fa.breakerState() == BreakerState.CLOSED && fb.breakerState() == BreakerState.CLOSED) break; + Thread.sleep(100); + } while (System.nanoTime() "changed".equals(row.message().key())).count(), "offline eviction was not journaled"); + String observed=cb.get("changed"); + System.out.println("RECOVERED sourceVersion="+source.get()+" observedByB="+observed + +" changedJournalRows="+changedRowsBefore); + assertNotEquals("v2", observed, "unstored source update cannot be reconstructed from the journal"); + } else { + long deadline=System.nanoTime()+TimeUnit.SECONDS.toNanos(45); + String observed; + do { + observed=cb.get("changed"); + // Both breakers must complete recovery before asserting a new distributed write. + ca.getOrCompute("recovery-probe-"+System.nanoTime(), k -> "probe"); + if ("v2".equals(observed) && mb.replayed.get()>0 && fb.breakerState() == BreakerState.CLOSED && fa.breakerState() == BreakerState.CLOSED) break; + Thread.sleep(100); + } while(System.nanoTime()0,"retained history must actually replay"); + int hits = mb.l1Hits.get(); + assertEquals("stable",cb.get("unaffected")); + assertEquals(hits+1, mb.l1Hits.get(), "verified-history recovery preserves unaffected L1 entries"); + assertEquals(BreakerState.CLOSED, fb.breakerState()); assertEquals(BreakerState.CLOSED, fa.breakerState()); + // New writes, locks (getOrCompute), journal and cross-instance delivery after promotion. + ca.put("after", "v2"); + assertEquals("v2", cb.get("after")); + ca.put("after", "v3"); + deadline=System.nanoTime()+TimeUnit.SECONDS.toNanos(10); + while(!"v3".equals(cb.get("after")) && System.nanoTime()0); + assertEquals("computed",ca.getOrCompute("rebuild", k -> "computed")); + var lockA = a.getBean(LettuceLockProvider.class).tryLock("sentinel-proof", Duration.ofSeconds(10)); + assertNotNull(lockA, "first starter lock connection recovered"); + try { + var lockB = b.getBean(LettuceLockProvider.class).tryLock("sentinel-proof", Duration.ofSeconds(10)); + try { assertNull(lockB, "both lock connections must coordinate on the elected primary"); } + finally { if (lockB != null) lockB.release(); } + } finally { lockA.release(); } + System.out.println("RECOVERED observedVersion="+observed+" replayed="+mb.replayed.get()); + } + signal("passed"); + } + } +} diff --git a/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/TierCacheSpringCacheTest.java b/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/TierCacheSpringCacheTest.java index 5c5a337..942be67 100644 --- a/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/TierCacheSpringCacheTest.java +++ b/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/TierCacheSpringCacheTest.java @@ -86,28 +86,30 @@ void concurrentValueLoadersCoalesce() throws Exception { @Test void retrieveIsMultilevel() throws Exception { CountingRemoteCache l2 = new CountingRemoteCache<>(); - TierCacheFactory factory = newFactory(l2, NullPolicy.deny()); - // Populate L2 (and L1 of the "writer" cache). - TierCacheSpringCache writer = new TierCacheSpringCache("writer", factory.getCache("writer")); - writer.put("k", "v"); - // The factory memoizes one core cache per name, so a different name - // yields a genuinely cold L1 over the same L2 (reusing "writer" would - // serve the read from its already-warm L1 and prove nothing). - TierCacheSpringCache reader = new TierCacheSpringCache("reader", factory.getCache("reader")); - - int l2GetsBefore = l2.gets.get(); - Cache.ValueWrapper wrapper = reader.retrieve("k").get(); - assertThat(wrapper.get()).isEqualTo("v"); - assertThat(l2.gets.get()).isGreaterThan(l2GetsBefore) - .as("cold L1 must be served from L2"); - - // The L2 hit must have warmed L1: a second retrieve stays on L1. - l2GetsBefore = l2.gets.get(); - assertThat(reader.retrieve("k").get().get()).isEqualTo("v"); - assertThat(l2.gets.get()).isEqualTo(l2GetsBefore) - .as("L2 hit must warm L1; second retrieve must not touch L2"); - - assertThat(reader.retrieve("absent").get()).isNull(); + try (TierCacheFactory factory = newFactory(l2, NullPolicy.deny())) { + // Populate L2 (and L1 of the "writer" cache). + TierCacheSpringCache writer = new TierCacheSpringCache("writer", factory.getCache("writer")); + writer.put("k", "v"); + // The factory memoizes one core cache per name, so a different name + // yields a genuinely cold L1 over the same L2 (reusing "writer" would + // serve the read from its already-warm L1 and prove nothing). + TierCacheSpringCache reader = new TierCacheSpringCache("reader", factory.getCache("reader"), + factory.asyncCache("reader")); + + int l2GetsBefore = l2.gets.get(); + Cache.ValueWrapper wrapper = reader.retrieve("k").get(); + assertThat(wrapper.get()).isEqualTo("v"); + assertThat(l2.gets.get()).isGreaterThan(l2GetsBefore) + .as("cold L1 must be served from L2"); + + // The L2 hit must have warmed L1: a second retrieve stays on L1. + l2GetsBefore = l2.gets.get(); + assertThat(reader.retrieve("k").get().get()).isEqualTo("v"); + assertThat(l2.gets.get()).isEqualTo(l2GetsBefore) + .as("L2 hit must warm L1; second retrieve must not touch L2"); + + assertThat(reader.retrieve("absent").get()).isNull(); + } } @Test diff --git a/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/TiercacheAutoConfigurationTest.java b/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/TiercacheAutoConfigurationTest.java index f5629b4..7f1731d 100644 --- a/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/TiercacheAutoConfigurationTest.java +++ b/tiercache-spring-boot-starter/src/test/java/io/tiercache/spring/TiercacheAutoConfigurationTest.java @@ -240,8 +240,8 @@ void defaultWiringIsolatesPerCacheL2Namespaces() { try (RedisClient probe = RedisClient.create(uri); io.lettuce.core.api.StatefulRedisConnection conn = probe.connect()) { - assertThat(conn.sync().keys("spring:demo:*")).isNotEmpty(); - assertThat(conn.sync().keys("spring:greetings:*")).isEmpty(); + assertThat(conn.sync().keys(new String(io.tiercache.redis.RedisKeyspace.dataPrefix("spring:demo"), java.nio.charset.StandardCharsets.US_ASCII) + "*")).isNotEmpty(); + assertThat(conn.sync().keys(new String(io.tiercache.redis.RedisKeyspace.dataPrefix("spring:greetings"), java.nio.charset.StandardCharsets.US_ASCII) + "*")).isEmpty(); } }); } diff --git a/tiercache-tck/build.gradle.kts b/tiercache-tck/build.gradle.kts index 6b92776..1f0e100 100644 --- a/tiercache-tck/build.gradle.kts +++ b/tiercache-tck/build.gradle.kts @@ -42,6 +42,9 @@ dependencies { implementation(libs.testcontainers.junit.jupiter) "vtStressImplementation"(project(":tiercache-core")) + "vtStressImplementation"(project(":tiercache-invalidation")) + "vtStressImplementation"(project(":tiercache-transport-redis")) + "vtStressImplementation"(libs.testcontainers) "vtStressImplementation"(testFixtures(project(":tiercache-core"))) "vtStressImplementation"(platform(libs.junit.bom)) "vtStressImplementation"(libs.junit.jupiter) @@ -66,11 +69,12 @@ configurations { // JDK 21+ toolchain for the virtual-thread stress gate. Resolution is lazy: // when no 21+ JDK is installed the compile/test tasks below are skipped with // a loud log line instead of failing (or downloading a JDK). +val vtJavaVersion = providers.gradleProperty("tiercacheVtJdk").map { it.toInt() }.orElse(21) val vtCompiler = javaToolchains.compilerFor { - languageVersion = JavaLanguageVersion.of(21) + languageVersion = JavaLanguageVersion.of(vtJavaVersion.get()) } val vtLauncher = javaToolchains.launcherFor { - languageVersion = JavaLanguageVersion.of(21) + languageVersion = JavaLanguageVersion.of(vtJavaVersion.get()) } fun vtToolchainAvailable(): Boolean = try { @@ -112,11 +116,13 @@ tasks.register("vtStressTest") { testClassesDirs = vtStress.output.classesDirs classpath = vtStress.runtimeClasspath javaLauncher = vtLauncher + systemProperty("tiercache.recovery.jfr", layout.buildDirectory.file( + "reports/recovery-jdk${vtJavaVersion.get()}.jfr").get().asFile.absolutePath) skipUnlessVtToolchain() } tasks.register("soakTest") { - description = "Soak gate: sustained churn against a real L2 container; memory and journal growth gates." + description = "Strict soak: separate post-GC heap/RSS <=5% growth, observed workers, bounded journal and JSON evidence." group = "verification" testClassesDirs = sourceSets.test.get().output.classesDirs classpath = sourceSets.test.get().runtimeClasspath @@ -125,6 +131,10 @@ tasks.register("soakTest") { } systemProperty("tiercache.soak.duration", providers.systemProperty("tiercache.soak.duration").orElse("PT10M").get()) + systemProperty("tiercache.soak.report", providers.systemProperty("tiercache.soak.report") + .orElse(layout.buildDirectory.file("reports/soak/report.json").map { it.asFile.absolutePath }).get()) + outputs.upToDateWhen { false } // An explicit release-gate invocation must run the workload again. + outputs.cacheIf { false } // Runtime evidence cannot be reused from a build cache. } // Compliance-suite artifact (TCK publication design D5): the TCK's value is diff --git a/tiercache-tck/src/test/java/io/tiercache/tck/DegradationLifetimeTest.java b/tiercache-tck/src/test/java/io/tiercache/tck/DegradationLifetimeTest.java new file mode 100644 index 0000000..4156618 --- /dev/null +++ b/tiercache-tck/src/test/java/io/tiercache/tck/DegradationLifetimeTest.java @@ -0,0 +1,65 @@ +package io.tiercache.tck; + +import io.tiercache.*; +import io.tiercache.internal.*; +import io.tiercache.redis.*; +import io.tiercache.spi.*; +import org.junit.jupiter.api.Test; +import org.testcontainers.containers.GenericContainer; +import org.testcontainers.utility.DockerImageName; +import java.time.Duration; +import java.util.*; +import java.util.concurrent.*; +import java.util.concurrent.atomic.*; +import static org.junit.jupiter.api.Assertions.*; + +class DegradationLifetimeTest { + @Test void realRedisOutageKeepsFreshnessAndBoundsCoalescedStaleFallback() throws Exception { + var now=new AtomicLong(1);var stale=new AtomicInteger();var loads=new AtomicInteger(); + var joined=new CountDownLatch(7);var releaseLoader=new CountDownLatch(1); + var settings=new CacheSettings(100,Duration.ofMinutes(30),null,Duration.ofHours(2),0, + NullPolicy.deny(),InvalidationMode.INVALIDATE,65536,Duration.ZERO,false, + Duration.ofSeconds(1),Duration.ofMinutes(30)); + try(var redis=new GenericContainer<>(DockerImageName.parse("redis:6.2-alpine")).withExposedPorts(6379)) { + redis.start();String uri="redis://"+redis.getHost()+":"+redis.getMappedPort(6379); + try(var remote=LettuceRemoteCache.builder(uri).commandTimeout(Duration.ofSeconds(1)).build()) { + var breaker=new CircuitBreaker(new CircuitBreaker.Config(1,1,1,Duration.ofSeconds(5),1), + new CircuitBreaker.Listener(){public void onOpen(){}public void onClose(){}}); + var l1=new CaffeineLocalCache(settings,now::get); + var cache=new DefaultTierCache<>("lifetime",l1,new CircuitBreakerRemoteCache<>(remote,breaker),settings,true, + null,null,new VersionGenerator(),null,breaker,new CacheMetricsListener(){ + public void onRequest(String c,Outcome o){if(o==Outcome.STALE_DEGRADED)stale.incrementAndGet();if(o==Outcome.COALESCED)joined.countDown();} + },null,new TtlJitter(),()->true,now::get); + cache.put("x","original");l1.clear();assertEquals("original",cache.get("x")); + assertNull(remote.get("x").localFreshness()); + redis.getDockerClient().pauseContainerCmd(redis.getContainerId()).exec(); + var pool=Executors.newFixedThreadPool(8); + try { + assertNull(cache.get("uncached-probe"));assertTrue(breaker.isOpen()); + now.set(1+Duration.ofMinutes(29).toNanos()); + assertEquals("original",cache.getOrCompute("x",k->{fail("fresh load");return null;}));assertEquals(0,stale.get()); + for(int minute=30;minute<60;minute++) { + now.set(1+Duration.ofMinutes(minute).toNanos()); + assertEquals("original",cache.getOrCompute("x",k->{fail("stale load");return null;})); + } + assertEquals(30,stale.get());now.set(1+Duration.ofMinutes(60).toNanos()); + var requests=new ArrayList>(); + for(int i=0;i<8;i++) requests.add(pool.submit(()->cache.getOrCompute("x",k->{ + loads.incrementAndGet();try{assertTrue(releaseLoader.await(5,TimeUnit.SECONDS));}catch(InterruptedException e){throw new AssertionError(e);}return "local-new"; + }))); + assertTrue(joined.await(3,TimeUnit.SECONDS));releaseLoader.countDown(); + for(var request:requests)assertEquals("local-new",request.get(3,TimeUnit.SECONDS));assertEquals(1,loads.get()); + } finally { + releaseLoader.countDown();pool.shutdownNow(); + redis.getDockerClient().unpauseContainerCmd(redis.getContainerId()).exec(); + } + long deadline=System.nanoTime()+Duration.ofSeconds(10).toNanos(); + while(breaker.state()!=BreakerState.CLOSED&&System.nanoTime() synchronous full L1 flush on reconnect. + // Overflow -> asynchronously scheduled baseline-and-clear on reconnect. b.transport.reconnect(); + waitFor(() -> { + var dropped = registry.find("tiercache.invalidation").tag("cache", CACHE) + .tag("direction", "dropped").counter(); + return dropped != null && dropped.count() > 0; + }); // After the flush, reads re-resolve from L2: B agrees with L2. + waitFor(() -> java.util.Objects.equals(l2Truth(b, "keep"), b.cache.get("keep"))); assertEquals(l2Truth(b, "keep"), b.cache.get("keep")); assertEquals("v0", b.cache.get("flood-0")); } finally { diff --git a/tiercache-tck/src/test/java/io/tiercache/tck/SoakGate.java b/tiercache-tck/src/test/java/io/tiercache/tck/SoakGate.java new file mode 100644 index 0000000..b35fba3 --- /dev/null +++ b/tiercache-tck/src/test/java/io/tiercache/tck/SoakGate.java @@ -0,0 +1,54 @@ +package io.tiercache.tck; + +import java.time.Duration; +import java.util.*; + +/** Fixed-baseline assessment; memory series are never added or compared step-to-step. */ +final class SoakGate { + static final double MAX_GROWTH = 0.05; + static final int WARMUP_PERCENT = 20; + record Sample(long elapsedNanos, long heapBytes, long rssBytes, long journalRows, + long operations, long explicitGcCompletions, List> workers) { } + record Assessment(int discarded, int steadySamples, long heapBaseline, long heapPeak, + double heapGrowth, long rssBaseline, long rssPeak, double rssGrowth, + long journalPeak, double journalFirstMean, double journalSecondMean, + List violations) { + void requirePass() { + if (!violations.isEmpty()) throw new AssertionError(String.join("; ", violations)); + } + } + static void validateDuration(Duration duration, Duration interval) { + if (interval.isZero() || interval.isNegative() || duration.compareTo(interval.multipliedBy(3)) < 0) { + throw new IllegalArgumentException("Soak duration must cover at least three sample intervals (minimum " + + interval.multipliedBy(3) + ") for three steady-state samples after warm-up"); + } + duration.toNanos(); + } + static Assessment assess(List samples, int capacity) { + int discard = (int) Math.ceil(samples.size() * WARMUP_PERCENT / 100.0); + if (samples.size() - discard < 3) throw new AssertionError("Need at least three usable steady-state samples after 20% warm-up; got " + (samples.size() - discard)); + long previous = -1, previousGc = 0; + for (var sample : samples) { + if (sample.elapsedNanos() < 0 || sample.elapsedNanos() <= previous || sample.heapBytes() <= 0 + || sample.rssBytes() <= 0 || sample.journalRows() < 0 || sample.explicitGcCompletions() <= previousGc) throw new AssertionError("Invalid memory/journal/time sample: " + sample); + previous = sample.elapsedNanos(); + previousGc = sample.explicitGcCompletions(); + } + var steady = samples.subList(discard, samples.size()); + long heapBase = steady.get(0).heapBytes(), rssBase = steady.get(0).rssBytes(); + long heapPeak = steady.stream().mapToLong(Sample::heapBytes).max().orElseThrow(); + long rssPeak = steady.stream().mapToLong(Sample::rssBytes).max().orElseThrow(); + double heapGrowth = (heapPeak - heapBase) / (double) heapBase, rssGrowth = (rssPeak - rssBase) / (double) rssBase; + long journalPeak = steady.stream().mapToLong(Sample::journalRows).max().orElseThrow(); + int half = steady.size() / 2; + double first = steady.subList(0, half).stream().mapToLong(Sample::journalRows).average().orElseThrow(); + double second = steady.subList(half, steady.size()).stream().mapToLong(Sample::journalRows).average().orElseThrow(); + List failures = new ArrayList<>(); + if (heapGrowth > MAX_GROWTH) failures.add("Post-GC heap peak exceeds fixed baseline by " + heapGrowth * 100 + "% (limit 5%)"); + if (rssGrowth > MAX_GROWTH) failures.add("RSS peak exceeds fixed baseline by " + rssGrowth * 100 + "% (limit 5%)"); + if (journalPeak > capacity * 2L) failures.add("Journal peak exceeds " + capacity * 2L + " rows"); + if (second > first + capacity * 0.25) failures.add("Journal second-half mean exceeds first-half mean plus " + capacity * 0.25 + " rows"); + return new Assessment(discard, steady.size(), heapBase, heapPeak, heapGrowth, rssBase, rssPeak, rssGrowth, + journalPeak, first, second, List.copyOf(failures)); + } +} diff --git a/tiercache-tck/src/test/java/io/tiercache/tck/SoakHarnessTest.java b/tiercache-tck/src/test/java/io/tiercache/tck/SoakHarnessTest.java new file mode 100644 index 0000000..fd1bbe6 --- /dev/null +++ b/tiercache-tck/src/test/java/io/tiercache/tck/SoakHarnessTest.java @@ -0,0 +1,131 @@ +package io.tiercache.tck; + +import org.junit.jupiter.api.Test; +import org.junit.jupiter.api.io.TempDir; +import java.nio.file.*; +import java.time.Duration; +import java.util.*; +import java.util.concurrent.*; +import java.util.concurrent.atomic.*; +import static org.junit.jupiter.api.Assertions.*; + +class SoakHarnessTest { + @TempDir Path directory; + static List samples(long[] heap, long[] rss, long[] journal) { + List result=new ArrayList<>(); + for(int i=0;iSoakGate.assess(samples(growth,stable,journal),2000).requirePass()); + assertThrows(AssertionError.class,()->SoakGate.assess(samples(stable,growth,journal),2000).requirePass()); + var okay=SoakGate.assess(samples(new long[]{9999,100,103,105},stable,journal),2000); + assertDoesNotThrow(okay::requirePass);assertEquals(100,okay.heapBaseline());assertEquals(1,okay.discarded());assertEquals(3,okay.steadySamples()); + } + @Test void insufficientSamplesInvalidBaselinesAndShortDurationsFail() { + assertThrows(AssertionError.class,()->SoakGate.assess(samples(new long[]{100,100,100},new long[]{100,100,100},new long[]{1,1,1}),2000)); + assertThrows(AssertionError.class,()->SoakGate.assess(samples(new long[]{100,0,100,100},new long[]{100,100,100,100},new long[]{1,1,1,1}),2000)); + assertThrows(AssertionError.class,()->SoakGate.assess(samples(new long[]{100,100,100,100},new long[]{100,100,-1,100},new long[]{1,1,1,1}),2000)); + for(long seconds:new long[]{-1,0,30,60,89})assertThrows(IllegalArgumentException.class,()->SoakGate.validateDuration(Duration.ofSeconds(seconds),Duration.ofSeconds(30))); + assertDoesNotThrow(()->SoakGate.validateDuration(Duration.ofSeconds(90),Duration.ofSeconds(30))); + } + @Test void journalPeakAndHalfMeanChecksRemainSeparateAndUnchanged() { + long[] memory={100,100,100,100}; + assertThrows(AssertionError.class,()->SoakGate.assess(samples(memory,memory,new long[]{2000,2000,2000,5000}),2000).requirePass()); + var trend=SoakGate.assess(samples(memory,memory,new long[]{2000,2000,2700,2700}),2000); + assertTrue(trend.journalPeak()<=4000);assertThrows(AssertionError.class,trend::requirePass); + } + @Test void rssReadersUseResidentKibibytesAndRejectUnavailableOrMalformedValues() throws Exception { + assertEquals(12*1024,SoakMemory.linuxRss("VmSize: 999999 kB\nVmRSS:\t12 kB\n")); + assertEquals(128*1024,SoakMemory.macRss(" 128\n")); + for(String text:new String[]{"", "-1", "0", "1.5", "NaN", "Infinity", "12 13", "9223372036854775807"})assertThrows(IllegalStateException.class,()->SoakMemory.macRss(text)); + for(String text:new String[]{"VmSize: 12 kB", "VmRSS: -1 kB", "VmRSS: 0 kB", "VmRSS: 12 MB", "VmRSS: 12 kB\nVmRSS: 13 kB"})assertThrows(IllegalStateException.class,()->SoakMemory.linuxRss(text)); + assertThrows(IllegalStateException.class,()->SoakMemory.rssReader("Windows",42,(c,t)->"12",()->"unused")); + var mac=SoakMemory.rssReader("Mac OS X",42,(command,timeout)->{ + assertEquals(List.of("/bin/ps","-o","rss=","-p","42"),command);assertEquals(Duration.ofSeconds(2),timeout);return "32"; + },()->{throw new AssertionError("wrong source");});assertEquals(32768,mac.bytes()); + var timeout=new IllegalStateException("RSS timeout"); + var failed=SoakMemory.rssReader("macOS",42,(c,t)->{throw timeout;},()->"unused");assertSame(timeout,assertThrows(IllegalStateException.class,failed::bytes)); + } + static boolean supportedHost() { + String os=System.getProperty("os.name").toLowerCase(Locale.ROOT); + return os.contains("linux")||os.contains("mac")||os.contains("darwin"); + } + @Test void commandTimeoutAndExitFailureAreBoundedAndNotZeroRss() { + if (!supportedHost()) { assertThrows(IllegalStateException.class,SoakMemory::systemRss); return; } + assertTimeoutPreemptively(Duration.ofSeconds(3),()->assertThrows(IllegalStateException.class, + ()->SoakMemory.runCommand(List.of("/bin/sleep","5"),Duration.ofMillis(20)))); + assertThrows(IllegalStateException.class,()->SoakMemory.runCommand(List.of("/bin/ps","--not-a-supported-option"),Duration.ofSeconds(1))); + } + static class FakeGc implements SoakMemory.GcAccess { + long completed,heap=123;Runnable request=()->{}; + public long completed(){return completed;}public long heapAfterGc(){return heap;}public void request(){request.run();} + } + @Test void gcProgressIsConfirmedAndIgnoredOrSlowExplicitGcFails() throws Exception { + var clock=new AtomicLong();var gc=new FakeGc(); + var reading=SoakMemory.sample(gc,()->456,clock::get,d->{clock.addAndGet(d.toNanos());gc.completed++;},Duration.ofMillis(20)); + assertEquals(123,reading.heapBytes());assertEquals(456,reading.rssBytes());assertEquals(1,reading.explicitGcCompletions()); + var ignored=new FakeGc();clock.set(0); + var failure=assertThrows(IllegalStateException.class,()->SoakMemory.sample(ignored,()->456,clock::get,d->clock.addAndGet(d.toNanos()),Duration.ofMillis(20))); + assertTrue(failure.getMessage().contains("DisableExplicitGC")); + var slow=new FakeGc();clock.set(0);slow.request=()->{clock.addAndGet(Duration.ofSeconds(1).toNanos());slow.completed++;}; + assertThrows(IllegalStateException.class,()->SoakMemory.sample(slow,()->456,clock::get,d->{},Duration.ofMillis(20))); + var invalid=new FakeGc();invalid.request=()->invalid.completed++;invalid.heap=0;clock.set(0); + assertThrows(IllegalStateException.class,()->SoakMemory.sample(invalid,()->456,clock::get,d->{},Duration.ofMillis(20))); + invalid.heap=123; + assertThrows(IllegalStateException.class,()->SoakMemory.sample(invalid,()->0,clock::get,d->{},Duration.ofMillis(20))); + } + @Test void realHostProvidesRssAndConfirmedPostGcHeap() throws Exception { + if (!supportedHost()) { assertThrows(IllegalStateException.class,SoakMemory::systemRss); return; } + var rss=SoakMemory.systemRss();assertTrue(rss.bytes()>0); + try(var gc=new SoakMemory.ExplicitGc()) { + var reading=SoakMemory.sample(gc,rss,System::nanoTime,d->TimeUnit.NANOSECONDS.sleep(d.toNanos()),Duration.ofSeconds(5)); + assertTrue(reading.heapBytes()>0);assertTrue(reading.rssBytes()>0);assertTrue(reading.explicitGcCompletions()>0); + } + } + @Test void futureCapturedErrorFailsEvenWhenExceptionCounterRemainsZero() throws Exception { + var errors=new AtomicInteger();var original=new AssertionError("worker assertion"); + try(var workers=new SoakWorkers(1,System.nanoTime()+TimeUnit.SECONDS.toNanos(10),System::nanoTime,w->{ + try{throw original;}catch(Exception error){errors.incrementAndGet();} + })) { + var failure=assertThrows(AssertionError.class,()->workers.awaitHealthy(Duration.ofSeconds(1))); + assertSame(original,failure.getCause());assertEquals(0,errors.get());assertEquals(true,workers.snapshots().get(0).get("futureObserved")); + } + } + @Test void earlyCompletionAndEmptyWorkerCannotPass() { + var clock=new AtomicLong(1); + try(var workers=new SoakWorkers(1,10,clock::get,w->w.succeeded())) { + assertTrue(assertThrows(AssertionError.class,()->workers.awaitHealthy(Duration.ofSeconds(1))).getCause().getMessage().contains("prematurely")); + } + try(var workers=new SoakWorkers(1,10,clock::get,w->clock.set(10))) { + assertTrue(assertThrows(AssertionError.class,()->workers.awaitHealthy(Duration.ofSeconds(1))).getCause().getMessage().contains("no successful")); + } + } + @Test void cancellationAndHungWorkerCannotPassAndCleanupObservesEveryFuture() throws Exception { + for(boolean cancelled:new boolean[]{false,true}) { + var entered=new CountDownLatch(1);var release=new CountDownLatch(1); + var workers=new SoakWorkers(2,System.nanoTime()+TimeUnit.SECONDS.toNanos(10),System::nanoTime,w->{ + w.succeeded();entered.countDown();while(release.getCount()>0)try{release.await(10,TimeUnit.MILLISECONDS);}catch(InterruptedException ignored){} + }); + try { + assertTrue(entered.await(1,TimeUnit.SECONDS));if(cancelled)workers.cancel(0); + assertTimeoutPreemptively(Duration.ofSeconds(2),()->assertThrows(AssertionError.class,()->workers.awaitHealthy(Duration.ofMillis(30)))); + } finally {release.countDown();workers.close();} + assertTrue(workers.terminated());assertTrue(workers.snapshots().stream().allMatch(row->Boolean.TRUE.equals(row.get("futureObserved")))); + } + } + @Test void successfulWorkHasObservedFuturesAndOperationCounts() throws Exception { + var now=new AtomicLong(1); + try(var workers=new SoakWorkers(1,10,now::get,w->{w.succeeded();now.set(10);})) { + workers.awaitHealthy(Duration.ofSeconds(1));assertEquals(1,workers.operations()); + assertEquals("COMPLETED",workers.snapshots().get(0).get("state"));assertEquals(true,workers.snapshots().get(0).get("futureObserved")); + } + } + @Test void failedReportRetainsUnitsAndEscapesDiagnostics() throws Exception { + var report=Map.of("status","FAIL","units",Map.of("memory","bytes"),"failure","quote\" slash\\ newline\n"); + Path path=directory.resolve("report.json");SoakReport.write(path,report);String json=Files.readString(path); + assertTrue(json.contains("\\\""));assertTrue(json.contains("\\\\"));assertTrue(json.contains("\\u000a"));assertTrue(json.contains("\"status\":\"FAIL\"")); + assertThrows(IllegalArgumentException.class,()->SoakReport.json(Double.NaN)); + } +} diff --git a/tiercache-tck/src/test/java/io/tiercache/tck/SoakMemory.java b/tiercache-tck/src/test/java/io/tiercache/tck/SoakMemory.java new file mode 100644 index 0000000..3a63142 --- /dev/null +++ b/tiercache-tck/src/test/java/io/tiercache/tck/SoakMemory.java @@ -0,0 +1,143 @@ +package io.tiercache.tck; + +import com.sun.management.GarbageCollectionNotificationInfo; +import javax.management.NotificationEmitter; +import javax.management.NotificationListener; +import javax.management.openmbean.CompositeData; +import java.lang.management.*; +import java.nio.charset.StandardCharsets; +import java.nio.file.*; +import java.time.Duration; +import java.util.*; +import java.util.concurrent.TimeUnit; +import java.util.concurrent.atomic.AtomicLong; +import java.util.function.LongSupplier; +import java.util.regex.Pattern; + +/** Independent, bounded memory acquisition for the strict soak gate. */ +final class SoakMemory { + interface RssReader { long bytes() throws Exception; } + interface CommandRunner { String run(List command, Duration timeout) throws Exception; } + interface GcAccess { + long completed(); + long heapAfterGc(); + void request(); + } + interface Sleeper { void sleep(Duration duration) throws InterruptedException; } + record Reading(long heapBytes, long rssBytes, long explicitGcCompletions) { } + + static RssReader rssReader(String os, long pid, CommandRunner runner, + java.util.function.Supplier linuxStatus) { + String platform = os.toLowerCase(Locale.ROOT); + if (platform.contains("linux")) return () -> linuxRss(linuxStatus.get()); + if (platform.contains("mac") || platform.contains("darwin")) { + return () -> macRss(runner.run(List.of("/bin/ps", "-o", "rss=", "-p", Long.toString(pid)), Duration.ofSeconds(2))); + } + throw new IllegalStateException("RSS measurement unsupported on " + os + "; run the strict soak on Linux or macOS"); + } + static RssReader systemRss() { + return rssReader(System.getProperty("os.name"), ProcessHandle.current().pid(), SoakMemory::runCommand, () -> { + try { return Files.readString(Path.of("/proc/self/status")); } + catch (java.io.IOException e) { throw new IllegalStateException("Cannot read /proc/self/status for RSS", e); } + }); + } + static long linuxRss(String status) { + var matcher = Pattern.compile("(?m)^VmRSS:[ \\t]+([0-9]+)[ \\t]+kB[ \\t]*$").matcher(status); + if (!matcher.find()) throw new IllegalStateException("Missing or malformed VmRSS (expected KiB) in /proc/self/status"); + long bytes = kibibytes(matcher.group(1)); + if (matcher.find()) throw new IllegalStateException("Duplicate VmRSS measurement"); + return bytes; + } + static long macRss(String output) { + String number = output.strip(); + if (!number.matches("[0-9]+")) throw new IllegalStateException("Malformed ps RSS; expected one positive KiB value"); + return kibibytes(number); + } + private static long kibibytes(String value) { + try { + long bytes = Math.multiplyExact(Long.parseLong(value), 1024L); + if (bytes <= 0) throw new IllegalArgumentException("nonpositive RSS"); + return bytes; + } catch (RuntimeException e) { throw new IllegalStateException("Invalid RSS KiB value: " + value, e); } + } + static String runCommand(List command, Duration timeout) throws Exception { + ProcessBuilder builder = new ProcessBuilder(command).redirectErrorStream(true); + builder.environment().put("LC_ALL", "C"); + builder.environment().put("LANG", "C"); + Process process = builder.start(); + try { + if (!process.waitFor(timeout.toMillis(), TimeUnit.MILLISECONDS)) { + throw new IllegalStateException("RSS command timed out after " + timeout + ": " + command.get(0)); + } + byte[] output = process.getInputStream().readNBytes(4097); + if (process.exitValue() != 0 || output.length > 4096) { + throw new IllegalStateException("RSS command failed or returned oversized output (exit=" + process.exitValue() + ")"); + } + return new String(output, StandardCharsets.US_ASCII); + } finally { + if (process.isAlive()) { + process.destroyForcibly(); + process.waitFor(1, TimeUnit.SECONDS); + } + process.getInputStream().close(); + process.getErrorStream().close(); + process.getOutputStream().close(); + } + } + static Reading sample(GcAccess gc, RssReader rss, LongSupplier clock, Sleeper sleeper, + Duration timeout) throws Exception { + long before = gc.completed(); + long start = clock.getAsLong(); + gc.request(); + while (true) { + if (clock.getAsLong() - start >= timeout.toNanos()) { + throw new IllegalStateException("Post-GC heap is inconclusive: no explicit GC completion within " + timeout + + "; enable explicit GC (remove -XX:+DisableExplicitGC) and use a collector reporting System.gc() completion"); + } + if (gc.completed() > before) break; + sleeper.sleep(Duration.ofMillis(10)); + } + long heap = gc.heapAfterGc(), resident = rss.bytes(); + if (heap <= 0 || resident <= 0) throw new IllegalStateException("Memory acquisition requires positive heap and RSS bytes"); + return new Reading(heap, resident, gc.completed()); + } + + /** Uses completion notifications, not an assumed sleep or an unrelated young GC. */ + static final class ExplicitGc implements GcAccess, AutoCloseable { + private final List emitters = new ArrayList<>(); + private final AtomicLong completed = new AtomicLong(), heap = new AtomicLong(); + private final Set heapPools = new HashSet<>(); + private final NotificationListener listener; + ExplicitGc() throws Exception { + for (var pool : ManagementFactory.getMemoryPoolMXBeans()) if (pool.getType() == MemoryType.HEAP) heapPools.add(pool.getName()); + listener = (notification, handback) -> { + if (!GarbageCollectionNotificationInfo.GARBAGE_COLLECTION_NOTIFICATION.equals(notification.getType())) return; + var info = GarbageCollectionNotificationInfo.from((CompositeData) notification.getUserData()); + if (!"System.gc()".equals(info.getGcCause())) return; + long used = info.getGcInfo().getMemoryUsageAfterGc().entrySet().stream() + .filter(entry -> heapPools.contains(entry.getKey())).mapToLong(entry -> entry.getValue().getUsed()).sum(); + heap.set(used); + completed.incrementAndGet(); + }; + try { + for (var collector : ManagementFactory.getGarbageCollectorMXBeans()) { + if (collector instanceof NotificationEmitter emitter) { + emitter.addNotificationListener(listener, null, null); + emitters.add(emitter); + } + } + if (emitters.isEmpty()) throw new IllegalStateException("Collector cannot report explicit GC completion; strict post-GC measurement unavailable"); + } catch (Exception e) { close(); throw e; } + } + public long completed() { return completed.get(); } + public long heapAfterGc() { return heap.get(); } + public void request() { System.gc(); } + public void close() { + for (var emitter : emitters) { + try { emitter.removeNotificationListener(listener); } + catch (javax.management.ListenerNotFoundException ignored) { } + } + emitters.clear(); + } + } +} diff --git a/tiercache-tck/src/test/java/io/tiercache/tck/SoakReport.java b/tiercache-tck/src/test/java/io/tiercache/tck/SoakReport.java new file mode 100644 index 0000000..f1682a8 --- /dev/null +++ b/tiercache-tck/src/test/java/io/tiercache/tck/SoakReport.java @@ -0,0 +1,61 @@ +package io.tiercache.tck; + +import java.io.IOException; +import java.nio.file.*; +import java.util.*; + +/** Small dependency-free JSON report, including failed/inconclusive runs. */ +final class SoakReport { + static void write(Path path, Map report) throws IOException { + Path target = path.toAbsolutePath(); + Files.createDirectories(target.getParent()); + Path temporary = Files.createTempFile(target.getParent(), "soak-", ".json.tmp"); + try { + Files.writeString(temporary, json(report) + "\n"); + Files.move(temporary, target, StandardCopyOption.REPLACE_EXISTING); + } finally { Files.deleteIfExists(temporary); } + } + static Map sample(SoakGate.Sample sample) { + return Map.of("elapsedNanos", sample.elapsedNanos(), "postGcHeapBytes", sample.heapBytes(), + "rssBytes", sample.rssBytes(), "journalRows", sample.journalRows(), + "completedOperations", sample.operations(), "explicitGcCompletions", sample.explicitGcCompletions(), "workers", sample.workers()); + } + static Map assessment(SoakGate.Assessment result) { + Map values = new LinkedHashMap<>(); + values.put("warmupSamplesDiscarded", result.discarded()); values.put("steadySamples", result.steadySamples()); + values.put("heapBaselineBytes", result.heapBaseline()); values.put("heapPeakBytes", result.heapPeak()); + values.put("heapGrowthFraction", result.heapGrowth()); values.put("rssBaselineBytes", result.rssBaseline()); + values.put("rssPeakBytes", result.rssPeak()); values.put("rssGrowthFraction", result.rssGrowth()); + values.put("journalPeakRows", result.journalPeak()); values.put("journalFirstHalfMean", result.journalFirstMean()); + values.put("journalSecondHalfMean", result.journalSecondMean()); values.put("violations", result.violations()); + return values; + } + static String json(Object value) { + if (value == null) return "null"; + if (value instanceof String text) { + StringBuilder out = new StringBuilder("\""); + for (char c : text.toCharArray()) { + if (c == '"' || c == '\\') out.append('\\').append(c); + else if (c < 32) out.append(String.format(Locale.ROOT, "\\u%04x", (int)c)); + else out.append(c); + } + return out.append('"').toString(); + } + if (value instanceof Number number) { + if (!Double.isFinite(number.doubleValue())) throw new IllegalArgumentException("Non-finite report number"); + return number.toString(); + } + if (value instanceof Boolean bool) return bool.toString(); + if (value instanceof Map map) { + var out = new StringJoiner(",", "{", "}"); + map.forEach((key, entry) -> out.add(json(key.toString()) + ":" + json(entry))); + return out.toString(); + } + if (value instanceof Iterable values) { + var out = new StringJoiner(",", "[", "]"); + values.forEach(entry -> out.add(json(entry))); + return out.toString(); + } + throw new IllegalArgumentException("Unsupported report type: " + value.getClass()); + } +} diff --git a/tiercache-tck/src/test/java/io/tiercache/tck/SoakTest.java b/tiercache-tck/src/test/java/io/tiercache/tck/SoakTest.java index ceae22f..dcef0ce 100644 --- a/tiercache-tck/src/test/java/io/tiercache/tck/SoakTest.java +++ b/tiercache-tck/src/test/java/io/tiercache/tck/SoakTest.java @@ -1,148 +1,141 @@ package io.tiercache.tck; -import com.sun.management.OperatingSystemMXBean; import io.lettuce.core.RedisClient; import io.lettuce.core.codec.ByteArrayCodec; -import io.tiercache.CacheSettings; -import io.tiercache.InvalidationMode; -import io.tiercache.NullPolicy; -import io.tiercache.TierCache; -import io.tiercache.TierCacheFactory; +import io.tiercache.*; import io.tiercache.invalidation.InvalidationService; -import io.tiercache.redis.JdkCacheSerializer; -import io.tiercache.redis.LettucePubSubInvalidationTransport; -import io.tiercache.redis.LettuceRemoteCache; -import io.tiercache.redis.RedisStreamJournal; +import io.tiercache.redis.*; import io.tiercache.spi.InvalidationListener; import org.junit.jupiter.api.Tag; import org.junit.jupiter.api.Test; import org.testcontainers.containers.GenericContainer; import org.testcontainers.utility.DockerImageName; - import java.lang.management.ManagementFactory; -import java.time.Duration; -import java.util.ArrayList; -import java.util.List; -import java.util.concurrent.ExecutorService; -import java.util.concurrent.Executors; -import java.util.concurrent.ThreadLocalRandom; -import java.util.concurrent.TimeUnit; -import java.util.concurrent.atomic.AtomicLong; - -import static org.junit.jupiter.api.Assertions.assertEquals; -import static org.junit.jupiter.api.Assertions.assertTrue; +import java.nio.file.Path; +import java.time.*; +import java.util.*; +import java.util.concurrent.*; -/** - * Soak: sustained churn against a real L2 container. A fixed worker pool runs - * a weighted workload (~80% reads, ~15% puts, ~5% evictions and tag - * invalidations) over a rotating key space with jittered TTLs, exercising the - * invalidation journal for real. Duration comes from the - * {@code tiercache.soak.duration} system property (ISO-8601, default PT10M). - * - *

    Every 30s the harness samples memory (heap used after an explicit GC, - * plus committed VM size via {@link OperatingSystemMXBean}) and the journal - * size. The first 20% of samples are discarded as warm-up; the gates on the - * remainder are: memory growth at most 5%, journal size bounded with no - * monotonic trend, and zero workload errors. - * - *

    Memory metric rationale (design D2): a true per-process RSS read is - * platform-specific, so the portable proxy is committed-VM size plus post-GC - * heap used. Post-GC heap bounds live-set growth (leaks), committed VM bounds - * total address-space growth the JVM is responsible for. The proxy can miss - * pure native leaks outside the JVM's books; on a Linux CI runner - * {@code /proc/self/status} can replace it without changing the gate shape. - * - *

    Runs only via the dedicated {@code soakTest} task (tag-filtered); the - * default {@code test} task excludes the {@code soak} tag. - */ +/** Strict, opt-in churn gate: separate post-GC heap/RSS, observed workers and retained evidence. */ @Tag("soak") class SoakTest { - private static final String CACHE = "soak"; - private static final Duration DEFAULT_DURATION = Duration.parse("PT10M"); + private static final Duration DEFAULT_DURATION = Duration.ofMinutes(10); private static final Duration SAMPLE_INTERVAL = Duration.ofSeconds(30); - private static final int WORKERS = 8; - private static final int KEY_SPACE = 5_000; - private static final int TAGS = 32; - private static final int JOURNAL_CAPACITY = 2_000; - private static final double MAX_MEMORY_GROWTH = 0.05; - private static final int WARMUP_DISCARD_PERCENT = 20; - - /** One memory/journal observation. */ - record Sample(long memoryBytes, long journalSize) { - } - - @Test - void churnStaysWithinMemoryAndJournalBudgets() throws Exception { - Duration duration = duration(); - try (var server = new GenericContainer<>(DockerImageName.parse("redis:6.2-alpine")) - .withExposedPorts(6379)) { - server.start(); - String uri = "redis://" + server.getHost() + ":" + server.getMappedPort(6379); - RedisClient client = RedisClient.create(uri); - RedisStreamJournal journal = new RedisStreamJournal( - client.connect(ByteArrayCodec.INSTANCE), JOURNAL_CAPACITY, new JdkCacheSerializer<>()); - LettuceRemoteCache l2 = LettuceRemoteCache.builder(uri) - .client(client) - .cacheName(CACHE) - .journal(journal) - .build(); - LettucePubSubInvalidationTransport transport = - new LettucePubSubInvalidationTransport(client, new JdkCacheSerializer<>()); - TierCacheFactory factory = TierCacheFactory.builder() - .defaults(new CacheSettings(KEY_SPACE, Duration.ofMinutes(2), null, - Duration.ofMinutes(5), 0.10, NullPolicy.deny(), - InvalidationMode.INVALIDATE, 64 * 1024)) - .remoteCache(l2) - .invalidation(versions -> new InvalidationService(transport, journal, - versions.instanceId(), InvalidationListener.NOOP)) - .build(); - TierCache cache = factory.getCache(CACHE); - try { - runChurn(cache, journal, duration); - } finally { - factory.close(); - l2.close(); - client.shutdown(); + private static final Duration GC_TIMEOUT = Duration.ofSeconds(5); + private static final Duration WORKER_TIMEOUT = Duration.ofSeconds(60); + private static final int WORKERS = 8, KEY_SPACE = 5_000, TAGS = 32, JOURNAL_CAPACITY = 2_000; + + @Test void churnStaysWithinMemoryAndJournalBudgets() throws Exception { + Path path = Path.of(System.getProperty("tiercache.soak.report", "build/reports/soak/report.json")); + Map report = new LinkedHashMap<>(); + report.put("status", "RUNNING"); report.put("startedAt", Instant.now().toString()); + report.put("durationRequested", System.getProperty("tiercache.soak.duration", DEFAULT_DURATION.toString())); + report.put("jdk", System.getProperty("java.runtime.version")); report.put("vm", System.getProperty("java.vm.name")); + report.put("os", System.getProperty("os.name")); report.put("pid", ProcessHandle.current().pid()); + report.put("collectors", ManagementFactory.getGarbageCollectorMXBeans().stream().map(bean -> bean.getName()).toList()); + var heap = ManagementFactory.getMemoryMXBean().getHeapMemoryUsage(); + report.put("heapInitialBytes", heap.getInit()); report.put("heapMaxBytes", heap.getMax()); + report.put("relevantVmFlags", ManagementFactory.getRuntimeMXBean().getInputArguments().stream() + .filter(arg -> arg.startsWith("-Xms") || arg.startsWith("-Xmx") || arg.contains("DisableExplicitGC") + || arg.contains("ExplicitGCInvokesConcurrent") || arg.matches("-XX:[+-]Use.*GC") + || arg.startsWith("-XX:MaxRAMPercentage") || arg.startsWith("-XX:InitialRAMPercentage")).toList()); + report.put("sampleIntervalNanos", SAMPLE_INTERVAL.toNanos()); report.put("units", Map.of("memory", "bytes", "elapsed", "nanoseconds", "journal", "rows")); + report.put("workload", Map.of("workers", WORKERS, "keySpace", KEY_SPACE, "tags", TAGS, + "readPercent", 80, "putPercent", 15, "evictPercent", 5, "journalCapacity", JOURNAL_CAPACITY)); + report.put("limits", Map.of("heapGrowthFraction", SoakGate.MAX_GROWTH, "rssGrowthFraction", SoakGate.MAX_GROWTH, + "warmupDiscardPercent", SoakGate.WARMUP_PERCENT, "minimumSteadySamples", 3, + "journalPeakRows", JOURNAL_CAPACITY * 2, "journalMeanIncreaseRows", JOURNAL_CAPACITY * 0.25, + "gcConfirmationTimeoutNanos", GC_TIMEOUT.toNanos(), "rssCommandTimeoutNanos", Duration.ofSeconds(2).toNanos(), + "workerTerminationTimeoutNanos", WORKER_TIMEOUT.toNanos(), "workerCleanupTimeoutNanos", Duration.ofSeconds(1).toNanos())); + report.put("samples", List.of()); report.put("workers", List.of()); + Throwable failed = null; + try { + Duration duration = duration(); SoakGate.validateDuration(duration, SAMPLE_INTERVAL); + report.put("durationNanos", duration.toNanos()); + var rss = SoakMemory.systemRss(); + report.put("rssSource", System.getProperty("os.name").toLowerCase(Locale.ROOT).contains("linux") + ? "/proc/self/status VmRSS (KiB to bytes)" : "/bin/ps -o rss= -p (KiB to bytes)"); + report.put("heapSource", "Heap pools in completed System.gc() JMX notifications"); + rss.bytes(); // Reject unsupported/unavailable RSS before starting the container/workload. + try (var gc = new SoakMemory.ExplicitGc()) { + var preflight = memory(gc, rss); + report.put("preflight", Map.of("heapBytes", preflight.heapBytes(), "rssBytes", preflight.rssBytes(), + "explicitGcCompletions", preflight.explicitGcCompletions())); + SoakReport.write(path, report); + try (var server = new GenericContainer<>(DockerImageName.parse("redis:6.2-alpine")).withExposedPorts(6379)) { + server.start(); String uri = "redis://" + server.getHost() + ":" + server.getMappedPort(6379); + try (var client = RedisClient.create(uri); var connection = client.connect(ByteArrayCodec.INSTANCE)) { + var journal = new RedisStreamJournal(connection, JOURNAL_CAPACITY, new JdkCacheSerializer<>()); + try (var l2 = LettuceRemoteCache.builder(uri).client(client).cacheName(CACHE).journal(journal).build()) { + var transport = new LettucePubSubInvalidationTransport(client, new JdkCacheSerializer<>()); + TierCacheFactory factory = null; + try { + factory = TierCacheFactory.builder() + .defaults(new CacheSettings(KEY_SPACE, Duration.ofMinutes(2), null, Duration.ofMinutes(5), + 0.10, NullPolicy.deny(), InvalidationMode.INVALIDATE, 64 * 1024)) + .remoteCache(l2).invalidation(versions -> new InvalidationService(transport, journal, + versions.instanceId(), InvalidationListener.NOOP)).build(); + runChurn(factory.getCache(CACHE), journal, duration, gc, rss, report, path); + } finally { if (factory != null) factory.close(); else transport.close(); } + } + } + } } + report.put("status", "PASS"); + } catch (Throwable failure) { + failed = failure; report.put("status", "FAIL"); + report.put("failureClass", failure.getClass().getName()); report.put("failureMessage", failure.getMessage()); + var stack = new java.io.StringWriter(); failure.printStackTrace(new java.io.PrintWriter(stack)); report.put("failureStackTrace", stack.toString()); + } finally { + report.put("finishedAt", Instant.now().toString()); + try { SoakReport.write(path, report); } + catch (Exception writeFailure) { if (failed != null) failed.addSuppressed(writeFailure); else throw writeFailure; } + System.out.println("SOAK " + report.get("status") + " report=" + path.toAbsolutePath()); } + if (failed instanceof Error error) throw error; + if (failed instanceof Exception exception) throw exception; + if (failed != null) throw new AssertionError(failed); } - private static Duration duration() { - String property = System.getProperty("tiercache.soak.duration"); - return property == null || property.isBlank() ? DEFAULT_DURATION : Duration.parse(property); + String value = System.getProperty("tiercache.soak.duration"); + return value == null || value.isBlank() ? DEFAULT_DURATION : Duration.parse(value); } - - private static void runChurn(TierCache cache, RedisStreamJournal journal, - Duration duration) throws Exception { - AtomicLong errors = new AtomicLong(); - long deadline = System.nanoTime() + duration.toNanos(); - ExecutorService pool = Executors.newFixedThreadPool(WORKERS); - for (int i = 0; i < WORKERS; i++) { - pool.submit(() -> { - while (System.nanoTime() < deadline) { - try { - churnOnce(cache); - } catch (Exception e) { - errors.incrementAndGet(); - } - } - }); - } - pool.shutdown(); - - List samples = new ArrayList<>(); - samples.add(sample(journal)); - while (System.nanoTime() < deadline) { - long remainingMillis = TimeUnit.NANOSECONDS.toMillis(deadline - System.nanoTime()); - Thread.sleep(Math.max(1, Math.min(SAMPLE_INTERVAL.toMillis(), remainingMillis))); - samples.add(sample(journal)); + private static SoakMemory.Reading memory(SoakMemory.GcAccess gc, SoakMemory.RssReader rss) throws Exception { + return SoakMemory.sample(gc, rss, System::nanoTime, duration -> TimeUnit.NANOSECONDS.sleep(duration.toNanos()), GC_TIMEOUT); + } + private static void runChurn(TierCache cache, RedisStreamJournal journal, Duration duration, + SoakMemory.GcAccess gc, SoakMemory.RssReader rss, Map report, Path path) throws Exception { + long start = System.nanoTime(), deadline = start + duration.toNanos(); + var samples = new ArrayList(); + var workers = new SoakWorkers(WORKERS, deadline, System::nanoTime, worker -> { + while (worker.keepRunning()) { churnOnce(cache); worker.succeeded(); } + }); + try { + long next = start; + while (true) { + workers.checkProgress(); + long delay = next - System.nanoTime(); + if (delay > 0) TimeUnit.NANOSECONDS.sleep(delay); + workers.checkProgress(); + var measured = memory(gc, rss); + var sample = new SoakGate.Sample(System.nanoTime() - start, measured.heapBytes(), measured.rssBytes(), + journal.size(CACHE), workers.operations(), measured.explicitGcCompletions(), workers.snapshots()); + samples.add(sample); report.put("samples", samples.stream().map(SoakReport::sample).toList()); + report.put("workers", workers.snapshots()); SoakReport.write(path, report); + System.out.printf(Locale.ROOT, "SOAK sample t=%.1fs heap=%d rss=%d journal=%d operations=%d%n", + sample.elapsedNanos() / 1e9, sample.heapBytes(), sample.rssBytes(), sample.journalRows(), sample.operations()); + if (System.nanoTime() - deadline >= 0) break; + next = Math.min(next + SAMPLE_INTERVAL.toNanos(), deadline); + } + workers.awaitHealthy(WORKER_TIMEOUT); + var assessment = SoakGate.assess(samples, JOURNAL_CAPACITY); + report.put("assessment", SoakReport.assessment(assessment)); assessment.requirePass(); + } finally { + workers.close(); report.put("workers", workers.snapshots()); report.put("workersTerminated", workers.terminated()); + report.put("completedOperations", workers.operations()); } - assertTrue(pool.awaitTermination(60, TimeUnit.SECONDS), "workers did not stop in time"); - assertEquals(0, errors.get(), "workload errors during the soak run"); - assertGates(samples); } - private static void churnOnce(TierCache cache) { ThreadLocalRandom random = ThreadLocalRandom.current(); String key = "k" + random.nextInt(KEY_SPACE); @@ -162,43 +155,4 @@ private static void churnOnce(TierCache cache) { } } - private static Sample sample(RedisStreamJournal journal) throws InterruptedException { - System.gc(); - System.gc(); - Thread.sleep(200); // let the explicit GCs finish before reading the heap - long heapUsed = ManagementFactory.getMemoryMXBean().getHeapMemoryUsage().getUsed(); - long committedVm = ((OperatingSystemMXBean) ManagementFactory.getOperatingSystemMXBean()) - .getCommittedVirtualMemorySize(); - return new Sample(heapUsed + committedVm, journal.size(CACHE)); - } - - private static void assertGates(List samples) { - assertTrue(samples.size() >= 3, () -> "too few samples to judge a trend: " + samples.size()); - int discard = (int) Math.ceil(samples.size() * WARMUP_DISCARD_PERCENT / 100.0); - List steady = samples.subList(discard, samples.size()); - - long baseline = steady.get(0).memoryBytes(); - long peak = steady.stream().mapToLong(Sample::memoryBytes).max().orElseThrow(); - double growth = (peak - baseline) / (double) baseline; - assertTrue(growth <= MAX_MEMORY_GROWTH, - () -> "post-warm-up memory growth %.2f%% exceeds the %.0f%% gate (samples=%s)" - .formatted(growth * 100, MAX_MEMORY_GROWTH * 100, samples)); - - // Bounded: approximate MAXLEN trimming keeps the stream near capacity; - // whole-node trimming may overshoot, so the bound carries headroom. - long journalPeak = steady.stream().mapToLong(Sample::journalSize).max().orElseThrow(); - assertTrue(journalPeak <= JOURNAL_CAPACITY * 2L, - () -> "journal size " + journalPeak + " exceeds the bounded gate (samples=" + samples + ")"); - - // No monotonic trend: the mean of the second half of the run must not - // exceed the first half beyond a small allowance. - int half = steady.size() / 2; - double firstHalfMean = steady.subList(0, half).stream() - .mapToLong(Sample::journalSize).average().orElse(0); - double secondHalfMean = steady.subList(half, steady.size()).stream() - .mapToLong(Sample::journalSize).average().orElse(0); - assertTrue(secondHalfMean <= firstHalfMean + JOURNAL_CAPACITY * 0.25, - () -> "journal size trends upward: first-half mean %.0f, second-half mean %.0f (samples=%s)" - .formatted(firstHalfMean, secondHalfMean, samples)); - } } diff --git a/tiercache-tck/src/test/java/io/tiercache/tck/SoakWorkers.java b/tiercache-tck/src/test/java/io/tiercache/tck/SoakWorkers.java new file mode 100644 index 0000000..f75676b --- /dev/null +++ b/tiercache-tck/src/test/java/io/tiercache/tck/SoakWorkers.java @@ -0,0 +1,124 @@ +package io.tiercache.tck; + +import java.time.Duration; +import java.util.*; +import java.util.concurrent.*; +import java.util.concurrent.atomic.AtomicLong; +import java.util.function.LongSupplier; + +/** Observes FutureTask failures as well as actual work and termination. */ +final class SoakWorkers implements AutoCloseable { + interface Work { void run(Worker worker) throws Exception; } + final class Worker { + final int id; + final AtomicLong operations = new AtomicLong(); + volatile boolean started; + volatile long startedNanos, finishedNanos; + volatile String state = "NOT_STARTED"; + volatile Throwable failure, futureFailure; + volatile boolean observed; + Future future; + Worker(int id) { this.id = id; } + boolean keepRunning() { return clock.getAsLong() - deadline < 0 && !Thread.currentThread().isInterrupted(); } + void succeeded() { operations.incrementAndGet(); } + } + private final List workers = new ArrayList<>(); + private final ExecutorService pool; + private final LongSupplier clock; + private final long deadline; + SoakWorkers(int count, long deadline, LongSupplier clock, Work work) { + this.deadline = deadline; + this.clock = clock; + pool = Executors.newFixedThreadPool(count, runnable -> { + Thread thread = new Thread(runnable, "tiercache-soak-worker"); + thread.setDaemon(true); // an uninterruptible failed worker cannot prevent the test JVM exiting + return thread; + }); + try { + for (int i = 0; i < count; i++) { + Worker worker = new Worker(i); + workers.add(worker); + worker.future = pool.submit(() -> { + worker.started = true; + worker.startedNanos = clock.getAsLong(); + worker.state = "RUNNING"; + try { + work.run(worker); + if (Thread.currentThread().isInterrupted()) throw new CancellationException("Worker interrupted before normal completion"); + if (clock.getAsLong() - deadline < 0) throw new AssertionError("Worker ended prematurely before the workload deadline"); + if (worker.operations.get() == 0) throw new AssertionError("Worker completed no successful operations"); + worker.state = "COMPLETED"; + } catch (Throwable failure) { + worker.failure = failure; + worker.state = "FAILED"; + if (failure instanceof Error error) throw error; + if (failure instanceof Exception exception) throw exception; + throw new AssertionError(failure); + } finally { worker.finishedNanos = clock.getAsLong(); } + return null; + }); + } + pool.shutdown(); + } catch (RuntimeException | Error failure) { close(); throw failure; } + } + void checkProgress() { + List failures = new ArrayList<>(); + for (Worker worker : workers) if (worker.future.isDone()) inspect(worker, failures); + requireHealthy(failures); + } + void awaitHealthy(Duration timeout) throws InterruptedException { + List failures = new ArrayList<>(); + if (!pool.awaitTermination(timeout.toNanos(), TimeUnit.NANOSECONDS)) { + failures.add(new TimeoutException("Soak workers did not terminate within " + timeout)); + for (Worker worker : workers) if (!worker.future.isDone()) worker.future.cancel(true); + pool.shutdownNow(); + } + // Includes cancelled futures; none may disappear merely because the pool terminated. + for (Worker worker : workers) inspect(worker, failures); + requireHealthy(failures); + } + private void inspect(Worker worker, List failures) { + try { + worker.future.get(); + if (!worker.started || worker.operations.get() == 0 || !"COMPLETED".equals(worker.state)) { + failures.add(new AssertionError("Incomplete soak worker " + worker.id + ": " + worker.state)); + } + } catch (ExecutionException e) { worker.futureFailure = e.getCause(); failures.add(e.getCause()); } + catch (CancellationException e) { worker.futureFailure = e; failures.add(e); } + catch (InterruptedException e) { Thread.currentThread().interrupt(); failures.add(e); } + finally { worker.observed = true; } + } + private static void requireHealthy(List failures) { + if (failures.isEmpty()) return; + AssertionError failure = new AssertionError("Incomplete or failed soak workload", failures.get(0)); + for (int i = 1; i < failures.size(); i++) failure.addSuppressed(failures.get(i)); + throw failure; + } + long operations() { return workers.stream().mapToLong(worker -> worker.operations.get()).sum(); } + List> snapshots() { + List> result = new ArrayList<>(); + for (Worker worker : workers) { + Map row = new LinkedHashMap<>(); + row.put("id", worker.id); row.put("started", worker.started); row.put("operations", worker.operations.get()); + row.put("state", worker.state); row.put("startedNanos", worker.startedNanos); row.put("finishedNanos", worker.finishedNanos); + row.put("futureDone", worker.future != null && worker.future.isDone()); + row.put("futureCancelled", worker.future != null && worker.future.isCancelled()); + row.put("futureObserved", worker.observed); + Throwable failure = worker.failure != null ? worker.failure : worker.futureFailure; + row.put("failure", failure == null ? null : failure.toString()); + result.add(row); + } + return result; + } + // Test seam for an externally cancelled workload. + void cancel(int worker) { workers.get(worker).future.cancel(true); } + boolean terminated() { return pool.isTerminated(); } + @Override public void close() { + for (Worker worker : workers) if (worker.future != null && !worker.future.isDone()) worker.future.cancel(true); + pool.shutdownNow(); + try { pool.awaitTermination(1, TimeUnit.SECONDS); } + catch (InterruptedException e) { Thread.currentThread().interrupt(); } + List observed = new ArrayList<>(); + for (Worker worker : workers) if (worker.future != null) inspect(worker, observed); + } +} diff --git a/tiercache-tck/src/vtStress/java/io/tiercache/tck/LockLifecycleJfrTest.java b/tiercache-tck/src/vtStress/java/io/tiercache/tck/LockLifecycleJfrTest.java new file mode 100644 index 0000000..b42616b --- /dev/null +++ b/tiercache-tck/src/vtStress/java/io/tiercache/tck/LockLifecycleJfrTest.java @@ -0,0 +1,55 @@ +package io.tiercache.tck; + +import io.lettuce.core.RedisClient; +import io.lettuce.core.RedisURI; +import io.lettuce.core.api.StatefulRedisConnection; +import io.tiercache.redis.LettuceLockProvider; +import io.tiercache.internal.LockProviderClosedException; +import jdk.jfr.Recording; +import jdk.jfr.consumer.RecordingFile; +import org.junit.jupiter.api.Test; +import org.testcontainers.containers.GenericContainer; +import org.testcontainers.utility.DockerImageName; +import java.time.Duration; +import java.nio.file.*; +import java.util.concurrent.*; +import java.util.concurrent.atomic.AtomicReference; +import static org.junit.jupiter.api.Assertions.*; + +class LockLifecycleJfrTest { + @Test void lazyConnectionCloseRaceDoesNotPinVirtualThreads() throws Exception { + var entered=new CountDownLatch(1); var resume=new CountDownLatch(1); + var connection=new AtomicReference>(); + try(var redis=new GenericContainer<>(DockerImageName.parse("redis:6.2-alpine")).withExposedPorts(6379)) { + redis.start(); + class PausedClient extends RedisClient { + PausedClient(){super(null,RedisURI.create("redis://"+redis.getHost()+":"+redis.getMappedPort(6379)));} + @Override public StatefulRedisConnection connect(){ + var c=super.connect(); connection.set(c); entered.countDown(); + try {assertTrue(resume.await(10,TimeUnit.SECONDS));}catch(InterruptedException e){throw new AssertionError(e);} + return c; + } + } + Path path=Path.of("build/reports/lock-lifecycle-jdk21.jfr");Files.createDirectories(path.getParent()); + try(var client=new PausedClient();var threads=Executors.newVirtualThreadPerTaskExecutor();var recording=new Recording()) { + recording.enable("jdk.VirtualThreadPinned").withThreshold(Duration.ZERO).withStackTrace(); + recording.setDestination(path);recording.start(); + var provider=new LettuceLockProvider(client); + var call=threads.submit(()->provider.tryLock("jfr-close",Duration.ofSeconds(10))); + try { + assertTrue(entered.await(10,TimeUnit.SECONDS)); + threads.submit(provider::close).get(1,TimeUnit.SECONDS); + } finally {resume.countDown();provider.close();} + assertInstanceOf(LockProviderClosedException.class, + assertThrows(ExecutionException.class,()->call.get(5,TimeUnit.SECONDS)).getCause()); + assertFalse(connection.get().isOpen());recording.stop(); + } + long pinned=RecordingFile.readAllEvents(path).stream() + .filter(e->e.getEventType().getName().equals("jdk.VirtualThreadPinned")) + .filter(e->e.getStackTrace()!=null && e.getStackTrace().getFrames().stream() + .anyMatch(f->f.getMethod().getType().getName().startsWith("io.tiercache.redis.LettuceLockProvider"))) + .count(); + assertEquals(0,pinned,"lazy connection/close must not hold an intrinsic monitor"); + } + } +} diff --git a/tiercache-tck/src/vtStress/java/io/tiercache/tck/RecoveryJfrTest.java b/tiercache-tck/src/vtStress/java/io/tiercache/tck/RecoveryJfrTest.java new file mode 100644 index 0000000..9242f51 --- /dev/null +++ b/tiercache-tck/src/vtStress/java/io/tiercache/tck/RecoveryJfrTest.java @@ -0,0 +1,120 @@ +package io.tiercache.tck; + +import io.lettuce.core.RedisClient; +import io.lettuce.core.codec.ByteArrayCodec; +import io.tiercache.*; +import io.tiercache.internal.CircuitBreaker; +import io.tiercache.invalidation.InvalidationService; +import io.tiercache.redis.*; +import io.tiercache.spi.*; +import jdk.jfr.Recording; +import jdk.jfr.consumer.RecordingFile; +import org.junit.jupiter.api.Test; +import org.testcontainers.containers.GenericContainer; +import org.testcontainers.utility.DockerImageName; +import com.sun.net.httpserver.HttpServer; + +import java.net.*; +import java.net.http.*; +import java.nio.file.*; +import java.time.Duration; +import java.util.*; +import java.util.concurrent.*; +import java.util.concurrent.atomic.*; +import static org.junit.jupiter.api.Assertions.*; + +/** Real Redis journal I/O, a virtual-thread HTTP probe and the registered reconnect callback. */ +class RecoveryJfrTest { + private static Object field(Object instance, String name) throws Exception { + var field = instance.getClass().getDeclaredField(name); field.setAccessible(true); return field.get(instance); + } + + @Test void slowJournalDoesNotPinOrBlockProbeAndReconnectCallback() throws Exception { + Path path = Path.of(System.getProperty("tiercache.recovery.jfr")); + Files.createDirectories(path.getParent()); + var entered = new CountDownLatch(1); var release = new CountDownLatch(1); + var reads = new AtomicInteger(); var held = new AtomicBoolean(); var virtualJournal = new AtomicBoolean(); + var serviceRef = new AtomicReference(); var breakerRef = new AtomicReference(); + try (var redis = new GenericContainer<>(DockerImageName.parse("redis:6.2-alpine")).withExposedPorts(6379)) { + redis.start(); String uri = "redis://" + redis.getHost() + ":" + redis.getMappedPort(6379); + try (var client = RedisClient.create(uri); var journalConnection = client.connect(ByteArrayCodec.INSTANCE); + var pauseConnection = client.connect(); + var remote = LettuceRemoteCache.builder(uri).client(client).cacheName("recovery-jfr").build(); + var virtual = Executors.newVirtualThreadPerTaskExecutor()) { + var journal = new RedisStreamJournal(journalConnection, 10000, new JdkCacheSerializer<>()); + InvalidationJournal observed = new InvalidationJournal() { + public String append(String c, InvalidationMessage m) { return journal.append(c, m); } + public List readRange(String c, String p) { return journal.readRange(c, p); } + public String endCursor(String c) { return journal.endCursor(c); } + public boolean isTrimmed(String c, String p) { return journal.isTrimmed(c, p); } + public CheckedRange checkedRead(String c, String p, int count) { + try { + Object state = ((Map) field(serviceRef.get(), "states")).get(c); + held.compareAndSet(false, Thread.holdsLock(state) || Thread.holdsLock(breakerRef.get())); + virtualJournal.compareAndSet(false, Thread.currentThread().isVirtual()); + if (reads.incrementAndGet() == 1) { + entered.countDown(); + if (!release.await(5, TimeUnit.SECONDS)) throw new AssertionError("journal gate timeout"); + } + // Delay the actual server read, not merely an in-memory journal call. + pauseConnection.sync().clientPause(25); + return journal.checkedRead(c, p, count); + } catch (Exception e) { throw new IllegalStateException(e); } + } + }; + var transport = new LettucePubSubInvalidationTransport(client, new JdkCacheSerializer<>()); + try (var factory = TierCacheFactory.builder().remoteCache(remote) + .circuitBreakerConfig(new CircuitBreaker.Config(2, 1, 1, Duration.ZERO, 1)) + .invalidation(v -> { + var service = new InvalidationService(transport, observed, v.instanceId(), InvalidationListener.NOOP); + serviceRef.set(service); return service; + }).build()) { + var cache = factory.getCache("c"); + CircuitBreaker breaker = (CircuitBreaker) field(factory, "breaker"); breakerRef.set(breaker); + remote.put("probe", StoredEntry.ofValue("available"), Duration.ofMinutes(1)); + UUID writer = UUID.randomUUID(); + for (int i = 1; i <= 1000; i++) journal.append("c", new InvalidationMessage("c", "k" + i, + new Version(i, writer), writer, InvalidationMessage.Type.INVALIDATE)); + HttpServer server = HttpServer.create(new InetSocketAddress(InetAddress.getLoopbackAddress(), 0), 0); + server.setExecutor(virtual); + server.createContext("/probe", exchange -> { + byte[] body = String.valueOf(cache.get("probe")).getBytes(java.nio.charset.StandardCharsets.UTF_8); + exchange.sendResponseHeaders(200, body.length); + try (var out = exchange.getResponseBody()) { out.write(body); } + }); + server.start(); + try (Recording recording = new Recording(); HttpClient http = HttpClient.newHttpClient()) { + recording.enable("jdk.VirtualThreadPinned").withThreshold(Duration.ZERO).withStackTrace(); + recording.setDestination(path); recording.start(); + breaker.onFailure(); + URI endpoint = new URI("http", null, server.getAddress().getAddress().getHostAddress(), server.getAddress().getPort(), "/probe", null, null); + var response = http.sendAsync(HttpRequest.newBuilder(endpoint).GET().build(), HttpResponse.BodyHandlers.ofString()); + try { + assertTrue(entered.await(5, TimeUnit.SECONDS)); + assertEquals("available", response.get(1, TimeUnit.SECONDS).body()); + assertEquals(BreakerState.HALF_OPEN, breaker.state()); + Runnable reconnect = (Runnable) field(transport, "reconnectListener"); + virtual.submit(reconnect).get(1, TimeUnit.SECONDS); + assertFalse(held.get()); assertFalse(virtualJournal.get()); + } finally { release.countDown(); } + long until = System.nanoTime() + Duration.ofSeconds(10).toNanos(); + while (breaker.state() != BreakerState.CLOSED && System.nanoTime() < until) Thread.sleep(10); + assertEquals(BreakerState.CLOSED, breaker.state()); assertTrue(reads.get() >= 5); + recording.stop(); + } finally { release.countDown(); server.stop(0); } + } + } + } + var pinned = RecordingFile.readAllEvents(path).stream() + .filter(e -> e.getEventType().getName().equals("jdk.VirtualThreadPinned")) + .filter(e -> e.getStackTrace() != null && e.getStackTrace().getFrames().stream().anyMatch(f -> { + String name = f.getMethod().getType().getName(); + return name.startsWith("io.tiercache.internal.") || name.startsWith("io.tiercache.invalidation.") + || name.startsWith("io.tiercache.redis.") || name.equals("io.tiercache.TierCacheFactory"); + })).toList(); + System.out.println("Recovery JFR: JDK=" + Runtime.version().feature() + ", real journal reads=" + reads.get() + + ", library pinning=" + pinned.size() + ", recording=" + path); + assertTrue(pinned.isEmpty(), () -> "library monitor pinning: " + pinned); + assertFalse(held.get()); + } +} diff --git a/tiercache-transport-redis/build.gradle.kts b/tiercache-transport-redis/build.gradle.kts index 3680647..2203713 100644 --- a/tiercache-transport-redis/build.gradle.kts +++ b/tiercache-transport-redis/build.gradle.kts @@ -1,3 +1,6 @@ +import java.util.Properties +import java.time.Duration + plugins { `java-library` alias(libs.plugins.vanniktech.publish) @@ -13,6 +16,8 @@ dependencies { api(project(":tiercache-core")) api(project(":tiercache-invalidation")) api(libs.lettuce.core) + // Publish alignment as well as using it locally; all Netty modules share the patched line. + api(platform(libs.netty.bom)) testImplementation(platform(libs.junit.bom)) testImplementation(libs.junit.jupiter) @@ -22,8 +27,15 @@ dependencies { testImplementation(libs.testcontainers.junit.jupiter) } +val serverProfiles = Properties().apply { + rootProject.file("compatibility/platforms.properties").inputStream().use { load(it) } +} tasks.withType { useJUnitPlatform() + val serverImage = providers.gradleProperty("serverImage").orElse(serverProfiles.getProperty("redis62")) + inputs.property("serverImage", serverImage) + systemProperty("tiercache.test.serverImage", serverImage.get()) + systemProperty("tiercache.test.valkeyImage", serverProfiles.getProperty("valkey")) } // --- Publishing (release automation): shared Central Portal target, license, @@ -37,3 +49,12 @@ mavenPublishing { ) } } + +// Run the same real-server contracts for each CI profile, with separate reports. +tasks.register("serverContractTest") { + testClassesDirs = sourceSets.test.get().output.classesDirs + classpath = sourceSets.test.get().runtimeClasspath + exclude("**/ValkeyLettuceContractTest.class") + outputs.upToDateWhen { false } + timeout.set(Duration.ofMinutes(15)) +} diff --git a/tiercache-transport-redis/src/main/java/io/tiercache/redis/LettuceLockProvider.java b/tiercache-transport-redis/src/main/java/io/tiercache/redis/LettuceLockProvider.java index 100e222..527cdd9 100644 --- a/tiercache-transport-redis/src/main/java/io/tiercache/redis/LettuceLockProvider.java +++ b/tiercache-transport-redis/src/main/java/io/tiercache/redis/LettuceLockProvider.java @@ -15,7 +15,13 @@ import java.util.concurrent.Executors; import java.util.concurrent.ScheduledExecutorService; import java.util.concurrent.TimeUnit; -import java.util.concurrent.atomic.AtomicInteger; +import java.util.concurrent.CompletableFuture; +import java.util.concurrent.CompletionException; +import java.util.concurrent.RejectedExecutionException; +import java.util.concurrent.locks.ReentrantLock; +import java.util.HashSet; +import java.util.Set; +import io.tiercache.internal.LockProviderClosedException; /** * {@link DistributedLockProvider} over Redis/Valkey via Lettuce, used for @@ -24,7 +30,7 @@ *

    Acquire: {@code SET name token PX lease NX}. Release: Lua * compare-and-delete on the ownership token, so a stale holder can never * release another's lock. Extend: Lua token-checked {@code PEXPIRE ... XX}. - * Locks live in the {@code tiercache:rebuild:*} keyspace, separate from data + * Locks live in the {@code tiercache:v2:rebuild:*} keyspace, separate from data * entries. * *

    Ambiguous acquire compensation. A failed acquire (client-side @@ -54,7 +60,7 @@ public final class LettuceLockProvider implements DistributedLockProvider, AutoC * * @since 0.1.0 */ - public static final String LOCK_KEYSPACE = "tiercache:rebuild:"; + public static final String LOCK_KEYSPACE = RedisKeyspace.LOCK; private static final String RELEASE_SCRIPT = "if redis.call('get', KEYS[1]) == ARGV[1] then return redis.call('del', KEYS[1])" @@ -78,8 +84,10 @@ public final class LettuceLockProvider implements DistributedLockProvider, AutoC private final RedisClient client; // non-null ⇒ connect lazily private final StatefulRedisConnection providedConnection; private volatile RedisCommands commands; - private final AtomicInteger pendingCompensations = new AtomicInteger(); - private final Object schedulerLock = new Object(); + private final ReentrantLock lifecycle = new ReentrantLock(); + private final Set compensations = new HashSet<>(); + private StatefulRedisConnection ownedConnection; + private CompletableFuture> initializing; private volatile ScheduledExecutorService compensationScheduler; private volatile boolean closed; @@ -110,175 +118,189 @@ public LettuceLockProvider(StatefulRedisConnection connection) { } private RedisCommands commands() { - RedisCommands resolved = commands; - if (resolved == null) { - synchronized (schedulerLock) { - resolved = commands; - if (resolved == null) { - resolved = (providedConnection != null ? providedConnection - : client.connect()).sync(); - commands = resolved; - } + CompletableFuture> attempt; + boolean creator = false; + lifecycle.lock(); + try { + if (closed) throw new LockProviderClosedException(); + if (commands != null) return commands; + attempt = initializing; + if (attempt == null) { + initializing = attempt = new CompletableFuture<>(); + creator = true; + } + } finally { lifecycle.unlock(); } + if (creator) { + StatefulRedisConnection created = null; + try { + var connection = providedConnection; + if (connection == null) connection = created = client.connect(); + var resolved = connection.sync(); + boolean accepted; + lifecycle.lock(); + try { + accepted = !closed && initializing == attempt; + if (accepted) { + commands = resolved; + ownedConnection = created; + created = null; // ownership transferred to close + initializing = null; + } + } finally { lifecycle.unlock(); } + if (accepted) attempt.complete(resolved); + else attempt.completeExceptionally(new LockProviderClosedException()); + } catch (RuntimeException | Error e) { + lifecycle.lock(); + try { if (initializing == attempt) initializing = null; } + finally { lifecycle.unlock(); } + attempt.completeExceptionally(e); + } finally { + if (created != null) closeConnection(created); } } - return resolved; + try { return attempt.join(); } + catch (CompletionException e) { + if (e.getCause() instanceof RuntimeException runtime) throw runtime; + if (e.getCause() instanceof Error error) throw error; + throw e; + } } @Override public DistributedLock tryLock(String name, Duration lease) { + var captured = commands(); String token = UUID.randomUUID().toString(); - String key = LOCK_KEYSPACE + name; + String key = RedisKeyspace.lock(name); + if (closed) throw new LockProviderClosedException(); // dispatch admission + String result; try { - String result = commands().set(key, token, SetArgs.Builder.px(lease).nx()); - return "OK".equals(result) ? new LettuceLock(key, token) : null; + result = captured.set(key, token, SetArgs.Builder.px(lease).nx()); } catch (RuntimeException e) { - // Ambiguous outcome: the command may have executed server-side. - scheduleCompensation(key, token, lease); + scheduleCompensation(captured, key, token, lease); throw e; } - } - - /** - * Schedules a best-effort compensating release for an ambiguous - * acquire: retried until DELETE=1 (terminal) or the window expires — - * never stopped by a zero delete. Overflow is dropped with a warning; - * a task that never gets an attempt within the window expires into the - * documented residual (an orphan self-expiring within one lease). - */ - private void scheduleCompensation(String key, String token, Duration lease) { if (closed) { - // No machinery is created and no retry is accepted after close; - // the caller still sees the original acquire exception. - return; - } - long windowMillis = Math.max(2 * lease.toMillis(), COMPENSATION_MIN_WINDOW.toMillis()); - long deadlineNanos = System.nanoTime() + TimeUnit.MILLISECONDS.toNanos(windowMillis); - if (pendingCompensations.incrementAndGet() > COMPENSATION_PENDING_CAP) { - pendingCompensations.decrementAndGet(); - log.warn("Too many pending lock compensations; dropping cleanup for '{}'. " - + "If the acquire executed, the orphan expires within its lease.", key); - return; - } - try { - scheduleCompensationAttempt(key, token, deadlineNanos); - } catch (java.util.concurrent.RejectedExecutionException e) { - // The scheduler shut down between the checks and the schedule: - // best-effort must never replace the original acquire failure. - pendingCompensations.decrementAndGet(); - log.debug("Lock compensation scheduling raced provider close for '{}'", key, e); + if ("OK".equals(result)) { + try { captured.eval(RELEASE_SCRIPT, ScriptOutputType.INTEGER, new String[]{key}, token); } + catch (RuntimeException ignored) { /* Lease bounds failed shutdown cleanup. */ } + } + throw new LockProviderClosedException(); } + return "OK".equals(result) ? new LettuceLock(key, token, captured) : null; } - private void scheduleCompensationAttempt(String key, String token, long deadlineNanos) { + private void scheduleCompensation(RedisCommands captured, + String key, String token, Duration lease) { + long window = Math.max(2 * lease.toMillis(), COMPENSATION_MIN_WINDOW.toMillis()); + var task = new Compensation(captured, key, token, + System.nanoTime() + TimeUnit.MILLISECONDS.toNanos(window)); + lifecycle.lock(); try { - compensationScheduler().schedule(() -> attemptCompensation(key, token, deadlineNanos), - COMPENSATION_RETRY_MILLIS, TimeUnit.MILLISECONDS); - } catch (java.util.concurrent.RejectedExecutionException e) { - // Closed between the checks and the (re)schedule: release the - // bookkeeping; best-effort stays silent. - pendingCompensations.decrementAndGet(); - log.debug("Lock compensation scheduling raced provider close for '{}'", key, e); - } + if (closed) return; + if (compensations.size() >= COMPENSATION_PENDING_CAP) { + log.warn("Too many pending lock compensations; cleanup falls back to lease expiry"); + return; + } + if (compensationScheduler == null) { + compensationScheduler = Executors.newScheduledThreadPool(COMPENSATION_THREADS, r -> { + Thread t = new Thread(r, "tiercache-lock-compensation"); t.setDaemon(true); return t; + }); + } + compensations.add(task); + task.scheduleLocked(); + } finally { lifecycle.unlock(); } } - private void attemptCompensation(String key, String token, long deadlineNanos) { - if (closed) { - pendingCompensations.decrementAndGet(); - return; - } - Long deleted = null; - try { - deleted = commands().eval(RELEASE_SCRIPT, ScriptOutputType.INTEGER, - new String[]{key}, token); - } catch (RuntimeException e) { - // Server still unreachable: retry within the window. + private final class Compensation implements Runnable { + final RedisCommands captured; + final String key, token; + final long deadline; + Compensation(RedisCommands captured, String key, String token, long deadline) { + this.captured = captured; this.key = key; this.token = token; this.deadline = deadline; } - if (deleted != null && deleted == 1L) { - pendingCompensations.decrementAndGet(); - return; // terminal: our lock existed and was removed + void scheduleLocked() { + if (closed || !compensations.contains(this)) { compensations.remove(this); return; } + try { compensationScheduler.schedule(this, COMPENSATION_RETRY_MILLIS, TimeUnit.MILLISECONDS); } + catch (RejectedExecutionException e) { compensations.remove(this); } } - if (System.nanoTime() >= deadlineNanos) { - pendingCompensations.decrementAndGet(); - log.warn("Lock compensation window expired for '{}'. If the acquire executed " - + "afterwards, the orphan expires within its lease.", key); - return; + @Override public void run() { + lifecycle.lock(); + try { + if (closed || !compensations.contains(this)) return; + if (System.nanoTime() >= deadline) { compensations.remove(this); return; } + } finally { lifecycle.unlock(); } + Long deleted = null; + try { deleted = captured.eval(RELEASE_SCRIPT, ScriptOutputType.INTEGER, new String[]{key}, token); } + catch (RuntimeException ignored) { /* Retry ambiguous cleanup within the window. */ } + lifecycle.lock(); + try { + if (closed || Long.valueOf(1).equals(deleted) || System.nanoTime() >= deadline) { + compensations.remove(this); + } else scheduleLocked(); + } finally { lifecycle.unlock(); } } - scheduleCompensationAttempt(key, token, deadlineNanos); } - private ScheduledExecutorService compensationScheduler() { - ScheduledExecutorService scheduler = compensationScheduler; - if (scheduler == null) { - synchronized (schedulerLock) { - scheduler = compensationScheduler; - if (scheduler == null && !closed) { - scheduler = Executors.newScheduledThreadPool(COMPENSATION_THREADS, runnable -> { - Thread thread = new Thread(runnable, "tiercache-lock-compensation"); - thread.setDaemon(true); - return thread; - }); - compensationScheduler = scheduler; - } - } - } - if (scheduler == null) { - throw new java.util.concurrent.RejectedExecutionException( - "lock provider is closed"); - } - return scheduler; - } - - /** - * Stops the compensation machinery. Pending compensations are - * discarded; the scheduler is shut down. The Redis connection is NOT - * closed (the caller keeps its ownership). - * - * @since 1.4.0 - */ + /** Stops auxiliary work and closes only the connection created by this provider. */ @Override public void close() { ScheduledExecutorService scheduler; - // The closed transition and the scheduler lookup happen under the - // same lifecycle lock as creation/publication: no scheduler can be - // published after a completed close. - synchronized (schedulerLock) { + StatefulRedisConnection connection; + CompletableFuture> attempt; + lifecycle.lock(); + try { + if (closed) return; closed = true; scheduler = compensationScheduler; - } - if (scheduler != null) { - scheduler.shutdownNow(); - } + connection = ownedConnection; + ownedConnection = null; + commands = null; + attempt = initializing; + initializing = null; + compensations.clear(); + } finally { lifecycle.unlock(); } + if (attempt != null) attempt.completeExceptionally(new LockProviderClosedException()); + if (scheduler != null) scheduler.shutdownNow(); + if (connection != null) closeConnection(connection); + } + + private static void closeConnection(StatefulRedisConnection connection) { + try { connection.close(); } + catch (RuntimeException e) { log.warn("Owned lock connection close failed ({})", e.getClass().getSimpleName()); } } /** Pending compensations (test/diagnostics seam). */ int pendingCompensations() { - return pendingCompensations.get(); + lifecycle.lock(); + try { return compensations.size(); } + finally { lifecycle.unlock(); } } - /** True after {@link #close()} (test/diagnostics seam). */ - boolean isClosed() { - return closed; - } + boolean isClosed() { return closed; } private final class LettuceLock implements DistributedLock { private final String fullName; private final String token; - LettuceLock(String fullName, String token) { + private final RedisCommands captured; + + LettuceLock(String fullName, String token, RedisCommands captured) { + this.captured = captured; this.fullName = fullName; this.token = token; } @Override public boolean extend(Duration lease) { - Long extended = commands().eval(EXTEND_SCRIPT, ScriptOutputType.INTEGER, + if (closed) return false; + Long extended = captured.eval(EXTEND_SCRIPT, ScriptOutputType.INTEGER, new String[]{fullName}, token, String.valueOf(lease.toMillis())); return extended != null && extended == 1L; } @Override public void release() { - commands().eval(RELEASE_SCRIPT, ScriptOutputType.INTEGER, + captured.eval(RELEASE_SCRIPT, ScriptOutputType.INTEGER, new String[]{fullName}, token); } } diff --git a/tiercache-transport-redis/src/main/java/io/tiercache/redis/LettucePubSubInvalidationTransport.java b/tiercache-transport-redis/src/main/java/io/tiercache/redis/LettucePubSubInvalidationTransport.java index cbf2df1..2c9687a 100644 --- a/tiercache-transport-redis/src/main/java/io/tiercache/redis/LettucePubSubInvalidationTransport.java +++ b/tiercache-transport-redis/src/main/java/io/tiercache/redis/LettucePubSubInvalidationTransport.java @@ -10,7 +10,6 @@ import io.tiercache.invalidation.MessageCodec; import io.tiercache.spi.InvalidationTransport; -import java.nio.charset.StandardCharsets; import java.util.Map; import java.util.UUID; import java.util.concurrent.ConcurrentHashMap; @@ -21,7 +20,7 @@ /** * Default invalidation transport profile: Redis Pub/Sub (minimal latency). * One Pub/Sub connection per instance; per-cache channels - * ({@code tiercache:inv:}); inbound events are dispatched on a + * ({@code tiercache:v2:inv:}); inbound events are dispatched on a * single daemon executor, off the I/O thread, preserving per-channel order. * *

    On reconnect (detected via the Lettuce event bus) the registered @@ -39,7 +38,7 @@ public final class LettucePubSubInvalidationTransport implements InvalidationTra * * @since 0.1.0 */ - public static final String CHANNEL_PREFIX = "tiercache:inv:"; + public static final String CHANNEL_PREFIX = RedisKeyspace.CHANNEL; private final RedisClient client; private final CacheSerializer keySerializer; @@ -112,33 +111,41 @@ public void onRedisConnected(RedisChannelHandler connection, } @Override - public void publish(InvalidationMessage message) { - byte[] keyBytes = message.key() != null ? keySerializer.toBytes(message.key()) : null; - InvalidationMessage toSend = message; - if (message.type() == InvalidationMessage.Type.UPDATE) { - byte[] payloadBytes = valueSerializer.toBytes(message.payload()); - if (payloadBytes.length > payloadCapBytes) { - // Oversized payload: degrade to plain INVALIDATE. - toSend = new InvalidationMessage(message.cache(), message.key(), - message.version(), message.originInstanceId(), - InvalidationMessage.Type.INVALIDATE); - } else { - toSend = new InvalidationMessage(message.cache(), message.key(), - message.version(), message.originInstanceId(), - message.type(), payloadBytes); + public void publish(InvalidationMessage message) { publishAsync(message); } + + @Override + public java.util.concurrent.CompletionStage publishAsync(InvalidationMessage message) { + try { + byte[] keyBytes = message.key() != null ? keySerializer.toBytes(message.key()) : null; + InvalidationMessage toSend = message; + if (message.type() == InvalidationMessage.Type.UPDATE) { + byte[] payloadBytes = valueSerializer.toBytes(message.payload()); + if (payloadBytes.length > payloadCapBytes) { + // Oversized payload: degrade to plain INVALIDATE. + toSend = new InvalidationMessage(message.cache(), message.key(), + message.version(), message.originInstanceId(), + InvalidationMessage.Type.INVALIDATE); + } else { + toSend = new InvalidationMessage(message.cache(), message.key(), + message.version(), message.originInstanceId(), + message.type(), payloadBytes); + } } + return connection.async().publish(channelName(message.cache()), + MessageCodec.encode(toSend, keyBytes)).thenApply(ignored -> io.tiercache.spi.PublicationOutcome.ACKNOWLEDGED); + } catch (RuntimeException e) { + return java.util.concurrent.CompletableFuture.failedFuture(e); } - connection.async().publish(channelName(message.cache()), - MessageCodec.encode(toSend, keyBytes)); } @Override public AutoCloseable subscribe(String cache, Consumer handler) { + byte[] channel = channelName(cache); // validate before registering a handler handlers.put(cache, handler); - connection.sync().subscribe(channelName(cache)); + connection.sync().subscribe(channel); return () -> { handlers.remove(cache); - connection.async().unsubscribe(channelName(cache)); + connection.async().unsubscribe(channel); }; } @@ -156,7 +163,7 @@ private InvalidationMessage toMessage(MessageCodec.Decoded decoded) { } private static byte[] channelName(String cache) { - return (CHANNEL_PREFIX + cache).getBytes(StandardCharsets.UTF_8); + return RedisKeyspace.channel(cache); } @Override diff --git a/tiercache-transport-redis/src/main/java/io/tiercache/redis/LettuceRemoteCache.java b/tiercache-transport-redis/src/main/java/io/tiercache/redis/LettuceRemoteCache.java index 6631fd6..03277c0 100644 --- a/tiercache-transport-redis/src/main/java/io/tiercache/redis/LettuceRemoteCache.java +++ b/tiercache-transport-redis/src/main/java/io/tiercache/redis/LettuceRemoteCache.java @@ -14,6 +14,7 @@ import io.tiercache.spi.LockProviderSource; import io.tiercache.spi.RemoteCache; import io.tiercache.spi.StoredEntry; +import io.tiercache.spi.TaggedWriteOutcome; import java.nio.charset.StandardCharsets; import java.time.Duration; @@ -90,6 +91,13 @@ public final class LettuceRemoteCache implements RemoteCache, LockPr private long payloadCapBytes = 64 * 1024; private LettuceRemoteCache(Builder builder) { + this.cacheName = builder.cacheName; + this.journalName = builder.journalName != null ? builder.journalName : builder.cacheName; + this.keyPrefix = RedisKeyspace.dataPrefix(cacheName); + // Validate all configured names before opening connections or issuing commands. + RedisKeyspace.tagPrefix(cacheName); + RedisKeyspace.journal(journalName); + RedisKeyspace.trims(journalName); this.ownsClient = builder.sharedClient == null; this.client = ownsClient ? RedisClient.create(builder.redisUri) : builder.sharedClient; if (ownsClient) { @@ -103,9 +111,6 @@ private LettuceRemoteCache(Builder builder) { } this.connection = client.connect(ByteArrayCodec.INSTANCE); this.commands = connection.sync(); - this.cacheName = builder.cacheName; - this.journalName = builder.journalName != null ? builder.journalName : builder.cacheName; - this.keyPrefix = (builder.cacheName + ":").getBytes(StandardCharsets.UTF_8); this.keySerializer = builder.keySerializer; this.valueSerializer = builder.valueSerializer; this.journal = builder.journal; @@ -201,8 +206,9 @@ private boolean putIfNewerInternal(K key, StoredEntry entry, Duration ttl, Du @Override public void evict(K key) { byte[] namespaced = namespaced(key); + byte[] reverse = keyTagsKey(namespaced); // validate before deleting data commands.del(namespaced); - pruneTags(namespaced); + pruneTags(namespaced, reverse); } /** @@ -216,7 +222,7 @@ public void evict(K key, Version version) { return; } byte[] namespaced = namespaced(key); - pruneTags(namespaced); + pruneTags(namespaced, keyTagsKey(namespaced)); commands.eval(Lua.VERSIONED_EVICT, io.lettuce.core.ScriptOutputType.INTEGER, new byte[][]{namespaced, RedisStreamJournal.streamKeyBytes(journalName), RedisStreamJournal.trimCounterKeyBytes(journalName)}, @@ -313,9 +319,9 @@ private byte[] encode(StoredEntry entry, boolean withWriteTimestamp) { return ValueFrame.encode(entry, valueSerializer, withWriteTimestamp); } - // --- Tag registry: tiercache:tags:: sets + reverse index - // tiercache:tagkeys::. Redis sets have no per-member TTL, - // so boundedness comes from two mechanisms (Lua.TAG_ADD/TAG_LIVE_MEMBERS): + // --- V2 tag sets contain full data keys; reverse indexes contain tag tokens. + // Redis sets have no per-member TTL, + // so boundedness comes from two mechanisms (Lua.TAGGED_WRITE/TAG_LIVE_MEMBERS): // each tag set carries an extend-only TTL (a shorter-lived entry never // shrinks the index, so a set outlives its longest member by a bounded // margin and then expires — including tags never touched again), and @@ -327,36 +333,60 @@ private byte[] encode(StoredEntry entry, boolean withWriteTimestamp) { private static final int TAG_JANITOR_SAMPLE = 8; private byte[] tagSetKey(String tag) { - return ("tiercache:tags:" + cacheName + ":" + tag).getBytes(StandardCharsets.UTF_8); + return RedisKeyspace.tagKey(cacheName, tag); } private byte[] keyTagsKey(byte[] namespacedKey) { - byte[] prefix = ("tiercache:tagkeys:").getBytes(StandardCharsets.UTF_8); - byte[] out = new byte[prefix.length + namespacedKey.length]; - System.arraycopy(prefix, 0, out, 0, prefix.length); - System.arraycopy(namespacedKey, 0, out, prefix.length, namespacedKey.length); - return out; + return RedisKeyspace.reverseKey(cacheName, + java.util.Arrays.copyOfRange(namespacedKey, keyPrefix.length, namespacedKey.length)); } @Override public void putTagged(K key, StoredEntry entry, Duration ttl, String[] tags) { - put(key, entry, ttl); - byte[] namespaced = namespaced(key); - byte[] keyTags = keyTagsKey(namespaced); - if (tags.length > 0) { - long ttlMillis = Math.max(1, ttl.toMillis()); - commands.del(keyTags); - commands.sadd(keyTags, java.util.Arrays.stream(tags) - .map(t -> t.getBytes(StandardCharsets.UTF_8)).toArray(byte[][]::new)); - commands.pexpire(keyTags, ttlMillis); - for (String tag : tags) { - commands.eval(Lua.TAG_ADD, io.lettuce.core.ScriptOutputType.INTEGER, - new byte[][]{tagSetKey(tag)}, - namespaced, - String.valueOf(ttlMillis).getBytes(StandardCharsets.UTF_8), - String.valueOf(TAG_JANITOR_SAMPLE).getBytes(StandardCharsets.UTF_8)); - } + putTaggedIfNewer(key, entry, ttl, tags); + } + + @Override + public boolean supportsTaggedWriteOutcomes() { + return true; + } + + @Override + public TaggedWriteOutcome putTaggedIfNewer(K key, StoredEntry entry, + Duration ttl, String[] tags) { + java.util.Objects.requireNonNull(entry, "entry"); + java.util.Objects.requireNonNull(tags, "tags"); + long ttlMillis = ttl.toMillis(); + if (ttlMillis <= 0) { + throw new IllegalArgumentException("Tagged entry TTL must be at least one millisecond"); + } + // Fully encode/validate before issuing any mutating Redis operation. + byte[] rawKey = keySerializer.toBytes(key); + byte[] namespaced = prefixKey(rawKey); + boolean conditional = journal != null && entry.version() != null; + java.util.List args = new java.util.ArrayList<>(); + args.add(entry.version() == null ? new byte[0] + : entry.version().toWire().getBytes(StandardCharsets.UTF_8)); + args.add(encode(entry, false)); + args.add(Long.toString(ttlMillis).getBytes(StandardCharsets.UTF_8)); + args.add(Integer.toString(journal == null ? 0 : journal.capacity()).getBytes(StandardCharsets.UTF_8)); + args.add(new byte[]{(byte) InvalidationMessage.Type.INVALIDATE.ordinal()}); + args.add(rawKey); + args.add(conditional ? journalPayload(entry) : new byte[0]); + args.add(new byte[]{(byte) (conditional ? '1' : '0')}); + args.add(RedisKeyspace.tagPrefix(cacheName)); + args.add(Integer.toString(TAG_JANITOR_SAMPLE).getBytes(StandardCharsets.UTF_8)); + for (String tag : new java.util.LinkedHashSet<>(java.util.Arrays.asList(tags))) { + byte[] tagToken = RedisKeyspace.token(java.util.Objects.requireNonNull(tag, "tag")) + .getBytes(StandardCharsets.US_ASCII); + RedisKeyspace.checkLength((long) args.get(8).length + tagToken.length); + args.add(tagToken); } + Long result = commands.eval(Lua.TAGGED_WRITE, io.lettuce.core.ScriptOutputType.INTEGER, + new byte[][]{namespaced, RedisStreamJournal.streamKeyBytes(journalName), + RedisStreamJournal.trimCounterKeyBytes(journalName), keyTagsKey(namespaced)}, + args.toArray(new byte[0][])); + return result != null && result == 1L ? TaggedWriteOutcome.WON : TaggedWriteOutcome.LOST; } @Override @@ -373,29 +403,28 @@ public java.util.List keysByTag(String tag) { } /** Removes tag bookkeeping for an evicted key. */ - private void pruneTags(byte[] namespaced) { - byte[] keyTags = keyTagsKey(namespaced); + private void pruneTags(byte[] namespaced, byte[] keyTags) { java.util.Set tags = commands.smembers(keyTags); if (tags != null && !tags.isEmpty()) { for (byte[] tag : tags) { - commands.srem(tagSetKey(new String(tag, StandardCharsets.UTF_8)), namespaced); + commands.srem(RedisKeyspace.join(RedisKeyspace.tagPrefix(cacheName), tag), namespaced); } commands.del(keyTags); } } private byte[] namespaced(K key) { - byte[] serialized = keySerializer.toBytes(key); - byte[] out = new byte[keyPrefix.length + serialized.length]; - System.arraycopy(keyPrefix, 0, out, 0, keyPrefix.length); - System.arraycopy(serialized, 0, out, keyPrefix.length, serialized.length); - return out; + return prefixKey(keySerializer.toBytes(key)); + } + + private byte[] prefixKey(byte[] serialized) { + return RedisKeyspace.join(keyPrefix, serialized); } @Override public DistributedLockProvider lockProvider() { // Shares the client; the provider owns its String-codec connection. - return new LettuceLockProvider(client.connect()); + return new LettuceLockProvider(client); } /** The connection used by this transport (for journal/wiring sharing). */ @@ -477,25 +506,47 @@ private static final class Lua { .replace("@FIELDS", "'t', ARGV[4], 'k', ARGV[5], 'v', ARGV[6], 'p', ARGV[7]") + "return 1"; - /** - * Tag write: add the member, then lift the set TTL extend-only - * (PTTL -1/-2 both fall below any real TTL, so the comparison - * covers them; a shorter-lived entry never shrinks the index — - * the 6.2-compatible emulation of PEXPIRE GT). The janitor runs - * in the same call: a random member sample, each SREM conditional - * on the data key being absent — atomically, so a concurrently - * rewritten key keeps its membership. - */ - static final String TAG_ADD = - "redis.call('sadd', KEYS[1], ARGV[1]) " - + "local ttl = tonumber(ARGV[2]) " - + "if redis.call('pttl', KEYS[1]) < ttl then " - + "redis.call('pexpire', KEYS[1], ttl) end " - + "local candidates = redis.call('srandmember', KEYS[1], tonumber(ARGV[3])) " - + "for _, m in ipairs(candidates) do " - + "if redis.call('exists', m) == 0 then redis.call('srem', KEYS[1], m) end " - + "end " - + "return 1"; + /** One acceptance decision for data, replacement tags and optional journal. */ + static final String TAGGED_WRITE = VERSION_COMPARE + """ + if ARGV[8] == '1' then + local curVer = curVersion(redis.call('get', KEYS[1])) + if curVer and newer(curVer, ARGV[1]) then return 0 end + end + local oldTags = redis.call('smembers', KEYS[4]) + local wanted = {} + for i = 11, #ARGV do wanted[ARGV[i]] = true end + local ttl = tonumber(ARGV[3]) + redis.replicate_commands() + local clock = redis.call('time') + local deadline = clock[1] * 1000 + math.floor(clock[2] / 1000) + ttl + redis.call('set', KEYS[1], ARGV[2], 'PXAT', deadline) + for _, tag in ipairs(oldTags) do + if not wanted[tag] then + redis.call('srem', ARGV[9] .. tag, KEYS[1]) + end + end + redis.call('del', KEYS[4]) + for i = 11, #ARGV do + local tag = ARGV[i] + local tagKey = ARGV[9] .. tag + redis.call('sadd', KEYS[4], tag) + redis.call('sadd', tagKey, KEYS[1]) + if redis.call('pttl', tagKey) < ttl then + redis.call('pexpire', tagKey, ttl) + end + local candidates = redis.call('srandmember', tagKey, tonumber(ARGV[10])) + for _, member in ipairs(candidates) do + if redis.call('exists', member) == 0 then + redis.call('srem', tagKey, member) + end + end + end + if #ARGV >= 11 then redis.call('pexpireat', KEYS[4], deadline) end + if ARGV[8] == '1' then + """ + + TRIM_COUNTED_XADD.replace("@CAP", "ARGV[4]") + .replace("@FIELDS", "'t', ARGV[5], 'k', ARGV[6], 'v', ARGV[1], 'p', ARGV[7]") + + "end return 1"; /** * Tag read: returns the live members only; dead members are diff --git a/tiercache-transport-redis/src/main/java/io/tiercache/redis/LettuceStreamsInvalidationTransport.java b/tiercache-transport-redis/src/main/java/io/tiercache/redis/LettuceStreamsInvalidationTransport.java index d261130..2cdac59 100644 --- a/tiercache-transport-redis/src/main/java/io/tiercache/redis/LettuceStreamsInvalidationTransport.java +++ b/tiercache-transport-redis/src/main/java/io/tiercache/redis/LettuceStreamsInvalidationTransport.java @@ -1,283 +1,307 @@ package io.tiercache.redis; -import io.lettuce.core.RedisClient; -import io.lettuce.core.StreamMessage; -import io.lettuce.core.XAutoClaimArgs; -import io.lettuce.core.XGroupCreateArgs; -import io.lettuce.core.XReadArgs; +import io.lettuce.core.*; import io.lettuce.core.api.StatefulRedisConnection; import io.lettuce.core.codec.ByteArrayCodec; import io.tiercache.InvalidationMessage; -import io.tiercache.Version; -import io.tiercache.spi.InvalidationTransport; +import io.tiercache.spi.*; +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; import java.nio.charset.StandardCharsets; -import java.util.List; -import java.util.Map; -import java.util.UUID; -import java.util.concurrent.ConcurrentHashMap; -import java.util.concurrent.ExecutorService; -import java.util.concurrent.Executors; -import java.util.concurrent.TimeUnit; +import java.util.*; +import java.util.concurrent.*; +import java.util.function.Consumer; /** - * Durable invalidation transport profile: receivers consume the per-cache - * journal stream through a consumer group PER INSTANCE (groups distribute, - * they do not fan out — one group per instance is the broadcast shape). - * The group cursor survives disconnects, so arbitrarily long partitions - * (within stream retention) heal without a full L1 flush. - * - *

    Internal — not part of the supported API. - * - * @since 0.1.0 + * Durable per-receiver Streams consumption. Pending rows are drained before + * new rows; unreadable history requires an authorized baseline-before-clear. + * Stable identities require one live owner. Internal transport API. */ public final class LettuceStreamsInvalidationTransport implements InvalidationTransport { - - /** - * Consumer-group keyspace: one group per instance per cache. - * - * @since 0.1.0 - */ - public static final String GROUP_PREFIX = "tiercache:cg:"; - + public static final String GROUP_PREFIX = RedisKeyspace.GROUP; + private static final Logger log = LoggerFactory.getLogger(LettuceStreamsInvalidationTransport.class); + private static final int BATCH = 50; + private static final byte[] MAIN = "main".getBytes(StandardCharsets.US_ASCII); private final StatefulRedisConnection connection; - private final io.lettuce.core.api.sync.RedisCommands commands; - private final CacheSerializer keySerializer; - private final CacheSerializer valueSerializer; + private final CacheSerializer keySerializer, valueSerializer; private final UUID instanceId; - private final Map> handlers = new ConcurrentHashMap<>(); - private final Map readers = new ConcurrentHashMap<>(); + private final boolean stable; + private final Map readers = new ConcurrentHashMap<>(); private volatile boolean closed; - volatile Throwable lastReaderError; // test diagnostics + private volatile InvalidationGapHandler gaps; + private volatile CacheMetricsListener metrics = CacheMetricsListener.NOOP; + volatile Throwable lastReaderError; // sanitized test/diagnostic state - /** - * Creates a transport with a random instance identity (a fresh consumer - * group cursor per start). - * - * @param client the Redis client to connect through - * @param keySerializer serializer for message keys - * @param valueSerializer serializer for UPDATE payloads - * @since 0.1.0 - */ - public LettuceStreamsInvalidationTransport(RedisClient client, - CacheSerializer keySerializer, CacheSerializer valueSerializer) { - this(client, keySerializer, valueSerializer, UUID.randomUUID()); + private final class Reader { + final String cache; + final byte[] stream, group; + final Consumer handler; + volatile boolean active = true, pending; + Thread thread; + RecoveryResult covered; + AutoCloseable gauge; + long nextLog; + Reader(String cache, Consumer handler) { + this.cache = cache; this.handler = handler; + stream = RedisKeyspace.journal(cache); group = RedisKeyspace.group(cache, instanceId); + } } - /** - * Stable instance identity: the same id reconnects to the same consumer - * group (durable cursor). Random per default (new instance). - * - * @param client the Redis client to connect through - * @param keySerializer serializer for message keys - * @param valueSerializer serializer for UPDATE payloads - * @param instanceId the stable identity of this instance - * @since 0.1.0 - */ - public LettuceStreamsInvalidationTransport(RedisClient client, - CacheSerializer keySerializer, CacheSerializer valueSerializer, - UUID instanceId) { - this.instanceId = instanceId; - this.connection = client.connect(ByteArrayCodec.INSTANCE); - this.commands = connection.sync(); - this.keySerializer = keySerializer; - this.valueSerializer = valueSerializer; + /** New volatile L1 identity; its group is retired best-effort on close. */ + public LettuceStreamsInvalidationTransport(RedisClient client, CacheSerializer keys, CacheSerializer values) { + this(client.connect(ByteArrayCodec.INSTANCE), keys, values, UUID.randomUUID(), false); } - - @Override - public void publish(InvalidationMessage message) { - // The journal row IS the durable event (written atomically with the - // data write by the L2 transport). Streams profile needs no channel. + /** Stable identity. The caller guarantees exclusive ownership across process restarts. */ + public LettuceStreamsInvalidationTransport(RedisClient client, CacheSerializer keys, + CacheSerializer values, UUID instanceId) { + this(client.connect(ByteArrayCodec.INSTANCE), keys, values, instanceId, true); } + // Connection seam for real-server ACK/failure tests; transport owns the supplied connection. + LettuceStreamsInvalidationTransport(StatefulRedisConnection connection, + CacheSerializer keys, CacheSerializer values, UUID instanceId, boolean stable) { + this.connection = connection; keySerializer = keys; valueSerializer = values; + this.instanceId = Objects.requireNonNull(instanceId); this.stable = stable; + } + @Override public void publish(InvalidationMessage message) { /* the journal row is the durable event */ } + @Override public java.util.concurrent.CompletionStage publishAsync(InvalidationMessage message) { + return java.util.concurrent.CompletableFuture.completedFuture(io.tiercache.spi.PublicationOutcome.NOT_REQUIRED); + } + + @Override public void setGapHandler(InvalidationGapHandler handler) { gaps = handler; } + @Override public void setMetricsListener(CacheMetricsListener listener) { metrics = listener == null ? CacheMetricsListener.NOOP : listener; } + @Override public boolean requiresRegistrationReset() { return true; } @Override - public AutoCloseable subscribe(String cache, java.util.function.Consumer handler) { - // Create the consumer group eagerly: its cursor starts at creation - // time, so entries published immediately after subscription are - // delivered. Lazy creation in the read loop left a permanent gap for - // anything published between subscribe() and the loop's first pass. - ensureGroup(RedisStreamJournal.streamKeyBytes(cache), group(cache), cache); - handlers.put(cache, handler); - readers.computeIfAbsent(cache, this::startReader); - return () -> { - handlers.remove(cache); - Thread reader = readers.remove(cache); - if (reader != null) { - reader.interrupt(); + public AutoCloseable subscribe(String cache, Consumer handler) { + if (closed) throw new IllegalStateException("Streams transport is closed"); + Reader reader = new Reader(cache, handler); + if (gaps != null) reader.covered = gaps.registrationBaseline(cache); + ensureGroup(reader, reader.covered == null ? "$" : reader.covered.baseline()); + if (readers.putIfAbsent(cache, reader) != null) throw new IllegalStateException("Cache already subscribed: " + cache); + synchronized (reader) { + if (closed) reader.active = false; + if (reader.active) { + reader.thread = new Thread(() -> readLoop(reader), "tiercache-streams-" + cache); + reader.thread.setDaemon(true); + reader.thread.start(); } - }; + } + AutoCloseable gauge = null; + try { gauge = metrics.registerRecovery(cache, () -> reader.pending); } + catch (Throwable e) { log.warn("Streams metric registration failed ({})", e.getClass().getSimpleName()); } + boolean discard; + synchronized (reader) { discard = !reader.active || closed; if (!discard) reader.gauge = gauge; } + if (discard) closeGauge(gauge); + if (!reader.active) retire(reader); + return () -> retire(reader); } - private Thread startReader(String cache) { - Thread thread = new Thread(() -> readLoop(cache), "tiercache-streams-" + cache); - thread.setDaemon(true); - thread.start(); - return thread; - } + private boolean active(Reader reader) { return !closed && reader.active; } - private byte[] group(String cache) { - return (GROUP_PREFIX + cache + ":" + instanceId).getBytes(StandardCharsets.UTF_8); - } + /** + * Read own pending IDs first. Before same-group XAUTOCLAIM, inspect missing + * payloads atomically: Redis 7+ may otherwise remove their PEL entries while + * claiming. This also handles the Redis 6.2 reply without deleted-ID fields. + */ + private static final String PENDING = """ + redis.replicate_commands() + local own = redis.call('xreadgroup','GROUP',ARGV[1],ARGV[2],'COUNT',50,'STREAMS',KEYS[1],'0-0') + if own and #own > 0 and #own[1][2] > 0 then return own[1][2] end + local pending = redis.call('xpending',KEYS[1],ARGV[1],'-','+',50) + local missing = {} + for _, p in ipairs(pending) do + local row = redis.call('xrange',KEYS[1],p[1],p[1]) + if #row == 0 then missing[#missing+1] = {p[1],{}} end + end + if #missing > 0 then return missing end + if #pending == 0 then return {} end + local claimed = redis.call('xautoclaim',KEYS[1],ARGV[1],ARGV[2],0,'0-0','COUNT',50) + return claimed[2] + """; - private void readLoop(String cache) { - try { - readLoopInner(cache); - } catch (Throwable t) { - lastReaderError = t; // surfaced for diagnostics (JMX/logs) + private List> pending(Reader reader) { + List rows = connection.sync().eval(PENDING, ScriptOutputType.MULTI, + new byte[][]{reader.stream}, reader.group, MAIN); + List> result = new ArrayList<>(rows.size()); + for (Object raw : rows) { + List row = (List) raw; + String id = new String((byte[]) row.get(0), StandardCharsets.US_ASCII); + Map body = new LinkedHashMap<>(); + if (row.get(1) instanceof List fields) { + for (int i = 0; i + 1 < fields.size(); i += 2) body.put((byte[]) fields.get(i), (byte[]) fields.get(i + 1)); + } + result.add(new StreamMessage<>(reader.stream, id, body)); } + return result; } - private void readLoopInner(String cache) { - - byte[] stream = RedisStreamJournal.streamKeyBytes(cache); - byte[] group = group(cache); - byte[] consumerName = "main".getBytes(StandardCharsets.UTF_8); - // Short-poll instead of BLOCK: a blocked sync call can outlive the - // connection's command timeout and never return on some stacks. - XReadArgs args = XReadArgs.Builder.count(50); - boolean groupReady = false; - while (!closed && handlers.containsKey(cache)) { - try { - if (!groupReady) { - ensureGroup(stream, group, cache); - claimDeadPending(stream, cache, group, consumerName); - groupReady = true; - } - List> messages = commands.xreadgroup( - io.lettuce.core.Consumer.from(group, consumerName), args, - XReadArgs.StreamOffset.lastConsumed(stream)); - - if (messages == null || messages.isEmpty()) { - sleepQuietly(50); - continue; - } - for (StreamMessage message : messages) { - apply(cache, message); - commands.xack(stream, group, message.getId()); - } - } catch (Throwable e) { - lastReaderError = e; - if (closed) { - return; + private void readLoop(Reader reader) { + int connectionFailures = 0; + try { + while (active(reader)) { + try { + List> batch = pending(reader); + if (batch.isEmpty()) batch = connection.sync().xreadgroup( + io.lettuce.core.Consumer.from(reader.group, MAIN), XReadArgs.Builder.count(BATCH), + XReadArgs.StreamOffset.lastConsumed(reader.stream)); + connectionFailures = 0; + if (batch == null || batch.isEmpty()) { reader.pending = false; pause(50); continue; } + // Retain this batch and its current row until application/settlement + // completes. Never abandon its remainder after Redis advanced >. + for (var row : batch) { + if (!active(reader)) return; + process(reader, row); + } + } catch (RuntimeException error) { + if (!active(reader)) return; + failure(reader, null, CacheMetricsListener.StreamResult.RESYNC_FAILED, error); + reader.pending = true; + if (String.valueOf(error.getMessage()).contains("NOGROUP")) { + if (recover(reader, null)) ensureGroup(reader, reader.covered.baseline()); + } + pause(backoff(++connectionFailures)); } - sleepQuietly(200); // transient failure: retry } + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); } } - private void apply(String cache, StreamMessage entry) { - java.util.function.Consumer handler = handlers.get(cache); - if (handler == null) { + private void process(Reader reader, StreamMessage row) throws InterruptedException { + String id = row.getId(); + refreshBaseline(reader); + if (gaps != null && reader.covered == null && !recover(reader, id)) return; + if (covered(reader, id)) { acknowledge(reader, id, true); return; } + InvalidationMessage message; + try { message = StreamRowDecoder.decode(reader.cache, id, row.getBody(), keySerializer, valueSerializer); } + catch (StreamRowCorruptionException error) { + failure(reader, id, CacheMetricsListener.StreamResult.DECODE_FAILED, error); + if (recover(reader, id)) acknowledge(reader, id, true); return; } - Map body = entry.getBody(); - byte[] type = field(body, "t"); - byte[] key = field(body, "k"); - byte[] version = field(body, "v"); - byte[] payload = field(body, "p"); - Version v = Version.fromWire(new String(version, StandardCharsets.UTF_8)); - Object k = key.length > 0 ? keySerializer.fromBytes(key) : null; - Object p = payload != null && payload.length > 0 ? valueSerializer.fromBytes(payload) : null; - InvalidationMessage.Type t = InvalidationMessage.Type.values()[type[0]]; - if (p != null && t == InvalidationMessage.Type.INVALIDATE) { - t = InvalidationMessage.Type.UPDATE; + boolean applied = false; + for (int attempt = 1; active(reader) && attempt <= 3; attempt++) { + try { + refreshBaseline(reader); + if (covered(reader, id)) { acknowledge(reader, id, true); return; } + // Dispatch is admitted before close, without executing arbitrary + // target/observer code under the transport state monitor. + synchronized (reader) { if (!active(reader)) return; } + reader.handler.accept(message); applied = true; break; + } catch (RuntimeException error) { + if (!active(reader)) return; + failure(reader, id, CacheMetricsListener.StreamResult.APPLY_FAILED, error); + if (attempt < 3) pause(attempt * 1000L); + } } - handler.accept(new InvalidationMessage(cache, k, v, v.instanceId(), t, p)); + if (applied) acknowledge(reader, id, false); + else if (active(reader) && recover(reader, id)) acknowledge(reader, id, true); } - private static byte[] field(Map body, String name) { - byte[] wanted = name.getBytes(StandardCharsets.UTF_8); - for (Map.Entry e : body.entrySet()) { - if (java.util.Arrays.equals(e.getKey(), wanted)) { - return e.getValue(); - } - } - return new byte[0]; + private void refreshBaseline(Reader reader) { + InvalidationGapHandler handler = gaps; + if (handler == null) return; + RecoveryResult latest = handler.registrationBaseline(reader.cache); + if (latest != null && handler.isCurrent(reader.cache, latest)) reader.covered = latest; + // A newer journal replay epoch alone does not require clearing L1. + // Only a row we intend to skip needs a still-current reset proof. } - private void ensureGroup(byte[] stream, byte[] group, String cache) { - try { - commands.xgroupCreate(XReadArgs.StreamOffset.latest(stream), group, - XGroupCreateArgs.Builder.mkstream()); - } catch (RuntimeException e) { - if (!String.valueOf(e.getMessage()).contains("BUSYGROUP")) { - throw e; // real failure: the read loop retries + private boolean covered(Reader reader, String id) { + return reader.covered != null && StreamRowDecoder.compareIds(id, reader.covered.baseline()) <= 0; + } + + private boolean recover(Reader reader, String row) throws InterruptedException { + reader.pending = true; + int failures = 0; + while (active(reader)) { + InvalidationGapHandler handler = gaps; + try { + if (handler == null) throw new IllegalStateException("No capable gap handler"); + RecoveryResult result = handler.reset(reader.cache).toCompletableFuture().get(); + if (!active(reader)) return false; + if (handler == gaps && result != null && result.status() == RecoveryResult.Status.RESET_SAFE + && result.generation() >= 0 && StreamRowDecoder.validId(result.baseline()) + && handler.isCurrent(reader.cache, result)) { + reader.covered = result; + return true; + } + throw new IllegalStateException("Recovery did not establish a current safe baseline"); + } catch (ExecutionException | RuntimeException error) { + if (!active(reader)) return false; + failure(reader, row, CacheMetricsListener.StreamResult.RESYNC_FAILED, error); + pause(backoff(++failures)); } - // BUSYGROUP: group exists — fine. } - // Consumers are registered implicitly on first XREADGROUP; no - // explicit CREATECONSUMER needed (it fails on a missing key). + return false; } - /** - * Claims pending entries from dead groups (zero registered consumers) of - * this cache, applies them, then destroys those groups. Conservative: - * live groups always have a registered consumer, so they are untouched. - */ - private void claimDeadPending(byte[] stream, String cache, byte[] group, byte[] consumerName) { - try { - List groups = commands.xinfoGroups(stream); - for (Object g : groups) { - List row = (List) g; - Map info = kvMap(row); - byte[] name = info.get("name") instanceof byte[] - ? (byte[]) info.get("name") : str(info.get("name")).getBytes(StandardCharsets.UTF_8); - long consumers = num(info.get("consumers")); - long pending = num(info.get("pending")); - if (name == null || java.util.Arrays.equals(name, group) || consumers > 0 || pending == 0) { - continue; + private void acknowledge(Reader reader, String id, boolean needsProof) throws InterruptedException { + int failures = 0; + while (active(reader)) { + if (needsProof && (gaps == null || !covered(reader, id) || !gaps.isCurrent(reader.cache, reader.covered))) { + if (!recover(reader, id)) return; + if (!covered(reader, id)) { // a reset must never cover a row after its baseline + failure(reader, id, CacheMetricsListener.StreamResult.RESYNC_FAILED, new IllegalStateException("Row is after baseline")); + pause(backoff(++failures)); continue; } - // Dead group with unprocessed entries: claim and apply. - var claimed = commands.xautoclaim(stream, new XAutoClaimArgs() - .minIdleTime(1) - .startId("0-0") - .count(1000) - .consumer(io.lettuce.core.Consumer.from(group, consumerName))); - for (StreamMessage m : claimed.getMessages()) { - apply(cache, m); - commands.xack(stream, group, m.getId()); + } + try { + RedisFuture ack; + synchronized (reader) { + if (!active(reader)) return; + if (needsProof && !gaps.isCurrent(reader.cache, reader.covered)) continue; + // Queue admission only. Waiting/network I/O is outside the gate. + ack = connection.async().xack(reader.stream, reader.group, id); } - commands.xgroupDestroy(stream, name); + ack.get(); reader.pending = false; return; + } catch (ExecutionException | RuntimeException error) { + if (!active(reader)) return; + failure(reader, id, CacheMetricsListener.StreamResult.ACK_FAILED, error); + reader.pending = true; pause(backoff(++failures)); } - } catch (RuntimeException e) { - // Janitor is best-effort; the stream trim bounds residue. } } - private static Map kvMap(List row) { - Map map = new java.util.HashMap<>(); - for (int i = 0; i + 1 < row.size(); i += 2) { - Object k = row.get(i); - map.put(k instanceof byte[] ? new String((byte[]) k, StandardCharsets.UTF_8) : String.valueOf(k), - row.get(i + 1)); + private void ensureGroup(Reader reader, String cursor) { + try { + connection.sync().xgroupCreate(XReadArgs.StreamOffset.from(reader.stream, cursor), reader.group, + XGroupCreateArgs.Builder.mkstream()); + } catch (RuntimeException error) { + if (!String.valueOf(error.getMessage()).contains("BUSYGROUP")) throw error; } - return map; } - private static String str(Object o) { - return o instanceof byte[] ? new String((byte[]) o, StandardCharsets.UTF_8) - : o != null ? String.valueOf(o) : null; + private void failure(Reader reader, String row, CacheMetricsListener.StreamResult result, Throwable error) { + lastReaderError = error instanceof StreamRowCorruptionException ? error + : new IllegalStateException("Streams " + result + ": " + error.getClass().getSimpleName()); + try { metrics.onStreamFailure(reader.cache, result); } + catch (Throwable ignored) { /* an observer cannot alter dispatch or ACK state */ } + long now = System.nanoTime(); + if (reader.nextLog == 0 || now - reader.nextLog >= 0) { + reader.nextLog = now + TimeUnit.SECONDS.toNanos(30); + log.warn("Streams recovery: cache={}, row={}, result={}, failure={}", reader.cache, row, result, error.getClass().getSimpleName()); + } } - - private static long num(Object o) { - return o instanceof Number ? ((Number) o).longValue() - : o != null ? Long.parseLong(str(o)) : 0; + private static long backoff(int failures) { return Math.min(30000, 1000L << Math.min(5, failures - 1)); } + private static void pause(long millis) throws InterruptedException { Thread.sleep(millis); } + private static void closeGauge(AutoCloseable gauge) { + if (gauge != null) try { gauge.close(); } catch (Throwable ignored) { } } - - private static void sleepQuietly(long millis) { - try { - Thread.sleep(millis); - } catch (InterruptedException e) { - Thread.currentThread().interrupt(); + private void retire(Reader reader) { + synchronized (reader) { + reader.active = false; + if (reader.thread != null) reader.thread.interrupt(); } + if (!readers.remove(reader.cache, reader)) return; + closeGauge(reader.gauge); + if (!stable) try { connection.sync().xgroupDestroy(reader.stream, reader.group); } + catch (RuntimeException error) { log.debug("Ephemeral Streams group cleanup failed for cache {} ({})", reader.cache, error.getClass().getSimpleName()); } } - - @Override - public void close() { + @Override public void close() { closed = true; - readers.values().forEach(Thread::interrupt); - readers.clear(); - handlers.clear(); + readers.values().forEach(this::retire); connection.close(); } } diff --git a/tiercache-transport-redis/src/main/java/io/tiercache/redis/RedisKeyspace.java b/tiercache-transport-redis/src/main/java/io/tiercache/redis/RedisKeyspace.java new file mode 100644 index 0000000..7014bf4 --- /dev/null +++ b/tiercache-transport-redis/src/main/java/io/tiercache/redis/RedisKeyspace.java @@ -0,0 +1,111 @@ +package io.tiercache.redis; + +import java.nio.ByteBuffer; +import java.nio.CharBuffer; +import java.nio.charset.CharacterCodingException; +import java.nio.charset.CodingErrorAction; +import java.nio.charset.StandardCharsets; +import java.util.Base64; +import java.util.Objects; +import java.util.UUID; + +/** + * Canonical v2 Redis addresses. Internal transport API, not an application API. + * Names are strict UTF-8 encoded as unpadded Base64URL tokens; application key + * bytes remain opaque. No v1 fallback or cleanup is performed. + */ +public final class RedisKeyspace { + static final String DATA = "tiercache:v2:data:"; + static final String TAGS = "tiercache:v2:tags:"; + static final String REVERSE = "tiercache:v2:tagkeys:"; + static final String JOURNAL = "tiercache:v2:journal:"; + static final String TRIMS = "tiercache:v2:journal-trims:"; + static final String CHANNEL = "tiercache:v2:inv:"; + static final String GROUP = "tiercache:v2:cg:"; + static final String LOCK = "tiercache:v2:rebuild:"; + private static final long MAX_KEY_BYTES = 512L * 1024 * 1024; + + private RedisKeyspace() { } + + /** Encodes a complete name without normalization, rejecting malformed UTF-16. */ + public static String token(String name) { + Objects.requireNonNull(name, "name"); + try { + ByteBuffer utf8 = StandardCharsets.UTF_8.newEncoder() + .onMalformedInput(CodingErrorAction.REPORT) + .onUnmappableCharacter(CodingErrorAction.REPORT) + .encode(CharBuffer.wrap(name)); + byte[] bytes = new byte[utf8.remaining()]; + utf8.get(bytes); + return byteToken(bytes); + } catch (CharacterCodingException e) { + throw new IllegalArgumentException("Redis namespace/name contains malformed Unicode", e); + } + } + + /** Encodes original bytes, without hashing or text conversion. */ + public static String byteToken(byte[] bytes) { + checkLength(((long) bytes.length * 4 + 2) / 3); + return Base64.getUrlEncoder().withoutPadding().encodeToString(bytes); + } + + /** Exact literal data prefix; only this prefix may be scanned by clear. */ + public static byte[] dataPrefix(String namespace) { return address(DATA + token(namespace) + ":"); } + + /** Data address, preserving serialized application key bytes. */ + public static byte[] dataKey(String namespace, byte[] key) { return join(dataPrefix(namespace), key); } + + /** Exact tag-set prefix of one physical namespace. */ + public static byte[] tagPrefix(String namespace) { return address(TAGS + token(namespace) + ":"); } + + /** Tag-set address; both namespace and tag are complete tokens. */ + public static byte[] tagKey(String namespace, String tag) { + return join(tagPrefix(namespace), address(token(tag))); + } + + /** Reverse-index address of an opaque serialized application key. */ + public static byte[] reverseKey(String namespace, byte[] key) { + return address(REVERSE + token(namespace) + ":" + byteToken(key)); + } + + /** Journal address uses the logical cache identity, not its physical data namespace. */ + public static byte[] journal(String logical) { return address(JOURNAL + token(logical)); } + + /** Trim counter paired with the logical cache's journal. */ + public static byte[] trims(String logical) { return address(TRIMS + token(logical)); } + + /** Pub/Sub channel for a logical cache. */ + public static byte[] channel(String logical) { return address(CHANNEL + token(logical)); } + + /** Consumer group for a logical cache and stable receiver identity. */ + public static byte[] group(String logical, UUID instance) { + return address(groupPrefix(logical) + Objects.requireNonNull(instance, "instance")); + } + + static String groupPrefix(String logical) { return GROUP + token(logical) + ":"; } + + /** Built-in lock address; the provider's lock name is opaque. */ + public static String lock(String name) { + String result = LOCK + token(name); + checkLength(result.length()); + return result; + } + + static byte[] join(byte[] prefix, byte[] suffix) { + checkLength((long) prefix.length + suffix.length); + byte[] result = java.util.Arrays.copyOf(prefix, prefix.length + suffix.length); + System.arraycopy(suffix, 0, result, prefix.length, suffix.length); + return result; + } + + private static byte[] address(String ascii) { + checkLength(ascii.length()); + return ascii.getBytes(StandardCharsets.US_ASCII); + } + + static void checkLength(long bytes) { + if (bytes < 0 || bytes > MAX_KEY_BYTES) { + throw new IllegalArgumentException("Redis key exceeds the 512 MiB key length limit"); + } + } +} diff --git a/tiercache-transport-redis/src/main/java/io/tiercache/redis/RedisStreamJournal.java b/tiercache-transport-redis/src/main/java/io/tiercache/redis/RedisStreamJournal.java index 48c52bc..2af832d 100644 --- a/tiercache-transport-redis/src/main/java/io/tiercache/redis/RedisStreamJournal.java +++ b/tiercache-transport-redis/src/main/java/io/tiercache/redis/RedisStreamJournal.java @@ -1,5 +1,6 @@ package io.tiercache.redis; +import io.tiercache.invalidation.JournalProtocol; import io.lettuce.core.Limit; import io.lettuce.core.Range; import io.lettuce.core.StreamMessage; @@ -18,11 +19,11 @@ /** * Bounded invalidation journal on Redis Streams: one stream per cache - * ({@code tiercache:journal:}), capacity-capped by approximate + * ({@code tiercache:v2:journal:}), capacity-capped by approximate * MAXLEN trimming. Writers append inside the same atomic Lua unit as the * data write (see {@link LettuceRemoteCache}), so there is no "wrote but * didn't journal" window; every append path also keeps the exact trim - * counter ({@code tiercache:journal-trims:}) for the + * counter ({@code tiercache:v2:journal-trims:}) for the * beginning-cursor trim check. * *

    Cursors are stream entry IDs ({@code millis-seq}); replay reads @@ -39,7 +40,7 @@ public final class RedisStreamJournal implements InvalidationJournal { * * @since 0.1.0 */ - public static final String JOURNAL_KEYSPACE = "tiercache:journal:"; + public static final String JOURNAL_KEYSPACE = RedisKeyspace.JOURNAL; /** * Keyspace prefix of the atomic trim counter (one per cache stream): @@ -49,7 +50,7 @@ public final class RedisStreamJournal implements InvalidationJournal { * * @since 1.3.0 */ - public static final String TRIMS_KEYSPACE = "tiercache:journal-trims:"; + public static final String TRIMS_KEYSPACE = RedisKeyspace.TRIMS; private static final org.slf4j.Logger log = org.slf4j.LoggerFactory.getLogger(RedisStreamJournal.class); @@ -70,7 +71,7 @@ public final class RedisStreamJournal implements InvalidationJournal { * * @param connection the connection to issue stream commands on * @param capacity maximum entries kept per cache stream (approximate - * MAXLEN trimming) + * MAXLEN trimming), at least {@value JournalProtocol#MIN_CAPACITY} * @param keySerializer serializer for message keys * @since 0.1.0 */ @@ -87,28 +88,28 @@ public RedisStreamJournal(StatefulRedisConnection connection, in * * @param connection the connection to issue stream commands on * @param capacity maximum entries kept per cache stream - * (approximate MAXLEN trimming) + * (approximate MAXLEN trimming), at least {@value JournalProtocol#MIN_CAPACITY} * @param keySerializer serializer for message keys * @param valueSerializer serializer for UPDATE payloads * @since 1.2.1 */ public RedisStreamJournal(StatefulRedisConnection connection, int capacity, CacheSerializer keySerializer, CacheSerializer valueSerializer) { + this.capacity = JournalProtocol.requireCapacity(capacity); this.commands = connection.sync(); - this.capacity = capacity; this.keySerializer = keySerializer; this.valueSerializer = valueSerializer; } static byte[] streamKey(String cache) { - return (JOURNAL_KEYSPACE + cache).getBytes(java.nio.charset.StandardCharsets.UTF_8); + return RedisKeyspace.journal(cache); } /** * Stream key bytes for a cache (public for the transport's Lua script). * * @param cache the cache name - * @return the stream key bytes ({@code tiercache:journal:}) + * @return the stream key bytes ({@code tiercache:v2:journal:}) * @since 0.1.0 */ public static byte[] streamKeyBytes(String cache) { @@ -116,7 +117,7 @@ public static byte[] streamKeyBytes(String cache) { } static byte[] trimCounterKey(String cache) { - return (TRIMS_KEYSPACE + cache).getBytes(java.nio.charset.StandardCharsets.UTF_8); + return RedisKeyspace.trims(cache); } /** @@ -124,7 +125,7 @@ static byte[] trimCounterKey(String cache) { * scripts, which increment it on every capped XADD that removed rows). * * @param cache the cache name - * @return the counter key bytes ({@code tiercache:journal-trims:}) + * @return the counter key bytes ({@code tiercache:v2:journal-trims:}) * @since 1.3.0 */ public static byte[] trimCounterKeyBytes(String cache) { @@ -191,6 +192,7 @@ public List readRange(String cache, String cursorExclusive) { @Override public CheckedRange checkedRead(String cache, String cursor, int maxRows) { + if (maxRows < 1) throw new IllegalArgumentException("maxRows must be positive"); if ("0-0".equals(cursor)) { return checkedReadFromBeginning(cache, maxRows); } @@ -198,10 +200,11 @@ public CheckedRange checkedRead(String cache, String cursor, int maxRows) { // row) and the range come from the same response. List> entries = commands.xrange(streamKey(cache), Range.from(Range.Boundary.including(cursor), Range.Boundary.unbounded()), - Limit.from(maxRows)); - List rows = toRows(cache, entries); - boolean intact = !rows.isEmpty() && rows.get(0).cursor().equals(cursor); - return new CheckedRange(intact, rows); + Limit.from((long) maxRows + 1)); + boolean intact = !entries.isEmpty() && entries.get(0).getId().equals(cursor); + // The anchor's raw ID proves integrity. Its payload was already accounted + // for and may be the poison row covered by the last safe reset. + return new CheckedRange(intact, intact ? toRows(cache, entries.subList(1, entries.size())) : List.of()); } private CheckedRange checkedReadFromBeginning(String cache, int maxRows) { @@ -212,6 +215,7 @@ private CheckedRange checkedReadFromBeginning(String cache, int maxRows) { Object trims = reply.get(0); boolean intact = trims == null || Long.parseLong(new String((byte[]) trims, java.nio.charset.StandardCharsets.UTF_8)) == 0; + if (!intact) return new CheckedRange(false, List.of()); List rows = new ArrayList<>(); for (Object entry : (List) reply.get(1)) { List pair = (List) entry; @@ -221,7 +225,7 @@ private CheckedRange checkedReadFromBeginning(String cache, int maxRows) { for (int f = 0; f + 1 < flatFields.size(); f += 2) { body.put((byte[]) flatFields.get(f), (byte[]) flatFields.get(f + 1)); } - rows.add(new JournalRow(id, toMessage(cache, body))); + rows.add(new JournalRow(id, StreamRowDecoder.decode(cache, id, body, keySerializer, valueSerializer))); } return new CheckedRange(intact, rows); } @@ -282,7 +286,7 @@ public boolean isTrimmed(String cache, String cursor) { private List toRows(String cache, List> entries) { List out = new ArrayList<>(entries.size()); for (StreamMessage entry : entries) { - out.add(new JournalRow(entry.getId(), toMessage(cache, entry.getBody()))); + out.add(new JournalRow(entry.getId(), StreamRowDecoder.decode(cache, entry.getId(), entry.getBody(), keySerializer, valueSerializer))); } return out; } @@ -297,46 +301,7 @@ private Map fields(byte[] keyBytes, Version version, return fields; } - private InvalidationMessage toMessage(String cache, Map body) { - byte[] typeOrd = get(body, FIELD_TYPE); - byte[] keyBytes = get(body, FIELD_KEY); - Version version = Version.fromWire(new String(get(body, FIELD_VERSION), - java.nio.charset.StandardCharsets.UTF_8)); - Object key = keyBytes.length > 0 ? keySerializer.fromBytes(keyBytes) : null; - byte[] payloadBytes = body.entrySet().stream() - .filter(e -> java.util.Arrays.equals(e.getKey(), FIELD_PAYLOAD)) - .map(Map.Entry::getValue).findFirst().orElse(new byte[0]); - // Replay must apply the same typed value as the live path (which - // deserializes in the transport): never hand raw bytes to L1. - Object payload = payloadBytes.length > 0 ? valueSerializer.fromBytes(payloadBytes) : null; - InvalidationMessage.Type type = InvalidationMessage.Type.values()[typeOrd[0]]; - if (payload != null && type == InvalidationMessage.Type.INVALIDATE) { - type = InvalidationMessage.Type.UPDATE; // payload implies update semantics - } - return new InvalidationMessage(cache, key, version, version.instanceId(), type, payload); - } - - private static byte[] get(Map body, byte[] field) { - for (Map.Entry e : body.entrySet()) { - if (java.util.Arrays.equals(e.getKey(), field)) { - return e.getValue(); - } - } - throw new IllegalStateException("journal entry missing field"); - } - static int compareIds(String a, String b) { - long[] pa = parse(a); - long[] pb = parse(b); - int byMillis = Long.compare(pa[0], pb[0]); - return byMillis != 0 ? byMillis : Long.compare(pa[1], pb[1]); - } - - private static long[] parse(String id) { - int dash = id.indexOf('-'); - if (dash < 0) { - return new long[]{Long.parseLong(id), 0}; - } - return new long[]{Long.parseLong(id.substring(0, dash)), Long.parseLong(id.substring(dash + 1))}; + return StreamRowDecoder.compareIds(a, b); } } diff --git a/tiercache-transport-redis/src/main/java/io/tiercache/redis/StreamRowCorruptionException.java b/tiercache-transport-redis/src/main/java/io/tiercache/redis/StreamRowCorruptionException.java new file mode 100644 index 0000000..75e0fdf --- /dev/null +++ b/tiercache-transport-redis/src/main/java/io/tiercache/redis/StreamRowCorruptionException.java @@ -0,0 +1,13 @@ +package io.tiercache.redis; + +/** Sanitized corruption metadata. Never retains serializer exceptions, keys or payload bytes. */ +public final class StreamRowCorruptionException extends io.tiercache.spi.JournalCorruptionException { + /** The structural/decoding failure category. */ + public enum Reason { MISSING_ROW, FIELDS, TYPE, VERSION, KEY, PAYLOAD, ROW_ID } + private final Reason reason; + StreamRowCorruptionException(String cache, String rowId, Reason reason) { + super(cache, rowId, reason.name()); + this.reason = reason; + } + public Reason reason() { return reason; } +} diff --git a/tiercache-transport-redis/src/main/java/io/tiercache/redis/StreamRowDecoder.java b/tiercache-transport-redis/src/main/java/io/tiercache/redis/StreamRowDecoder.java new file mode 100644 index 0000000..011f62c --- /dev/null +++ b/tiercache-transport-redis/src/main/java/io/tiercache/redis/StreamRowDecoder.java @@ -0,0 +1,73 @@ +package io.tiercache.redis; + +import io.tiercache.InvalidationMessage; +import io.tiercache.Version; +import java.nio.charset.StandardCharsets; +import java.util.Arrays; +import java.util.Map; +import static io.tiercache.redis.StreamRowCorruptionException.Reason; + +/** One validation policy for live Streams and journal replay. */ +final class StreamRowDecoder { + private StreamRowDecoder() { } + static InvalidationMessage decode(String cache, String row, Map body, + CacheSerializer keys, CacheSerializer values) { + if (!validId(row)) throw bad(cache, row, Reason.ROW_ID); + if (body == null || body.isEmpty()) throw bad(cache, row, Reason.MISSING_ROW); + byte[] type = field(cache, row, body, "t", true); + byte[] version = field(cache, row, body, "v", true); + byte[] key = field(cache, row, body, "k", true); + byte[] payload = field(cache, row, body, "p", false); + if (type.length != 1 || Byte.toUnsignedInt(type[0]) >= InvalidationMessage.Type.values().length) throw bad(cache, row, Reason.TYPE); + var kind = InvalidationMessage.Type.values()[Byte.toUnsignedInt(type[0])]; + Version parsed; + try { + String wire = new String(version, StandardCharsets.UTF_8); + parsed = Version.fromWire(wire); + if (!parsed.toWire().equalsIgnoreCase(wire)) throw new IllegalArgumentException(); + } catch (RuntimeException e) { throw bad(cache, row, Reason.VERSION); } + if (kind == InvalidationMessage.Type.EVICT_ALL) { + if (key.length != 0 || (payload != null && payload.length != 0)) throw bad(cache, row, Reason.FIELDS); + return new InvalidationMessage(cache, null, parsed, parsed.instanceId(), kind); + } + Object decodedKey; + try { decodedKey = keys.fromBytes(key); if (decodedKey == null) throw new IllegalArgumentException(); } + catch (RuntimeException e) { throw bad(cache, row, Reason.KEY); } + Object value = null; + boolean update = kind == InvalidationMessage.Type.UPDATE || (payload != null && payload.length > 0); + if (update) { + if (payload == null) throw bad(cache, row, Reason.PAYLOAD); + try { value = values.fromBytes(payload); if (value == null) throw new IllegalArgumentException(); } + catch (RuntimeException e) { throw bad(cache, row, Reason.PAYLOAD); } + kind = InvalidationMessage.Type.UPDATE; // legacy journal payload implies UPDATE + } + return new InvalidationMessage(cache, decodedKey, parsed, parsed.instanceId(), kind, value); + } + private static byte[] field(String cache, String row, Map body, String name, boolean required) { + byte[] result = null; boolean found = false; + byte[] wanted = name.getBytes(StandardCharsets.US_ASCII); + for (var entry : body.entrySet()) if (Arrays.equals(wanted, entry.getKey())) { + if (found || entry.getValue() == null) throw bad(cache, row, Reason.FIELDS); + found = true; result = entry.getValue(); + } + if (!found && required) throw bad(cache, row, Reason.FIELDS); + return result; + } + static boolean validId(String id) { + if (id == null || !id.matches("[0-9]+-[0-9]+")) return false; + try { parts(id); return true; } catch (RuntimeException e) { return false; } + } + static int compareIds(String left, String right) { + long[] a = parts(left), b = parts(right); + int first = Long.compareUnsigned(a[0], b[0]); + return first != 0 ? first : Long.compareUnsigned(a[1], b[1]); + } + private static long[] parts(String id) { + int dash = id.indexOf('-'); + if (dash < 0) return new long[]{Long.parseUnsignedLong(id), 0}; + return new long[]{Long.parseUnsignedLong(id.substring(0, dash)), Long.parseUnsignedLong(id.substring(dash + 1))}; + } + private static StreamRowCorruptionException bad(String cache, String row, Reason reason) { + return new StreamRowCorruptionException(cache, row, reason); + } +} diff --git a/tiercache-transport-redis/src/test/java/io/tiercache/redis/AbstractLettuceContractTest.java b/tiercache-transport-redis/src/test/java/io/tiercache/redis/AbstractLettuceContractTest.java index a3af9f5..88dd696 100644 --- a/tiercache-transport-redis/src/test/java/io/tiercache/redis/AbstractLettuceContractTest.java +++ b/tiercache-transport-redis/src/test/java/io/tiercache/redis/AbstractLettuceContractTest.java @@ -9,6 +9,19 @@ import org.testcontainers.utility.DockerImageName; import java.util.UUID; +import java.time.Duration; +import java.util.List; +import java.util.ArrayList; +import java.nio.charset.StandardCharsets; +import io.lettuce.core.codec.ByteArrayCodec; +import io.lettuce.core.ScriptOutputType; +import io.tiercache.Version; +import io.tiercache.InvalidationMode; +import io.tiercache.spi.StoredEntry; +import io.tiercache.spi.TaggedWriteOutcome; +import org.junit.jupiter.api.Test; +import static org.junit.jupiter.api.Assertions.*; +import static io.tiercache.spi.TaggedWriteOutcome.*; /** * Runs the shared {@link RemoteCacheContractTest} suite against a real @@ -28,6 +41,10 @@ abstract class AbstractLettuceContractTest extends RemoteCacheContractTest { void startServer() { server = new GenericContainer<>(image()).withExposedPorts(REDIS_PORT); server.start(); + System.out.println("SERVER_EVIDENCE image=" + server.getDockerImageName() + + " imageId=" + server.getContainerInfo().getImageId() + + " jdk=" + System.getProperty("java.version") + + " lettuce=" + RedisClient.class.getPackage().getImplementationVersion()); client = RedisClient.create( "redis://" + server.getHost() + ":" + server.getMappedPort(REDIS_PORT)); } @@ -58,4 +75,255 @@ String redisUri() { RedisClient sharedClient() { return client; } + @org.junit.jupiter.api.Test + void rejectedTaggedWriteKeepsWinningMembership() { + String name = "tagged-loss-" + UUID.randomUUID(); + try (var connection = client.connect(io.lettuce.core.codec.ByteArrayCodec.INSTANCE)) { + var journal = new RedisStreamJournal(connection, 1000, new JdkCacheSerializer<>()); + try (var cache = LettuceRemoteCache.builder(redisUri()) + .client(client).cacheName(name).journal(journal).build()) { + UUID writer = UUID.randomUUID(); + cache.putTagged("k", io.tiercache.spi.StoredEntry.ofValue("new", + new io.tiercache.Version(20, writer)), java.time.Duration.ofMinutes(1), + new String[]{"winner"}); + long rows = journal.size(name); + cache.putTagged("k", io.tiercache.spi.StoredEntry.ofValue("old", + new io.tiercache.Version(10, writer)), java.time.Duration.ofMinutes(1), + new String[]{"loser"}); + org.junit.jupiter.api.Assertions.assertEquals("new", cache.get("k").value()); + org.junit.jupiter.api.Assertions.assertTrue(cache.keysByTag("loser").isEmpty(), + "a rejected tagged candidate cannot acquire membership"); + org.junit.jupiter.api.Assertions.assertEquals(java.util.List.of("k"), cache.keysByTag("winner")); + org.junit.jupiter.api.Assertions.assertEquals(rows, journal.size(name)); + } + } + } + + @org.junit.jupiter.api.Test + void acceptedRetaggingRemovesSupersededMembership() { + try (var cache = newCache()) { + cache.putTagged("k", io.tiercache.spi.StoredEntry.ofValue("first"), + java.time.Duration.ofMinutes(1), new String[]{"A"}); + cache.putTagged("k", io.tiercache.spi.StoredEntry.ofValue("second"), + java.time.Duration.ofMinutes(1), new String[]{"B"}); + org.junit.jupiter.api.Assertions.assertTrue(cache.keysByTag("A").isEmpty(), + "an accepted retag must remove previous membership"); + org.junit.jupiter.api.Assertions.assertEquals(java.util.List.of("k"), cache.keysByTag("B")); + } + } + + private static byte[] bytes(String value) { return value.getBytes(StandardCharsets.UTF_8); } + + private static byte[] dataKey(String name, String key) { + return RedisKeyspace.dataKey(name, new JdkCacheSerializer().toBytes(key)); + } + + private static byte[] reverseKey(String name, String key) { + return RedisKeyspace.reverseKey(name, new JdkCacheSerializer().toBytes(key)); + } + + @Test + void losingWriteLeavesAllBytesAndExpirationsUntouchedIncludingTombstones() { + String name = "snapshot-" + UUID.randomUUID(); + try (var connection = client.connect(ByteArrayCodec.INSTANCE)) { + var cmd = connection.sync(); + var journal = new RedisStreamJournal(connection, 128, new JdkCacheSerializer<>()); + try (var cache = LettuceRemoteCache.builder(redisUri()).client(client) + .cacheName(name).journal(journal).build()) { + var id = UUID.randomUUID(); + assertEquals(WON, cache.putTaggedIfNewer("k", StoredEntry.ofValue("new", new Version(20, id)), + Duration.ofMinutes(1), new String[]{"A"})); + byte[] data = dataKey(name, "k"); + byte[][] keys = {data, reverseKey(name, "k"), RedisKeyspace.tagKey(name, "A"), + RedisKeyspace.tagKey(name, "B"), RedisStreamJournal.streamKey(name), + RedisStreamJournal.trimCounterKey(name)}; + for (boolean tombstone : new boolean[]{false, true}) { + if (tombstone) cache.evict("k", new Version(30, id)); + List snapshot = new ArrayList<>(); + for (byte[] key : keys) snapshot.add(cmd.dump(key)); + long ttl = cmd.pttl(data); + assertEquals(LOST, cache.putTaggedIfNewer("k", StoredEntry.ofValue("old", new Version(10, id)), + Duration.ofHours(1), new String[]{"B"})); + for (int i = 0; i < keys.length; i++) assertArrayEquals(snapshot.get(i), cmd.dump(keys[i])); + assertTrue(cmd.pttl(data) <= ttl, "a loss cannot extend expiry"); + assertTrue(cache.keysByTag("B").isEmpty()); + if (tombstone) assertNull(cache.get("k")); + else assertEquals("new", cache.get("k").value()); + } + } + } + } + + @Test + void retaggingPreservesSharedMembershipExactReverseTtlAndExtendOnlyTagTtl() { + String name = "ttl-" + UUID.randomUUID(); + try (var connection = client.connect(ByteArrayCodec.INSTANCE); + var cache = LettuceRemoteCache.builder(redisUri()).client(client).cacheName(name).build()) { + var cmd = connection.sync(); + cache.putTagged("other", StoredEntry.ofValue("shared"), Duration.ofMinutes(2), new String[]{"A"}); + cache.putTagged("k", StoredEntry.ofValue("old"), Duration.ofMinutes(2), new String[]{"A", "keep"}); + long before = cmd.pttl(RedisKeyspace.tagKey(name, "keep")); + cache.putTagged("k", StoredEntry.ofValue("new"), Duration.ofSeconds(30), new String[]{"B", "keep", "B"}); + assertEquals(List.of("other"), cache.keysByTag("A")); + assertEquals(List.of("k"), cache.keysByTag("B")); + byte[] data = dataKey(name, "k"); + assertSameExpiry(cmd, data, reverseKey(name, "k")); + long after = cmd.pttl(RedisKeyspace.tagKey(name, "keep")); + assertTrue(after > 90000 && after <= before, "retained tags must not shrink to the new 30s TTL"); + assertEquals(java.util.Set.of(RedisKeyspace.token("B"), RedisKeyspace.token("keep")), cmd.smembers(reverseKey(name, "k")).stream() + .map(b -> new String(b, StandardCharsets.UTF_8)).collect(java.util.stream.Collectors.toSet())); + for (String key : cache.keysByTag("A")) cache.evict(key); + assertEquals("new", cache.get("k").value(), "old tag eviction must not delete the retagged value"); + for (String key : cache.keysByTag("B")) cache.evict(key); + assertNull(cache.get("k")); assertTrue(cache.keysByTag("keep").isEmpty()); + assertEquals(0, cmd.exists(reverseKey(name, "k"))); + } + } + + @Test + void concurrentTaggedWritersShareOneAcceptanceDecision() throws Exception { + String name = "concurrent-" + UUID.randomUUID(); + try (var connection = client.connect(ByteArrayCodec.INSTANCE)) { + var journal = new RedisStreamJournal(connection, 1000, new JdkCacheSerializer<>()); + try (var first = LettuceRemoteCache.builder(redisUri()).client(client).cacheName(name).journal(journal).build(); + var second = LettuceRemoteCache.builder(redisUri()).client(client).cacheName(name).journal(journal).build()) { + var executor = java.util.concurrent.Executors.newFixedThreadPool(2); + var start = new java.util.concurrent.CountDownLatch(1); + var id = UUID.randomUUID(); + try { + var low = executor.submit(() -> { start.await(); int won = 0; + for (int i = 1; i <= 100; i++) if (first.putTaggedIfNewer("k", StoredEntry.ofValue("v" + i, + new Version(i, id)), Duration.ofMinutes(1), new String[]{"t" + i}) == WON) won++; + return won; + }); + var high = executor.submit(() -> { start.await(); int won = 0; + for (int i = 200; i > 100; i--) if (second.putTaggedIfNewer("k", StoredEntry.ofValue("v" + i, + new Version(i, id)), Duration.ofMinutes(1), new String[]{"t" + i}) == WON) won++; + return won; + }); + start.countDown(); + int accepted = low.get(15, java.util.concurrent.TimeUnit.SECONDS) + high.get(15, java.util.concurrent.TimeUnit.SECONDS); + assertEquals(accepted, journal.size(name)); + assertEquals("v200", first.get("k").value()); + assertEquals(List.of("k"), first.keysByTag("t200")); + for (int i = 1; i < 200; i++) assertTrue(first.keysByTag("t" + i).isEmpty()); + } finally { executor.shutdownNow(); } + } + } + } + + @Test + void updateJournalAndTrimAccountingRemainCompatible() { + String name = "journal-" + UUID.randomUUID(); + try (var connection = client.connect(ByteArrayCodec.INSTANCE)) { + var journal = new RedisStreamJournal(connection, 128, new JdkCacheSerializer<>()); + try (var cache = LettuceRemoteCache.builder(redisUri()).client(client).cacheName(name) + .journal(journal).invalidationMode(InvalidationMode.UPDATE, 65536).build()) { + var id = UUID.randomUUID(); + for (int i = 1; i <= 350; i++) assertEquals(WON, cache.putTaggedIfNewer("k", + StoredEntry.ofValue("value" + i, new Version(i, id)), Duration.ofMinutes(1), new String[]{"tag"})); + var rows = journal.readRange(name, "0-0"); + var last = rows.get(rows.size() - 1).message(); + assertEquals("k", last.key()); assertEquals(new Version(350, id), last.version()); + assertEquals(io.tiercache.InvalidationMessage.Type.UPDATE, last.type()); + assertEquals("value350", last.payload()); + byte[] trims = connection.sync().get(RedisStreamJournal.trimCounterKey(name)); + assertNotNull(trims); assertTrue(Long.parseLong(new String(trims, StandardCharsets.UTF_8)) > 0); + long size = journal.size(name); + assertEquals(LOST, cache.putTaggedIfNewer("k", StoredEntry.ofValue("old", new Version(1, id)), + Duration.ofMinutes(1), new String[]{"bad"})); + assertEquals(size, journal.size(name)); + assertArrayEquals(trims, connection.sync().get(RedisStreamJournal.trimCounterKey(name))); + assertEquals(WON, cache.putTaggedIfNewer("k", StoredEntry.nullMarker(new Version(351, id)), + Duration.ofMinutes(1), new String[]{"null"})); + assertTrue(cache.get("k").isNullMarker()); assertTrue(cache.keysByTag("tag").isEmpty()); + } + } + } + + @Test + void noJournalAndUnversionedTaggedWritesRemainUnfenced() { + String name = "unfenced-" + UUID.randomUUID(); + try (var cache = LettuceRemoteCache.builder(redisUri()).client(client).cacheName(name).build()) { + var id = UUID.randomUUID(); + cache.putTagged("k", StoredEntry.ofValue("new", new Version(20, id)), Duration.ofMinutes(1), new String[]{"A"}); + assertEquals(WON, cache.putTaggedIfNewer("k", StoredEntry.ofValue("old", new Version(10, id)), + Duration.ofMinutes(1), new String[]{"B"})); + assertEquals("old", cache.get("k").value()); assertTrue(cache.keysByTag("A").isEmpty()); + } + try (var connection = client.connect(ByteArrayCodec.INSTANCE)) { + var journal = new RedisStreamJournal(connection, 128, new JdkCacheSerializer<>()); + try (var cache = LettuceRemoteCache.builder(redisUri()).client(client).cacheName(name).journal(journal).build()) { + assertEquals(WON, cache.putTaggedIfNewer("k", StoredEntry.ofValue("plain"), Duration.ofMinutes(1), new String[]{"C"})); + assertEquals(0, journal.size(name)); assertEquals("plain", cache.get("k").value()); + assertTrue(cache.keysByTag("B").isEmpty()); + } + } + } + + @Test + void invalidTaggedArgumentsDoNotPartiallyMutateRedis() { + try (var cache = newCache()) { + cache.putTagged("k", StoredEntry.ofValue("old"), Duration.ofMinutes(1), new String[]{"A"}); + assertThrows(NullPointerException.class, () -> cache.putTaggedIfNewer("k", StoredEntry.ofValue("bad"), + Duration.ofMinutes(1), new String[]{"B", null})); + assertThrows(IllegalArgumentException.class, () -> cache.putTaggedIfNewer("k", StoredEntry.ofValue("bad"), + Duration.ZERO, new String[]{"B"})); + assertEquals("old", cache.get("k").value()); + assertEquals(List.of("k"), cache.keysByTag("A")); assertTrue(cache.keysByTag("B").isEmpty()); + } + } + @Test + void clearDoesNotCrossHierarchicalNamespace() { + assertClearIsolation("user", "user:roles"); + } + + @Test + void clearDoesNotInterpretNamespaceGlob() { + assertClearIsolation("a?", "a1"); + } + + private void assertClearIsolation(String cleared, String retained) { + String suffix = "-" + UUID.randomUUID(); + try (var first = LettuceRemoteCache.builder(redisUri()).client(client).cacheName(cleared).build(); + var second = LettuceRemoteCache.builder(redisUri()).client(client).cacheName(retained).build()) { + first.put(suffix, StoredEntry.ofValue("own"), Duration.ofMinutes(1)); + second.put(suffix, StoredEntry.ofValue("other"), Duration.ofMinutes(1)); + first.clear(); + assertNull(first.get(suffix)); + assertNotNull(second.get(suffix), "clear must preserve another complete cache namespace"); + assertEquals("other", second.get(suffix).value()); + second.evict(suffix); + } + } + + @Test + void largeTaggedWriteUsesOneExpirationInstant() { + String name = "expiry-" + UUID.randomUUID(); + try (var connection = client.connect(ByteArrayCodec.INSTANCE); + var cache = LettuceRemoteCache.builder(redisUri()).client(client).cacheName(name).build()) { + String[] tags = java.util.stream.IntStream.range(0, 3000).mapToObj(i -> "t" + i).toArray(String[]::new); + cache.putTagged("k", StoredEntry.ofValue("v"), Duration.ofSeconds(30), tags); + assertSameExpiry(connection.sync(), dataKey(name, "k"), reverseKey(name, "k")); + } + } + + private static void assertSameExpiry(io.lettuce.core.api.sync.RedisCommands commands, + byte[] data, byte[] reverse) { + // Redis 6.2 updates time during Lua execution. Compare PTTLs only when + // both were sampled inside one server millisecond, without a tolerance. + String script = "local a=redis.call('time'); local d=redis.call('pttl',KEYS[1]); " + + "local r=redis.call('pttl',KEYS[2]); local b=redis.call('time'); " + + "return {a[1]*1000+math.floor(a[2]/1000), b[1]*1000+math.floor(b[2]/1000), d, r}"; + for (int i = 0; i < 20; i++) { + List sample = commands.eval(script, ScriptOutputType.MULTI, new byte[][]{data, reverse}); + if (sample.get(0).equals(sample.get(1))) { + assertTrue(sample.get(2) > 0, "data must still be live"); + assertEquals(sample.get(2), sample.get(3), "data and reverse index must expire at exactly the same instant"); + return; + } + } + fail("could not sample both PTTLs within one Redis millisecond"); + } + } diff --git a/tiercache-transport-redis/src/test/java/io/tiercache/redis/AbstractLettuceStreamsIT.java b/tiercache-transport-redis/src/test/java/io/tiercache/redis/AbstractLettuceStreamsIT.java index 629b083..b4620cb 100644 --- a/tiercache-transport-redis/src/test/java/io/tiercache/redis/AbstractLettuceStreamsIT.java +++ b/tiercache-transport-redis/src/test/java/io/tiercache/redis/AbstractLettuceStreamsIT.java @@ -24,7 +24,7 @@ /** * Spec: invalidation — durable Streams profile over real Redis: events flow, - * disconnects heal by consuming the journal stream (no full flush). + * pending resumes within retained history or a safe registration/reset baseline. */ abstract class AbstractLettuceStreamsIT { @@ -109,8 +109,7 @@ void eventsFlowThroughStreamsProfile() throws Exception { void subscriptionCreatesConsumerGroupEagerly() { factoryB.getCache("streams"); byte[] stream = RedisStreamJournal.streamKeyBytes("streams"); - byte[] group = (LettuceStreamsInvalidationTransport.GROUP_PREFIX + "streams:" + instanceB) - .getBytes(StandardCharsets.UTF_8); + byte[] group = RedisKeyspace.group("streams", instanceB); var connection = clientB.connect(ByteArrayCodec.INSTANCE); try { assertTrue(groupExists(connection.sync(), stream, group), @@ -121,7 +120,7 @@ void subscriptionCreatesConsumerGroupEagerly() { } @Test - void disconnectHealsWithoutFullFlush() throws Exception { + void stableResumeEstablishesSafeRegistrationBaseline() throws Exception { TierCache a = factoryA.getCache("streams"); TierCache b = factoryB.getCache("streams"); awaitConsumerGroup("streams"); @@ -152,7 +151,7 @@ void disconnectHealsWithoutFullFlush() throws Exception { service.registerTarget("streams", (io.tiercache.spi.InvalidationTarget) factoryB.getCache("streams")); waitFor(() -> b.get("keep") == null); assertEquals("v0", b.get("flood-0")); - reconnected.close(); + service.close(); } /** @@ -162,8 +161,7 @@ void disconnectHealsWithoutFullFlush() throws Exception { */ private void awaitConsumerGroup(String cache) throws InterruptedException { byte[] stream = RedisStreamJournal.streamKeyBytes(cache); - byte[] group = (LettuceStreamsInvalidationTransport.GROUP_PREFIX + cache + ":" + instanceB) - .getBytes(StandardCharsets.UTF_8); + byte[] group = RedisKeyspace.group(cache, instanceB); var connection = clientB.connect(ByteArrayCodec.INSTANCE); try { var sync = connection.sync(); @@ -211,6 +209,41 @@ private static void waitFor(Check check) throws InterruptedException { } } + @Test + void poisonBatchCannotStrandPendingOrLeaveAStaleVictim() throws Exception { + TierCache a = factoryA.getCache("streams"); + TierCache b = factoryB.getCache("streams"); + var observed = new java.util.concurrent.CopyOnWriteArrayList(); + var field = TierCacheFactory.class.getDeclaredField("invalidation"); field.setAccessible(true); + ((io.tiercache.spi.InvalidationHandler) field.get(factoryB)).setEventListener((c, event) -> observed.add(event.key())); + a.put("victim", "old"); a.put("batch", "old"); + waitFor(() -> observed.contains("victim") && observed.contains("batch")); + assertEquals("old", b.get("victim")); assertEquals("old", b.get("batch")); observed.clear(); + byte[] stream = RedisKeyspace.journal("streams"); byte[] group = RedisKeyspace.group("streams", instanceB); + var serializer = new JdkCacheSerializer(); + var version = new io.tiercache.Version(Long.MAX_VALUE - 1, java.util.UUID.randomUUID()); + try (var connection = clientA.connect(ByteArrayCodec.INSTANCE)) { + var cmd = connection.sync(); + cmd.del(RedisKeyspace.dataKey("streams", serializer.toBytes("victim")), + RedisKeyspace.dataKey("streams", serializer.toBytes("batch"))); + cmd.eval("redis.call('xadd',KEYS[1],'*','t',string.char(99),'k',ARGV[1],'v',ARGV[3]); " + + "redis.call('xadd',KEYS[1],'*','t',string.char(0),'k',ARGV[2],'v',ARGV[3]); return 1", + io.lettuce.core.ScriptOutputType.INTEGER, new byte[][]{stream}, + serializer.toBytes("victim"), serializer.toBytes("batch"), version.toWire().getBytes(StandardCharsets.UTF_8)); + waitFor(() -> transportB.lastReaderError != null); + var journal = new RedisStreamJournal(connection, 1000, new JdkCacheSerializer<>()); + journal.append("streams", new io.tiercache.InvalidationMessage("streams", "later", version, + version.instanceId(), io.tiercache.InvalidationMessage.Type.INVALIDATE)); + try { waitFor(() -> cmd.xpending(stream, group).getCount() == 0); } + finally { System.out.println("Poison batch: pending=" + cmd.xpending(stream, group).getCount() + ", delivered=" + observed); } + org.junit.jupiter.api.Assertions.assertNull(b.get("victim"), "ACK/skip without a clear leaves the corrupt-only victim stale"); + org.junit.jupiter.api.Assertions.assertNull(b.get("batch"), "the batch remainder must be applied or covered by a safe clear"); + journal.append("streams", new io.tiercache.InvalidationMessage("streams", "after", version, + version.instanceId(), io.tiercache.InvalidationMessage.Type.INVALIDATE)); + waitFor(() -> observed.contains("after")); + } + } + private interface Check { boolean ok(); } @@ -220,7 +253,7 @@ private interface Check { class RedisStreamsIT extends AbstractLettuceStreamsIT { @Override DockerImageName image() { - return DockerImageName.parse("redis:6.2-alpine"); + return ServerProfile.image(); } } diff --git a/tiercache-transport-redis/src/test/java/io/tiercache/redis/AbstractNamespaceContractTest.java b/tiercache-transport-redis/src/test/java/io/tiercache/redis/AbstractNamespaceContractTest.java new file mode 100644 index 0000000..81680cf --- /dev/null +++ b/tiercache-transport-redis/src/test/java/io/tiercache/redis/AbstractNamespaceContractTest.java @@ -0,0 +1,220 @@ +package io.tiercache.redis; + +import io.lettuce.core.codec.ByteArrayCodec; +import io.lettuce.core.XReadArgs; +import io.tiercache.InvalidationMessage; +import io.tiercache.Version; +import io.tiercache.TierCacheFactory; +import io.tiercache.invalidation.MessageCodec; +import io.tiercache.spi.StoredEntry; +import org.junit.jupiter.api.Test; + +import java.nio.charset.StandardCharsets; +import java.time.Duration; +import java.util.*; +import java.util.concurrent.*; +import java.util.concurrent.atomic.AtomicInteger; + +import static org.junit.jupiter.api.Assertions.*; + +/** Namespace and cold-cutover contract, inherited by both supported server suites. */ +abstract class AbstractNamespaceContractTest extends AbstractLettuceContractTest { + private static final Duration TTL = Duration.ofMinutes(2); + private static byte[] bytes(String s) { return s.getBytes(StandardCharsets.UTF_8); } + private static final JdkCacheSerializer STRINGS = new JdkCacheSerializer<>(); + + @Test void allNamesClearOnlyTheirOwnDataAndPreserveControlFamilies() { + String[] names = {"", "user", "user:roles", "a?", "a1", "a*", "ab", "a[12]", "a2", "a\\b", + "tiercache", "tiercache:v2", "spring:users", "micronaut:users", "é", "e\u0301", "用户", "😀"}; + List> caches = new ArrayList<>(); + try (var connection = sharedClient().connect(ByteArrayCodec.INSTANCE)) { + byte[][] control = {RedisKeyspace.journal("users"), RedisKeyspace.trims("users"), + RedisKeyspace.tagKey("users", "group"), RedisKeyspace.reverseKey("users", bytes("k")), + bytes(RedisKeyspace.lock("users:k")), bytes("tiercache:journal:users"), + bytes("tiercache:tags:users:group"), bytes("tiercache:rebuild:users:k"), + bytes("user:legacy"), RedisKeyspace.channel("users"), RedisKeyspace.group("users", UUID.randomUUID())}; + for (byte[] key : control) connection.sync().set(key, bytes("sentinel")); + for (String name : names) { + var cache = LettuceRemoteCache.builder(redisUri()).client(sharedClient()).cacheName(name).build(); + caches.add(cache); cache.put("k", StoredEntry.ofValue(name), TTL); + } + for (int i = 0; i < caches.size(); i++) { + caches.get(i).clear(); assertNull(caches.get(i).get("k")); + for (int j = 0; j < caches.size(); j++) if (i != j) assertEquals(names[j], caches.get(j).get("k").value()); + for (byte[] key : control) assertArrayEquals(bytes("sentinel"), connection.sync().get(key)); + caches.get(i).put("k", StoredEntry.ofValue(names[i]), TTL); + } + for (var cache : caches) cache.clear(); + connection.sync().del(control); + } finally { caches.forEach(LettuceRemoteCache::close); } + } + + @Test void binaryApplicationKeysAndDelimiterLookingTagsRemainOpaque() { + String name = "spring:user:?" + UUID.randomUUID(); + byte[] key = {0, (byte) 255, ':', '*', '[', ']', '\\', 0, ':'}; + CacheSerializer identity = new CacheSerializer<>() { + public byte[] toBytes(byte[] value) { return value; } + public byte[] fromBytes(byte[] value) { return value; } + }; + try (var connection = sharedClient().connect(ByteArrayCodec.INSTANCE); + var cache = LettuceRemoteCache.builder(redisUri()).client(sharedClient()) + .cacheName(name).keySerializer(identity).build()) { + cache.putTagged(key, StoredEntry.ofValue("v"), TTL, new String[]{"A:B", "A?", "用户"}); + byte[] physical = RedisKeyspace.dataKey(name, key); + assertNotNull(connection.sync().get(physical)); + assertTrue(connection.sync().exists(RedisKeyspace.reverseKey(name, key)) == 1); + assertArrayEquals(key, cache.keysByTag("A:B").get(0)); + assertEquals("v", cache.get(key).value()); + cache.putTagged(key, StoredEntry.ofValue("new"), TTL, new String[]{"B:A"}); + assertTrue(cache.keysByTag("A:B").isEmpty()); + assertArrayEquals(physical, connection.sync().smembers(RedisKeyspace.tagKey(name, "B:A")).iterator().next()); + cache.evict(key); + assertNull(connection.sync().get(physical)); + assertEquals(0, connection.sync().exists(RedisKeyspace.reverseKey(name, key))); + assertTrue(cache.keysByTag("B:A").isEmpty()); + } + } + + @Test void malformedNamesAndTagsFailBeforeMutation() { + String name = "malformed-" + UUID.randomUUID(); + String bad = "x\ud800"; + assertThrows(IllegalArgumentException.class, () -> LettuceRemoteCache.builder(redisUri()) + .client(sharedClient()).cacheName(bad).build()); + assertThrows(IllegalArgumentException.class, () -> LettuceRemoteCache.builder(redisUri()) + .client(sharedClient()).cacheName(name).journalName(bad).build()); + try (var connection = sharedClient().connect(ByteArrayCodec.INSTANCE); + var cache = LettuceRemoteCache.builder(redisUri()).client(sharedClient()).cacheName(name).build(); + var pubsub = new LettucePubSubInvalidationTransport(sharedClient(), new JdkCacheSerializer<>()); + var locks = new LettuceLockProvider(sharedClient())) { + cache.putTagged("k", StoredEntry.ofValue("old"), TTL, new String[]{"A"}); + assertThrows(IllegalArgumentException.class, () -> cache.putTagged("k", StoredEntry.ofValue("bad"), TTL, new String[]{"B", bad})); + assertEquals("old", cache.get("k").value()); assertEquals(List.of("k"), cache.keysByTag("A")); + assertTrue(cache.keysByTag("B").isEmpty()); + assertThrows(IllegalArgumentException.class, () -> pubsub.subscribe(bad, m -> fail("unexpected event"))); + assertThrows(IllegalArgumentException.class, () -> locks.tryLock(bad, TTL)); + var journal = new RedisStreamJournal(connection, 1000, new JdkCacheSerializer<>()); + var v = new Version(1, UUID.randomUUID()); + assertThrows(IllegalArgumentException.class, () -> journal.append(bad, + new InvalidationMessage(bad, "k", v, v.instanceId(), InvalidationMessage.Type.INVALIDATE))); + } + } + + @Test void legacyDataIsNeitherReadNorDeletedDuringColdCutover() { + String name = "cutover-" + UUID.randomUUID(); + byte[] v2 = RedisKeyspace.dataKey(name, STRINGS.toBytes("k")); + byte[] legacy = RedisKeyspace.join(bytes(name + ":"), STRINGS.toBytes("k")); + try (var connection = sharedClient().connect(ByteArrayCodec.INSTANCE); + var remote = LettuceRemoteCache.builder(redisUri()).client(sharedClient()).cacheName(name).build()) { + remote.put("k", StoredEntry.ofValue("retired"), TTL); + byte[] frame = connection.sync().get(v2); + connection.sync().set(legacy, frame); connection.sync().del(v2); + AtomicInteger loads = new AtomicInteger(); + try (var factory = TierCacheFactory.builder().remoteCache(remote).build()) { + var cache = factory.getCache(name); + assertEquals("source", cache.getOrCompute("k", k -> { loads.incrementAndGet(); return "source"; })); + assertEquals(1, loads.get()); + cache.evictAll(); + assertArrayEquals(frame, connection.sync().get(legacy)); + assertNull(connection.sync().get(v2)); + } + connection.sync().del(legacy); + } + } + + @Test void pubsubUsesV2ChannelAndIgnoresLegacyEvents() throws Exception { + String name = "channel:?" + UUID.randomUUID(); + var id = UUID.randomUUID(); + var old = new InvalidationMessage(name, "old", new Version(1, id), id, InvalidationMessage.Type.INVALIDATE); + var current = new InvalidationMessage(name, "new", new Version(2, id), id, InvalidationMessage.Type.INVALIDATE); + var received = new CopyOnWriteArrayList(); + var delivered = new CountDownLatch(1); + try (var connection = sharedClient().connect(ByteArrayCodec.INSTANCE); + var transport = new LettucePubSubInvalidationTransport(sharedClient(), new JdkCacheSerializer<>()); + var subscription = transport.subscribe(name, message -> { received.add(message); delivered.countDown(); })) { + assertEquals(0, connection.sync().publish(bytes("tiercache:inv:" + name), MessageCodec.encode(old, STRINGS.toBytes("old")))); + transport.publish(current); + assertTrue(delivered.await(5, TimeUnit.SECONDS)); + assertEquals(List.of(current), received); + assertEquals(1, connection.sync().pubsubNumsub(RedisKeyspace.channel(name)).values().iterator().next()); + } + } + + @Test void streamsStartInV2AndLeaveLegacyJournalAndGroupsUntouched() throws Exception { + String name = "stream:?" + UUID.randomUUID(); UUID id = UUID.randomUUID(); + byte[] legacy = bytes("tiercache:journal:" + name); + byte[] legacyGroup = bytes("tiercache:cg:" + name + ":" + id); + var delivered = new CountDownLatch(1); var received = new CopyOnWriteArrayList(); + try (var connection = sharedClient().connect(ByteArrayCodec.INSTANCE)) { + var cmd = connection.sync(); + cmd.xadd(legacy, Map.of(bytes("retired"), bytes("event"))); + cmd.xgroupCreate(XReadArgs.StreamOffset.from(legacy, "0-0"), legacyGroup); + byte[] before = cmd.dump(legacy); + var journal = new RedisStreamJournal(connection, 1000, new JdkCacheSerializer<>()); + try (var transport = new LettuceStreamsInvalidationTransport(sharedClient(), new JdkCacheSerializer<>(), new JdkCacheSerializer<>(), id); + var subscription = transport.subscribe(name, message -> { received.add(message); delivered.countDown(); })) { + var v = new Version(1, UUID.randomUUID()); + var event = new InvalidationMessage(name, "new", v, v.instanceId(), InvalidationMessage.Type.INVALIDATE); + journal.append(name, event); + assertTrue(delivered.await(5, TimeUnit.SECONDS)); assertEquals(List.of(event), received); + // An exact group operation proves subscription used this v2 identity. + assertNotNull(cmd.xpending(RedisKeyspace.journal(name), RedisKeyspace.group(name, id))); + assertArrayEquals(before, cmd.dump(legacy)); + assertEquals(1, journal.size(name)); + } + cmd.del(legacy, RedisKeyspace.journal(name)); + } + } + + @Test void lockAcquireExtendAndReleaseUseV2WithoutTouchingLegacyLock() { + String name = "lock:?" + UUID.randomUUID(); String legacy = "tiercache:rebuild:" + name; + try (var connection = sharedClient().connect(); var provider = new LettuceLockProvider(connection)) { + connection.sync().set(legacy, "legacy-owner"); + var lock = provider.tryLock(name, Duration.ofSeconds(5)); assertNotNull(lock); + String key = RedisKeyspace.lock(name); + assertNotNull(connection.sync().get(key)); + assertNull(provider.tryLock(name, Duration.ofSeconds(5))); + assertTrue(lock.extend(TTL)); assertTrue(connection.sync().pttl(key) > 60000); + lock.release(); assertNull(connection.sync().get(key)); + assertEquals("legacy-owner", connection.sync().get(legacy)); + connection.sync().del(legacy); + } + } + @Test + @SuppressWarnings("unchecked") + void uncertainLockAcquireCompensatesOnlyTheV2Token() throws Exception { + String name = "compensate:?" + UUID.randomUUID(); + String legacy = "tiercache:rebuild:" + name; + try (var connection = sharedClient().connect()) { + connection.sync().set(legacy, "old-owner"); + var captured = new java.util.concurrent.atomic.AtomicReference(); + var commands = (io.lettuce.core.api.sync.RedisCommands) java.lang.reflect.Proxy.newProxyInstance( + getClass().getClassLoader(), new Class[]{io.lettuce.core.api.sync.RedisCommands.class}, + (proxy, method, args) -> { + Object result; + try { result = method.invoke(connection.sync(), args); } + catch (java.lang.reflect.InvocationTargetException e) { throw e.getCause(); } + if ("set".equals(method.getName()) && args.length == 3 && args[2] instanceof io.lettuce.core.SetArgs) { + captured.set((String) args[0]); + throw new io.lettuce.core.RedisCommandTimeoutException("accepted SET, lost reply"); + } + return result; + }); + var wrapped = (io.lettuce.core.api.StatefulRedisConnection) java.lang.reflect.Proxy.newProxyInstance( + getClass().getClassLoader(), new Class[]{io.lettuce.core.api.StatefulRedisConnection.class}, + (proxy, method, args) -> { + if ("sync".equals(method.getName())) return commands; + throw new UnsupportedOperationException(method.getName()); + }); + try (var provider = new LettuceLockProvider(wrapped)) { + assertThrows(io.lettuce.core.RedisCommandTimeoutException.class, () -> provider.tryLock(name, TTL)); + assertEquals(RedisKeyspace.lock(name), captured.get()); + long deadline = System.nanoTime() + Duration.ofSeconds(5).toNanos(); + while (connection.sync().get(captured.get()) != null && System.nanoTime() < deadline) Thread.sleep(10); + assertNull(connection.sync().get(captured.get())); + assertEquals("old-owner", connection.sync().get(legacy)); + } + connection.sync().del(legacy); + } + } + +} diff --git a/tiercache-transport-redis/src/test/java/io/tiercache/redis/JournalCapacityTest.java b/tiercache-transport-redis/src/test/java/io/tiercache/redis/JournalCapacityTest.java new file mode 100644 index 0000000..e880ef3 --- /dev/null +++ b/tiercache-transport-redis/src/test/java/io/tiercache/redis/JournalCapacityTest.java @@ -0,0 +1,36 @@ +package io.tiercache.redis; + +import io.lettuce.core.api.StatefulRedisConnection; +import org.junit.jupiter.params.ParameterizedTest; +import org.junit.jupiter.params.provider.ValueSource; +import java.lang.reflect.Proxy; +import java.util.concurrent.atomic.AtomicInteger; +import static org.junit.jupiter.api.Assertions.*; + +class JournalCapacityTest { + @SuppressWarnings("unchecked") + static StatefulRedisConnection connection(AtomicInteger calls) { + return (StatefulRedisConnection)Proxy.newProxyInstance(JournalCapacityTest.class.getClassLoader(), + new Class[]{StatefulRedisConnection.class},(p,m,a)->{calls.incrementAndGet();return null;}); + } + @ParameterizedTest @ValueSource(ints={-1,0,1,32,63,64}) + void invalidBothConstructorsDoNotAccessConnection(int capacity) { + var calls=new AtomicInteger();var connection=connection(calls);var serializer=new JdkCacheSerializer(); + for(boolean separate:new boolean[]{false,true}) { + var error=assertThrows(IllegalArgumentException.class,()->{ + if(separate)new RedisStreamJournal(connection,capacity,serializer,serializer); + else new RedisStreamJournal(connection,capacity,serializer); + }); + assertTrue(error.getMessage().contains("tiercache.invalidation.journal-capacity")); + assertTrue(error.getMessage().contains(String.valueOf(capacity)));assertTrue(error.getMessage().contains("65")); + assertTrue(error.getMessage().contains("64"));assertEquals(0,calls.get()); + } + } + @ParameterizedTest @ValueSource(ints={65,10000}) + void validBoundariesReachConnectionWithoutClamping(int capacity) { + var calls=new AtomicInteger();var serializer=new JdkCacheSerializer(); + assertEquals(capacity,new RedisStreamJournal(connection(calls),capacity,serializer).capacity()); + assertEquals(capacity,new RedisStreamJournal(connection(calls),capacity,serializer,serializer).capacity()); + assertEquals(2,calls.get()); + } +} diff --git a/tiercache-transport-redis/src/test/java/io/tiercache/redis/LettuceLockProviderCompensationTest.java b/tiercache-transport-redis/src/test/java/io/tiercache/redis/LettuceLockProviderCompensationTest.java index ccd780c..c0e80c0 100644 --- a/tiercache-transport-redis/src/test/java/io/tiercache/redis/LettuceLockProviderCompensationTest.java +++ b/tiercache-transport-redis/src/test/java/io/tiercache/redis/LettuceLockProviderCompensationTest.java @@ -41,7 +41,7 @@ class LettuceLockProviderCompensationTest { @BeforeAll static void startServer() { - server = new GenericContainer<>(DockerImageName.parse("redis:6.2-alpine")) + server = new GenericContainer<>(ServerProfile.image()) .withExposedPorts(6379); server.start(); redisUri = "redis://" + server.getHost() + ":" + server.getMappedPort(6379); @@ -91,7 +91,7 @@ private static void awaitTrue(Supplier check, long timeoutMillis, Strin } private static String lockKey(String name) { - return LettuceLockProvider.LOCK_KEYSPACE + name; + return RedisKeyspace.lock(name); } /** @@ -396,32 +396,34 @@ void closeDuringAcquireCreatesNoMachineryAndKeepsTheTimeout() throws Exception { @Test void closeWithLiveSchedulerSwallowsTheSchedulingRace() throws Exception { var connection = client.connect(); - // The failing proxy always reports a client timeout on acquire. + var entered = new java.util.concurrent.CountDownLatch(1); + var resume = new java.util.concurrent.CountDownLatch(1); + var calls = new java.util.concurrent.atomic.AtomicInteger(); + var original = new RedisCommandTimeoutException("simulated client timeout"); RedisCommands failing = proxy(connection, (args, method) -> { - if ("set".equals(method.getName()) && args != null && args.length == 3 - && args[2] instanceof SetArgs) { - throw new RedisCommandTimeoutException("simulated client timeout"); + if ("set".equals(method.getName()) && args != null && args.length == 3) { + if (calls.incrementAndGet() == 2) { + entered.countDown(); + try { assertTrue(resume.await(5, TimeUnit.SECONDS)); } + catch (InterruptedException e) { throw new AssertionError(e); } + } + throw original; } return passthrough(); }); - LettuceLockProvider provider = new LettuceLockProvider(connectionTo(failing)); - // First ambiguous acquire spins the scheduler up. - assertThrows(RedisCommandTimeoutException.class, - () -> provider.tryLock("close-live-1", Duration.ofSeconds(30))); - - provider.close(); - // Another ambiguous acquire on the closed provider: the scheduling - // race must be swallowed and the ORIGINAL timeout must surface. + var provider = new LettuceLockProvider(connectionTo(failing)); + var pool = java.util.concurrent.Executors.newSingleThreadExecutor(); try { - provider.tryLock("close-live-2", Duration.ofSeconds(30)); - throw new AssertionError("the acquire must fail"); - } catch (Throwable t) { - assertTrue(t instanceof RedisCommandTimeoutException, - "the original timeout surfaces even with a live scheduler, got " + t); - } - assertTrue(provider.pendingCompensations() <= 1, - "bookkeeping stays bounded after close, got " + provider.pendingCompensations()); - connection.close(); + assertThrows(RedisCommandTimeoutException.class, + () -> provider.tryLock("close-live-1", Duration.ofSeconds(30))); + var call = pool.submit(() -> provider.tryLock("close-live-2", Duration.ofSeconds(30))); + assertTrue(entered.await(5, TimeUnit.SECONDS)); + provider.close(); resume.countDown(); + var failure = assertThrows(java.util.concurrent.ExecutionException.class, + () -> call.get(5, TimeUnit.SECONDS)); + org.junit.jupiter.api.Assertions.assertSame(original, failure.getCause()); + assertEquals(0, provider.pendingCompensations()); + } finally { resume.countDown(); provider.close(); pool.shutdownNow(); connection.close(); } } /** diff --git a/tiercache-transport-redis/src/test/java/io/tiercache/redis/LettuceLockProviderTest.java b/tiercache-transport-redis/src/test/java/io/tiercache/redis/LettuceLockProviderTest.java index fab756c..22961a7 100644 --- a/tiercache-transport-redis/src/test/java/io/tiercache/redis/LettuceLockProviderTest.java +++ b/tiercache-transport-redis/src/test/java/io/tiercache/redis/LettuceLockProviderTest.java @@ -33,7 +33,7 @@ class LettuceLockProviderTest { @BeforeAll static void startServer() { - server = new GenericContainer<>(DockerImageName.parse("redis:6.2-alpine")) + server = new GenericContainer<>(ServerProfile.image()) .withExposedPorts(6379); server.start(); client = RedisClient.create( @@ -147,4 +147,32 @@ void concurrentAcquireHasSingleWinner() throws Exception { pool.shutdown(); assertEquals(1, wins.get()); } + @Test + void derivedProvidersAreLazyIndependentAndOwnedByTheirFactories() throws Exception { + try (var remote = LettuceRemoteCache.builder("redis://unused").client(client).build()) { + var a = io.tiercache.TierCacheFactory.builder().remoteCache(remote).build(); + var b = io.tiercache.TierCacheFactory.builder().remoteCache(remote).build(); + var field = io.tiercache.TierCacheFactory.class.getDeclaredField("lockProvider"); field.setAccessible(true); + var wrapper = io.tiercache.internal.BreakerLockProvider.class.getDeclaredField("delegate"); wrapper.setAccessible(true); + var pa = (LettuceLockProvider) wrapper.get(field.get(a)); + var pb = (LettuceLockProvider) wrapper.get(field.get(b)); + var connection = LettuceLockProvider.class.getDeclaredField("ownedConnection"); connection.setAccessible(true); + assertNull(connection.get(pa)); assertNull(connection.get(pb)); + try { + pa.tryLock("owned-a",Duration.ofSeconds(10)).release(); pb.tryLock("owned-b",Duration.ofSeconds(10)).release(); + var ca=(io.lettuce.core.api.StatefulRedisConnection)connection.get(pa); + var cb=(io.lettuce.core.api.StatefulRedisConnection)connection.get(pb); + a.close(); a.close(); assertFalse(ca.isOpen()); assertTrue(cb.isOpen()); + pb.tryLock("still-live",Duration.ofSeconds(10)).release(); + remote.put("alive",io.tiercache.spi.StoredEntry.ofValue("v"),Duration.ofMinutes(1)); + assertEquals("v",remote.get("alive").value()); + try(var borrowed=client.connect()) { + var p=new LettuceLockProvider(borrowed); + var old=p.tryLock("borrowed-live",Duration.ofSeconds(10)); p.close(); old.release(); + assertEquals("PONG",borrowed.sync().ping()); + } + } finally {a.close();b.close();} + } + } + } diff --git a/tiercache-transport-redis/src/test/java/io/tiercache/redis/LettuceRemoteCacheTest.java b/tiercache-transport-redis/src/test/java/io/tiercache/redis/LettuceRemoteCacheTest.java index 0e61a84..08f3132 100644 --- a/tiercache-transport-redis/src/test/java/io/tiercache/redis/LettuceRemoteCacheTest.java +++ b/tiercache-transport-redis/src/test/java/io/tiercache/redis/LettuceRemoteCacheTest.java @@ -33,7 +33,7 @@ class LettuceRemoteCacheTest { @BeforeAll static void startServer() { - server = new GenericContainer<>(DockerImageName.parse("redis:6.2-alpine")) + server = new GenericContainer<>(ServerProfile.image()) .withExposedPorts(6379); server.start(); redisUri = "redis://" + server.getHost() + ":" + server.getMappedPort(6379); @@ -234,12 +234,8 @@ private interface Check { } private static byte[] rawValue(LettuceRemoteCache cache, String cacheName, String key) { - byte[] prefix = (cacheName + ":").getBytes(StandardCharsets.UTF_8); - byte[] serializedKey = new JdkCacheSerializer().toBytes(key); - byte[] namespaced = new byte[prefix.length + serializedKey.length]; - System.arraycopy(prefix, 0, namespaced, 0, prefix.length); - System.arraycopy(serializedKey, 0, namespaced, prefix.length, serializedKey.length); - return cache.connection().sync().get(namespaced); + return cache.connection().sync().get(RedisKeyspace.dataKey(cacheName, + new JdkCacheSerializer().toBytes(key))); } private static LettuceRemoteCache sharedNamespaceInstance() { @@ -258,7 +254,7 @@ void tagIndexesExpireWithTheData() throws Exception { Duration.ofMillis(400), new String[]{"g"}); try (io.lettuce.core.RedisClient probe = io.lettuce.core.RedisClient.create(redisUri); var conn = probe.connect()) { - Long pttl = conn.sync().pttl("tiercache:tags:tag-ttl:g"); + Long pttl = conn.sync().pttl(new String(RedisKeyspace.tagKey("tag-ttl", "g"), StandardCharsets.US_ASCII)); assertTrue(pttl != null && pttl > 0 && pttl <= 400, "tag set must be TTL-bounded from write time, got PTTL=" + pttl); } @@ -266,7 +262,7 @@ void tagIndexesExpireWithTheData() throws Exception { // A dead member with a long-lived index row is filtered and pruned on lookup. try (io.lettuce.core.RedisClient probe = io.lettuce.core.RedisClient.create(redisUri); var conn = probe.connect()) { -conn.sync().sadd("tiercache:tags:tag-ttl:g", "tag-ttl:ghost"); +conn.sync().sadd(new String(RedisKeyspace.tagKey("tag-ttl", "g"), StandardCharsets.US_ASCII), "tag-ttl:ghost"); } assertTrue(cache.keysByTag("g").stream().noneMatch("ghost"::equals), "phantom members must not be returned"); @@ -276,7 +272,7 @@ void tagIndexesExpireWithTheData() throws Exception { assertTrue(cache.keysByTag("g").isEmpty()); try (io.lettuce.core.RedisClient probe = io.lettuce.core.RedisClient.create(redisUri); var conn = probe.connect()) { - assertTrue(conn.sync().keys("tiercache:tags:tag-ttl:*").isEmpty(), + assertTrue(conn.sync().keys(new String(RedisKeyspace.tagPrefix("tag-ttl"), StandardCharsets.US_ASCII) + "*").isEmpty(), "tag set must expire with the data"); } cache.close(); @@ -333,7 +329,7 @@ void tagSetSizeStaysBoundedUnderContinuousWrites() throws Exception { int writesBefore = writes.get(); java.util.List result = probe.sync().eval(measure, io.lettuce.core.ScriptOutputType.MULTI, - new byte[][]{"tiercache:tags:tag-hot:hot".getBytes(StandardCharsets.UTF_8)}); + new byte[][]{RedisKeyspace.tagKey("tag-hot", "hot")}); // The eval itself is milliseconds — too short to guarantee a // write inside it. Prove liveness instead: the counter must // advance right after the measurement, within a bounded wait. @@ -381,7 +377,7 @@ void idleUniqueTagsExpire() throws Exception { Thread.sleep(600); try (io.lettuce.core.RedisClient probe = io.lettuce.core.RedisClient.create(redisUri); var conn = probe.connect()) { - assertTrue(conn.sync().keys("tiercache:tags:tag-idle:*").isEmpty(), + assertTrue(conn.sync().keys(new String(RedisKeyspace.tagPrefix("tag-idle"), StandardCharsets.US_ASCII) + "*").isEmpty(), "idle tag sets must expire within the longest member TTL"); } cache.close(); @@ -438,7 +434,7 @@ void janitorNeverRemovesLiveMembership() throws Exception { } try (io.lettuce.core.RedisClient probeClient = io.lettuce.core.RedisClient.create(redisUri); var probe = probeClient.connect(io.lettuce.core.codec.ByteArrayCodec.INSTANCE)) { - byte[] setKey = ("tiercache:tags:tag-race:" + tag).getBytes(StandardCharsets.UTF_8); + byte[] setKey = RedisKeyspace.tagKey("tag-race", tag); // A long-lived tag (mixed-TTL reality): the SET outlives the // short-lived members, so the dead membership rows persist. probe.sync().pexpire(setKey, 30_000); @@ -461,7 +457,7 @@ void janitorNeverRemovesLiveMembership() throws Exception { } try (io.lettuce.core.RedisClient probeClient = io.lettuce.core.RedisClient.create(redisUri); var probe = probeClient.connect(io.lettuce.core.codec.ByteArrayCodec.INSTANCE)) { - byte[] setKey = ("tiercache:tags:tag-race:" + tag).getBytes(StandardCharsets.UTF_8); + byte[] setKey = RedisKeyspace.tagKey("tag-race", tag); long members = probe.sync().scard(setKey); assertTrue(members >= filler + victims / 3 && members <= filler + victims, "round " + round + ": filler plus most dead victim rows must be present " diff --git a/tiercache-transport-redis/src/test/java/io/tiercache/redis/LocalFreshnessFrameTest.java b/tiercache-transport-redis/src/test/java/io/tiercache/redis/LocalFreshnessFrameTest.java new file mode 100644 index 0000000..f9e2ab6 --- /dev/null +++ b/tiercache-transport-redis/src/test/java/io/tiercache/redis/LocalFreshnessFrameTest.java @@ -0,0 +1,32 @@ +package io.tiercache.redis; + +import io.tiercache.Version; +import io.tiercache.spi.StoredEntry; +import org.junit.jupiter.api.Test; +import java.util.*; +import java.nio.ByteBuffer; +import static org.junit.jupiter.api.Assertions.*; + +class LocalFreshnessFrameTest { + @Test void localStateNeverChangesLegacyVersionedOrTimestampedFrameLayout() { + var version=new Version(123,UUID.randomUUID());var serializer=new JdkCacheSerializer(); + for(var plain:List.of(StoredEntry.ofValue("v"),StoredEntry.nullMarker(), + StoredEntry.ofValue("v",version),StoredEntry.nullMarker(version), + StoredEntry.ofValue("v",version,321),StoredEntry.nullMarker(version,321))) { + var decorated=plain.withLocalFreshness(new StoredEntry.LocalFreshness(11,22,33,44,version)); + for(boolean timestamp:new boolean[]{false,true}) { + byte[] expected=ValueFrame.encode(plain,serializer,timestamp),actual=ValueFrame.encode(decorated,serializer,timestamp); + if(timestamp) { + int offset=5+ByteBuffer.wrap(expected,1,4).getInt(); + Arrays.fill(expected,offset,offset+8,(byte)0);Arrays.fill(actual,offset,offset+8,(byte)0); + } + assertArrayEquals(expected,actual); + var decoded=ValueFrame.decode(actual,serializer);assertNull(decoded.localFreshness()); + assertEquals(plain.version(),decoded.version());assertEquals(plain.isNullMarker(),decoded.isNullMarker()); + } + assertNull(plain.localFreshness());assertNotSame(plain,decorated); + assertEquals(plain.hasWriteTimestamp(),decorated.hasWriteTimestamp()); + if(plain.hasWriteTimestamp())assertEquals(plain.writeTimestampMillis(),decorated.writeTimestampMillis()); + } + } +} diff --git a/tiercache-transport-redis/src/test/java/io/tiercache/redis/LockOwnershipTest.java b/tiercache-transport-redis/src/test/java/io/tiercache/redis/LockOwnershipTest.java new file mode 100644 index 0000000..88d0194 --- /dev/null +++ b/tiercache-transport-redis/src/test/java/io/tiercache/redis/LockOwnershipTest.java @@ -0,0 +1,147 @@ +package io.tiercache.redis; + +import io.lettuce.core.RedisClient; +import io.lettuce.core.api.StatefulRedisConnection; +import io.lettuce.core.api.sync.RedisCommands; +import org.junit.jupiter.api.Test; +import java.lang.reflect.Proxy; +import java.time.Duration; +import java.util.concurrent.*; +import java.util.concurrent.atomic.AtomicInteger; +import static org.junit.jupiter.api.Assertions.*; + +class LockOwnershipTest { + static final Duration LEASE = Duration.ofSeconds(30); + @SuppressWarnings("unchecked") + static T proxy(Class type, java.lang.reflect.InvocationHandler handler) { + return (T) Proxy.newProxyInstance(type.getClassLoader(), new Class[]{type}, handler); + } + static class Client extends RedisClient { + final AtomicInteger connects = new AtomicInteger(), closes = new AtomicInteger(), sets = new AtomicInteger(); + volatile Runnable connecting = () -> {}; + volatile Runnable resolving = () -> {}; + final RedisCommands commands = proxy(RedisCommands.class, (p,m,a) -> { + if (m.getName().equals("set")) { sets.incrementAndGet(); return "OK"; } + if (m.getName().equals("eval")) return 1L; + return null; + }); + final StatefulRedisConnection connection = proxy(StatefulRedisConnection.class, (p,m,a) -> { + if (m.getName().equals("sync")) { resolving.run(); return commands; } + if (m.getName().equals("close")) closes.incrementAndGet(); + return null; + }); + @Override public StatefulRedisConnection connect() { + connects.incrementAndGet(); connecting.run(); return connection; + } + } + static void await(CountDownLatch latch) { + try { assertTrue(latch.await(5, TimeUnit.SECONDS)); } + catch (InterruptedException e) { throw new AssertionError(e); } + } + @Test void ownsConnectionButNotClient() { + var c = new Client(); + try { + var p = new LettuceLockProvider(c); + p.tryLock("x", LEASE).release(); p.close(); p.close(); + assertEquals(1,c.closes.get()); + assertFalse(c.getResources().eventExecutorGroup().isShuttingDown()); + } finally { c.shutdown(); } + } + @Test void borrowedConnectionSurvivesAndOldHandleDoesNotRenew() { + var c = new Client(); + try { + var p = new LettuceLockProvider(c.connection); + var lock = p.tryLock("x",LEASE); p.close(); + assertFalse(lock.extend(LEASE)); lock.release(); + assertEquals(0,c.closes.get()); assertEquals(0,c.connects.get()); + } finally { c.shutdown(); } + } + @Test void closeBeforeFirstUseNeverConnects() { + var c = new Client(); + try { + var p = new LettuceLockProvider(c); p.close(); + assertThrows(IllegalStateException.class, () -> p.tryLock("x",LEASE)); + assertEquals(0,c.connects.get()); assertEquals(0,p.pendingCompensations()); + } finally { c.shutdown(); } + } + @Test void lateConnectCannotPublish() throws Exception { + var c = new Client(); var pool = Executors.newFixedThreadPool(2); + var entered = new CountDownLatch(1); var resume = new CountDownLatch(1); + c.connecting = () -> { entered.countDown(); await(resume); }; + var p = new LettuceLockProvider(c); + try { + var call = pool.submit(() -> p.tryLock("x",LEASE)); await(entered); + p.close(); resume.countDown(); + assertThrows(ExecutionException.class, () -> call.get(5,TimeUnit.SECONDS)); + assertEquals(1,c.closes.get()); assertEquals(0,c.sets.get()); + } finally { resume.countDown(); p.close(); pool.shutdownNow(); c.shutdown(); } + } + @Test void proxyFailureDisposesOwnedConnectionAndAllowsRetry() { + var c = new Client(); var p = new LettuceLockProvider(c); + try { + c.resolving = () -> { throw new IllegalStateException("proxy"); }; + assertThrows(IllegalStateException.class, () -> p.tryLock("x",LEASE)); + assertEquals(1,c.closes.get()); assertEquals(0,p.pendingCompensations()); + c.resolving = () -> {}; assertNotNull(p.tryLock("x",LEASE)); + p.close(); assertEquals(2,c.closes.get()); + } finally { p.close(); c.shutdown(); } + } + @Test void concurrentFirstUseSharesOneInitialization() throws Exception { + var c=new Client(); var p=new LettuceLockProvider(c); var pool=Executors.newFixedThreadPool(8); + var entered=new CountDownLatch(1); var resume=new CountDownLatch(1); + c.connecting=()->{entered.countDown();await(resume);}; + try { + var futures=new java.util.ArrayList>(); + for(int i=0;i<8;i++) futures.add(pool.submit(()->p.tryLock("x",LEASE))); + await(entered); resume.countDown(); + for(var f:futures) assertNotNull(f.get(5,TimeUnit.SECONDS)); + assertEquals(1,c.connects.get()); + p.close(); assertEquals(1,c.closes.get()); + } finally {resume.countDown();p.close();pool.shutdownNow();c.shutdown();} + } + @Test void closeWhileResolvingProxyClosesLateOwnedConnection() throws Exception { + var c=new Client(); var p=new LettuceLockProvider(c); var pool=Executors.newFixedThreadPool(2); + var entered=new CountDownLatch(1); var resume=new CountDownLatch(1); + c.resolving=()->{entered.countDown();await(resume);}; + try { + var creator=pool.submit(()->p.tryLock("x",LEASE)); await(entered); + var waiter=pool.submit(()->p.tryLock("y",LEASE)); + p.close(); + assertThrows(ExecutionException.class,()->waiter.get(1,TimeUnit.SECONDS)); + resume.countDown(); assertThrows(ExecutionException.class,()->creator.get(5,TimeUnit.SECONDS)); + assertEquals(1,c.closes.get()); assertEquals(0,c.sets.get()); + } finally {resume.countDown();p.close();pool.shutdownNow();c.shutdown();} + } + @Test void acquireCompletingAfterCloseCleansCapturedToken() throws Exception { + var entered=new CountDownLatch(1); var resume=new CountDownLatch(1); var deletes=new AtomicInteger(); + RedisCommands commands=proxy(RedisCommands.class,(o,m,a)->{ + if(m.getName().equals("set")){entered.countDown();await(resume);return "OK";} + if(m.getName().equals("eval")){deletes.incrementAndGet();return 1L;} + return null; + }); + StatefulRedisConnection connection=proxy(StatefulRedisConnection.class,(o,m,a)->m.getName().equals("sync")?commands:null); + var p=new LettuceLockProvider(connection);var pool=Executors.newSingleThreadExecutor(); + try { + var call=pool.submit(()->p.tryLock("x",LEASE));await(entered);p.close();resume.countDown(); + var failure=assertThrows(ExecutionException.class,()->call.get(5,TimeUnit.SECONDS)); + assertInstanceOf(io.tiercache.internal.LockProviderClosedException.class,failure.getCause()); + assertEquals(1,deletes.get()); assertEquals(0,p.pendingCompensations()); + } finally {resume.countDown();p.close();pool.shutdownNow();} + } + @Test void closeDuringCompensationRetiresBookkeepingOnce() throws Exception { + var entered=new CountDownLatch(1); var resume=new CountDownLatch(1); + var original=new IllegalArgumentException("acquire"); + RedisCommands commands=proxy(RedisCommands.class,(o,m,a)->{ + if(m.getName().equals("set")) throw original; + if(m.getName().equals("eval")){entered.countDown();try{resume.await(5,TimeUnit.SECONDS);}catch(InterruptedException ignored){}return 0L;} + return null; + }); + StatefulRedisConnection connection=proxy(StatefulRedisConnection.class,(o,m,a)->m.getName().equals("sync")?commands:null); + var p=new LettuceLockProvider(connection); + try { + assertSame(original,assertThrows(IllegalArgumentException.class,()->p.tryLock("x",LEASE))); + await(entered);p.close();resume.countDown();p.close();assertEquals(0,p.pendingCompensations()); + } finally {resume.countDown();p.close();} + } + +} diff --git a/tiercache-transport-redis/src/test/java/io/tiercache/redis/PubSubInvalidationIT.java b/tiercache-transport-redis/src/test/java/io/tiercache/redis/PubSubInvalidationIT.java index 10c47b4..ed2c849 100644 --- a/tiercache-transport-redis/src/test/java/io/tiercache/redis/PubSubInvalidationIT.java +++ b/tiercache-transport-redis/src/test/java/io/tiercache/redis/PubSubInvalidationIT.java @@ -34,7 +34,7 @@ class PubSubInvalidationIT { @BeforeEach void startServer() { - server = new GenericContainer<>(DockerImageName.parse("redis:6.2-alpine")) + server = new GenericContainer<>(ServerProfile.image()) .withExposedPorts(6379); server.start(); String uri = "redis://" + server.getHost() + ":" + server.getMappedPort(6379); diff --git a/tiercache-transport-redis/src/test/java/io/tiercache/redis/PublicationOutcomeTest.java b/tiercache-transport-redis/src/test/java/io/tiercache/redis/PublicationOutcomeTest.java new file mode 100644 index 0000000..5701830 --- /dev/null +++ b/tiercache-transport-redis/src/test/java/io/tiercache/redis/PublicationOutcomeTest.java @@ -0,0 +1,59 @@ +package io.tiercache.redis; +import io.lettuce.core.RedisClient; +import io.lettuce.core.api.StatefulConnection; +import io.tiercache.*; +import io.tiercache.invalidation.InvalidationService; +import io.tiercache.spi.*; +import org.junit.jupiter.api.Test; +import org.testcontainers.containers.GenericContainer; +import org.testcontainers.utility.DockerImageName; +import java.util.*; +import java.util.concurrent.*; +import java.util.concurrent.atomic.*; +import static org.junit.jupiter.api.Assertions.*; + +class PublicationOutcomeTest { + static InvalidationMessage message(){var id=UUID.randomUUID();return new InvalidationMessage("no-subscribers-"+id,"key",new Version(1,id),id,InvalidationMessage.Type.INVALIDATE);} + @Test void nativeSuccessWithNoSubscribersAndClosedConnectionFailure() throws Exception { + try(var server=new GenericContainer<>(ServerProfile.image()).withExposedPorts(6379)) { + server.start();String uri="redis://"+server.getHost()+":"+server.getMappedPort(6379); + try(var client=RedisClient.create(uri);var remote=LettuceRemoteCache.builder(uri).client(client).build()) { + var transport=new LettucePubSubInvalidationTransport(client,new JdkCacheSerializer<>()); + assertEquals(PublicationOutcome.ACKNOWLEDGED,transport.publishAsync(message()).toCompletableFuture().get(3,TimeUnit.SECONDS)); + var original=new IllegalArgumentException("serializer payload must never be logged"); + CacheSerializer failingSerializer=new CacheSerializer<>() { + public byte[] toBytes(Object value){throw original;} + public Object fromBytes(byte[] bytes){throw new UnsupportedOperationException();} + }; + try(var failing=new LettucePubSubInvalidationTransport(client,failingSerializer)) { + var error=assertThrows(CompletionException.class,()->failing.publishAsync(message()).toCompletableFuture().join()); + assertSame(original,error.getCause());assertDoesNotThrow(()->failing.publish(message())); + } + var base=message();var update=new InvalidationMessage(base.cache(),base.key(),base.version(),base.originInstanceId(),InvalidationMessage.Type.UPDATE,"oversized"); + var received=new CompletableFuture(); + try(var capped=new LettucePubSubInvalidationTransport(client,new JdkCacheSerializer<>(),new JdkCacheSerializer<>(),1); + var subscription=transport.subscribe(update.cache(),received::complete)) { + assertEquals(PublicationOutcome.ACKNOWLEDGED,capped.publishAsync(update).toCompletableFuture().get(3,TimeUnit.SECONDS)); + assertEquals(InvalidationMessage.Type.INVALIDATE,received.get(3,TimeUnit.SECONDS).type()); + } + var outcomes=new AtomicLongArray(4);var sent=new AtomicInteger(); + var metrics=new CacheMetricsListener(){ + public void onPublication(String c,PublicationOutcome o,long count){outcomes.addAndGet(o.ordinal(),count);} + public void onInvalidation(String c,Direction d){if(d==Direction.SENT)sent.incrementAndGet();} + }; + try(var factory=TierCacheFactory.builder().remoteCache(remote) + .invalidation(v->new InvalidationService(transport,null,v.instanceId(),InvalidationListener.NOOP,metrics)).build()) { + var cache=factory.getCache("c");var field=LettucePubSubInvalidationTransport.class.getDeclaredField("connection");field.setAccessible(true); + ((StatefulConnection)field.get(transport)).close(); + assertDoesNotThrow(()->cache.put("x","committed"));assertEquals("committed",remote.get("x").value()); + long until=System.nanoTime()+TimeUnit.SECONDS.toNanos(5); + while(outcomes.get(PublicationOutcome.FAILED.ordinal())==0 && System.nanoTime()(),new JdkCacheSerializer<>())) { + assertEquals(PublicationOutcome.NOT_REQUIRED,streams.publishAsync(message()).toCompletableFuture().get()); + } + } + } + } +} diff --git a/tiercache-transport-redis/src/test/java/io/tiercache/redis/RedisKeyspaceTest.java b/tiercache-transport-redis/src/test/java/io/tiercache/redis/RedisKeyspaceTest.java new file mode 100644 index 0000000..e5b0c41 --- /dev/null +++ b/tiercache-transport-redis/src/test/java/io/tiercache/redis/RedisKeyspaceTest.java @@ -0,0 +1,47 @@ +package io.tiercache.redis; + +import org.junit.jupiter.api.Test; +import java.nio.charset.StandardCharsets; +import java.util.UUID; +import static org.junit.jupiter.api.Assertions.*; + +class RedisKeyspaceTest { + private static String text(byte[] bytes) { return new String(bytes, StandardCharsets.US_ASCII); } + + @Test void exactWireLayoutUsesCompleteTokensAndRawDataKeys() { + byte[] key = {0, (byte) 255, ':'}; + assertEquals("dXNlcjpyb2xlcw", RedisKeyspace.token("user:roles")); + assertEquals("", RedisKeyspace.token("")); + assertEquals("tiercache:v2:data:dXNlcjpyb2xlcw:", text(RedisKeyspace.dataPrefix("user:roles"))); + byte[] data = RedisKeyspace.dataKey("user:roles", key); + assertArrayEquals(key, java.util.Arrays.copyOfRange(data, data.length - key.length, data.length)); + assertEquals("tiercache:v2:tagkeys:dXNlcjpyb2xlcw:AP86", text(RedisKeyspace.reverseKey("user:roles", key))); + assertEquals("tiercache:v2:tags:dXNlcjpyb2xlcw:QTpC", text(RedisKeyspace.tagKey("user:roles", "A:B"))); + assertEquals("tiercache:v2:journal:dXNlcnM", text(RedisKeyspace.journal("users"))); + assertEquals("tiercache:v2:journal-trims:dXNlcnM", text(RedisKeyspace.trims("users"))); + assertEquals("tiercache:v2:inv:dXNlcnM", text(RedisKeyspace.channel("users"))); + UUID id = UUID.fromString("00000000-0000-0000-0000-000000000001"); + assertEquals("tiercache:v2:cg:dXNlcnM:" + id, text(RedisKeyspace.group("users", id))); + assertEquals("tiercache:v2:rebuild:dXNlcjpr", RedisKeyspace.lock("user:k")); + } + + @Test void namesRemainDistinctWithoutNormalizationAndCannotInjectDelimiters() { + String[] names = {"", "user", "user:roles", "a?", "a*", "a[12]", "a\\b", "é", "e\u0301", "Users", "users", "用户", "😀"}; + var tokens = new java.util.HashSet(); + for (String name : names) { + String token = RedisKeyspace.token(name); + assertTrue(token.matches("[A-Za-z0-9_-]*")); + assertTrue(tokens.add(token), name); + assertEquals(name, new String(java.util.Base64.getUrlDecoder().decode(token), StandardCharsets.UTF_8)); + } + assertThrows(IllegalArgumentException.class, () -> RedisKeyspace.token("\ud800")); + assertThrows(IllegalArgumentException.class, () -> RedisKeyspace.token("x\udc00")); + } + + @Test void redisLengthLimitUsesLongArithmeticWithoutAllocatingHugeKeys() { + assertDoesNotThrow(() -> RedisKeyspace.checkLength(512L * 1024 * 1024)); + assertThrows(IllegalArgumentException.class, () -> RedisKeyspace.checkLength(512L * 1024 * 1024 + 1)); + assertThrows(IllegalArgumentException.class, () -> RedisKeyspace.checkLength(Long.MAX_VALUE)); + assertThrows(IllegalArgumentException.class, () -> RedisKeyspace.checkLength(-1)); + } +} diff --git a/tiercache-transport-redis/src/test/java/io/tiercache/redis/RedisLettuceContractTest.java b/tiercache-transport-redis/src/test/java/io/tiercache/redis/RedisLettuceContractTest.java index a8be524..5376a33 100644 --- a/tiercache-transport-redis/src/test/java/io/tiercache/redis/RedisLettuceContractTest.java +++ b/tiercache-transport-redis/src/test/java/io/tiercache/redis/RedisLettuceContractTest.java @@ -3,10 +3,10 @@ import org.testcontainers.utility.DockerImageName; /** Contract suite against Redis 6.2 (the supported baseline). */ -class RedisLettuceContractTest extends AbstractLettuceContractTest { +class RedisLettuceContractTest extends AbstractNamespaceContractTest { @Override DockerImageName image() { - return DockerImageName.parse("redis:6.2-alpine"); + return ServerProfile.image(); } } diff --git a/tiercache-transport-redis/src/test/java/io/tiercache/redis/RedisStreamJournalTest.java b/tiercache-transport-redis/src/test/java/io/tiercache/redis/RedisStreamJournalTest.java index 457fba9..761ca3a 100644 --- a/tiercache-transport-redis/src/test/java/io/tiercache/redis/RedisStreamJournalTest.java +++ b/tiercache-transport-redis/src/test/java/io/tiercache/redis/RedisStreamJournalTest.java @@ -31,7 +31,7 @@ class RedisStreamJournalTest { @BeforeAll static void startServer() { - server = new GenericContainer<>(DockerImageName.parse("redis:6.2-alpine")) + server = new GenericContainer<>(ServerProfile.image()) .withExposedPorts(6379); server.start(); redisUri = "redis://" + server.getHost() + ":" + server.getMappedPort(6379); @@ -140,11 +140,11 @@ void extendedFrameStaysVersionComparableInLua() { @Test void trimmedCursorIsDetected() { - RedisStreamJournal journal = new RedisStreamJournal(client.connect(ByteArrayCodec.INSTANCE), 3, + RedisStreamJournal journal = new RedisStreamJournal(client.connect(ByteArrayCodec.INSTANCE), 65, new JdkCacheSerializer<>()); UUID origin = UUID.randomUUID(); // Approximate MAXLEN trims lazily, so drive the stream far past the - // window: 200 appends to a capacity-3 journal guarantee real trims. + // window: 200 appends to a capacity-65 journal guarantee real trims. for (int i = 1; i <= 200; i++) { journal.append("trimmed", new InvalidationMessage("trimmed", "k" + i, new Version(i, origin), origin, InvalidationMessage.Type.INVALIDATE)); @@ -173,8 +173,8 @@ void trimIsDetectedEvenAtExactCapacityLength() { var probe = client.connect(ByteArrayCodec.INSTANCE); try { var sync = probe.sync(); - byte[] streamKey = "tiercache:journal:boundary".getBytes(); - byte[] counter = sync.get("tiercache:journal-trims:boundary".getBytes()); + byte[] streamKey = RedisStreamJournal.streamKeyBytes("boundary"); + byte[] counter = sync.get(RedisStreamJournal.trimCounterKeyBytes("boundary")); assertTrue(counter != null && Long.parseLong(new String(counter)) > 0, "the drive must have counted real trims"); // Land exactly on the boundary the length heuristic missed. @@ -195,19 +195,19 @@ void trimIsDetectedEvenAtExactCapacityLength() { */ @Test void trimCounterIncrementsOnTheLuaWritePaths() { - RedisStreamJournal journal = new RedisStreamJournal(client.connect(ByteArrayCodec.INSTANCE), 3, + RedisStreamJournal journal = new RedisStreamJournal(client.connect(ByteArrayCodec.INSTANCE), 65, new JdkCacheSerializer<>()); LettuceRemoteCache cache = LettuceRemoteCache.builder(redisUri) .cacheName("jctr") .journal(journal) .build(); io.tiercache.VersionGenerator versions = new io.tiercache.VersionGenerator(); - for (int i = 0; i < 200; i++) { // capacity 3: trims are guaranteed + for (int i = 0; i < 200; i++) { // capacity 65: trims are guaranteed cache.put("k" + i, StoredEntry.ofValue("v", versions.next()), Duration.ofMinutes(1)); } var probe = client.connect(ByteArrayCodec.INSTANCE); try { - byte[] counter = probe.sync().get("tiercache:journal-trims:jctr".getBytes()); + byte[] counter = probe.sync().get(RedisStreamJournal.trimCounterKeyBytes("jctr")); assertTrue(counter != null && Long.parseLong(new String(counter)) > 0, "the conditional-write Lua path must count its trims"); } finally { @@ -224,7 +224,7 @@ void trimCounterIncrementsOnTheLuaWritePaths() { */ @Test void checkedReadNeverReturnsATornPrefix() throws Exception { - RedisStreamJournal journal = new RedisStreamJournal(client.connect(ByteArrayCodec.INSTANCE), 10, + RedisStreamJournal journal = new RedisStreamJournal(client.connect(ByteArrayCodec.INSTANCE), 65, new JdkCacheSerializer<>()); UUID origin = UUID.randomUUID(); String cursor = null; @@ -252,8 +252,9 @@ void checkedReadNeverReturnsATornPrefix() throws Exception { for (int i = 0; i < 2000 && !sawTrimmedCursor; i++) { io.tiercache.spi.CheckedRange range = journal.checkedRead("torn", fixedCursor, 50); if (range.startIntact()) { - assertEquals(fixedCursor, range.rows().get(0).cursor(), - "an intact read's head must be the cursor row itself"); + assertTrue(range.rows().stream().allMatch(row -> RedisStreamJournal.compareIds(fixedCursor, row.cursor()) < 0), + "the raw-validated anchor is omitted; events start strictly after it"); + assertTrue(range.rows().size() <= 50); } else if (!range.rows().isEmpty()) { assertTrue(RedisStreamJournal.compareIds(fixedCursor, range.rows().get(0).cursor()) < 0, "a non-intact read must not pose as a contiguous prefix"); diff --git a/tiercache-transport-redis/src/test/java/io/tiercache/redis/ServerProfile.java b/tiercache-transport-redis/src/test/java/io/tiercache/redis/ServerProfile.java new file mode 100644 index 0000000..528c011 --- /dev/null +++ b/tiercache-transport-redis/src/test/java/io/tiercache/redis/ServerProfile.java @@ -0,0 +1,11 @@ +package io.tiercache.redis; + +import org.testcontainers.utility.DockerImageName; + +/** Pinned test inputs shared by all transport integration tests. */ +final class ServerProfile { + private ServerProfile() {} + static DockerImageName image() { + return DockerImageName.parse(System.getProperty("tiercache.test.serverImage", "redis:6.2.24-alpine")); + } +} diff --git a/tiercache-transport-redis/src/test/java/io/tiercache/redis/StreamRowDecoderTest.java b/tiercache-transport-redis/src/test/java/io/tiercache/redis/StreamRowDecoderTest.java new file mode 100644 index 0000000..5858393 --- /dev/null +++ b/tiercache-transport-redis/src/test/java/io/tiercache/redis/StreamRowDecoderTest.java @@ -0,0 +1,52 @@ +package io.tiercache.redis; +import io.tiercache.*; +import org.junit.jupiter.api.Test; +import java.util.*; +import java.nio.charset.StandardCharsets; +import static org.junit.jupiter.api.Assertions.*; + +class StreamRowDecoderTest { + static final JdkCacheSerializer CODEC = new JdkCacheSerializer<>(); + static byte[] b(String value) { return value.getBytes(StandardCharsets.UTF_8); } + static Map row() { + Map result = new LinkedHashMap<>(); + result.put(b("t"),new byte[]{0}); result.put(b("k"),CODEC.toBytes("k")); + result.put(b("v"),b(new Version(1,UUID.randomUUID()).toWire())); return result; + } + static void replace(Map row, String field, byte[] value) { + row.keySet().removeIf(k -> Arrays.equals(k,b(field))); row.put(b(field),value); + } + @Test void malformedRowsHaveTypedSanitizedFailures() { + for (String kind : List.of("missing","type","version","key","payload","duplicate")) { + Map row = row(); + switch(kind) { + case "missing" -> row.clear(); + case "type" -> replace(row,"t",new byte[]{(byte)255}); + case "version" -> replace(row,"v",b("secret-invalid-version")); + case "key" -> replace(row,"k",b("secret-invalid-key")); + case "payload" -> replace(row,"p",b("secret-invalid-payload")); + case "duplicate" -> row.put(b("t"),new byte[]{0}); + } + var error = assertThrows(StreamRowCorruptionException.class, () -> StreamRowDecoder.decode("c","1-0",row,CODEC,CODEC)); + assertEquals("c",error.cache()); assertEquals("1-0",error.rowId()); assertNull(error.getCause()); + assertFalse(error.toString().contains("secret")); + } + assertThrows(StreamRowCorruptionException.class, () -> StreamRowDecoder.decode("c","1-0",null,CODEC,CODEC)); + } + @Test void validPayloadAndControlShapesStayCompatible() { + Map row = row(); row.put(b("p"),CODEC.toBytes("value")); + var message = StreamRowDecoder.decode("c","1-0",row,CODEC,CODEC); + assertEquals(InvalidationMessage.Type.UPDATE,message.type()); assertEquals("value",message.payload()); + replace(row,"t",new byte[]{1}); replace(row,"k",new byte[0]); replace(row,"p",new byte[0]); + assertEquals(InvalidationMessage.Type.EVICT_ALL,StreamRowDecoder.decode("c","2-0",row,CODEC,CODEC).type()); + replace(row,"p",CODEC.toBytes("unexpected")); + assertThrows(StreamRowCorruptionException.class,()->StreamRowDecoder.decode("c","2-0",row,CODEC,CODEC)); + var missingPayload=row(); replace(missingPayload,"t",new byte[]{2}); + assertThrows(StreamRowCorruptionException.class,()->StreamRowDecoder.decode("c","3-0",missingPayload,CODEC,CODEC)); + } + @Test void serializerNullsAreNotMistakenForSuccessfulInvalidation() { + CacheSerializer nulls=new CacheSerializer<>() { public byte[] toBytes(Object v){return new byte[0];} public Object fromBytes(byte[] b){return null;} }; + var row=row(); assertThrows(StreamRowCorruptionException.class,()->StreamRowDecoder.decode("c","1-0",row,nulls,CODEC)); + row.put(b("p"),CODEC.toBytes("value")); assertThrows(StreamRowCorruptionException.class,()->StreamRowDecoder.decode("c","1-0",row,CODEC,nulls)); + } +} diff --git a/tiercache-transport-redis/src/test/java/io/tiercache/redis/StreamsRecoveryTest.java b/tiercache-transport-redis/src/test/java/io/tiercache/redis/StreamsRecoveryTest.java new file mode 100644 index 0000000..c453405 --- /dev/null +++ b/tiercache-transport-redis/src/test/java/io/tiercache/redis/StreamsRecoveryTest.java @@ -0,0 +1,384 @@ +package io.tiercache.redis; + +import io.lettuce.core.*; +import io.lettuce.core.api.StatefulRedisConnection; +import io.lettuce.core.codec.ByteArrayCodec; +import io.tiercache.*; +import io.tiercache.invalidation.InvalidationService; +import io.tiercache.spi.*; +import org.junit.jupiter.api.*; +import org.testcontainers.containers.GenericContainer; +import org.testcontainers.utility.DockerImageName; +import java.nio.charset.StandardCharsets; +import java.util.*; +import java.util.concurrent.*; +import java.util.concurrent.atomic.*; +import java.util.function.*; +import static org.junit.jupiter.api.Assertions.*; + +@TestInstance(TestInstance.Lifecycle.PER_CLASS) +abstract class StreamsRecoveryTest { + abstract String image(); + GenericContainer server; RedisClient client; + static final JdkCacheSerializer CODEC = new JdkCacheSerializer<>(); + static byte[] bytes(String s) { return s.getBytes(StandardCharsets.UTF_8); } + @BeforeAll void start() { + server = new GenericContainer<>(DockerImageName.parse(image())).withExposedPorts(6379); server.start(); + client = RedisClient.create("redis://" + server.getHost() + ":" + server.getMappedPort(6379)); + } + @AfterAll void stop() { client.shutdown(); server.stop(); } + static void await(BooleanSupplier condition) throws Exception { + long deadline = System.nanoTime() + TimeUnit.SECONDS.toNanos(15); + while (!condition.getAsBoolean() && System.nanoTime() < deadline) Thread.sleep(10); + assertTrue(condition.getAsBoolean(), "condition did not settle"); + } + static void gate(CountDownLatch latch) { + try { if (!latch.await(10, TimeUnit.SECONDS)) throw new AssertionError("gate timeout"); } + catch (InterruptedException e) { Thread.currentThread().interrupt(); throw new IllegalStateException(e); } + } + static class Target implements InvalidationTarget { + final Map values = new ConcurrentHashMap<>(); final AtomicLong generation = new AtomicLong(); + final AtomicInteger clears = new AtomicInteger(); volatile boolean failClear; volatile Runnable beforeClear = () -> { }; + public Version versionOfL1Entry(Object key) { return null; } + public long recoveryGeneration() { return generation.get(); } + public void evictL1IfNewer(Object key, Version v) { values.remove(key); } + public void applyUpdateL1(Object key, Object value, Version v) { values.put(key, value); } + public void evictAllL1() { + beforeClear.run(); if (failClear) throw new IllegalStateException("clear failed"); + values.clear(); generation.incrementAndGet(); clears.incrementAndGet(); + } + } + static class Metrics implements CacheMetricsListener { + final Map failures = new ConcurrentHashMap<>(); + final List pending = new CopyOnWriteArrayList<>(); + public void onStreamFailure(String cache, StreamResult result) { failures.computeIfAbsent(result, r -> new AtomicInteger()).incrementAndGet(); } + public AutoCloseable registerRecovery(String cache, BooleanSupplier value) { pending.add(value); return () -> pending.remove(value); } + int count(StreamResult result) { return failures.getOrDefault(result, new AtomicInteger()).get(); } + boolean pending() { return pending.stream().anyMatch(BooleanSupplier::getAsBoolean); } + } + final class Harness implements AutoCloseable { + final String cache = "recovery-" + UUID.randomUUID(); final UUID id = UUID.randomUUID(); + final byte[] stream = RedisKeyspace.journal(cache), group = RedisKeyspace.group(cache, id); + final StatefulRedisConnection connection = client.connect(ByteArrayCodec.INSTANCE); + final RedisStreamJournal journal = new RedisStreamJournal(connection, 1000, CODEC); + final Target target = new Target(); final Metrics metrics = new Metrics(); + final AtomicLong sequence = new AtomicLong(); final AtomicReference baseline = new AtomicReference<>(); + volatile boolean failBaseline; volatile Runnable afterBaseline = () -> { }; + final LettuceStreamsInvalidationTransport transport = new LettuceStreamsInvalidationTransport(client, CODEC, CODEC, id); + final InvalidationService service; + Harness() { + InvalidationJournal observed = new InvalidationJournal() { + public String append(String c, InvalidationMessage m) { return journal.append(c, m); } + public List readRange(String c, String p) { return journal.readRange(c, p); } + public CheckedRange checkedRead(String c, String p, int n) { return journal.checkedRead(c, p, n); } + public boolean isTrimmed(String c, String p) { return journal.isTrimmed(c, p); } + public String endCursor(String c) { + String value = journal.endCursor(c); baseline.set(value); + if (failBaseline) throw new IllegalStateException("baseline unavailable"); + afterBaseline.run(); return value; + } + }; + service = new InvalidationService(transport, observed, UUID.randomUUID(), InvalidationListener.NOOP, metrics); + service.registerTarget(cache, target); + } + String put(String key, String value) { + var version = new Version(sequence.incrementAndGet(), id); + return journal.append(cache, new InvalidationMessage(cache, key, version, id, InvalidationMessage.Type.UPDATE, value)); + } + List poisonBatch() { + target.values.put("victim", "stale"); + return connection.sync().eval("local a=redis.call('xadd',KEYS[1],'*','t',string.char(99),'k',ARGV[1],'v',ARGV[3]); " + + "local b=redis.call('xadd',KEYS[1],'*','t',string.char(2),'k',ARGV[2],'v',ARGV[4],'p',ARGV[5]); return {a,b}", + ScriptOutputType.MULTI, new byte[][]{stream}, CODEC.toBytes("victim"), CODEC.toBytes("covered"), + bytes(new Version(sequence.incrementAndGet(), id).toWire()), bytes(new Version(sequence.incrementAndGet(), id).toWire()), CODEC.toBytes("obsolete")); + } + long pending() { return connection.sync().xpending(stream, group).getCount(); } + public void close() { service.close(); connection.close(); } + } + + @Test void baselineAndClearMustFinishBeforeAckAndLaterRowsStillApply() throws Exception { + try (Harness h = new Harness()) { + var captured = new CountDownLatch(1); var releaseBaseline = new CountDownLatch(1); + var clearing = new CountDownLatch(1); var releaseClear = new CountDownLatch(1); + h.afterBaseline = () -> { captured.countDown(); gate(releaseBaseline); }; + h.target.beforeClear = () -> { clearing.countDown(); gate(releaseClear); }; + int before = h.target.clears.get(); h.poisonBatch(); + try { + assertTrue(captured.await(5, TimeUnit.SECONDS)); assertEquals(2, h.pending()); + assertEquals("stale", h.target.values.get("victim")); assertTrue(h.metrics.pending()); + String after = h.put("after-baseline", "new"); + assertTrue(RedisStreamJournal.compareIds(after, h.baseline.get()) > 0); + releaseBaseline.countDown(); assertTrue(clearing.await(5, TimeUnit.SECONDS)); + assertEquals(2, h.pending(), "clear is not yet committed, so poison cannot be ACKed"); + } finally { releaseBaseline.countDown(); releaseClear.countDown(); } + await(() -> h.pending() == 0 && "new".equals(h.target.values.get("after-baseline"))); + assertFalse(h.target.values.containsKey("victim")); assertFalse(h.target.values.containsKey("covered")); + assertEquals(before + 1, h.target.clears.get()); + assertEquals(1, h.metrics.count(CacheMetricsListener.StreamResult.DECODE_FAILED)); + } + } + + @Test void failedBaselineAndFailedClearDoNotAuthorizeAck() throws Exception { + for (boolean baselineFailure : new boolean[]{true, false}) { + try (Harness h = new Harness()) { + int initialClears = h.target.clears.get(); + h.failBaseline = baselineFailure; h.target.failClear = !baselineFailure; + h.poisonBatch(); + await(() -> h.metrics.count(CacheMetricsListener.StreamResult.RESYNC_FAILED) > 0); + assertEquals(2, h.pending()); assertTrue(h.metrics.pending()); + assertEquals(initialClears + (baselineFailure ? 1 : 0), h.target.clears.get()); + if (baselineFailure) assertFalse(h.target.values.containsKey("victim")); + h.failBaseline = false; h.target.failClear = false; + await(() -> h.pending() == 0); + assertFalse(h.target.values.containsKey("victim")); + h.put("after", "ok"); await(() -> "ok".equals(h.target.values.get("after"))); + } + } + } + + @Test void corruptTailIsAnOpaqueCheckedReadAnchorAndReplayRecovers() throws Exception { + String cache = "anchor-" + UUID.randomUUID(); var target = new Target(); + try (var connection = client.connect(ByteArrayCodec.INSTANCE)) { + var journal = new RedisStreamJournal(connection, 1000, CODEC); + var pubsub = new LettucePubSubInvalidationTransport(client, CODEC); + try (var service = new InvalidationService(pubsub, journal, UUID.randomUUID(), InvalidationListener.NOOP)) { + service.registerTarget(cache, target); target.values.put("victim", "stale"); + String bad = connection.sync().xadd(RedisKeyspace.journal(cache), Map.of(bytes("t"), new byte[]{99}, bytes("k"), CODEC.toBytes("victim"), bytes("v"), bytes("broken-secret"))); + var error = assertThrows(StreamRowCorruptionException.class, () -> journal.checkedRead(cache, "0-0", 10)); + assertEquals(cache, error.cache()); assertEquals(bad, error.rowId()); assertNull(error.getCause()); + assertFalse(error.toString().contains("broken-secret")); + assertTrue(service.recoverAsync(Runnable::run).toCompletableFuture().get(5, TimeUnit.SECONDS)); + assertFalse(target.values.containsKey("victim")); + CheckedRange atTail = journal.checkedRead(cache, bad, 1); + assertTrue(atTail.startIntact()); assertTrue(atTail.rows().isEmpty()); + var v = new Version(1, UUID.randomUUID()); + String next = journal.append(cache, new InvalidationMessage(cache, "later", v, v.instanceId(), InvalidationMessage.Type.UPDATE, "typed")); + CheckedRange after = journal.checkedRead(cache, bad, 1); + assertTrue(after.startIntact()); assertEquals(1, after.rows().size()); + assertEquals(next, after.rows().get(0).cursor()); assertEquals("typed", after.rows().get(0).message().payload()); + assertTrue(service.recoverAsync(Runnable::run).toCompletableFuture().get(5, TimeUnit.SECONDS)); + assertEquals("typed", target.values.get("later")); assertEquals(1, target.clears.get()); + connection.sync().xdel(RedisKeyspace.journal(cache), bad); + assertFalse(journal.checkedRead(cache, bad, 1).startIntact()); + } + } + } + + @Test void stableResumeClaimsOnlyItsOwnGroupAndSkipsCoveredHistoricalUpdates() throws Exception { + for (String oldConsumer : List.of("main", "previous")) { + String cache = "resume-" + UUID.randomUUID(); UUID owner = UUID.randomUUID(), spectator = UUID.randomUUID(); + byte[] stream = RedisKeyspace.journal(cache), own = RedisKeyspace.group(cache, owner), other = RedisKeyspace.group(cache, spectator); + try (var connection = client.connect(ByteArrayCodec.INSTANCE)) { + var commands = connection.sync(); var journal = new RedisStreamJournal(connection, 1000, CODEC); + var v = new Version(1, UUID.randomUUID()); + journal.append(cache, new InvalidationMessage(cache, "historical", v, v.instanceId(), InvalidationMessage.Type.UPDATE, "stale")); + commands.xgroupCreate(XReadArgs.StreamOffset.from(stream, "0-0"), own); + commands.xgroupCreate(XReadArgs.StreamOffset.from(stream, "0-0"), other); + commands.xreadgroup(io.lettuce.core.Consumer.from(own, bytes(oldConsumer)), XReadArgs.StreamOffset.lastConsumed(stream)); + commands.xreadgroup(io.lettuce.core.Consumer.from(other, bytes("spectator")), XReadArgs.StreamOffset.lastConsumed(stream)); + var transport = new LettuceStreamsInvalidationTransport(client, CODEC, CODEC, owner); + var target = new Target(); + try (var service = new InvalidationService(transport, journal, UUID.randomUUID(), InvalidationListener.NOOP)) { + service.registerTarget(cache, target); + await(() -> commands.xpending(stream, own).getCount() == 0); + assertFalse(target.values.containsKey("historical")); + assertEquals(1, commands.xpending(stream, other).getCount(), "another receiver's PEL must be untouched"); + journal.append(cache, new InvalidationMessage(cache, "new", new Version(2, v.instanceId()), v.instanceId(), InvalidationMessage.Type.UPDATE, "fresh")); + await(() -> "fresh".equals(target.values.get("new"))); + } + assertNotNull(commands.xpending(stream, own), "stable group survives close"); + assertEquals(1, commands.xpending(stream, other).getCount()); + } + } + } + + @Test void missingPendingPayloadIsNotRemovedBeforeSafeReset() throws Exception { + for (String oldConsumer : List.of("main", "previous")) { + String cache = "missing-" + UUID.randomUUID(); UUID id = UUID.randomUUID(); + byte[] stream = RedisKeyspace.journal(cache), group = RedisKeyspace.group(cache, id); + try (var connection = client.connect(ByteArrayCodec.INSTANCE)) { + var commands = connection.sync(); var journal = new RedisStreamJournal(connection, 1000, CODEC); + var v = new Version(1, UUID.randomUUID()); + String missing = journal.append(cache, new InvalidationMessage(cache, "victim", v, v.instanceId(), InvalidationMessage.Type.INVALIDATE)); + commands.xgroupCreate(XReadArgs.StreamOffset.from(stream, "0-0"), group); + commands.xreadgroup(io.lettuce.core.Consumer.from(group, bytes(oldConsumer)), XReadArgs.StreamOffset.lastConsumed(stream)); + commands.xdel(stream, missing); + String baseline = journal.append(cache, new InvalidationMessage(cache, "tail", v, v.instanceId(), InvalidationMessage.Type.INVALIDATE)); + var future = new CompletableFuture(); var asked = new CountDownLatch(1); + var delivered = new CountDownLatch(1); + var initial = new RecoveryResult(RecoveryResult.Status.RESET_SAFE, "0-0", 1); + var current = new AtomicReference<>(initial); + try (var transport = new LettuceStreamsInvalidationTransport(client, CODEC, CODEC, id)) { + transport.setGapHandler(new InvalidationGapHandler() { + public CompletionStage reset(String c) { asked.countDown(); return future; } + public RecoveryResult registrationBaseline(String c) { return initial; } + public boolean isCurrent(String c, RecoveryResult r) { return r == current.get(); } + }); + transport.subscribe(cache, message -> { + if ("after-reset".equals(message.key())) delivered.countDown(); + }); + assertTrue(asked.await(5, TimeUnit.SECONDS)); + assertEquals(1, commands.xpending(stream, group).getCount(), "claim must not silently delete the missing PEL entry"); + var safe = new RecoveryResult(RecoveryResult.Status.RESET_SAFE, baseline, 2); current.set(safe); future.complete(safe); + // A zero PEL can be transient before the reader fetches the tail. + // Require delivery past the reset baseline before checking drainage. + journal.append(cache, new InvalidationMessage(cache, "after-reset", + new Version(2, v.instanceId()), v.instanceId(), InvalidationMessage.Type.INVALIDATE)); + assertTrue(delivered.await(15, TimeUnit.SECONDS)); + await(() -> commands.xpending(stream, group).getCount() == 0); + } + } + } + } + + @Test void standalonePoisonRemainsPendingWhileAnotherCacheMakesProgress() throws Exception { + String bad = "blocked-" + UUID.randomUUID(), good = "healthy-" + UUID.randomUUID(); UUID id = UUID.randomUUID(); + Metrics metrics = new Metrics(); var delivered = new CountDownLatch(1); + try (var connection = client.connect(ByteArrayCodec.INSTANCE); + var transport = new LettuceStreamsInvalidationTransport(client, CODEC, CODEC, id)) { + transport.setMetricsListener(metrics); transport.subscribe(bad, message -> fail("poison dispatched")); + transport.subscribe(good, message -> delivered.countDown()); + connection.sync().xadd(RedisKeyspace.journal(bad), Map.of(bytes("t"), new byte[]{99})); + await(() -> metrics.count(CacheMetricsListener.StreamResult.RESYNC_FAILED) > 0); + assertTrue(metrics.pending()); + var journal = new RedisStreamJournal(connection, 1000, CODEC); var v = new Version(1, UUID.randomUUID()); + journal.append(good, new InvalidationMessage(good, "k", v, v.instanceId(), InvalidationMessage.Type.INVALIDATE)); + assertTrue(delivered.await(1, TimeUnit.SECONDS)); + Thread.sleep(250); + assertEquals(1, metrics.count(CacheMetricsListener.StreamResult.DECODE_FAILED), "known poison is not repeatedly deserialized"); + assertEquals(1, connection.sync().xpending(RedisKeyspace.journal(bad), RedisKeyspace.group(bad, id)).getCount()); + } + assertTrue(metrics.pending.isEmpty(), "both gauge registrations must be removed"); + } + + @Test void lateAndSupersededResetResultsCannotAcknowledgeRows() throws Exception { + for (boolean close : new boolean[]{true, false}) { + String cache = "late-" + UUID.randomUUID(); UUID id = UUID.randomUUID(); + byte[] stream = RedisKeyspace.journal(cache), group = RedisKeyspace.group(cache, id); + var first = new CompletableFuture(); var second = new CompletableFuture(); + var count = new AtomicInteger(); var initial = new RecoveryResult(RecoveryResult.Status.RESET_SAFE, "0-0", 1); + var current = new AtomicReference<>(initial); Metrics metrics = new Metrics(); + try (var connection = client.connect(ByteArrayCodec.INSTANCE); + var transport = new LettuceStreamsInvalidationTransport(client, CODEC, CODEC, id)) { + transport.setMetricsListener(metrics); + transport.setGapHandler(new InvalidationGapHandler() { + public CompletionStage reset(String c) { return count.incrementAndGet() == 1 ? first : second; } + public RecoveryResult registrationBaseline(String c) { return initial; } + public boolean isCurrent(String c, RecoveryResult r) { return r == current.get(); } + }); + transport.subscribe(cache, message -> { }); + String row = connection.sync().xadd(stream, Map.of(bytes("t"), new byte[]{99})); + await(() -> count.get() == 1); + var old = new RecoveryResult(RecoveryResult.Status.RESET_SAFE, row, 2); + if (close) { current.set(old); transport.close(); } + first.complete(old); + if (!close) await(() -> metrics.count(CacheMetricsListener.StreamResult.RESYNC_FAILED) > 0); + assertEquals(1, connection.sync().xpending(stream, group).getCount()); + if (!close) { + await(() -> count.get() == 2); + var safe = new RecoveryResult(RecoveryResult.Status.RESET_SAFE, row, 3); current.set(safe); second.complete(safe); + await(() -> connection.sync().xpending(stream, group).getCount() == 0); + } + } + } + } + + @Test void ackFailureDoesNotReapplyOrLoseTheRestOfTheBatch() throws Exception { + for (boolean replyLost : new boolean[]{false, true}) { + String cache = "ack-" + UUID.randomUUID(); UUID id = UUID.randomUUID(); + byte[] stream = RedisKeyspace.journal(cache), group = RedisKeyspace.group(cache, id); + try (var probe = client.connect(ByteArrayCodec.INSTANCE)) { + var real = client.connect(ByteArrayCodec.INSTANCE); var once = new AtomicBoolean(); + var applied = new ConcurrentHashMap(); Metrics metrics = new Metrics(); + var wrapped = ackFailureConnection(real, once, replyLost); + try (var transport = new LettuceStreamsInvalidationTransport(wrapped, CODEC, CODEC, id, true)) { + transport.setMetricsListener(metrics); + transport.subscribe(cache, message -> applied.computeIfAbsent(message.key(), k -> new AtomicInteger()).incrementAndGet()); + var v = new Version(1, UUID.randomUUID()); + probe.sync().eval("redis.call('xadd',KEYS[1],'*','t',string.char(0),'k',ARGV[1],'v',ARGV[3]); " + + "redis.call('xadd',KEYS[1],'*','t',string.char(0),'k',ARGV[2],'v',ARGV[3]); return 1", + ScriptOutputType.INTEGER, new byte[][]{stream}, CODEC.toBytes("first"), CODEC.toBytes("second"), bytes(v.toWire())); + await(() -> applied.containsKey("second") && probe.sync().xpending(stream, group).getCount() == 0); + assertEquals(1, applied.get("first").get()); assertEquals(1, applied.get("second").get()); + assertEquals(1, metrics.count(CacheMetricsListener.StreamResult.ACK_FAILED)); + } + } + } + } + + @Test void throwingObserverDoesNotRetryCommittedApplicationAsPoison() throws Exception { + try (Harness h = new Harness()) { + h.service.setEventListener((cache, message) -> { throw new IllegalStateException("observer failed"); }); + int clears = h.target.clears.get(); h.put("k", "value"); + await(() -> "value".equals(h.target.values.get("k")) && h.pending() == 0); + assertEquals(0, h.metrics.count(CacheMetricsListener.StreamResult.APPLY_FAILED)); + assertEquals(clears, h.target.clears.get()); + } + } + + @Test void applicationFailureGetsThreeAttemptsThenSafeResync() throws Exception { + String cache = "apply-" + UUID.randomUUID(); UUID id = UUID.randomUUID(); + byte[] stream = RedisKeyspace.journal(cache), group = RedisKeyspace.group(cache, id); + var times = new CopyOnWriteArrayList(); var resets = new AtomicInteger(); + var current = new AtomicReference<>(new RecoveryResult(RecoveryResult.Status.RESET_SAFE, "0-0", 1)); + Metrics metrics = new Metrics(); + try (var connection = client.connect(ByteArrayCodec.INSTANCE); + var transport = new LettuceStreamsInvalidationTransport(client, CODEC, CODEC, id)) { + var journal = new RedisStreamJournal(connection, 1000, CODEC); + transport.setMetricsListener(metrics); + transport.setGapHandler(new InvalidationGapHandler() { + public CompletionStage reset(String c) { + resets.incrementAndGet(); var safe = new RecoveryResult(RecoveryResult.Status.RESET_SAFE, journal.endCursor(c), 2); + current.set(safe); return CompletableFuture.completedFuture(safe); + } + public RecoveryResult registrationBaseline(String c) { return current.get(); } + public boolean isCurrent(String c, RecoveryResult r) { return r == current.get(); } + }); + transport.subscribe(cache, message -> { times.add(System.nanoTime()); throw new IllegalStateException("target temporarily failed"); }); + var v = new Version(1, UUID.randomUUID()); journal.append(cache, new InvalidationMessage(cache, "k", v, v.instanceId(), InvalidationMessage.Type.INVALIDATE)); + await(() -> resets.get() == 1 && connection.sync().xpending(stream, group).getCount() == 0); + assertEquals(3, times.size()); assertEquals(3, metrics.count(CacheMetricsListener.StreamResult.APPLY_FAILED)); + assertTrue(times.get(1) - times.get(0) >= TimeUnit.MILLISECONDS.toNanos(900)); + assertTrue(times.get(2) - times.get(1) >= TimeUnit.MILLISECONDS.toNanos(1900)); + } + } + + @Test void ephemeralCloseRetiresOnlyItsOwnGroup() throws Exception { + String cache = "ephemeral-" + UUID.randomUUID(); byte[] stream = RedisKeyspace.journal(cache); + try (var connection = client.connect(ByteArrayCodec.INSTANCE)) { + var transport = new LettuceStreamsInvalidationTransport(client, CODEC, CODEC); + var field = LettuceStreamsInvalidationTransport.class.getDeclaredField("instanceId"); field.setAccessible(true); + byte[] own = RedisKeyspace.group(cache, (UUID) field.get(transport)); + var subscription = transport.subscribe(cache, message -> { }); + byte[] other = RedisKeyspace.group(cache, UUID.randomUUID()); + connection.sync().xgroupCreate(XReadArgs.StreamOffset.from(stream, "0-0"), other); + subscription.close(); subscription.close(); transport.close(); + assertThrows(RedisCommandExecutionException.class, () -> connection.sync().xpending(stream, own)); + assertNotNull(connection.sync().xpending(stream, other)); + } + } + + @SuppressWarnings("unchecked") + static StatefulRedisConnection ackFailureConnection(StatefulRedisConnection real, + AtomicBoolean once, boolean replyLost) { + Object async = java.lang.reflect.Proxy.newProxyInstance(StreamsRecoveryTest.class.getClassLoader(), + new Class[]{io.lettuce.core.api.async.RedisAsyncCommands.class}, (proxy, method, args) -> { + if (method.getName().equals("xack") && once.compareAndSet(false, true)) { + if (replyLost) ((RedisFuture) method.invoke(real.async(), args)).get(); + var failed = CompletableFuture.failedFuture(new IllegalStateException("ACK unavailable")); + return java.lang.reflect.Proxy.newProxyInstance(StreamsRecoveryTest.class.getClassLoader(), new Class[]{RedisFuture.class}, + (p, m, a) -> { try { return CompletableFuture.class.getMethod(m.getName(), m.getParameterTypes()).invoke(failed, a); } + catch (java.lang.reflect.InvocationTargetException e) { throw e.getCause(); } }); + } + try { return method.invoke(real.async(), args); } catch (java.lang.reflect.InvocationTargetException e) { throw e.getCause(); } + }); + return (StatefulRedisConnection) java.lang.reflect.Proxy.newProxyInstance(StreamsRecoveryTest.class.getClassLoader(), + new Class[]{StatefulRedisConnection.class}, (proxy, method, args) -> { + if (method.getName().equals("async")) return async; + try { return method.invoke(real, args); } catch (java.lang.reflect.InvocationTargetException e) { throw e.getCause(); } + }); + } +} + +class RedisStreamsRecoveryTest extends StreamsRecoveryTest { String image() { return "redis:6.2-alpine"; } } +class ValkeyStreamsRecoveryTest extends StreamsRecoveryTest { String image() { return "valkey/valkey:8.0-alpine"; } } diff --git a/tiercache-transport-redis/src/test/java/io/tiercache/redis/ValkeyLettuceContractTest.java b/tiercache-transport-redis/src/test/java/io/tiercache/redis/ValkeyLettuceContractTest.java index 8b2f5d4..e133d84 100644 --- a/tiercache-transport-redis/src/test/java/io/tiercache/redis/ValkeyLettuceContractTest.java +++ b/tiercache-transport-redis/src/test/java/io/tiercache/redis/ValkeyLettuceContractTest.java @@ -3,10 +3,10 @@ import org.testcontainers.utility.DockerImageName; /** Contract suite against Valkey. */ -class ValkeyLettuceContractTest extends AbstractLettuceContractTest { +class ValkeyLettuceContractTest extends AbstractNamespaceContractTest { @Override DockerImageName image() { - return DockerImageName.parse("valkey/valkey:8.0-alpine"); + return DockerImageName.parse(System.getProperty("tiercache.test.valkeyImage", "valkey/valkey:9.1.2-alpine")); } }