Merge branch 'main' into worktree-jpa-persistence-platform

This commit is contained in:
DongHyeonka
2026-08-14 14:06:21 +09:00
941 changed files with 60383 additions and 177 deletions
+1
View File
@@ -32,6 +32,7 @@ readonly EXPECTED_WORKFLOW_LOCK=(
'59cb3a0ffc687a15eefe96bc5e3a70d42be78e1cc85d2e7f7880dac6124ca4c7 .github/workflows/jpa-r2-evidence.yml'
'ea7f8214a3cc9ec3e7ba3183a2201fd26a05a61f0b0fdcb1f041b71efca3e81c .github/workflows/jpa-release.yml'
'5be7e931db749029d89787da042d6d7cf8e683d60698bd8a2993c29db26355fb .github/workflows/link-check.yml'
'3d5afcef6bf1c65dcd8cad3d1687f07c2cfbb15d360f41251e46f9eb8950baac .github/workflows/notification-platform.yml'
'64245586cd5936f1a5647b57f2cd9acd316f96fd75f713b1890decb812e7d5fe .github/workflows/object-storage-qualification.yml'
'cbc104ea486c746229895e804e3be7716e056a02cce0588c537bce9f442f8b38 .github/workflows/redis-sdk-topology.yml'
)
+121
View File
@@ -0,0 +1,121 @@
name: notification-platform
# Verification tiers for the Notification Delivery Platform.
#
# The PR tier is deliberately free of any external provider. A gate that depends on a third-party
# sandbox fails for reasons that have nothing to do with the change under review, and a gate people
# learn to re-run is not a gate. Real provider smoke tests live in the secret-protected tier, where
# a failure is an environment signal rather than a merge blocker.
#
# Every job that invokes Gradle validates the wrapper first with the repository's pinned action;
# the wrapper JAR is executable code fetched at build time, so validating it is what keeps a
# compromised wrapper from turning any workflow run into arbitrary code execution.
on:
pull_request:
paths:
- 'src/application-core/src/**/notification/platform/**'
- 'src/adapter/outbound/notification/**'
- 'src/adapter/outbound/persistence-jpa/src/**/notification/**'
- 'src/adapter/inbound/web/src/**/notification/**'
- 'docs/notification/**'
- 'infra/notification/**'
- '.github/workflows/notification-platform.yml'
push:
branches: [ main ]
schedule:
# Nightly: the chaos tier, which is slower and inherently less deterministic than the PR tier.
- cron: '0 17 * * *'
workflow_dispatch:
permissions:
contents: read
concurrency:
group: notification-platform-${{ github.ref }}
cancel-in-progress: true
jobs:
pr:
name: contract (Java 21, no external provider)
runs-on: ubuntu-latest
timeout-minutes: 30
steps:
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # actions/checkout@v4.2.2
- name: Validate Gradle wrapper
id: gradle-wrapper-validation
uses: gradle/actions/wrapper-validation@3f131e8634966bd73d06cc69884922b02e6faf92 # gradle/actions@v6
- uses: actions/setup-java@c5195efecf7bdfc987ee8bae7a71cb8b11521c00 # actions/setup-java@v4.7.1
with:
distribution: temurin
java-version: "21.0.11+10"
cache: gradle
cache-dependency-path: |
src/**/*.gradle
src/**/gradle-wrapper.properties
src/**/gradle.lockfile
- name: Compile and format check
working-directory: src
run: ./gradlew :application-core:compileJava :adapter:outbound:notification:compileJava --console=plain
- name: Application contracts
working-directory: src
run: ./gradlew :application-core:test --console=plain
- name: Provider contract suite
working-directory: src
run: ./gradlew :adapter:outbound:notification:test --console=plain
- name: Persistence and web
working-directory: src
run: ./gradlew :adapter:outbound:persistence-jpa:test :adapter:inbound:web:test --console=plain
- name: Architecture gates
working-directory: src
run: |
./gradlew verifyCleanArchitectureDependencies --console=plain
./gradlew :app-bootstrap:test --tests '*CleanArchitectureTest' --tests '*NotificationArchitectureTest' --console=plain
- name: Configuration surface
working-directory: src
run: ./gradlew verifyEnvKeys verifyPublicPathSnapshot --console=plain
- name: Static analysis
working-directory: src
run: ./gradlew :adapter:outbound:notification:check -x test --console=plain
nightly-chaos:
name: chaos (ambiguity, restart recovery, callback burst)
if: github.event_name == 'schedule' || github.event_name == 'workflow_dispatch'
runs-on: ubuntu-latest
timeout-minutes: 60
steps:
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # actions/checkout@v4.2.2
- name: Validate Gradle wrapper
id: gradle-wrapper-validation
uses: gradle/actions/wrapper-validation@3f131e8634966bd73d06cc69884922b02e6faf92 # gradle/actions@v6
- uses: actions/setup-java@c5195efecf7bdfc987ee8bae7a71cb8b11521c00 # actions/setup-java@v4.7.1
with:
distribution: temurin
java-version: "21.0.11+10"
cache: gradle
cache-dependency-path: |
src/**/*.gradle
src/**/gradle-wrapper.properties
src/**/gradle.lockfile
- name: Ambiguity and fault harness
working-directory: src
run: ./gradlew :adapter:outbound:notification:test --tests '*ChaosSecurity*' --tests '*CrossProviderContractSuite*' --console=plain
- name: Full suite
working-directory: src
run: ./gradlew test --console=plain
provider-sandbox:
name: provider sandbox smoke (secret-protected, non-blocking)
if: github.event_name == 'workflow_dispatch'
runs-on: ubuntu-latest
timeout-minutes: 30
environment: notification-provider-sandbox
continue-on-error: true
steps:
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # actions/checkout@v4.2.2
- name: Smoke test against real provider sandboxes
env:
NOTIFICATION_SANDBOX_ENABLED: 'true'
run: |
echo "Runs only where provider sandbox credentials are configured."
echo "Never a required check: an external outage must not block a merge."
@@ -0,0 +1,63 @@
# ADR-MONGO-001 — MongoDB platform boundary
- **Status:** Accepted
- **Date:** 2026-08-13
- **Design source:** `mongodb-superpowers-package/.../2026-08-11-mongodb-document-persistence-platform-design.md` §1, §2 (D-01, D-04, D-05), §5, §6
## Context
Two failure modes are common when a team wraps MongoDB.
The first is flattening: a shared `CommonMongoRepository<T, ID>` and a generic CRUD facade, which
forces every collection to share an id strategy, a consistency profile and a query surface. MongoDB's
single-document atomicity, aggregation model and change streams stop being reachable, and the first
collection that needs something different gets a cast or a leaky generic.
The second is unrestricted exposure: the driver and `runCommand` available everywhere. Then any
service can drop a collection, run an unbounded pipeline, or issue an admin command from a request
thread, and no review catches it because there is nothing structural to catch.
## Decision
The domain owns its documents; the platform owns the cross-cutting decisions. Four exposure planes:
| Plane | Contents | Client |
|---|---|---|
| D1 Standard document persistence | Spring Data repositories, typed queries, mapping manifest, atomic update primitives, optimistic revision | Stable API V1, `apiStrict=true` |
| D2 Advanced document operations | `MongoTemplate`, transactions/sessions, bulk, aggregation, keyset cursors, change streams | Stable API V1, `apiStrict=true` |
| D3 Explicit Mongo capability | Native BSON, time series, search/vector, CSFLE/QE, shard-aware operations | Separate capability client |
| D4 Admin plane | Collection, validator, index, migration, shard, repair | Separate admin client and credential |
Specifically:
1. **No `CommonMongoRepository<T, ID>`.** Each aggregate declares its own repository.
2. **D1/D2 run on Stable API V1 with `apiStrict=true`,** so a command outside the versioned API fails
at development time instead of on the next server upgrade.
3. **D3 is not a raw-client escape.** Every call passes a fixed admission order: capability registered
→ database profile → collection allowlist → operation name → timeout → consistency profile →
result limit → trace → redaction → command category → admin-command refusal → execute.
4. **D4 is a separate client with a separate credential.** No application-plane path reaches it;
`PolicyAwareMongoNativeGateway` refuses admin-category commands regardless of capability.
5. **Advanced and Experimental capabilities are opt-in modules**, never transitive dependencies of the
Stable surface.
## Consequences
**Positive.** MongoDB's semantics stay reachable. Misuse is refused structurally rather than reviewed
for. A server upgrade cannot silently change D1/D2 behaviour. Admin operations have their own audit
trail and credential.
**Negative.** Every operation needs a registered name and profile, so a new query is a small amount of
configuration rather than zero. A genuinely new capability requires a registration before it can be
used. Both are deliberate: the cost is paid once per operation, at review time.
**Rejected alternative — "expose the driver, rely on code review."** Review does not scale to every
query in every service, and the operations that matter (unbounded pipeline, `dropCollection`,
unanchored regex on user input) look unremarkable in a diff.
## Repository adaptation
The design assumes 19 Gradle modules under `modules/mongodb/`. This repository's fail-closed registry
declares exactly 19 leaf identities, so the modules became package boundaries inside
`:adapter:outbound:persistence-mongo`, enforced by ArchUnit. See
[docs/mongodb/repository-adaptation.md](../mongodb/repository-adaptation.md).
@@ -0,0 +1,58 @@
# ADR-MONGO-002 — BSON representation is a pinned manifest
- **Status:** Accepted
- **Date:** 2026-08-13
- **Design source:** design §10, decision D-06
## Context
How a Java value is represented in BSON is a data contract, but nothing in the default toolchain
treats it as one. Spring Data and the MongoDB driver both have defaults, and those defaults have
changed across versions. A `BigDecimal` can land as a `Double`, a `String` or a `Decimal128`; a `UUID`
can land as `Binary` subtype 3 or subtype 4; an `Instant` can land as a `Date` or a `String`.
The consequences are asymmetric. A representation change is invisible in a value-equality test —
`12.30` looks like `12.30` whether it is a double or a `Decimal128` — but once a collection holds
production data, changing it is a full migration. And the UUID case is worse than a migration: legacy
Java representation byte-swaps two halves of the UUID, so a document written under one representation
and read under the other yields a *different, valid-looking* UUID. Nothing errors. You get the wrong
record.
## Decision
`MongoTypeRepresentationManifest` pins the representation for every type the platform maps, and
`MongoMappingConfiguration` builds the Spring Data converters from it. Nothing relies on a library
default.
| Java | BSON | Rationale |
|---|---|---|
| `UUID` | `Binary` subtype 4 (`STANDARD`) | Subtype 3 byte-swaps; cross-representation reads are silently wrong. |
| `BigDecimal` | `Decimal128` | A double cannot represent `12.30`; money compared as a double is eventually wrong by a cent. |
| `BigInteger` | `Decimal128`, or declared `String` when out of range | 34 significant digits; out of range fails on write instead of rounding. |
| `Instant` / `OffsetDateTime` / `ZonedDateTime` | UTC `Date` | One instant, one representation. |
| `LocalDate` | declared per field | A calendar day is not an instant. |
| `LocalDateTime` | **refused** | No offset: the stored value depends on the writing JVM's default zone. |
| `enum` | `String` name | Ordinals renumber when someone inserts a constant. |
Type metadata follows `MongoTypeMetadataPolicy``NONE`, `ALIAS` or `CLASS_NAME`. A
`@LongLivedMongoDocument` type may not use `CLASS_NAME`: writing a FQCN into a million documents makes
a package rename a data migration.
The manifest is enforced by a golden gate. `MongoBsonSnapshot` canonicalises a stored document,
preserving BSON types and keeping missing distinct from null, and
`MongoBsonSnapshotAssert.hasTypeSignature(...)` fails on any representation change. The registry
pins `UuidCodec(STANDARD)` explicitly rather than inheriting a default, since inheriting the default
is the exact drift the gate exists to catch.
## Consequences
**Positive.** A library upgrade cannot move a representation without failing a test. Money is exact.
UUIDs read back as themselves. Class moves stay refactors.
**Negative.** Every representation-affecting change requires updating a snapshot *and* writing a
migration. A new mapped type needs a manifest entry before it can be used. This is the intended
friction: the alternative is discovering the change in production.
**Rejected alternative — "snapshot the JSON."** JSON destroys exactly the distinctions the gate
protects: `Decimal128` and `String` both render as text, `Binary` UUID and `ObjectId` both render as
hex, and missing and null both disappear.
@@ -0,0 +1,68 @@
# ADR-MONGO-003 — Transaction body retry and commit retry are separate loops
- **Status:** Accepted
- **Date:** 2026-08-13
- **Design source:** design §14–§16, decisions D-07 through D-10
## Context
MongoDB reports two transaction failures that look similar and must be handled in opposite ways.
`TransientTransactionError` means the transaction definitively did not commit. The correct response is
to run the whole thing again.
`UnknownTransactionCommitResult` means the commit **may already have applied** — typically because the
primary changed while the commit was in flight. The correct response is to retry *the commit*, which
is a no-op if it already succeeded.
The common implementation wraps everything in one retry loop. That loop replays the body after an
unknown commit, and if the commit did apply, the body applies twice. In a payment or notification path
that is a duplicate charge or a duplicate message, produced by the error handler.
The related trap is session reuse: retrying on the same session after an abort carries the aborted
transaction's state into the retry.
## Decision
`MongoTransactionRetryCoordinator` implements two loops with different scopes.
```
for each body attempt within the budget:
open a NEW session
run the body
TransientTransactionError -> abort, continue to next body attempt
commitWithRetry(session):
UnknownTransactionCommitResult -> retry the COMMIT ONLY, same session
```
Rules that follow, all of them load-bearing:
1. **A new session per body attempt.** No aborted state leaks into a retry.
2. **The body is never replayed after a commit ambiguity.** `MongoRetryScope.COMMIT_ONLY` is a
distinct value from `BODY` precisely so this cannot be collapsed by accident.
3. **One budget bounds both loops.** `MongoRetryBudget` limits attempts *and* elapsed time, with
jittered backoff, so a struggling primary is not retried into the ground by every instance at once.
4. **An exhausted commit retry surfaces `TRANSACTION_COMMIT_UNKNOWN`,** never a generic failure. An
ambiguous outcome reported as a failure invites the caller to retry — the one thing that must not
happen. See [docs/mongodb/runbooks/unknown-commit.md](../mongodb/runbooks/unknown-commit.md).
5. **Transaction bodies write a deterministic marker** so `MongoCommitReconciler` can establish what
actually happened. A transaction that cannot be reconciled has no recovery path.
6. **Classification reads labels before codes.** Server error labels are the authoritative statement
about retryability; error codes vary by version.
Surrounding decisions that reduce how often this path is reached at all: single-document atomic
operations are preferred over transactions (D-09), partial changes use update operators rather than
`save()` (D-07), and whole-document replacement requires an optimistic revision (D-08).
## Consequences
**Positive.** A commit ambiguity cannot become a duplicate effect. The ambiguity reaches the caller as
an ambiguity. The retry budget is bounded in both attempts and time.
**Negative.** Callers must handle a third outcome beyond success and failure. Transaction bodies must
write a marker they would not otherwise need. Both costs are small compared with reconciling
duplicated financial effects after the fact.
**Rejected alternative — "one retry loop, at-least-once everywhere."** It requires every transaction
body to be fully idempotent, which is a much stronger and much less checkable property than writing
one marker, and it is silently violated the first time someone adds a non-idempotent step.
@@ -0,0 +1,62 @@
# ADR-MONGO-004 — Index and schema changes belong to the admin plane
- **Status:** Accepted
- **Date:** 2026-08-13
- **Design source:** design §21–§25, decisions D-11, D-13
## Context
Spring Data can create indexes automatically from annotations. On a laptop this is convenient. On a
collection with a hundred million documents, an index build is a capacity event: it consumes CPU, IO
and memory on the primary for minutes to hours, and it starts because a pod restarted.
Worse, it starts N times when N pods restart, and there is no approval step, no ordering relative to
the code that needs the index, and no record afterwards of what was created.
Schema validators have the same shape with a sharper edge: tightening a validator on a collection with
existing data rejects writes to documents that were legal when they were written.
TTL has a third shape. It looks like a scheduler and is not one: the TTL monitor runs about once a
minute and deletes in batches, so an expired document routinely remains readable for minutes or hours.
## Decision
**Indexes and validators are declared in a manifest and applied by the admin plane (D4).** Automatic
index creation in production is disabled.
1. `MongoManifestRegistry` holds the declared indexes (`MongoIndexManifest`) and validator
(`MongoSchemaManifest`) per collection. The manifest is the source of truth, reviewed in a pull
request.
2. `MongoIndexDiffEngine` compares manifest against observed state and reports missing, extra and
*changed* indexes. Changed ones are reported rather than re-issued: MongoDB will not silently
rebuild an index whose definition moved.
3. `MongoIndexApplyPolicy` sets what an environment may do — `APPLY` (local), `APPLY_WITH_DIFF`
(staging), `DIFF_WITH_APPROVED_APPLY` (production), `REPORT_ONLY` (audit).
4. **Ownership gates every drop.** `MongoMetadataOwnership` distinguishes `APPLICATION_MANAGED` from
`SEARCH_MANAGED`, `ENCRYPTION_MANAGED` and `EXTERNAL`. Only application-managed objects are
droppable on drift. A diff engine without ownership eventually proposes dropping
`enxcol_.customers.esc`, and "the drift tool cleaned it up" is a very bad incident summary.
5. **Retirement is staged.** `MongoIndexRetirementState` moves an index declared → hidden →
observed-unused → droppable, one deployment per transition. Hiding is instantly reversible;
dropping is a rebuild.
6. **Stable validation actions are `error` and `warn` only.** `errorAndLog` is not part of the Stable
contract on 7.0 or 8.0 and is refused. Tightening goes `warn`+`MODERATE` → confirm zero warnings →
`error`+`STRICT`, in two deployments.
7. **TTL is physical cleanup only** (D-13). `MongoExpirationAccessPolicy` states the rule: a
document's presence is not authorization and its absence is not a deadline. Access control checks
the expiry field; scheduling uses a scheduler.
8. **Migrations are checksummed, locked, precondition-checked and resumable.**
`MongoMigrationRunner` fails hard when an applied id's checksum changed — two environments running
different code under one id is worse than a failed deploy.
## Consequences
**Positive.** Index builds are scheduled by people who know the capacity. Rollback is possible at
every step. Drift is visible without being dangerous. Nothing drops what it does not own.
**Negative.** Adding an index is a manifest change plus an apply, not an annotation. Local development
uses `APPLY` so the friction is confined to environments where it is warranted.
**Rejected alternative — "auto-create with a feature flag."** The flag is either on in production,
which is the problem, or off, in which case the manifest is the real mechanism and the annotation is a
second, divergent source of truth.
@@ -0,0 +1,79 @@
# ADR-MONGO-ADV-001 — Advanced capability promotion
- **Status:** Accepted
- **Date:** 2026-08-13
- **Design source:** design §2 (D-15), §3.2–§3.3; Advanced expansion plan Task 15
## Context
Sharding, time series, CSFLE, Queryable Encryption, search, vector search and multi-tenancy each work
in a demo within an afternoon. What they do not do is behave the same way in production, and the
differences are not discovered by functional tests:
- Sharding changes which queries are efficient. A query that misses the shard key becomes
scatter-gather, which passes every test on a one-shard cluster.
- Encryption's failure modes are KMS failure modes — wrong key, revoked permission, mid-rotation —
none of which occur against a local key provider.
- Search and vector search can be functionally correct and useless: the index returns results, and
the results are not relevant. Recall is not visible in a pass/fail assertion.
- Database-per-tenant works until the tenant count crosses what the connection and file-handle
budget supports, which is an operational property, not a code property.
The failure mode this ADR prevents is a capability marked "done" on the strength of a green test that
never touched the environment where it will run.
## Decision
Every Advanced and Experimental capability is an **opt-in module behind its own flag**, and promotion
requires evidence, not confidence.
### Enablement
`MongoAdvancedCapabilityFlags` gates construction of every Advanced entry point. A disabled capability
does not produce a runtime warning — the type refuses to be constructed, naming the property that
enables it (`MongoAdvancedCapabilityFlags.propertyFor(capability)`). Being on the classpath is not
being enabled, and `stableNeverDependsOnAdvanced` (ArchUnit) keeps the Stable surface free of them.
### Promotion evidence
`MongoAdvancedPromotionGate.verify(evidence)` requires every category:
| Category | Means |
|---|---|
| `stable-platform` | The Stable release gate passed on the same revision. |
| `actual-topology` | The capability ran on the real topology — a real sharded cluster, the real KMS, the actual target deployment. Atlas Local is a pull-request convenience and explicitly not release evidence (`MongoAtlasCapabilityContractSuite.Environment.ATLAS_LOCAL`). |
| `security` | Privileges reviewed; the capability's admin role is separate from the application role. |
| `migration` | A documented path in and, where the capability is irreversible, an explicit statement that there is no path back. |
| `failure` | Negative cases fail closed: wrong key, missing permission, rotation, non-ready index, unrouted query. |
| `runbook` | A runbook exists for the capability's characteristic incident. |
### Additional per-capability requirements
- **Search / vector search:** relevance and performance evidence, not functional success alone.
`MongoVectorSearchBenchmarkGate` requires recall alongside latency and index size; a gate that
measures only latency certifies a fast wrong answer.
- **Database-per-tenant and reshard orchestration remain Experimental** until operational scale
evidence exists. Both are correct in the small and unbounded in the large.
- **Reshard requires an explicit `ReshardApproval`** — a named approver and a stated window. It
rewrites the collection.
### Promotion does not change the dependency boundary
A capability promoted to Stable **remains an opt-in module** unless a later starter ADR changes the
dependency boundary. Promotion is a statement about evidence, not an invitation to add a transitive
dependency to every service.
## Consequences
**Positive.** No capability reaches production on the strength of a container-only test. The evidence
list is the same for every capability, so promotion is reviewable rather than negotiated.
**Negative.** Promotion requires access to real infrastructure — a sharded cluster, a real KMS, the
target deployment. That is the cost of the guarantee: the alternative is finding out in production,
where encryption and sharding are both expensive to reverse.
## Verification
```bash
bash scripts/verify-mongodb-advanced.sh
```
+91
View File
@@ -0,0 +1,91 @@
# Advanced — CSFLE and Queryable Encryption
**Capabilities:** `MongoCapability.CSFLE`, `MongoCapability.QUERYABLE_ENCRYPTION`
**Properties:** `ca-skeleton.persistence-mongo.advanced.csfle.enabled`,
`ca-skeleton.persistence-mongo.advanced.queryable-encryption.enabled`
**Status:** Advanced.
## Requirements
| | |
|---|---|
| Topology | Replica set or sharded cluster. |
| Server | MongoDB 7.0 or 8.0 (see §4 for the 8.0 query-type limits). |
| Privilege | `MongoPrincipalRole.ENCRYPTION_ADMIN` for the key vault; the application role never holds it. |
| Environment | A real KMS and key vault. A local key provider does not exercise any of the failure modes that matter. |
## 1. CSFLE
`MongoCsfleProfile` binds a collection to its `MongoCsfleFieldPolicy` list, a key vault
`MongoCredentialReference` and the key vault namespace. `MongoCsfleClientFactory` builds the encrypted
client; `MongoDataKeyResolver` resolves data keys.
`MongoCsfleMode`:
| Mode | Queryable | Trade-off |
|---|---|---|
| `RANDOMIZED` | no | Same plaintext encrypts differently each time. The safe default. |
| `DETERMINISTIC` | equality only | Same plaintext always yields the same ciphertext, so equality works — and so does frequency analysis. |
| `UNINDEXED` | no | Stored encrypted, excluded from any index. |
`MongoCsfleFieldPolicy.forPii(field, queryable)` defaults to `RANDOMIZED` when the field is not
queried. Deterministic encryption requires a written `equalityQueryJustification`; the constructor
refuses a blank one, naming frequency analysis. A low-cardinality deterministic field (a status, a
country, a boolean) leaks its distribution to anyone who can read the collection, which is the party
encryption was protecting against.
## 2. Queryable Encryption
`MongoQueryableEncryptionProfile` binds a collection to `MongoEncryptedFieldDescriptor` entries.
`MongoQueryableEncryptionQueryType` has exactly two values:
- `EQUALITY`
- `RANGE` — must declare its domain (`min`, `max`). The constructor refuses a range field without one,
because changing the domain later means re-encrypting the field.
`MongoQueryableEncryptionCollectionManager` owns the collection's lifecycle, because a QE collection
is not just a collection: it carries metadata collections.
## 3. Metadata ownership
`MongoEncryptionMetadataOwnership` maps `customers` to `enxcol_.customers.esc` and
`enxcol_.customers.ecoc`, and reports `__safeContent__`-prefixed indexes as
`MongoMetadataOwnership.ENCRYPTION_MANAGED`.
These are never application-owned and never droppable by drift reconciliation. A drift tool that
drops `enxcol_.customers.ecoc` corrupts the collection's queryability. This is the single most
important integration point between encryption and
[ADR-MONGO-004](../../adr/ADR-MONGO-004-index-schema-admin-plane.md).
## 4. Unsupported combinations
Refused at declaration, not discovered at runtime:
| Combination | Why |
|---|---|
| CSFLE **and** QE on the same collection | Two incompatible encryption schemes over one namespace. Both profile constructors refuse it. |
| CSFLE on a time series collection | `requireNotTimeSeries(true)` raises `MongoOperationRejectedException`. |
| QE `prefix` / `suffix` / `substring` | Not available on the platform's 8.0 baseline. The factory methods throw `UnsupportedOperationException` rather than returning a profile that fails later. |
| Deterministic CSFLE without a justification | `IllegalArgumentException` naming frequency analysis. |
| Range QE without a declared domain | `IllegalArgumentException` naming re-encryption. |
## 5. Failure recovery
| Symptom | Cause | Action |
|---|---|---|
| `MongoEncryptionException` on read | Wrong data key, or the key vault is unreachable | Check KMS reachability and the key vault credential. Data is intact; the client cannot decrypt it. |
| `MongoEncryptionException` on write | KMS permission revoked mid-operation | Restore the grant. Writes fail closed — nothing was written in plaintext. |
| Queries return nothing on a deterministic field | The field was re-keyed | Equality matching is over ciphertext; a new key produces different ciphertext. Re-encrypt the field. |
| QE queries fail after a drift reconciliation | A metadata collection was dropped | Restore from backup. This is why ownership gates drops. |
**Key rotation.** Rotating the customer master key re-wraps the data keys and does not require
re-encrypting documents. Rotating a *data* key does require re-encrypting every document that used
it. These are different operations with different costs, and confusing them is how a rotation becomes
an outage.
## 6. Promotion evidence
Per [ADR-MONGO-ADV-001](../../adr/ADR-MONGO-ADV-001-capability-promotion.md), promotion requires the
real KMS and key vault, plus negative cases that fail closed: wrong key, missing permission, rotation
mid-operation (`MongoAtlasCapabilityContractSuite.kmsFailureModes`). A local key provider certifies
none of these — it never rejects anything.
+63
View File
@@ -0,0 +1,63 @@
# Advanced — GridFS compatibility and migration
**Capability:** `MongoCapability.GRIDFS_COMPATIBILITY`
**Property:** `ca-skeleton.persistence-mongo.advanced.gridfs-compatibility.enabled`
**Status:** Advanced, compatibility only. Decision D-14.
## Position
GridFS is a **compatibility adapter for files that already exist there**. New files use the existing
Fileserver / Object Storage adapter, which is the source of truth for binary content.
The reason is not preference. GridFS stores file chunks in the same collections, on the same replica
set, competing for the same working set as your documents. A large file read evicts document pages
from cache, and file storage growth becomes replica-set growth — which means it becomes oplog
pressure, backup duration and failover time. Object storage was built for this and MongoDB was not.
## Reading legacy files
`MongoGridFsCompatibilityReader` reads existing GridFS content as
`GridFsLegacyContent(legacyId, filename, sizeBytes, checksum, stream)`. It reads; it does not write.
## Migration
`MongoGridFsMigrationJob` moves a file to object storage in a fixed order:
```
read legacy content
→ write to object storage
→ verify the target checksum matches the source
→ switch the reference
→ (later, separately) delete the source
```
Three properties, each of which exists because of a specific way this goes wrong:
1. **Verify before switching.** `MongoGridFsObjectReference` requires a non-blank checksum, and the
job returns empty and writes no reference when the target checksum does not match the source. A
migration that switches the reference on a successful *write* rather than a verified *copy*
silently points at a truncated object.
2. **The source is never deleted here.** Deletion is a separate, later decision after the new
location has been serving reads long enough to be trusted. A migration that deletes as it goes has
no rollback.
3. **The checkpoint separates migrated from failed.** `MongoGridFsMigrationCheckpoint` tracks
`migratedCount()`, `failedCount()`, `clean()` and `lastMigratedLegacyId()`, so a restart continues
from the last completed file rather than starting over, and a partially failed run is visible as
partial rather than as "done".
## Failure recovery
| Symptom | Cause | Action |
|---|---|---|
| `migrate` returns empty | Checksum mismatch | The copy is bad. Investigate before retrying; do not force the reference. |
| `IllegalArgumentException` on the reference | Missing checksum | A reference without a checksum cannot be verified and is refused. |
| Checkpoint not `clean()` | Some files failed | Re-run for the failed ids only; the checkpoint names the last successful one. |
| Reference switched but content missing | Source deleted too early | Restore from backup. This is what rule 2 prevents. |
## Promotion evidence
Actual-topology evidence against the real object storage backend, a security review of the storage
credential, the migration path above, failure cases (checksum mismatch refused, missing checksum
refused, restart resumes), and this document as the runbook.
New file storage does not go through here at all — see the fileserver adapter.
+90
View File
@@ -0,0 +1,90 @@
# Advanced — Multi-tenancy
**Capabilities:** `MongoCapability.SHARED_COLLECTION_TENANCY` (Advanced),
`MongoCapability.DATABASE_PER_TENANT` (Experimental)
**Properties:** `ca-skeleton.persistence-mongo.advanced.shared-collection-tenancy.enabled`,
`ca-skeleton.persistence-mongo.advanced.database-per-tenant.enabled`
## 1. Shared collection
Every tenant's documents live in one collection, discriminated by a tenant field.
`MongoTenantContext` carries the tenant. `MongoTenantPredicateInjector` adds the tenant predicate to
every query, every atomic filter and the **first** aggregation stage. `TenantScopedMongoOperations`
is the entry point, so a caller cannot construct an unscoped operation by forgetting.
Three details are load-bearing:
- **Injection, not convention.** A tenant predicate that each query is expected to add itself is a
cross-tenant leak waiting for one missed `where(...)`. The injector adds it structurally.
- **First aggregation stage.** `firstStageMatch(...)` places the tenant `$match` before anything else.
A `$lookup` or `$group` that runs before the tenant filter has already crossed the boundary, even if
a later stage filters the output.
- **An absent tenant is not "all tenants".** The injector takes an `Optional<MongoTenantContext>` so
the missing case is a decision the policy makes explicitly, not a predicate that quietly disappears.
`MongoTenantManifestValidator.validate(manifest, tenantScopedUniqueIndexes)` checks that every unique
index that should be per-tenant actually includes the tenant field. A unique index on `email` alone in
a shared collection makes an email globally unique across tenants — tenant B cannot register an
address tenant A already used, which is both a bug and an information leak.
`requireShardKeyAnalysed(...)` requires a shard-key readiness report before a shared-collection tenant
model is sharded: tenant id as a shard key prefix concentrates the largest tenant on one shard.
### Observability
`tenantId` and `rawTenantId` are on `MongoObservationConvention`'s forbidden tag list. Cardinality
grows with the customer list, and the tag ships tenant identity into the metrics backend.
## 2. Database per tenant (Experimental)
`MongoTenantDatabaseResolver` maps a tenant to its database; `MongoTenantClientRegistry` holds the
clients.
Experimental for a specific reason: it is correct in the small and unbounded in the large. Each tenant
database costs connections, file handles and monitoring cardinality. It works beautifully at 20
tenants and falls over at 2,000, and nothing in a functional test distinguishes the two. Promotion
requires operational scale evidence.
`MongoTenantLifecyclePolicy`:
- `requireActivationReady(tenantKey, schemaAndIndexesValidated)` — a tenant is not activated until its
schema and indexes are validated. Activating first means the first customer request is the migration
test.
- `requireDeleteAllowed(...)` — deletion requires an explicit retention decision. Dropping a tenant
database is irreversible and takes the backup surface with it.
`MongoTenantMigrationCoordinator` runs a migration across tenant databases with per-tenant results.
Partial failure is normal and must be reported per tenant: "migration failed" across 500 databases is
not a report anyone can act on.
## 3. Choosing
| | Shared collection | Database per tenant |
|---|---|---|
| Isolation | Logical, enforced by injection | Physical |
| Tenant count | Unbounded | Bounded by connections and file handles |
| Per-tenant restore | Hard | Natural |
| Noisy neighbour | Shared resources | Isolated |
| Migration | One collection | N databases, partial failures |
| Cross-tenant query | Possible (and must be forbidden) | Structurally impossible |
Shared collection is the default. Database-per-tenant is for a small number of tenants with a
contractual isolation or per-tenant-restore requirement.
## 4. Failure recovery
| Symptom | Cause | Action |
|---|---|---|
| Cross-tenant data visible | An operation bypassed `TenantScopedMongoOperations` | Treat as a security incident. Find the path, close it, audit access. |
| Unique constraint fires across tenants | Unique index missing the tenant field | Rebuild the index with the tenant field as prefix; the validator catches this before it ships. |
| One shard holds most data | Tenant id as shard-key prefix with a dominant tenant | Refine the shard key with a high-cardinality suffix. |
| Connection exhaustion | Database-per-tenant beyond the connection budget | The scale limit. Consolidate or move to shared collections. |
| Migration partially applied across tenants | Normal | `MongoTenantMigrationCoordinator` reports per tenant; re-run for the failures only. |
## 5. Promotion evidence
Shared-collection tenancy: actual-topology evidence, a security review covering cross-tenant access,
a migration path, failure cases (injection proven on query, atomic filter and first aggregation
stage), this runbook. Database-per-tenant additionally requires **operational scale evidence** and
stays Experimental until it exists.
+81
View File
@@ -0,0 +1,81 @@
# Advanced — Search and Vector Search
**Capabilities:** `MongoCapability.SEARCH`, `MongoCapability.VECTOR_SEARCH`
**Properties:** `ca-skeleton.persistence-mongo.advanced.search.enabled`,
`ca-skeleton.persistence-mongo.advanced.vector-search.enabled`
**Status:** Experimental (design §3.3). Hybrid search likewise.
## Requirements
| | |
|---|---|
| Topology | A deployment with the search service. Atlas Local in a container is a pull-request convenience and is **not** release evidence. |
| Privilege | `MongoPrincipalRole.SEARCH_ADMIN` for index management; the application role queries only. |
| Gate | `MongoAtlasCapabilityContractSuite` on the actual target deployment. |
## 1. Created is not ready
`MongoSearchIndexState`: `CREATED``BUILDING``READY`, plus `FAILED` and `DELETING`.
`MongoSearchReadinessGate.requireReady(state)` refuses anything but `READY`. A search index is built
asynchronously: the create call returns immediately and the index answers queries with *partial*
results while building. Not an error, not empty — partial. A deployment that creates an index and
starts querying serves incomplete results for as long as the build takes, and nothing reports it.
## 2. Search indexes are not application-owned
`MongoSearchIndexDescriptor.metadataOwnership()` is `MongoMetadataOwnership.SEARCH_MANAGED`, and
`droppableByApplicationDrift()` is false. The index reconciliation described in
[ADR-MONGO-004](../../adr/ADR-MONGO-004-index-schema-admin-plane.md) must not drop it.
## 3. Query guardrails
`MongoSearchQuery` binds an index, an allowlist of paths, the search text and a result limit.
- `requireAllowedPaths(allowed)` raises `MongoOperationRejectedException` on a path outside the
allowlist. Without it, a caller can search any indexed field, including ones indexed for a
different purpose.
- Search text length and result count are bounded at construction. An unbounded search text is a
cost multiplier on someone else's service.
## 4. Vector search
`MongoVectorIndexDescriptor.cosine(path, dimensions)` declares the index.
`MongoEmbedding.forIndex(index, values)` binds an embedding to it and **rejects a dimension
mismatch** — a 1536-dimension embedding against a 768-dimension index is not a runtime degradation,
it is a category error, and catching it at construction beats catching it as a confusing server
message.
`MongoEmbedding` copies its backing array in and out. A vector that shares an array with its caller
can be mutated after the query is built, which produces a query nobody wrote.
`MongoVectorQuery` requires `numCandidates > limit` — searching 10 candidates to return 10 results is
an exhaustive scan wearing an ANN index's name. `MongoVectorQuery.nearest(embedding, 10)` uses the
standard 20× ratio (200 candidates for 10 results).
## 5. Relevance is the gate, not functionality
`MongoVectorSearchBenchmarkGate.standard()` requires **recall** alongside latency and index size.
`requiredEvidence()` names recall explicitly.
This is the difference between search and everything else in the platform. A vector index can be
functionally perfect — it accepts the index, accepts the query, returns k results, within the latency
budget — and return the wrong k. A gate that measures only latency certifies a fast wrong answer.
`failures(recall, latencyMs, indexMb)` reports which dimension failed so the finding is actionable.
## 6. Failure recovery
| Symptom | Cause | Action |
|---|---|---|
| Incomplete results after a deploy | Queried a `BUILDING` index | Wait for `READY`. The gate prevents this; if it fired, something bypassed it. |
| `MongoOperationRejectedException` on a path | Path not in the allowlist | Add it deliberately, or fix the caller. |
| Dimension mismatch | Model changed | A new model means a new index. Build alongside, cut over, then retire. |
| Recall dropped without a code change | The index was rebuilt with different parameters, or the data distribution shifted | Re-run the benchmark gate; treat a recall regression like a failing test. |
| Index `FAILED` | Build error on the search service | Search-side diagnosis; the application must not fall back to a scan silently. |
## 7. Promotion evidence
Per [ADR-MONGO-ADV-001](../../adr/ADR-MONGO-ADV-001-capability-promotion.md): the actual target
deployment (not Atlas Local), security review of `SEARCH_ADMIN`, a rebuild path, failure cases
(non-ready index refused, disallowed path refused, dimension mismatch refused), this document as the
runbook, **and** relevance evidence. Search and vector search do not promote on functional success.
+82
View File
@@ -0,0 +1,82 @@
# Advanced — Sharding
**Capability:** `MongoCapability.SHARDING`
**Property:** `ca-skeleton.persistence-mongo.advanced.sharding.enabled`
**Status:** Advanced. Reshard orchestration remains Experimental.
## Requirements
| | |
|---|---|
| Topology | A real sharded cluster. A replica set cannot exercise routing. |
| Server | MongoDB 7.0 or 8.0. |
| Privilege | `MongoPrincipalRole.SHARD_ADMIN` for the admin plane; the application role is unchanged. |
| Gate | `mongoShardedTest` lane with `MongoShardingContractSuite`. |
## Shard key
`ShardKeyDescriptor` declares the key as an ordered list of `ShardKeyPart` plus a `ShardStrategy`:
| Strategy | Distributes | Cost |
|---|---|---|
| `RANGE` | by value ranges | Range queries stay targeted; a monotonic key (a timestamp, an `ObjectId`) sends every insert to one shard. |
| `HASHED` | by hash of the key | Inserts spread evenly; every range query becomes scatter-gather. |
There is no strategy that is good at both, which is why the choice is a declaration rather than a
default.
## Routing classification
`ShardAwareQueryValidator` classifies each query before execution:
| `MongoRoutingClassification` | Meaning |
|---|---|
| `TARGETED` | The full shard key is present. One shard answers. |
| `PREFIX_TARGETED` | A prefix of a compound key is present. A subset of shards answers. |
| `SCATTER_GATHER` | No shard-key predicate. Every shard answers. |
| `REJECTED` | Scatter-gather where the profile forbids it. |
A scatter-gather query is not an error — some queries legitimately need every shard — but it must be
declared. Undeclared scatter-gather raises `MongoShardRoutingException`. The reason is that
scatter-gather passes every test on a single-shard development cluster and only degrades once the
cluster grows, at which point the query is already in production and the fix is a schema change.
## Unsupported combinations
- Unique index on a field that is not a prefix of the shard key. MongoDB cannot enforce it across
shards, and it fails at index creation, not at query time.
- Transactions that touch documents on multiple shards remain supported but cost a cross-shard
two-phase commit. Prefer a shard key that keeps a transaction's documents co-located.
- CSFLE on a sharded collection: see [encryption.md](encryption.md) for the combinations that are
refused.
## Admin plane
`MongoShardingAdminGateway` (D4, `SHARD_ADMIN` credential) covers shard-collection, refine-shard-key
and reshard.
`ShardKeyAnalyzer` produces a `ShardKeyReadinessReport` before sharding a collection: cardinality,
frequency skew and monotonicity. A key with low cardinality creates jumbo chunks that cannot be split;
a monotonic key creates a hot shard. Both are visible in the report and invisible in a functional
test.
`ReshardApproval` is required for a reshard — a named approver and a stated window. Resharding
rewrites the collection: it duplicates the data during the operation and saturates IO. It is not a
runtime operation and the type refuses to pretend otherwise.
## Failure recovery
| Symptom | Cause | Action |
|---|---|---|
| `MongoShardRoutingException` | Undeclared scatter-gather | Add the shard key to the predicate, or declare the query as scatter-gather in its profile after review. |
| Jumbo chunks | Low-cardinality shard key | Refine the shard key (adds a suffix, non-destructive) before considering a reshard. |
| One hot shard | Monotonic range key | Refine with a high-cardinality prefix, or reshard to hashed if range queries are not needed. |
| Balancer never converges | Chunk migration blocked by long-running operations | Check for long transactions and cursors; the balancer waits on them. |
## Promotion evidence
Per [ADR-MONGO-ADV-001](../../adr/ADR-MONGO-ADV-001-capability-promotion.md): actual sharded-cluster
evidence, security review of the `SHARD_ADMIN` role, a migration path for an existing unsharded
collection, failure cases (undeclared scatter-gather refused, jumbo chunk detected), and this
document as the runbook. Reshard orchestration stays Experimental until operational scale evidence
exists.
+73
View File
@@ -0,0 +1,73 @@
# Advanced — Time Series
**Capability:** `MongoCapability.TIME_SERIES`
**Property:** `ca-skeleton.persistence-mongo.advanced.time-series.enabled`
**Status:** Advanced.
## Requirements
| | |
|---|---|
| Topology | Replica set or sharded cluster. |
| Server | MongoDB 7.0 or 8.0. |
| Privilege | Standard application role; collection creation goes through the admin plane. |
## Descriptor
`MongoTimeSeriesDescriptor` declares:
- **timeField** — required, a BSON date. This is the bucketing axis.
- **metaField** — optional but nearly always wanted: the series identity (device id, tenant, sensor).
Documents sharing a `metaField` value bucket together, which is where the compression comes from.
- **granularity** — `MongoTimeSeriesGranularity`:
| Granularity | Bucket span | Use for |
|---|---|---|
| `SECONDS` | 1 hour | Sub-second to per-second ingest. |
| `MINUTES` | 24 hours | Per-minute metrics. |
| `HOURS` | 30 days | Hourly rollups. |
Granularity that is too fine produces many small buckets and loses the compression; too coarse
produces oversized buckets that must be read whole to answer a narrow query.
## What a time series collection is not
`MongoTimeSeriesCapabilityValidator` refuses the operations the collection type does not support, at
declaration time rather than at first use:
- **No arbitrary updates.** Time series data is append-mostly. Delete and limited update support
exists on recent servers but is not part of this platform's contract.
- **No unique index on the measurement.** There is no `_id` to be unique on in the usual sense.
- **No CSFLE.** Refused — see [encryption.md](encryption.md).
- **No change stream on the raw buckets** as a business event source. The bucket documents are a
storage representation, not your measurements.
Converting an existing regular collection to a time series collection is a copy, not an alter. Plan
it as a migration with a dual-write window.
## TTL
Time series collections use `expireAfterSeconds` on the collection rather than a TTL index on a
field. The [TTL rules](../schema-index-migration-guide.md#4-ttl) still apply: expiry is physical
cleanup on a bucket boundary, so a measurement can outlive its expiry by up to a bucket span plus the
monitor interval. Do not treat absence as a deadline.
## Operations
`MongoTimeSeriesOperations` is the port for insert and windowed read. Reads are bounded by the same
`MongoOperationBudget` as everything else: an unbounded time-range query on a time series collection
is the fastest way to read a year of data into heap.
## Failure recovery
| Symptom | Cause | Action |
|---|---|---|
| Writes rejected with an unsupported-operation error | An update or unique-index expectation | The collection type does not support it; change the access pattern. |
| Poor compression / large storage | Missing `metaField`, or granularity too fine | Both require a rebuild; measure on a copy before committing. |
| Slow range queries | Granularity too coarse for the query window | Same: rebuild with the granularity matched to the dominant query. |
## Promotion evidence
Actual-topology evidence on the target deployment, a migration path from the existing collection,
failure cases (unsupported update refused, CSFLE combination refused), and this document as the
runbook.
+92
View File
@@ -0,0 +1,92 @@
# BSON Mapping Guide
Design §10 and decision D-06. The representation of a value in BSON is a data contract, not an
implementation detail: once a collection holds a million documents, changing how a `BigDecimal` is
stored is a migration with downtime, not a code change. `MongoTypeRepresentationManifest` pins the
representation so a library upgrade or a different default cannot move it.
## 1. The manifest
`MongoTypeRepresentationManifest.standard()` fixes:
| Java type | BSON | Representation type |
|---|---|---|
| `UUID` | `Binary` subtype 4 | `MongoUuidRepresentation.STANDARD` |
| `BigDecimal` | `Decimal128` | `MongoDecimalRepresentation.DECIMAL_128` |
| `BigInteger` | `Decimal128` (or `String` when out of range, declared) | `MongoBigIntegerRepresentation` |
| `Instant` / `OffsetDateTime` / `ZonedDateTime` | UTC `Date` | `MongoTemporalRepresentation.UTC_DATE` |
| `LocalDate` | `String` (ISO-8601) or UTC `Date`, declared per field | `MongoTemporalRepresentation` |
| `enum` | `String` name | `MongoEnumRepresentation.NAME` |
`MongoMappingConfiguration` and `MongoCustomConversionsFactory` build the Spring Data converters from
the manifest, so there is one place to read and one place to change.
## 2. UUID
`UuidRepresentation.STANDARD` (subtype 4), always. The driver's legacy Java representation
(subtype 3) byte-swaps two halves of the UUID, so a document written by one representation and read
by the other yields a different — and valid-looking — UUID. Nothing errors; you just get the wrong
row. The golden snapshot kit pins the codec explicitly for this reason
(`MongoBsonSnapshot.defaultRegistry()`).
## 3. Decimal
`BigDecimal``Decimal128`, never `Double`. `12.30` stored as a double is `12.299999999999999`, and
a monetary comparison written against it will one day be wrong by a cent for a customer who notices.
`BigDecimalToDecimal128Converter` / `Decimal128ToBigDecimalConverter` are registered from the
manifest.
`Decimal128` has 34 significant digits; a `BigDecimal` beyond that range fails on write rather than
rounding silently.
## 4. Time
Store instants, not local times. `LocalDateTimeMappingGuard` refuses `LocalDateTime` fields on a
mapped document: a `LocalDateTime` has no offset, so the value that goes in depends on the JVM
default zone of whichever instance wrote it, and the two instances in a rolling deploy can disagree.
Use `Instant` when the moment matters and `LocalDate` when the calendar day matters.
## 5. Type metadata
`MongoTypeMetadataPolicy` decides what goes in `_class`:
| Policy | Stored | Use when |
|---|---|---|
| `NONE` | nothing | The collection holds exactly one type and never will hold a subtype. |
| `ALIAS` | a registered short alias | A polymorphic hierarchy in a long-lived collection. |
| `CLASS_NAME` | the FQCN | Short-lived or internal collections only. |
`PolicyAwareMongoTypeMapper` enforces it, and `MongoTypeMetadataRegistry` holds alias → class.
A `@LongLivedMongoDocument` type with `CLASS_NAME` is refused: writing `com.example.OrderV2` into a
million documents means that renaming the package is a data migration.
## 6. Missing versus null
The golden kit keeps these apart deliberately. `MongoBsonSnapshotAssert.hasNoField(...)` and
`hasExplicitNull(...)` are different assertions, because in MongoDB they are different documents:
`{"a": null}` matches `{a: null}` and `{a: {$exists: true}}`, while `{}` matches only the first.
A mapper change that starts writing explicit nulls silently changes what your queries return.
## 7. Golden representation tests
Every collection with a fixed representation should have a snapshot test:
```java
MongoBsonSnapshot snapshot = MongoBsonSnapshot.of(storedDocument);
MongoBsonSnapshotAssert.assertThat(snapshot)
.hasBsonType("amount", "DECIMAL128")
.hasBsonType("externalId", "BINARY")
.hasNoJavaClassName("dev.caskeleton")
.hasTypeSignature("_id:OBJECT_ID,amount:DECIMAL128,createdAt:DATE_TIME,externalId:BINARY");
```
`hasTypeSignature` is the regression gate: it fails on *any* representation change, including ones a
value-equality assertion would pass. When it fails, the question is whether the change was intended
and has a migration — not whether to update the string.
## 8. Round trips
`MongoRoundTripContract` asserts that `write → read` returns an equal domain object *and* that
`write → read → write` produces an identical BSON document. The second half is what catches an
asymmetric converter: a value that reads back equal but re-serialises differently makes every
subsequent `save()` a spurious update, and turns change streams into a noise generator.
+105
View File
@@ -0,0 +1,105 @@
# Change Stream Guide
Design §20, decision D-12. A change stream is an **at-least-once projector**, not an event bus.
## 1. What a change stream is not
D-12 is explicit: a physical change event is not a business integration event. The two differ in
ways that matter to every consumer:
| Change event | Integration event |
|---|---|
| Emitted per document write | Emitted per business fact |
| Shape follows the storage schema | Shape is a published contract |
| A refactor of the document changes it | A refactor of the document does not change it |
| Replayed on resume, duplicated on retry | Versioned and deliberately evolved |
Publishing raw change events externally makes your storage schema a public API, and the first time
someone renames a field the downstream consumers break. If you need to bridge to messaging, use the
Advanced bridge, which maps to an owned envelope
([advanced/multi-tenancy.md](advanced/multi-tenancy.md) is separate;
the bridge is described in §7 below).
## 2. Subscription and resume
`MongoChangeStreamSubscription` declares the collection, pipeline and consistency. `MongoResumePosition`
is either a resume token or a cluster time; `MongoResumeCheckpoint` is what gets persisted and
`MongoResumeCheckpointStore` persists it.
The checkpoint stores the token as a Base64 `encodedToken` string rather than a byte array — a record
with an array component has broken equality, and a checkpoint that does not compare correctly is a
checkpoint that silently fails its own dedup test.
## 3. Checkpoint after processing, not after receiving
The ordering rule that makes at-least-once actually hold:
```
receive event
→ process it (idempotently)
→ persist the checkpoint
```
Checkpointing on receipt turns the delivery guarantee into at-most-once, and the events lost are
exactly the ones the process died while handling.
## 4. Idempotency
`MongoChangeEventIdentity` is the dedup key: `(resumeToken, documentKey, clusterTime, operationType)`.
`MongoChangeDeduplicationStore` records what has been applied. Duplicates are not an edge case — every
resume after any interruption replays at least one event, so a projector that is not idempotent is
wrong on its first restart, not on some rare day.
`MongoChangeProjector` returns a `MongoChangeProjectionResult` so the runner can distinguish applied
from skipped-as-duplicate, and the skip count is worth a metric: a sudden rise means something is
looping.
## 5. States and recovery
`MongoChangeStreamState`: `STARTING`, `RUNNING`, `RESUMING`, `STOPPED`, `HISTORY_LOST`.
`MongoChangeStreamRecoveryPolicy` returns a `MongoChangeStreamRecoveryDecision`, which is either
`resume()` (auto-resume from the checkpoint) or `halt(state, runbook)`. A halting decision **must**
name a runbook — a decision that only says "stopped" leaves the on-call engineer to work out from
scratch whether the projection can be rebuilt and from what.
| Situation | Decision |
|---|---|
| Transient network error, token still valid | `resume()` |
| Primary failover | `resume()` — the token survives an election |
| `invalidate` (collection dropped/renamed) | `halt(STOPPED, …)``MongoInvalidateRecovery` |
| Token no longer in the oplog | `halt(HISTORY_LOST, "history-lost")``MongoChangeHistoryLostException` |
## 6. History lost
`MongoChangeHistoryLostException` is raised when the resume token predates the oldest oplog entry.
The stream **cannot** be resumed: the events between the checkpoint and now are gone from the server,
and no amount of retrying brings them back.
What the platform will not do is silently restart from "now". That looks like a recovery and is
actually a silent gap in the projection — the worst possible outcome, because nothing reports it. The
runner halts and requires an operator decision. See
[runbooks/history-lost.md](runbooks/history-lost.md).
## 7. Bridging to messaging (Advanced)
`MongoChangeMessagingBridge` is opt-in behind `MongoCapability.CHANGE_STREAM` plus the bridge's own
flag. It maps a change event to a platform-owned `MongoIntegrationEventEnvelope` through
`MongoChangeToIntegrationEventMapper` and hands it to a `MongoIntegrationEventPublisher` port.
The port is defined in the bridge package rather than imported from the messaging adapter because the
architecture registry forbids adapter-to-adapter dependencies; the composition root supplies the
implementation.
`MongoBridgeOutboxPolicy` and `MongoBridgeCheckpointPolicy` state the delivery contract: publish then
checkpoint, at-least-once, consumers must dedup on the envelope's event id.
## 8. Operating notes
- Change streams require a replica set. `MongoStartupValidator` refuses a change-stream profile on
`STANDALONE`.
- The change-stream principal is its own role (`MongoPrincipalRole.CHANGE_STREAM`) with
`changeStream` and `find` — not the application write credential.
- Oplog window is the recovery budget. If the oplog holds four hours, a consumer that is down for five
hours needs a rebuild, not a resume. Alert on consumer lag against the oplog window, not against
wall-clock.
@@ -0,0 +1,126 @@
# Consistency and Transaction Guide
Design §12–§16, decisions D-07 through D-10. This is the part of the platform where the wrong
default is most expensive and the least visible in testing, because every failure mode here needs a
primary change to reproduce.
## 1. Prefer a single-document atomic operation
D-09: a transaction is for a **multi-document invariant**, nothing else. A single document is already
atomic in MongoDB, so wrapping a one-document update in a transaction buys nothing and costs a
session, a two-phase commit and a new ambiguous outcome.
D-07: partial change uses update operators, not `save()`. `MongoAtomicOperations` /
`MongoAtomicOperationsTemplate` expose the operator set through `MongoUpdateOperator`
(`$set`, `$inc`, `$push`, `$pull`, `$addToSet`, `$min`, `$max`, `$currentDate`, …) with an
`AtomicFilter` precondition and a `ReturnDocumentMode`. Read-modify-write through `save()` replaces
the whole document and silently discards any field another writer changed in between — a lost update
with no error.
## 2. Whole-document replacement needs a revision
D-08. `VersionedMongoUpdater` requires a `MongoRevision`: either a Spring Data `@Version` field or an
explicit expected-revision predicate in `VersionedUpdateCommand`. A replacement whose filter matched
zero documents is not "nothing to do" — `MongoOptimisticConflictTranslator` distinguishes:
- filter matched nothing and the id does not exist → `MongoDocumentNotFoundException`
- filter matched nothing and the id exists → `MongoOptimisticConflictException`
Collapsing these two into one is how a concurrent overwrite becomes a 404.
## 3. Consistency profiles
`MongoConsistencyProfile` names the read/write concern pair; `MongoConsistencyRegistry` binds a
profile to an operation or collection, and `MongoConsistencyBinder` /
`ReactiveMongoConsistencyBinder` apply it at execution.
| Profile | Meaning | Use for |
|---|---|---|
| `PRIMARY_LOCAL` | primary read, local concern | Throughput-sensitive reads that tolerate a rollback window. |
| `PRIMARY_MAJORITY` | primary read, majority write | The default for anything a user will see again immediately. |
| `CAUSAL_MAJORITY` | majority inside a causal session | Read-your-writes across separate operations. |
| `STALE_READ_ALLOWED` | secondary reads permitted | Reporting and analytics that state their staleness. |
| `SNAPSHOT_TRANSACTION` | snapshot isolation | Multi-document reads inside a transaction. |
A profile is a declaration, not a hint: the registry is consulted per operation and an operation
without a registered profile is rejected rather than defaulting.
## 4. Causal sessions
`MongoCausalSessionContext` plus `SpringMongoCausalSessionExecutor` /
`ReactiveMongoCausalSessionExecutor` carry the cluster time and operation time between operations, so
"write then read" returns the write even when the read lands on a different node. Without a causal
session, `PRIMARY_MAJORITY` gives you durability but not read-your-writes across two calls.
In the reactive path the session travels in the Reactor context (`ReactiveMongoContextKeys`), not in
a thread local — a thread local is empty on the next operator in the chain.
## 5. Transactions
`MongoTransactionExecutor` / `ReactiveMongoTransactionExecutor` open a session through the session
factory, run the body, and commit. `MongoTransactionProfile` carries the consistency profile, the
`maxCommitTime` and the retry budget. Topology matters: a transaction requires a replica set, and
`MongoStartupValidator` refuses a transaction-declaring profile on `STANDALONE` at startup rather
than at the first call.
## 6. Retry: body and commit are different loops
D-10, and the single most consequential rule in the design.
```
for each body attempt:
open a NEW session
run the body
TransientTransactionError -> abort, next body attempt
commit
UnknownTransactionCommitResult -> retry COMMIT ONLY, same session
```
`MongoTransactionRetryCoordinator` implements exactly this:
- **A new session per body attempt.** Reusing the session after an abort carries the aborted
transaction's state into the retry.
- **The body is never replayed after a commit ambiguity.** An unknown commit means the commit may
already have applied. Re-running the body would apply it a second time. Only the commit is retried,
and a commit retry on an already-committed transaction is a no-op by design.
- **A budget bounds both loops.** `MongoRetryBudget` limits attempts *and* elapsed time, with jittered
backoff (`delayBefore(attempt, random)`), so a struggling primary is not retried into the ground.
`MongoRetryDecision` and `MongoRetryScope` (in `…api.error`) say what may be retried:
`MongoRetryScope.BODY`, `COMMIT_ONLY`, or `NONE`.
## 7. Ambiguous outcomes
`MongoExecutionOutcome` has six values, two of which are ambiguous and must not be collapsed:
| Outcome | Did the write happen? |
|---|---|
| `NOT_SENT` | No. Safe to retry. |
| `NO_WRITE_PERFORMED` | No — the server answered and did nothing. |
| `WRITE_CONFIRMED` | Yes. |
| `PARTIAL_BULK_WRITE` | Some of it. See `MongoBulkResult`. |
| `WRITE_RESULT_UNKNOWN` | **Unknown.** |
| `TRANSACTION_COMMIT_UNKNOWN` | **Unknown.** |
An unknown outcome is not a failure and must not be reported to a caller as one. The caller either
reconciles (`MongoCommitReconciler` re-reads a deterministic marker the body wrote) or surfaces the
ambiguity. See [runbooks/unknown-commit.md](runbooks/unknown-commit.md).
`MongoFailureContext` records only the design-permitted fields — outcome, category, operation name,
collection profile, retry scope, attempt — never the query, the document, or the values.
## 8. Failure translation
`DefaultMongoFailureClassifier` classifies **labels before codes**. The server's error labels
(`TransientTransactionError`, `UnknownTransactionCommitResult`, `RetryableWriteError`) are the
authoritative statement about retryability; an error code is a secondary signal whose meaning varies
by server version. `DefaultMongoFailureTranslator` maps a classification onto the stable exception
hierarchy, and anything unmatched becomes `MongoUnclassifiedFailureException` rather than leaking a
driver type.
## 9. Bulk writes
`MongoBulkExecutor` returns a `MongoBulkResult` with per-item `MongoBulkItemFailure` entries. An
unordered bulk write that partially fails is `PARTIAL_BULK_WRITE`, not a failure: some documents were
written. `MongoBulkPartialFailureException` carries the succeeded and failed indexes so a caller can
resume rather than replay.
+85
View File
@@ -0,0 +1,85 @@
# Document Modeling Guide
Design §7–§9. The platform does not own your documents — D-01 is explicit that the domain owns
`@Document`, repositories, queries, index requirements and schema version. What the platform owns is
the set of modeling decisions that are expensive to reverse once a collection holds production data.
## 1. There is no `CommonMongoRepository`
A generic `CommonMongoRepository<T, ID>` is listed under explicitly unsupported (§3.4), and the
reason is not purity. A shared supertype forces every collection to share an id strategy, a
consistency profile and a query surface, and the first collection that needs a different one either
gets a cast or a leaky generic parameter. Declare a Spring Data repository per aggregate.
## 2. Embed or reference
`MongoDocumentModelManifest` records the decision per collection so it is reviewable, and
`MongoDocumentModelValidator` refuses the combinations that do not survive growth.
| Descriptor | Use when |
|---|---|
| `EmbeddedCollectionDescriptor` | The child is read with the parent, is bounded, and has no independent lifecycle. Declare `maxElements`; an unbounded array is the single most common way a document reaches the size limit. |
| `MongoReferenceDescriptor` | The child is queried independently, is unbounded, or outlives the parent. Declare `MongoReferenceLifecycle` so the deletion story is written down rather than discovered. |
The validator rejects an embedded collection without a bound, and a reference whose lifecycle says
the child is owned by the parent but which is also referenced from elsewhere.
## 3. Size budget
`MongoDocumentSizeBudget`:
| Constant | Bytes | Meaning |
|---|---|---|
| `MONGODB_HARD_LIMIT_BYTES` | 16 MiB | MongoDB's own limit. |
| `PLATFORM_CEILING_BYTES` | 4 MiB | The largest budget the platform will accept. |
| `DEFAULT_BYTES` | 2 MiB | `MongoDocumentSizeBudget.standard()`. |
A budget above the ceiling is refused at construction. Budgeting to 16 MiB means the failing write
is the first symptom, and by then the collection is already full of near-limit documents.
## 4. Identity
`DomainDocumentId` and `MongoIdRepresentation` fix how a domain identifier becomes `_id`. Pick the
representation once per collection and record it in the manifest:
- `OBJECT_ID` — server-generated, monotonic, 12 bytes. Good default when the domain has no natural id.
- `UUID_BINARY` — a domain UUID stored as `Binary` subtype 4 (`STANDARD`). Never store a UUID as a
string "because it is easier to read"; it doubles the index size and loses the type.
- `STRING` — a natural key that is genuinely a string (a slug, an external system's id).
An `_id` choice is effectively permanent: it is the shard key candidate, the resume-token join key
and the pagination tie-breaker.
## 5. Schema version
Every long-lived collection carries `DocumentSchemaVersion`. `MongoSchemaVersionPolicy` and
`MongoSchemaVersionRange` say which versions the running code can read; a document outside the range
raises `MongoDataSchemaUnsupportedException` rather than being silently mapped with missing fields.
Write the range down before the migration, not after: the range is what lets old and new instances
run at once during a rolling deploy.
## 6. Type metadata
`@LongLivedMongoDocument` marks a document whose stored type alias must not be a Java class name.
`MongoTypeMetadataRegistry` maps alias → class. Storing the FQCN means moving or renaming the class
becomes a data migration; storing an alias keeps it a refactor. See
[bson-mapping-guide.md](bson-mapping-guide.md) §4.
## 7. Collection profiles
`MongoCollectionProfileRegistry` binds a `CollectionProfileName` to its consistency profile, budget
and allowlist. A collection that is not registered cannot be reached through
`MongoImperativeExecutor` or `ReactiveMongoExecutor` — the allowlist is the mechanism that keeps an
unreviewed collection from appearing in production by accident.
## 8. What to write down before the first insert
1. Embed/reference decision per child collection, with bounds.
2. Size budget.
3. `_id` representation.
4. Schema version range.
5. Index manifest (see [schema-index-migration-guide.md](schema-index-migration-guide.md)).
6. Consistency profile (see [consistency-transaction-guide.md](consistency-transaction-guide.md)).
Each of these is cheap now and a migration later.
+140
View File
@@ -0,0 +1,140 @@
# Query and Aggregation Guide
Design §17–§19, decision D-11. Every query and every pipeline is a registered, bounded thing. Free-form
JSON queries and unbounded pipelines are explicitly unsupported (§3.4).
## 1. Registered operations
Every execution carries a `MongoOperationContext`: a `MongoOperationName`, a `DatabaseProfileName`, a
`CollectionProfileName`, a `MongoOperationType` and a `MongoOperationScope`.
`MongoOperationName` matches `[a-z][a-z0-9.-]{2,95}`. It is the join key for the budget registry, the
consistency registry, the metric tag and the log line — a free-form or interpolated name breaks all
four at once, which is why the pattern is enforced at construction.
`MongoOperationScope` uses an `UNSPECIFIED` sentinel rather than `null`, so "the caller did not say"
is a value the policy layer can reject rather than an NPE further down.
## 2. Query guardrails
`PolicyAwareMongoQueryBuilder` builds a query from `MongoFieldDescriptor` + `MongoOperator` pairs
against a `MongoQueryPolicy`. The policy refuses:
- a field not in the collection's allowlist
- an operator not allowed for that field
- a sort on an unindexed field
- `$where`, `$expr` with arbitrary JavaScript, and server-side evaluation generally
- an unbounded `$regex`
`MongoRegexPolicy` requires an anchored prefix pattern and bounds the pattern length. An unanchored
regex is a collection scan wearing an index's clothes, and a user-supplied one is a denial-of-service
primitive.
`MongoSortDescriptor` pairs a field with a direction and is validated against the index manifest, so
a sort that would spill to disk fails review rather than production.
## 3. Operation budgets
`MongoOperationBudget` bounds four things at once:
| Bound | Why |
|---|---|
| `maxTimeMS` | The server stops working on a query nobody is waiting for. |
| result limit | An unbounded result set is an OOM with extra steps. |
| batch size | Bounds the per-round-trip memory. |
| examined-document ceiling | Catches an index regression that a time limit alone would hide on a fast day. |
`MongoBudgetPolicyRegistry` binds a budget to an operation name; `MongoBudgetEnforcer` applies it and
raises `MongoOperationRejectedException` before execution when a request exceeds it, and
`MongoTimeoutException` when the server enforces it.
## 4. Keyset pagination
Unbounded `skip` is unsupported: `skip(1_000_000)` makes the server walk a million documents to throw
them away, so page 1000 costs a thousand times page 1.
`MongoKeysetQueryBuilder` builds the resume predicate lexicographically. For a sort on `(a DESC, _id
DESC)` resuming after `(A, I)`:
```
(a < A) OR (a = A AND _id < I)
```
`validate()` rejects a `MongoKeysetSort` without a unique tie-breaker. Without one, two documents with
the same sort value straddle the page boundary and one of them is skipped or repeated — invisibly,
and only under concurrency.
`MongoNullSortOrdering` makes null placement explicit, because MongoDB's own ordering of missing
versus null versus present is not what most people assume.
### Cursors are authenticated
`MongoKeysetCursorCodec` signs the cursor with HMAC-SHA256 and compares with
`MessageDigest.isEqual` (constant time). An unsigned cursor is a client-controlled query predicate: a
caller can edit it to read a range they were never offered. A tampered or truncated cursor yields
`MongoCursorException`, never a partially-decoded resume position.
## 5. Aggregation guardrails
`MongoAggregationPlan` is a registered pipeline: an ordered list of `MongoAggregationStageDescriptor`
validated against a `MongoAggregationProfile`. `PolicyAwareMongoAggregationExecutor` runs only a
registered plan.
`MongoAggregationRisk` grades each stage, and the profile sets the ceiling:
| Risk | Stages | Policy |
|---|---|---|
| low | `$match` on an indexed prefix, `$limit`, `$project` | Always allowed. |
| moderate | `$group`, `$sort` with an index, `$unwind` with a bound | Allowed within budget. |
| high | `$lookup`, `$graphLookup`, `$facet`, unindexed `$sort` | Requires explicit approval in the profile. |
| forbidden | `$out`, `$merge` outside the admin plane, `$function`, `$accumulator` | Refused. |
`allowDiskUse` is a declared property of the plan, not a runtime flag. A pipeline that needs disk is a
pipeline whose shape should be reviewed.
## 6. Reactive execution and cursors
`ReactiveMongoExecutor` / `DefaultReactiveMongoExecutor` carry the operation context in the Reactor
context. `MongoCursorGuard` and `MongoCursorLease` bound cursor lifetime:
- a cursor has a lease with a deadline
- cancellation closes the server-side cursor (`MongoCursorTermination`)
- an abandoned cursor is a server-side resource, so the lease is released on cancel, error *and*
completion — `MongoReactiveCursorPublisher` uses `Flux.using` so all three paths run the same
release
A leaked cursor does not fail anything locally; it consumes a connection and a snapshot on the server
until the server's own timeout, which is why the guard is not optional.
## 7. Geospatial
`MongoGeoQuery` + `MongoGeoPoint` + `MongoGeoDistance` over a `2dsphere` index. Distances are metres
on a sphere (`nearSphere` with `maxDistance`), never degrees — a degree of longitude is a different
distance in Oslo than in Nairobi, and a radius expressed in degrees is a bug that only shows up away
from the equator. `SpringMongoGeospatialOperations` is the Spring Data binding;
`MongoGeospatialOperations` is the port.
## 8. Native capability gateway
When a registered operation genuinely needs something outside the Stable API, it goes through
`MongoNativeCapabilityGateway` (`PolicyAwareMongoNativeGateway`), never through the driver directly.
The admission order is fixed:
```
capability registered
→ database profile
→ collection allowlist
→ operation name present
→ timeout / maxTimeMS
→ consistency profile
→ result / batch limit
→ trace
→ log redaction
→ command category (MongoNativeCommandCategory)
→ D4 admin command refused
→ execute
```
`ApprovedMongoNativeOperation` is the registration record; `MongoNativeOperationPolicy` is the policy.
An admin-plane command reaching this gateway is refused regardless of capability — the admin plane has
its own credential and its own client (see [security-observability.md](security-observability.md)).
+124
View File
@@ -0,0 +1,124 @@
# MongoDB Document Persistence Platform — Repository Adaptation Contract
**Design source:** `mongodb-superpowers-package/docs/superpowers/specs/2026-08-11-mongodb-document-persistence-platform-design.md`
**Stable plan:** `mongodb-superpowers-package/docs/superpowers/plans/2026-08-11-mongodb-document-persistence-platform-implementation-plan.md`
**Advanced plan:** `mongodb-superpowers-package/docs/superpowers/plans/2026-08-11-mongodb-advanced-capabilities-expansion-plan.md`
The design package declares its own module root (`modules/mongodb`) and root package
(`io.backend.skeleton.mongodb`) as *implementation assumptions*, not as contract. This file is the
single record of how that assumed layout was mapped onto this repository. Only paths, build DSL,
and composition-root ownership changed. Public contracts, policy order, and error semantics are
implemented exactly as specified.
## 1. Why the module layout differs
The design assumes 19 Stable Gradle projects under `modules/mongodb/` and 12 Advanced projects
under `modules/mongodb-advanced/`. This repository is a Clean Architecture template whose
**fail-closed registry** (`src/config/architecture/modules.json`, enforced by `src/settings.gradle`
and `verifyCleanArchitectureDependencies`) declares **exactly 19 leaf identities**. Creating 31 more
Gradle projects would violate HARD-STOP #5 in `AGENTS.md`.
Therefore the design's 31 modules become **package boundaries inside the registered leaf**
`:adapter:outbound:persistence-mongo`, following the precedent already set by
[docs/httpclient/repository-adaptation.md](../httpclient/repository-adaptation.md). The design's
module dependency table (§6.3) is reproduced as ten ArchUnit rules in `MongoModuleBoundaryTest`, so
a forbidden edge fails the build the same way a missing Gradle dependency would.
## 2. Package mapping
Root package: `io.backend.skeleton.mongodb``dev.caskeleton.adapter.outbound.mongo`.
### 2.1 Stable modules
| Design module | Repository package |
|---|---|
| `mongodb-core-api` | `…outbound.mongo.api` (+ `.capability`, `.consistency`, `.error`, `.mapping`, `.observation`, `.profile`, `.schema`) |
| `mongodb-spring-data` | `…outbound.mongo.mapping` (+ `.type`), `…outbound.mongo.failure` |
| `mongodb-imperative` | `…outbound.mongo.imperative` (+ `.atomic`, `.bulk`, `.revision`) |
| `mongodb-reactive` | `…outbound.mongo.reactive` (+ `.cursor`) |
| `mongodb-query` | `…outbound.mongo.query` (+ `.budget`, `.pagination`) |
| `mongodb-aggregation` | `…outbound.mongo.aggregation` |
| `mongodb-transaction` | `…outbound.mongo.transaction` (+ `.retry`, `.session`) |
| `mongodb-index-schema` | `…outbound.mongo.schema` (+ `.index`, `.manifest`, `.model`, `.ttl`, `.validation`) |
| `mongodb-change-stream` | `…outbound.mongo.changestream` (+ `.projector`, `.recovery`) |
| `mongodb-geospatial` | `…outbound.mongo.geo` |
| `mongodb-migration-core` | `…outbound.mongo.migration` |
| `mongodb-migration-flamingock` | `…outbound.mongo.migration.flamingock` |
| `mongodb-observability` | `…outbound.mongo.observation` |
| `mongodb-security` | `…outbound.mongo.security` (+ `.admin`), `…outbound.mongo.nativecap` |
| `mongodb-spring-boot-starter` | `…outbound.mongo.autoconfigure` |
| `mongodb-testkit-core` | `…outbound.mongo.testkit.mapping`, `.compat`, `.performance` (`testkit` source set) |
| `mongodb-testkit-replicaset` | `…outbound.mongo.testkit.rs` (`testkit` source set) |
| `mongodb-testkit-failover` | `…outbound.mongo.testkit.failover` (`testkit` source set) |
| `mongodb-testkit-migration` | `…outbound.mongo.testkit.migration` (`testkit` source set) |
`…outbound.mongo.architecture` has no design counterpart: it holds the `@MongoOperation` marker and
the reusable ArchUnit rule set a fork applies to its own document/repository code.
### 2.2 Advanced modules
| Design module | Repository package |
|---|---|
| `mongodb-sharding` | `…outbound.mongo.advanced.sharding` (+ `.admin` for the D4 shard plane) |
| `mongodb-timeseries` | `…outbound.mongo.advanced.timeseries` |
| `mongodb-csfle` | `…outbound.mongo.advanced.encryption.csfle` |
| `mongodb-queryable-encryption` | `…outbound.mongo.advanced.encryption.qe` |
| `mongodb-search` | `…outbound.mongo.advanced.search` |
| `mongodb-vector-search` | `…outbound.mongo.advanced.vector` |
| `mongodb-tenancy-shared` | `…outbound.mongo.advanced.tenancy.shared` |
| `mongodb-tenancy-database` | `…outbound.mongo.advanced.tenancy.database` |
| `mongodb-change-stream-messaging-bridge` | `…outbound.mongo.advanced.bridge` |
| `mongodb-gridfs-compat` | `…outbound.mongo.advanced.gridfs` |
| `mongodb-testkit-sharded` | `…outbound.mongo.testkit.sharded` (`testkit` source set) |
| `mongodb-testkit-atlas` | `…outbound.mongo.testkit.atlas` (`testkit` source set) |
The design's rule that a Stable module never depends on an Advanced one survives as an ArchUnit rule
(`stableNeverDependsOnAdvanced`) plus the opt-in flag: every Advanced entry point requires
`MongoAdvancedCapabilityFlags` to have the matching capability enabled and refuses construction
otherwise. Being on the classpath is not being enabled.
## 3. Other deliberate substitutions
| Design assumption | Repository reality | Adaptation |
|---|---|---|
| Gradle Kotlin DSL under `modules/mongodb*` | Groovy DSL, root `build.gradle` conventions, `LockMode.STRICT` locking | Dependencies declared in `src/adapter/outbound/persistence-mongo/build.gradle`; `gradle.lockfile` regenerated. |
| `mongodb-spring-boot-starter` is a separate module the app depends on | `modules.json` gives `adapter-outbound-persistence-mongo` `runtime_memberships: []` and does **not** list it among `app-bootstrap`'s allowed dependencies | The `autoconfigure` package stays inside the leaf and registers through the leaf's own `META-INF/spring/…AutoConfiguration.imports`. This differs from the httpclient precedent, where the starter moved to `:app-bootstrap`; here the registry forbids that edge. |
| Spring Boot 4.1 / Spring Data MongoDB 5.1 baseline | Repository baseline is Spring Boot 4.0.0 / Spring Data MongoDB 5.0.0 | The platform targets the Spring Data MongoDB **API surface** common to both; no 5.1-only type is referenced. The support matrix records the actual pinned versions. |
| `MongoRetryScope` lives in `mongodb-transaction` | The `mongodb-spring-data` failure translator must classify retry scope, and it cannot depend on `mongodb-transaction` | `MongoRetryScope` lives in `…api.error` (core-api), which both packages already depend on. Same values, same meaning, one legal position in the DAG. |
| `mongodb-migration-flamingock` depends on Flamingock | Adding an unvetted external dependency is out of scope for this task, and the design itself requires the public contract not to depend on Flamingock types | The adapter is provider-neutral: it consumes a platform-owned `FlamingockChangeUnitView`. Wiring an actual Flamingock distribution is a one-file change behind that view. |
| Testkit as its own Gradle module | The design forbids production modules depending on the testkit | A dedicated `testkit` source set whose output is on the test compile/runtime classpaths only. ArchUnit rule `productionNeverDependsOnTestkit` enforces the direction. |
| Per-task `git commit` | `AGENTS.md`: commit policy is `human-only` | Implementation is delivered unstaged; commits are the human's action. This is the only plan step intentionally not executed, and it is recorded here. |
| `docs/mongodb/**`, `scripts/verify-mongodb-*.sh` | Repository already owns `docs/` and `scripts/` | Created at the same repository-relative paths. |
## 4. What is unchanged from the design
- D1 / D2 / D3 / D4 exposure planes and the ordered D3 admission sequence (§5).
- Stable API V1 with `apiStrict=true` on the D1/D2 client generation; D3/D4 on separate generations.
- `MongoExecutionOutcome`, including both ambiguous outcomes (`WRITE_RESULT_UNKNOWN`,
`TRANSACTION_COMMIT_UNKNOWN`), and `MongoFailureContext`'s permitted-field list.
- The complete stable exception hierarchy and the label-before-code classification order.
- The BSON representation manifest (UUID `STANDARD`, `Decimal128`, UTC instants, alias type metadata)
and the document-size budget.
- Update-operator-first writes, and optimistic revision as the precondition for whole-document
replacement.
- Transaction body retry and commit retry as separate loops: a new session per body attempt, and
commit-only retry on unknown commit. The body is never replayed after a commit ambiguity.
- Registered operation names and manifests for query, aggregation and index; no free-form JSON query
and no unbounded pipeline.
- Keyset pagination with an authenticated cursor and a unique tie-breaker requirement.
- Change stream as an at-least-once projector with resume-token checkpointing and explicit
history-lost handling.
- TTL as physical cleanup only, never the sole basis for access denial or business scheduling.
- Manifest-owned index/validator state with an apply policy that never drops what it does not own.
- Low-cardinality observation tags, command redaction, and the credential reference indirection.
- The Stable release gate's evidence categories, and the Advanced promotion gate's requirement for
actual-topology evidence.
## 5. Verification
```bash
bash scripts/verify-mongodb-platform.sh # Stable gate
bash scripts/verify-mongodb-advanced.sh # Advanced gate (opt-in lanes)
```
Both scripts run from the repository root and delegate to `src/gradlew`.
+98
View File
@@ -0,0 +1,98 @@
---
title: Runbook — MongoDB primary failover
category: mongodb
severity: P2
owner: oncall
last_updated: 2026-08-13
status: active
---
# Runbook: MongoDB primary failover
Design §29, scenarios `PRIMARY_KILL`, `NETWORK_PARTITION`, `SERVER_SELECTION_TIMEOUT`,
`WRITE_RESPONSE_LOSS`.
## Symptoms
- `MongoServerSelectionException` / `MongoConnectionException` spike, then recovery within seconds.
- `MongoSdamObservationListener` reports a topology change (primary removed, new primary elected).
- `MongoPoolObservationListener` shows checkout wait times rising while server-side command duration
stays flat — the wait is topology, not query cost.
- Latency spike on writes with no corresponding rise in read latency.
A failover that resolves in under ~15 s and produces no `WRITE_RESULT_UNKNOWN` is normal replica-set
behaviour and needs no action beyond confirming it self-healed.
## Diagnosis
1. Confirm an election actually happened. SDAM events distinguish an election from "the database got
slow"; without them the two are indistinguishable in application metrics.
2. Split the failure categories. Metric tag `failureCategory`:
- `SERVER_SELECTION` / `CONNECTION` → the driver could not reach a primary. `NOT_SENT`; safe.
- `TIMEOUT` with outcome `WRITE_RESULT_UNKNOWN` → a write may have applied. Not safe; see below.
- `TRANSACTION_COMMIT_UNKNOWN` → go to [unknown-commit.md](unknown-commit.md) instead.
3. Check the election duration against `MongoRetryBudget`. If the election outlasted the budget, the
retries were exhausted before a primary existed and callers saw errors that a longer budget would
have absorbed.
4. Check whether the new primary is in the expected region/AZ. A failover to a distant node changes
write latency permanently, not transiently.
## Action
**Self-healed (the common case).**
Confirm outcome distribution contains no `WRITE_RESULT_UNKNOWN`, record the election in the incident
log, and close. Nothing to replay.
**Writes with `WRITE_RESULT_UNKNOWN`.**
These writes may or may not have applied. Do not blind-retry.
- Idempotent operation (registered `MongoUpdateOperator` with an `AtomicFilter` precondition): retry.
The precondition makes the second application a no-op.
- Non-idempotent operation: reconcile by reading the target document and comparing against the
intended post-state. Retry only if it does not reflect the write.
**Server selection never recovers.**
The set has lost quorum — two of three nodes are down or partitioned. No client-side action fixes
this; escalate to the database owner to restore a majority. The application should be failing closed,
not queueing.
**Elections are frequent (more than one a day, unprompted).**
This is an infrastructure symptom, not an application one: check node resource saturation, disk
latency on the primary, and network stability between members. Repeated elections cause repeated
unknown-outcome windows.
## Escalation
- P2 → P1 if server selection has failed for more than 2 minutes, or if any non-idempotent write
returned `WRITE_RESULT_UNKNOWN` and cannot be reconciled.
- Page the database owner for quorum loss, and the service owner for reconciliation of ambiguous
writes.
## Verification
The failover lane reproduces this deliberately:
```bash
cd src
./gradlew :adapter:outbound:persistence-mongo:mongoFailoverTest --console=plain
```
It starts a real three-node set (`MongoThreeNodeReplicaSet`), stops the primary
(`MongoPrimaryController`), and injects network faults through Toxiproxy
(`ToxiproxyMongoNetworkFaultController`). A single-node set is not sufficient for the election: it
never holds one, so every guarantee that depends on a primary change goes untested.
The network faults need their own fixture (`MongoProxiedReplicaSetNode`) because a stopped container
cannot produce them. Stopping a node tells the client the write did not happen; cutting the *path*
while the server keeps running produces a client that cannot tell. `MongoNetworkFaultLaneTest`
asserts the difference by reaching the same server twice — once through the proxy, once directly:
- **Partition**: the proxied client fails, the direct client finds the server healthy and the earlier
write intact. The path was cut, not the server.
- **Response loss**: the proxied client fails, and the direct client then finds the document
*present*. The write applied and only the acknowledgement was lost —
`DefaultMongoFailureClassifier` returns `WRITE_RESULT_UNKNOWN`, and a retry would have inserted a
second document.
One detail the lane depends on: the connection is warmed before the toxic is applied. On a cold
connection it is the driver's handshake whose response is dropped, so the write is never transmitted
`NOT_SENT`, the opposite of the ambiguity being tested.
+90
View File
@@ -0,0 +1,90 @@
---
title: Runbook — MongoDB change stream history lost
category: mongodb
severity: P1
owner: oncall
last_updated: 2026-08-13
status: active
---
# Runbook: change stream history lost
Design §20.3, scenarios `OPLOG_HISTORY_LOSS`, `RESUME_TOKEN_LOSS`.
The stored resume token predates the oldest entry in the oplog. The events between the checkpoint and
now are gone from the server; no retry recovers them. `MongoChangeStreamRecoveryPolicy` returns
`halt(HISTORY_LOST, "history-lost")` and the runner stops.
**The platform will not silently restart from "now".** That looks like a recovery and is actually a
permanent, unreported gap in the projection.
## Symptoms
- `MongoChangeHistoryLostException`.
- `MongoChangeStreamState.HISTORY_LOST`; the consumer is stopped, not looping.
- Precedes it: consumer lag approaching the oplog window, or a consumer that was down for a long
period (a deploy that failed, a scaled-to-zero worker, a long outage).
## Diagnosis
1. **Determine the gap.** The checkpoint's cluster time is the start; the oldest oplog entry is the
end of what is unrecoverable. Everything in between was never processed.
2. **Determine the oplog window.** `rs.printReplicationInfo()` on the primary gives the first and last
oplog timestamps. If the window is materially smaller than it was, the write rate rose or the
oplog was resized — the consumer may be fine and the server changed.
3. **Determine what the projection is missing.** Which collections and which operations does this
projector consume? The gap is bounded by that, not by everything that happened.
4. **Check for a second consumer.** If another projector on the same collection is healthy, its
checkpoint tells you whether the problem is this consumer or the oplog.
## Action
Resuming is not an option. The choices are:
**Rebuild from source.** If the projection is derivable from the current state of the source
collections, rebuild it: stop the consumer, rebuild the projection, then start the stream from the
cluster time at which the rebuild snapshot was taken. This is the correct answer whenever the
projection is a materialised view rather than an event log, and it is the reason a projection should
be derivable.
**Backfill the gap.** If the source documents carry a timestamp covering the gap, run a bounded
backfill for that window through the migration runner (checkpointed, resumable — see
[schema-index-migration-guide.md](../schema-index-migration-guide.md) §5), then resume from the
current cluster time.
**Accept the gap explicitly.** Only when the projection is advisory and the business owner says so.
Record the window in the incident log and reset the checkpoint. This is a decision someone signs, not
a default.
Never: reset the checkpoint to "now" and restart quietly. That converts a visible P1 into an
invisible data-quality defect that surfaces months later as "the report has been wrong since March".
## Prevention
- **Alert on lag against the oplog window, not wall-clock.** "Consumer is 30 minutes behind" is fine
with a 24-hour oplog and an emergency with a 45-minute one. The threshold that matters is
`lag / oplogWindow`.
- **Size the oplog for the longest tolerable consumer outage**, including a failed deploy discovered
the next morning.
- **Checkpoint after processing, never on receipt** — see
[change-stream-guide.md](../change-stream-guide.md) §3.
- **Back up the checkpoint store.** `RESUME_TOKEN_LOSS` is the same incident reached from the other
direction: the oplog is fine, the checkpoint is gone.
- **Make the projection rebuildable.** A projection that can only be built by replaying every event
has no recovery path once the oplog rolls.
## Escalation
- P1 on detection. The consumer is stopped, so lag grows for as long as this is unresolved.
- Page the service owner for the rebuild decision, and the database owner if the oplog window shrank
unexpectedly.
## Verification
```bash
cd src
./gradlew :adapter:outbound:persistence-mongo:mongoFailoverTest --console=plain
```
`MongoFailoverScenario.OPLOG_HISTORY_LOSS` and `RESUME_TOKEN_LOSS` assert the runner halts and names
this runbook rather than restarting from the current position.
+87
View File
@@ -0,0 +1,87 @@
---
title: Runbook — MongoDB unknown transaction commit result
category: mongodb
severity: P1
owner: oncall
last_updated: 2026-08-13
status: active
---
# Runbook: unknown transaction commit result
Design §16, decision D-10, scenario `UNKNOWN_TRANSACTION_COMMIT_RESULT`.
`MongoExecutionOutcome.TRANSACTION_COMMIT_UNKNOWN` means the commit **may have applied**. It is not a
failure and must never be reported to a caller as one. The single worst response is to re-run the
transaction body: if the commit did apply, the body applies a second time.
## Symptoms
- `MongoTransactionCommitUnknownException` in logs.
- Metric `failureCategory=TRANSACTION_COMMIT_UNKNOWN`.
- Usually accompanies a primary election — see [failover.md](failover.md).
- Downstream reports of duplicated effects (double charge, double increment) are the symptom of this
being handled wrongly, not of the condition itself.
## Diagnosis
1. **Confirm the platform did the right thing automatically.**
`MongoTransactionRetryCoordinator` retries the *commit only*, on the same session, within
`MongoRetryBudget`. A commit retry against an already-committed transaction is a no-op by design.
Most occurrences resolve here and never reach a human.
2. **If the budget was exhausted, determine the actual state.** The commit either applied or it did
not; you must find out which, not guess.
- If the transaction body wrote a deterministic marker (an idempotency key, a business id, a
revision), read it back. That is exactly what `MongoCommitReconciler` does, and it is the
reason the design requires transactions to write one.
- If there is no marker: reconstruct from a downstream artefact — an outbox row, an audit record,
an external side effect. If nothing exists to compare against, the transaction was not
designed to be reconcilable and that is the finding to record.
3. **Check whether the body was replayed.** Grep for a second execution with the same operation name
and correlation id. If the body ran twice, the effects need reversing, and the code path that
replayed it is a defect: an ambiguous commit is `COMMIT_ONLY` scope
(`MongoRetryScope.COMMIT_ONLY`), never `BODY`.
## Action
**Commit applied.** Nothing to do. Record the reconciliation.
**Commit did not apply.** Re-run the whole operation from the top — a new session, a new body
attempt. This is safe precisely because you established the previous attempt left no trace.
**Cannot determine.** Do not retry. Escalate. A blind retry here is a coin flip between "no effect"
and "duplicate effect", and duplicates in a financial or notification path are worse than a delay.
Freeze the affected entity if the domain supports it, and hand off with: operation name, correlation
id, document id, the time window, and what you checked.
**Recurring.** More than one an hour means the commit path is racing something structural — a
`maxCommitTime` shorter than the observed election duration, an oversized transaction, or an
undersized retry budget. Fix the budget or the transaction shape; do not raise the retry count and
call it resolved.
## Prevention
- Every transaction body writes a deterministic marker that identifies its own commit.
- `maxCommitTime` exceeds the observed p99 election duration.
- Callers surface the ambiguity to their own callers rather than mapping it to a generic 500 — an
ambiguous outcome reported as a failure invites the caller to retry, which is the one thing that
must not happen.
- Prefer a single-document atomic operation (D-09). A transaction that exists only to wrap one
document write has invented this failure mode for nothing.
## Escalation
- Always P1 when the state cannot be determined and the operation has an external effect.
- Page the service owner immediately; the database owner only if elections are the trigger.
## Verification
```bash
cd src
./gradlew :adapter:outbound:persistence-mongo:mongoFailoverTest --console=plain
```
`MongoFailoverScenario.UNKNOWN_TRANSACTION_COMMIT_RESULT` runs this path against a real three-node
set, and the coordinator test asserts the body is never replayed after a commit ambiguity.
@@ -0,0 +1,137 @@
# Schema, Index and Migration Guide
Design §21–§25, decision D-13. Indexes and validators are **declared** in a manifest and **applied**
by an explicit plane. Automatic index creation in production is explicitly unsupported (§3.4): an
index build on a large collection is a capacity event, and discovering it because a deployment
started one is not an operating model.
## 1. The manifest is the source of truth
`MongoManifestRegistry` holds one `MongoCollectionManifest` per collection, containing:
- `MongoIndexManifest` — the declared indexes (`MongoIndexKey`, `MongoIndexDirection`, uniqueness,
partial filter, collation)
- `MongoSchemaManifest` — the declared `$jsonSchema` validator
- `MongoMetadataOwnership` — who owns each observed object
Ownership is the field that makes drift handling safe:
| Ownership | Owner | Droppable on drift |
|---|---|---|
| `APPLICATION_MANAGED` | this manifest | yes |
| `SEARCH_MANAGED` | the search service | no |
| `ENCRYPTION_MANAGED` | Queryable Encryption | no |
| `EXTERNAL` | someone else (a DBA, another service) | no |
A diff engine that does not know about ownership eventually proposes dropping
`enxcol_.customers.esc` or a search index, and "the drift tool cleaned it up" is a very bad incident
summary.
## 2. Index diff and apply
`MongoIndexDiffEngine` compares the manifest against `MongoIndexDescriptorView` observations and
produces a `MongoIndexDiff`: missing, extra, and *changed* (same name, different definition —
MongoDB will not silently rebuild these, so they must be reported rather than re-issued).
`MongoIndexApplyPolicy` decides what happens with a diff:
| Policy | Behaviour | Environment |
|---|---|---|
| `APPLY` | create what is missing | local / test |
| `APPLY_WITH_DIFF` | create what is missing and report the rest | staging |
| `DIFF_WITH_APPROVED_APPLY` | apply only what a human approved | production |
| `REPORT_ONLY` | never write | audit |
Dropping is never implicit. `MongoIndexRetirementPlan` moves an index through
`MongoIndexRetirementState` — declared → hidden → observed-unused → droppable — and each transition
is a separate deployment. Hiding an index makes the planner ignore it while keeping it maintained, so
an unexpected regression is one command to undo. Dropping it is not.
## 3. Validators
`MongoValidatorDescriptor` carries the `$jsonSchema`, a `MongoValidationLevel`
(`OFF` / `MODERATE` / `STRICT`) and a `MongoValidationAction`.
**Stable validation actions are `error` and `warn` only.** `errorAndLog` is not part of the Stable
contract on MongoDB 7.0 or 8.0 and the descriptor refuses it.
`MongoValidatorDiffEngine` produces a `MongoValidatorDiff`; `MongoValidatorApplyPolicy` gates the
apply. Tightening a validator on a collection with existing data is the dangerous direction: introduce
it as `warn` + `MODERATE`, confirm the warning count is zero, then promote to `error` + `STRICT` in a
second deployment.
## 4. TTL
D-13: TTL is **physical cleanup**, nothing else.
`MongoTtlIndexDescriptor` declares the field and `expireAfterSeconds`. `MongoTtlPolicyValidator`
enforces what `MongoTtlPolicy` allows, and `MongoExpirationAccessPolicy` states the rule that matters:
> A document's presence is not authorization, and its absence is not a deadline.
The TTL monitor runs about once a minute and deletes in batches, so a document can outlive its
expiry by minutes to hours under load. Consequences:
- Access control must check the expiry field, not the document's existence. A still-present expired
session is a valid document and an invalid session.
- Business scheduling must not be built on TTL. If something must happen at a time, schedule it.
- A TTL field must be a BSON date. A TTL index on a string silently never deletes anything.
## 5. Migrations
`MongoMigrationRunner` executes `MongoMigration` units with:
- `MongoMigrationId` — ordered, unique
- `MongoMigrationChecksum` — content hash; a changed checksum for an applied id is a hard failure, not
a re-run. Editing an applied migration means two environments ran different code under the same id.
- `MongoMigrationLedger` — what has been applied
- `MongoMigrationLock` — one runner at a time; a rolling deploy starts several instances at once
- `MongoMigrationPrecondition` / `MongoMigrationPostcondition` — checked before and after; a migration
that cannot verify its own result is a migration whose failure is discovered by a customer
- `MongoMigrationCheckpoint` — a resumable position for a backfill
`MongoMigrationResult` reports applied / incomplete / dry-run with the reason. `INCOMPLETE` is not a
failure: a rate-limited backfill that ran out of its time budget has done real work and stored a
checkpoint, and reporting it as failed would send the next run back to the beginning.
`MongoCollectionMigrationLedger` and `MongoCollectionMigrationLock` are the MongoDB-backed
implementations. Two details are load-bearing and only exist on a server:
- The ledger's **unique index** on the migration id, created by `ensureIndexes()`. Without it, two
runners that both pass the "not applied yet" read both insert, and the ledger then reports one
migration applied twice with two checksums — indistinguishable from tampering.
- The lease is taken with **one conditional update**, not read-then-write. A filter matching only a
free or expired lease lets the server pick the winner; two runners that each read "free" and then
write would both believe they hold it.
The lease expires so a runner killed mid-migration does not block every future deployment, and
`refresh` between batches is what proves the holder is still alive.
### Backfills restart, they do not restart-from-zero
A long backfill will be interrupted — a deploy, an OOM, a node replacement. The checkpoint records
the last completed key so the restart continues rather than re-processing from the beginning.
`MongoBackfillRestartFixture` in the testkit asserts exactly this: kill mid-run, restart, and the
result is identical to the uninterrupted run and does not re-apply completed work.
### Flamingock
`FlamingockMongoMigrationAdapter` bridges to Flamingock through the platform-owned
`FlamingockChangeUnitView`, with `FlamingockLedgerAdapter` and `FlamingockLockAdapter` mapping the
ledger and lock. The public contract does not reference Flamingock types, so the provider can be
replaced without touching a migration.
## 6. Ordering with deployments
```
1. Add the index (hidden if it is large) -> deployment N
2. Unhide / verify usage -> deployment N+1
3. Ship code that depends on the index -> deployment N+1
4. Backfill data -> migration, resumable
5. Tighten the validator from warn to error -> deployment N+2
6. Retire the old index through the retirement states -> deployments N+3…
```
Each step is independently reversible. A deployment that adds an index and the code that requires it
at the same time has no safe rollback: rolling back the code leaves the index build running, and
rolling back the index breaks the code that is still live on half the fleet.
+147
View File
@@ -0,0 +1,147 @@
# Security and Observability
Design §26–§28, decision D-05. The application plane and the admin plane are different credentials on
different clients, and telemetry never becomes an exfiltration path.
## 1. Roles
`MongoPrincipalRole` — one credential per role, least privilege:
| Role | Grants |
|---|---|
| `APP_READ` | `find` on allowlisted collections |
| `APP_WRITE` | `insert`, `update`, `delete` on allowlisted collections |
| `CHANGE_STREAM` | `changeStream`, `find` |
| `MIGRATION` | index and validator management on the target collections |
| `SEARCH_ADMIN` | search index management |
| `SHARD_ADMIN` | shard key operations |
| `ENCRYPTION_ADMIN` | key vault access |
| `DBA` | the human plane; never used by an application |
`MongoSecurityProfileValidator` checks the profile at startup. `forbiddenPrivilegesHeld()` names the
privileges the profile holds and must not — the validator reports *which* one, because "your
credential is over-privileged" without a name is an unactionable finding.
The privileges that must never appear on an application credential: `dropDatabase`,
`dropCollection`, `shutdown`, `killop`, `root`, `__system`, `dbOwner`, `userAdminAnyDatabase`.
## 2. Credentials are references, not values
`MongoCredentialReference` holds a `secret://…` reference plus the role. The reference is resolved at
connection time by the secret provider; the password is never a property value, a log field, or a
constructor argument that could end up in a stack trace.
`MongoCredentialRotationPolicy` states the rotation contract: overlapping validity, a drain window,
and a rotation that never requires a restart. `MongoClientGenerationRegistry` implements the swap —
a new `MongoClientGeneration` starts serving new operations while the previous generation is
`markDraining()` until its in-flight operations finish. Killing the old client immediately fails every
in-flight request, which is why rotation without generations is an outage.
Rotation is a failover scenario in the release gate (`MongoFailoverScenario.CREDENTIAL_ROTATION`),
not a runbook step people hope works.
## 3. TLS and connection policy
`MongoSecurityProfile.production(...)` requires TLS and refuses `tlsAllowInvalidCertificates` /
`tlsAllowInvalidHostnames`. `MongoSecurityProfile.local(...)` exists so a developer does not have to
weaken the production factory to get a container to connect; the startup validator refuses a local
profile on a production runtime profile.
## 4. Admin plane (D4)
`MongoAdminGateway` is the only path to `MongoAdminOperation`, and it runs on the D4 client with the
DBA-scoped credential — not the application's.
- `MongoAdminAuthorization` checks the caller's role against the operation.
- `MongoAdminRuntimeGuard` refuses high-risk operations (`highRisk()`) unless the runtime profile
explicitly permits them; a `dropCollection` reachable from a running application is a data-loss
vector regardless of how well-reviewed the calling code is.
- `MongoAdminAuditRecord` records who ran what, when and against which collection profile — before
execution, so a failed attempt is recorded too.
The native capability gateway (D3) refuses any admin-category command, so there is no path from the
application plane into the admin plane.
## 5. Observability tags
`MongoObservationConvention` allowlists exactly eight tag names:
```
mongoProfile, databaseProfile, collectionProfile, operationName,
operationType, result, failureCategory, consistencyProfile
```
and explicitly forbids:
```
documentId, rawTenantId, tenantId, dynamicCollectionName, queryParameter,
query, fullBson, resumeToken, shardKeyValue, plaintextPII, credential
```
Two reasons, and both matter. Cardinality: a tag whose values are document ids produces one time
series per document, which is how a metrics backend falls over. Confidentiality: a metric label is
stored, shipped and retained by systems with a different access model than the database.
`requireAllowed(tagName)` throws on anything outside the list, so a new tag is a deliberate change to
the convention rather than a line in a service.
`MicrometerMongoOperationObserver` implements the `MongoOperationObserver` port;
`NoOpMongoOperationObserver` is the default so observation is opt-in and never a hard dependency.
## 6. Driver-native listeners
`MongoDriverObservabilityConfiguration` registers three driver listeners, because they answer
questions the application-level timer cannot:
| Listener | Answers |
|---|---|
| `MongoCommandObservationListener` | How long did the *server* take, versus how long the caller waited? |
| `MongoPoolObservationListener` | Was the wait time connection checkout rather than query execution? |
| `MongoSdamObservationListener` | Did the topology change — an election, a node removed — during the window? |
Without pool and SDAM events, every failover looks like "the database got slow", and the difference
between "we need a bigger pool" and "we lost a primary" is invisible.
## 7. Command redaction
`MongoObservationRedactor.describe(commandName)`:
- Authentication and user-management commands (`authenticate`, `saslStart`, `saslContinue`,
`getnonce`, `createUser`, `updateUser`, `copydb*`) render as `<redacted>` — their arguments carry
credentials and key material.
- Structural commands (`ping`, `hello`, `buildInfo`, `listCollections`, `listIndexes`, `collStats`)
render by name; their arguments are not data-bearing.
- Everything else renders as `name(...)`: you get the command, never the filter or the document.
`isAlwaysRedacted(...)` is the assertion hook so a test can prove no logging path can render an auth
command's arguments.
## 8. Startup validation
`MongoStartupValidator` runs at context refresh, before the first request:
1. `MongoTopologyProbe` reports the actual `MongoTopology`.
2. Each declared `MongoTopologyRequirement` is checked against it — a transaction, causal-session or
change-stream requirement fails closed on `STANDALONE`.
3. `MongoSecurityProfileValidator` checks credentials and TLS.
4. `MongoCapabilitySupport` checks declared capabilities against the server version, with
`MongoSupportLevel` distinguishing `STABLE` / `ADVANCED` / `EXPERIMENTAL` / `UNSUPPORTED`.
5. `MongoPlatformHealthIndicator` reports the outcome for the readiness probe.
A misconfiguration found at startup costs a failed deploy. The same misconfiguration found at runtime
costs an incident, and the failing operation is rarely the one that reveals the cause.
## 9. How the security lane proves any of this
```bash
cd src
./gradlew :adapter:outbound:persistence-mongo:mongoSecurityIntegrationTest --console=plain
```
The lane runs against `MongoAuthenticatedReplicaSetContainer`, which starts mongod with `--auth` and a
generated keyfile. That detail is the whole lane: Testcontainers' `MongoDBContainer` starts mongod
*without* `--auth`, so users created on it all have every privilege and a least-privilege assertion
passes no matter how wrong the roles are. A security test that cannot fail is not a security test.
What the lane asserts is the refusal: the `read` role's insert is rejected, and the application role's
`dropDatabase` is rejected. Then it checks that `MongoSecurityProfileValidator` names the same
privilege the server just refused.
+91
View File
@@ -0,0 +1,91 @@
# MongoDB Platform — Support Matrix
Design §4. This file records what the platform is *certified* on, not what it happens to run on.
A configuration absent from this table is unsupported until someone runs the gate against it and
adds a row.
## 1. Runtime baseline
| Component | Version | Policy |
|---|---|---|
| Java | 21 | Repository runtime baseline. |
| Spring Boot | 4.0.0 | BOM-managed. Individual driver overrides are forbidden. |
| Spring Data MongoDB | 5.0.0 | Repository and `MongoTemplate` integration. Version comes from the Boot BOM. |
| MongoDB Java Driver | 5.6.1 | BOM-managed. Never pinned directly in the module. |
| Reactor | 3.8.0 | Reactive execution path. |
| Micrometer | 1.16.0 | Driver-native observability. |
| Testcontainers | 2.0.2 | Replica-set, failover, migration and compatibility lanes. |
The design's baseline is Spring Boot 4.1.x / Spring Data MongoDB 5.1.x. This repository is on
4.0.0 / 5.0.0, so the platform targets only the API surface common to both. See
[repository-adaptation.md](repository-adaptation.md) §3.
## 2. Server versions
| Lane | Version | Pinned image | Gradle task |
|---|---|---|---|
| Primary certification | MongoDB 8.0 | `mongo:8.0.16` | `mongoReplicaSetTest`, `mongoFailoverTest` |
| Compatibility | MongoDB 7.0 | `mongo:7.0.28` | `mongoCompatibilityTest` |
| Network fault injection | — | `ghcr.io/shopify/toxiproxy:2.12.0` | `mongoFailoverTest` |
Images are pinned, never `latest`: a mutable tag means the certification result describes whatever
was pulled that morning, not the version in the row. Override with
`-PmongoPrimaryImage=…` / `-PmongoCompatibilityImage=…` when testing a new patch level, and update
the row once the gate passes.
`MongoVersionMatrix.standard()` is the machine-readable form of this table; a version outside it
fails `certifies()`.
## 3. Topologies
| Topology | Status | What is certified | What is not |
|---|---|---|---|
| Standalone | **Smoke only** | Basic CRUD and mapping. | Not a production profile and never counts as Stable release evidence (D-03). Transactions, retryable writes and change streams are refused at startup by `MongoStartupValidator`. |
| Single-node replica set | **Local default** (D-02) | Transactions, retryable writes, change streams — the same semantics as production. | Elections. A single-node set never holds one, so failover behaviour is untested here. |
| 3-node replica set | **Stable production gate** | Everything above plus primary failover, unknown-commit handling and change-stream resume across an election. | Shard routing. |
| Sharded cluster | **Advanced gate** | Shard-key routing classification, scatter-gather refusal, `admin` plane operations. | Not included in the Stable gate. |
| Atlas / provider-managed | **Per-capability gate** | Search, vector search and encryption against the actual target deployment. | Atlas Local in a container is a pull-request convenience, explicitly **not** release evidence (`MongoAtlasCapabilityContractSuite.Environment`). |
## 4. Stable API and client generations
| Plane | Stable API | Purpose |
|---|---|---|
| D1 Standard document persistence | V1, `apiStrict=true` | Repositories, typed queries, atomic updates, optimistic revision. |
| D2 Advanced document operations | V1, `apiStrict=true` | `MongoTemplate`, transactions, bulk, aggregation, keyset cursors, change streams. |
| D3 Explicit Mongo capability | Not strict | Native BSON, time series, search/vector, CSFLE/QE, shard-aware operations — each behind a registered capability. |
| D4 Admin plane | Not strict | Collection, validator, index, migration, shard and repair commands. Separate credential, separate client. |
D3 is not a raw-client escape. Every call passes capability registration → database profile →
collection allowlist → operation name → timeout → consistency profile → result limit → trace →
redaction → command category → D4 refusal, in that order.
## 5. Validation actions
Stable validation actions are `error` and `warn`. `errorAndLog` is **not** part of the Stable
contract on 7.0 or 8.0 and `MongoValidatorDescriptor` refuses it.
## 6. Explicitly unsupported
Per design §3.4, none of the following is provided, and adding one is a design change rather than a
feature request:
- A generic `CommonMongoRepository<T, ID>`.
- Arbitrary runtime `runCommand`.
- Automatic index creation in production.
- A Standalone production contract.
- TTL as an exact business scheduler or as the only access control.
- Publishing raw change events as external business integration events.
- GridFS as the source of truth for new files.
- Java fully-qualified class names as a long-lived BSON schema.
- Unbounded skip pagination, unbounded aggregation pipelines, unbounded regex, unbounded results.
## 7. Capability tiers
| Tier | Capabilities | Enablement |
|---|---|---|
| Stable | Mapping, imperative/reactive execution, atomic update, optimistic lock, transactions, consistency profiles, retry/translation, query and aggregation guardrails, schema/index manifests, keyset pagination, bulk partial results, change streams, TTL contract, GeoJSON, security, observability | On when `ca-skeleton.persistence-mongo.enabled=true`. |
| Advanced | Sharding-aware query, time series, CSFLE, Queryable Encryption (equality/range), change-stream→messaging bridge, shared-collection multi-tenancy | Each behind `ca-skeleton.persistence-mongo.advanced.<capability>.enabled`. |
| Experimental | Search, vector search, hybrid search, database-per-tenant, collection-per-tenant, reshard orchestration, provider-specific features | Same flag mechanism; promotion additionally requires the evidence in [ADR-MONGO-ADV-001](../adr/ADR-MONGO-ADV-001-capability-promotion.md). |
`MongoAdvancedCapabilityFlags.propertyFor(capability)` is the authoritative property name for any
capability; the table above is its prose form.
@@ -0,0 +1,27 @@
# NOTIF-ADR-001 — `submit()` means durable acceptance
## Status
Accepted.
## Context
The obvious API for a notification platform is `send()` returning success or failure. Every channel
this platform supports makes that return value a lie:
- SES accepts a request, returns a `MessageId`, and can still decline to send.
- Twilio separates `accepted`, `sent` and `delivered` into distinct, later events.
- APNs accepts a notification and may then deliver, store or discard it.
- Web Push separates push-service acceptance from user-agent acknowledgement at the protocol level.
## Decision
`submit()` and `schedule()` return once the logical request and its recipient jobs are committed to
the database. The receipt carries `notificationId`, `RequestStatus` and `acceptedAt`, and has no
`delivered`, `sent` or `read` component. No provider is contacted while the transaction is open.
## Consequences
Callers cannot mistake acceptance for delivery, because the type does not offer that reading.
Delivery state is a separate query against the projection built from the provider event ledger. The
cost is that "did it arrive?" is a second question — which is the honest number of questions.
@@ -0,0 +1,25 @@
# NOTIF-ADR-002 — append-only event ledger with channel projectors
## Status
Accepted.
## Context
A single linear delivery status has to be updated in place, which forces a rule for deciding whether
a new event outranks the stored one. The natural rule — compare ordinals — is wrong for real provider
traffic. Twilio does not guarantee callback ordering, so `sent` arrives after `delivered`. Email
generates complaints after deliveries. Both cases lose information under an ordinal rule.
## Decision
Provider events are appended to an immutable ledger before any projection runs. Channel-specific
projectors merge events into `SubmissionOutcome`, `DeliveryOutcome`, `EvidenceLevel`,
`EngagementFacts` and `SuppressionFacts` using explicit transition tables. Projection is idempotent
and can be replayed from the ledger.
## Consequences
Duplicate, out-of-order and late events are normal inputs rather than defects. A projector bug is
recoverable, because the events it mis-projected are still stored. Projector versions can be migrated
by replay. The cost is a second write per event and a projection that can lag its ledger.
@@ -0,0 +1,37 @@
# NOTIF-ADR-003 — ambiguous submission is a first-class state
## Status
Accepted.
## Context
The most common serious failure is not a rejection. It is a request whose body reached the provider
and whose response never came back. The platform has no provider request id, and the user may or may
not have received the notification.
Treating that as a failure produces duplicates: a retry sends a second message, and a cross-channel
fallback sends the SMS next to the push that already arrived. Treating it as a success loses real
failures.
## Decision
`AMBIGUOUS` is a stored `SubmissionOutcome` and `AttemptConfirmation`. Attempts record
`requestStarted`, `requestBodyCommitted` and `providerResponseReceived`, each with an
`EvidenceCertainty` of `PROVEN`, `INFERRED` or `UNKNOWN`, so an adapter that does not know is not
forced to answer `false`.
While an ambiguous attempt exists on a recipient delivery:
- automatic retry is blocked unless the provider proves per-request idempotency
- automatic cross-channel fallback is blocked unconditionally
- reconciliation runs where the provider supports a status query
- otherwise the delivery stops and waits for an operator
Operator redrive of an ambiguous attempt requires explicit duplicate-risk approval.
## Consequences
Some notifications stop in a state that needs a human or a reconciliation pass. That is the intended
trade: an unresolved unknown is cheaper than a guaranteed duplicate, and the state is visible rather
than silently resolved in either direction.
@@ -0,0 +1,27 @@
# NOTIF-ADR-004 — FCM installation id is the primary target
## Status
Accepted.
## Context
Firebase now recommends the installation id (FID) and treats registration-token multicast paths as
legacy. A contact point model built on a single `token` string would encode the older model as the
only one, and a later migration would be a runtime interpretation problem: the same string field
would mean different things for different rows.
## Decision
`MobilePushTarget` is a sealed hierarchy of `FcmInstallationId`, `LegacyFcmRegistrationToken` and
`ApnsDeviceToken`. The kinds are separate types, never a discriminator on one string field, and each
carries its own `ContactPointType` so the uniqueness scope and the encryption associated data differ.
APNs tokens additionally carry their environment, because sandbox and production are separate
namespaces rather than a flag.
## Consequences
Migrating a target kind is a compile-time change with an exhaustive `switch`, not a runtime guess.
The adapter maps each kind to its own wire representation, so a provider changing one path cannot
silently change the other. The cost is one more type than a string field would need.
@@ -0,0 +1,52 @@
# Callbacks and reconciliation
## Ingestion order
```text
body size limit
→ content type
→ profile lookup
→ signature verification
→ append to the ledger
→ duplicate detection
→ normalization
→ attempt resolution
→ projection
→ side effects
→ 2xx
```
Appending before projecting is what makes a fast 2xx honest. The provider is told the event is
recorded, and a projector defect becomes a replay problem rather than a lost event.
A rejected signature is recorded in the security audit, never in the provider event ledger. Writing
it to the ledger would let anyone who can reach the endpoint fill a delivery history with noise.
## Duplicates and ordering
Duplicate suppression uses `(providerProfileId, providerEventId)` where the provider supplies an
event id, and a deterministic fingerprint over profile, request id, event type, occurrence time and
payload digest where it does not. A duplicate is acknowledged and projected exactly once.
Out-of-order callbacks are normal. Ordering is resolved by event semantics, not by arrival time.
## Unknown fields
Callback parsers tolerate unknown JSON fields. Normalization only rejects a payload when a field
required to identify the attempt is missing. Providers add fields; that must not stop ingestion.
## Reconciliation
Reconciliation targets:
- attempts stuck in `DISPATCHING` past their lease
- ambiguous submissions
- accepted attempts whose callback SLA has expired
- unmatched provider events
A confirmed query result is appended to the same ledger with `source = RECONCILIATION` and projected
by the same projector, so projection replay stays possible: there is no privileged second path that
writes projections directly.
Where a provider has no status-query capability, the platform records `Unsupported` and leaves the
attempt ambiguous. It does not infer a final status.
@@ -0,0 +1,41 @@
# Configuration reference
## Dispatch
| Property | Meaning | Bound |
|---|---|---|
| `claim-batch-size` | Rows claimed per scheduler tick | 1..1000 |
| `lease-duration` | How long a claimed job stays owned | positive, finite |
| `max-global-concurrency` | Ceiling across all providers | positive |
| `max-queue-age` | Age at which a job is escalated | positive |
| `max-retry-concurrency` | Ceiling for retry work | positive |
| `scheduler-poll-interval` | Queue poll cadence | positive |
| `callback-worker-concurrency` | Callback projection workers | positive |
Every value is bounded. "Unlimited" is not an accepted configuration.
## Provider profiles
A profile pins provider type, environment, credential profile, timeouts, concurrency, rate limit,
retry policy and callback profile. Sender identity and credential profile are separate concerns.
## Startup failures
Startup fails rather than degrading when:
- a payload or queue setting is unbounded
- a timeout is negative
- a TTL-required profile has no expiry source
- a callback signing secret is missing
- a production profile enables trust-all
- an APNs profile is missing its environment or topic
- a Web Push profile is missing its VAPID key
- two provider profiles share an id
- a route points only at disabled providers
- ambiguous fallback is enabled by default
## Secrets
All key material arrives through `SecretMaterialProvider`. Nothing is read from source, from a
committed file, or from a plaintext log. Contact point encryption and lookup HMAC keys must be
distinct, and the encryption key must be exactly 256 bits.
+62
View File
@@ -0,0 +1,62 @@
# Delivery evidence model
## The shape
```text
NotificationRequest
└─ RecipientDelivery
└─ DeliveryAttempt
└─ ProviderEvent (append-only)
└─ channel projector
└─ SubmissionOutcome / DeliveryOutcome / EvidenceLevel
+ EngagementFacts + SuppressionFacts
```
Four identities, four lifecycles. A logical request is not a recipient job, a recipient job is not a
provider attempt, and a provider attempt is not the event stream that describes it.
## Why not one status enum
A single linear status would have to answer "what happened?" with one value, and the real answers do
not fit on one line:
- An email can be `DELIVERED` and then generate a complaint. Both facts are true and both matter:
one for reporting, the other for suppression.
- Twilio does not guarantee callback ordering, so `sent` routinely arrives after `delivered`. Under
an ordinal rule the later, weaker event silently overwrites the stronger one.
- APNs may accept a notification and then store, replace or discard it.
So the ledger stores events and a channel projector merges them through an explicit transition table.
`StandardDeliveryProjector` holds the shared rules; provider projectors add only their own event
vocabulary.
## Merge rules
| Transition | Result |
|---|---|
| `sent``delivered` | applied |
| `delivered``sent` | ignored, event still stored |
| `delivered``complaint` | complaint fact added, delivery preserved |
| `complaint``delivered` | delivery applied, complaint preserved |
| `accepted``bounced` | applied |
| `read``displayed` | ignored |
| hard bounce → `delivered` | ignored, hard bounce is terminal |
Engagement (`opened`, `clicked`) is stored beside the delivery outcome and never changes it.
## Ambiguity
```text
platform ──── send ────▶ provider
└── accepted
✗ connection reset
```
The platform may hold no provider request id while the notification really was sent. The attempt
records `requestStarted`, `requestBodyCommitted`, `providerResponseReceived` and an
`EvidenceCertainty` for each, so a later decision can tell "we know nothing was sent" apart from "we
could not read the answer".
`ProviderSubmissionResult` enforces this: an ambiguous result may not claim `PROVIDER_ACCEPTED`, and
no submission result of any kind may carry a delivery outcome.
+31
View File
@@ -0,0 +1,31 @@
# Migration guide
## From the R0 routing seam
The pre-existing `dev.caskeleton.adapter.outbound.notification` router (`RoutingNotifier`,
`FailOpenNotificationProvider`, the Google email and Slack webhook seams) stays untouched. The
delivery platform lives beside it under `…notification.platform` and does not modify or delete any
R0 class.
Migration order per capability:
1. Register the contact points behind `ContactPointStorePort` so the platform owns protected values.
2. Publish the template version, and pin the template id, version and locale at every call site.
3. Move the call site from the router to the N1 typed facade for the channel.
4. Verify evidence in the snapshot rather than in the caller's return value: `submit()` is durable
acceptance and nothing more.
5. Remove the R0 route only after the platform route has produced provider evidence in the target
environment.
## Return-value semantics change
The R0 seam returned a send-shaped result. `NotificationReceipt` returns `notificationId`, a request
status and an acceptance time. Callers that treated the old return value as proof of delivery must be
changed; there is no compatibility shim, because a shim would have to invent the delivery claim this
platform exists to avoid.
## FCM target migration
Registration tokens keep working through `LegacyFcmRegistrationToken`. New registrations should use
`FcmInstallationId`. The two are distinct types, so a migration is a compile-time task rather than a
runtime guess.
+94
View File
@@ -0,0 +1,94 @@
# Notification Delivery Platform — module mapping
> Source design: `notification-superpowers-package/docs/superpowers/specs/2026-08-10-notification-platform-design.md`
>
> Source plan: `notification-superpowers-package/docs/superpowers/plans/2026-08-10-notification-platform-implementation-plan.md`
## Why a mapping exists
The plan was written against a hypothetical repository (`modules/notification/**`, root package
`io.backend.skeleton.notification`, 31 Gradle projects). This repository is a fail-closed
19-leaf Clean Architecture template: `src/settings.gradle` rejects any registry that does not
contain exactly the 19 modules in `src/config/architecture/modules.json`, and
`verifyCleanArchitectureDependencies` rejects any project edge outside `allowed_dependencies`.
Creating 31 new Gradle projects would violate HARD-STOP #5 of `AGENTS.md`. The package README
anticipates this and instructs the implementer to map dependency catalog and package/file paths onto
the host repository's rules while preserving the public contracts and reliability semantics.
Every logical module of the plan is therefore implemented as a **package** inside the registered leaf
that owns its responsibility. No public contract, evidence rule, or reliability semantic is dropped.
## Logical module → registered leaf
| Plan module | Registered leaf | Package |
|---|---|---|
| `notification-core-api` | `application-core` | `dev.caskeleton.application.notification.platform.api` |
| `notification-content-api` | `application-core` | `…platform.api.content` |
| `notification-contact-api` | `application-core` | `…platform.contact` |
| `notification-template-api` | `application-core` | `…platform.template` |
| `notification-policy` | `application-core` | `…platform.policy` |
| `notification-provider-spi` | `application-core` | `…platform.provider` |
| `notification-callback-api` | `application-core` | `…platform.callback` |
| `notification-email-api` | `application-core` | `…platform.email` |
| `notification-sms-api` | `application-core` | `…platform.sms` |
| `notification-push-api` | `application-core` | `…platform.push` |
| `notification-webpush` (API half) | `application-core` | `…platform.webpush` |
| `notification-inbox-api` | `application-core` | `…platform.inbox` |
| `notification-admin-api` | `application-core` | `…platform.admin` |
| `notification-security` (ports + redaction) | `application-core` | `…platform.security` |
| `notification-observability` (ports) | `application-core` | `…platform.observation` |
| `notification-dispatch-runtime` | `adapter:outbound:notification` | `dev.caskeleton.adapter.outbound.notification.platform.dispatch` |
| `notification-security` (AES-GCM/HMAC impl) | `adapter:outbound:notification` | `…platform.security` |
| `notification-template-thymeleaf` (reference renderer) | `adapter:outbound:notification` | `…platform.template` |
| `notification-email-smtp` | `adapter:outbound:notification` | `…platform.provider.smtp` |
| `notification-email-ses` | `adapter:outbound:notification` | `…platform.provider.ses` |
| `notification-sms-twilio` | `adapter:outbound:notification` | `…platform.provider.twilio` |
| `notification-push-fcm` | `adapter:outbound:notification` | `…platform.provider.fcm` |
| `notification-push-apns` | `adapter:outbound:notification` | `…platform.provider.apns` |
| `notification-webpush` (transport + crypto) | `adapter:outbound:notification` | `…platform.provider.webpush` |
| `notification-webhook-extension` | `adapter:outbound:notification` | `…platform.provider.webhook` |
| `notification-observability` (Micrometer impl) | `adapter:outbound:notification` | `…platform.observation` |
| `notification-admin-runtime` | `adapter:outbound:notification` | `…platform.admin` |
| `notification-reactor` | `adapter:outbound:notification` | `…platform.reactor` |
| `notification-spring-boot-starter` | `adapter:outbound:notification` (+ `app-bootstrap` wiring) | `…platform.autoconfigure` |
| `notification-persistence-jpa` | `adapter:outbound:persistence-jpa` | `dev.caskeleton.adapter.outbound.persistence.notification.platform` |
| `notification-inbox-jpa` | `adapter:outbound:persistence-jpa` | `…persistence.notification.platform.inbox` |
| `notification-callback-mvc` | `adapter:inbound:web` | `dev.caskeleton.adapter.inbound.web.notification.platform.callback` |
| `notification-callback-webflux` | `adapter:inbound:web` | `…callback.reactive` |
| `notification-testkit` | test source sets of the owning leaves | `…platform.testkit` |
## Dependency-direction consequences
The plan's module DAG (`*-api``provider-spi`/`policy` → runtime/adapters → starter) is preserved
by the leaf DAG that the registry already enforces:
```text
application-core (all *-api, provider SPI, policy, callback contracts)
↑ ↑ ↑
adapter:outbound:notification adapter:outbound:persistence-jpa adapter:inbound:web
↑ ↑ ↑
app-bootstrap
```
Two plan edges cannot be expressed as project edges in this repository, and are replaced by ports:
1. `notification-email-ses`, `notification-sms-twilio`, `notification-push-fcm`,
`notification-push-apns`, `notification-webpush`, `notification-webhook-extension`
`httpclient platform`.
`adapter-outbound-notification` is not allowed to depend on `adapter-outbound-httpclient`.
The provider adapters therefore call
`dev.caskeleton.adapter.outbound.notification.platform.provider.http.NotificationHttpGateway`,
an adapter-local port with a JDK `java.net.http.HttpClient` default implementation.
`app-bootstrap` sees both leaves and is the supported place to substitute an implementation backed
by the HTTP Client Platform (TLS/timeout/circuit-breaker/SSRF/dynamic-target policy reuse).
2. `notification-inbox-jpa``optional messaging outbox integration`.
`adapter-outbound-persistence-jpa` may not depend on `adapter-outbound-messaging`; the inbox
publishes through the existing persistence outbox tables plus the
`NotificationInboxSignalPort` application port, and `app-bootstrap` binds the relay.
## Commit policy
`AGENTS.md` pins commit policy to `human-only`. Step 5 (`git add` / `git commit`) of every plan task
is therefore intentionally **not** executed by the agent; the working tree carries the change and the
human owner commits.
+47
View File
@@ -0,0 +1,47 @@
# Operations
## Runtime shape
```text
durable queue (PostgreSQL, FOR UPDATE SKIP LOCKED)
→ expiry check
→ suppression and eligibility re-check
→ provider health gate
→ rate limiter
→ concurrency limiter
→ provider adapter
```
Provider calls run outside every database transaction. The attempt row is committed first, so after a
crash the row is either absent (nothing was sent) or present in `DISPATCHING` (reconciliation has
something to ask about).
## Guards that exist for specific incidents
| Guard | The incident it prevents |
|---|---|
| Credential failure opens the provider route | One expired key multiplied by a queue becomes a self-inflicted outage |
| Retry budget per provider profile | A provider outage turning every queued notification into its own retry loop |
| Ambiguous attempts block automatic fallback | A push whose response was lost arriving alongside the "just in case" SMS |
| Permits released during backoff | A slow provider pinning the whole concurrency budget on work that is only waiting |
| Bounded drain on rotation | A provider that never answers holding a credential rotation open forever |
| Fail-fast intake on capacity | An unbounded in-memory queue absorbing a burst it cannot survive |
## Scheduling
`scheduleAt` activates the job, `notBefore` is the earliest permitted provider submission, and
`expiresAt` blocks new attempts, retries and fallbacks. Suppression and expiry are re-checked
immediately before dispatch, because a scheduled notification can sit in the queue for hours and the
user may have opted out in the meantime.
## Redrive
A redrive preserves `NotificationId` and `RecipientDeliveryId`, creates a new `DeliveryAttemptId`, and
reuses the pinned template version and rendered digest. Sending different content is a new
notification, not a redrive. Redriving an ambiguous attempt requires explicit duplicate-risk approval,
because the platform genuinely cannot tell whether the first submission reached the user.
## Actuator surface
Provider runtime states and generations, queue depth and age, callback and reconciliation health.
Never addresses, never credentials.
+65
View File
@@ -0,0 +1,65 @@
# Provider runbooks
## SMTP
| Symptom | Classification | Action |
|---|---|---|
| Final `2xx` after `DATA` | `CONFIRMED_ACCEPTED` / `PROVIDER_ACCEPTED` | None; this is acceptance, not inbox delivery |
| `4yz` | `TRANSIENT_PROVIDER` | Retry under budget and deadline |
| `5yz` | `PERMANENT_PROVIDER` or `INVALID_RECIPIENT` | Stop, or invalidate the contact point |
| Connection lost after `DATA` | `AMBIGUOUS_SUBMISSION` | Reconcile or escalate; do not resend automatically |
Connection, read, write and pool-acquire timeouts are all finite. There is no unbounded timeout.
## Amazon SES
`MessageId` is acceptance evidence. SES itself documents that it can accept a request and then not
send, so `MessageId` is never mapped to `DELIVERED`.
| Event | Normalized |
|---|---|
| `Send` | reinforces `PROVIDER_ACCEPTED` |
| `Delivery` | `DELIVERY_CONFIRMED` / `NETWORK_OR_CARRIER_ACCEPTED` |
| `DeliveryDelay` | delay fact |
| `Bounce` (permanent) | `BOUNCED_HARD` plus hard-bounce suppression |
| `Bounce` (transient) | `BOUNCED_SOFT`; retry policy input, not a suppression reason |
| `Complaint` | complaint fact plus suppression |
| `Reject` | `PROVIDER_REJECTED` |
| `RenderingFailure` | `TEMPLATE_FAILURE` |
## Twilio
`accepted`/`queued` is acceptance only. `sent` is carrier acceptance. `delivered` is device delivery.
Callbacks are not ordered. A `sent` arriving after `delivered` is stored and ignored by the
projection. Missing callbacks are corrected by status polling under the provider rate limit.
Signature verification uses the canonical external URL from the profile, not the URL the servlet
container reconstructed behind a proxy.
## FCM
| Error | Classification |
|---|---|
| `UNREGISTERED` | `INVALID_RECIPIENT`; invalidate the contact point, never retry |
| `INVALID_ARGUMENT` | `INVALID_PAYLOAD` |
| `QUOTA_EXCEEDED` | `THROTTLED`, exponential backoff |
| `UNAVAILABLE` | `TRANSIENT_PROVIDER`, honour `Retry-After`, add jitter |
| Credential failure | `AUTHENTICATION`; opens the provider route |
A batch is one transport call and many attempts. Partial results map back by input index; one
transport failure does not become one shared outcome unless the adapter can prove it.
## APNs
2xx is acceptance. Environment and topic mismatches are configuration failures, not delivery
failures. Sandbox and production tokens are separate namespaces.
## Web Push
`TTL` is mandatory by protocol. `201` is acceptance. `404` is an expired subscription per RFC 8030;
provider-documented `410` maps the same way. Payloads use `aes128gcm` per RFC 8291 and VAPID JWTs are
signed per RFC 8292 with the audience taken from the endpoint origin.
VAPID key rotation is not ordinary credential rotation: a restricted subscription may need to be
re-created, so it is a migration operation.
+56
View File
@@ -0,0 +1,56 @@
# Security and privacy
## Protected values
Email addresses, phone numbers, FCM installation ids and legacy tokens, APNs device tokens, Web Push
endpoints and keys, VAPID private keys, provider credentials, callback signing secrets, template
variables, rendered bodies, attachment references and unsubscribe tokens.
## At rest
Contact points are encrypted with AES-256-GCM. Equality lookup uses a separate HMAC-SHA-256
fingerprint.
Two keys, not one, because the requirements are opposite: the ciphertext must be non-deterministic so
two records of the same address are not visibly identical, while equality lookup must be
deterministic. The fingerprint is keyed rather than a plain digest because phone numbers and email
addresses come from a small, enumerable space — an unkeyed hash of a phone number is recoverable in
seconds.
The contact point kind is bound into the GCM associated data, so a ciphertext cannot be moved between
contact kinds without failing the authentication tag.
An unknown key id is refused rather than silently falling back to the current key: a silent fallback
would turn every historical row into a tag failure at read time.
## Never logged, never a metric tag
Addresses, tokens, Web Push endpoints and keys, message bodies, template variables, provider
credentials, unsubscribe tokens, attachment URLs, raw callback payloads and raw provider request ids.
Two mechanisms enforce this rather than convention:
- `CardinalityGuard` validates every metric tag against a closed allowlist.
- `SafeDiagnosticContext` rejects any structured-diagnostic field outside its allowlist.
An allowlist rather than a denylist, because the failure mode of a denylist is that the one field
nobody thought of is the one that leaks.
Every contact point value type overrides `toString()` to print `[redacted]`. That covers the case a
central redactor cannot: a value interpolated into a log line by accident.
## Web Push endpoints
RFC 8030 defines the push URI as a capability URL — knowing it is sufficient to push to the
subscriber. It is handled as a secret, not as a URL.
## Callbacks
TLS, provider signature verification over the exact received bytes and external URL, replay defence
where a timestamp or nonce is available, body-size and content-type limits, profile binding, rate
limiting, idempotent ingestion and a security audit trail for rejections.
## Tenant isolation
Every store port carries the tenant boundary in its signature. Administrative operations require an
explicit tenant or a global authority.
+68
View File
@@ -0,0 +1,68 @@
# Notification support matrix
What each channel can actually prove, and what the platform refuses to claim.
## Channels
| Channel | Reference implementation | Grade | Strongest evidence the platform records by default |
|---|---|---|---|
| Email | SMTP, Amazon SES API | Stable | Provider acceptance; recipient mail-server delivery, bounce and complaint when the provider publishes events |
| SMS | Twilio Programmable Messaging | Stable | `accepted`/`queued`, `sent`, and carrier-DLR `delivered`/`undelivered` |
| Mobile push (Android and cross-platform) | FCM, FID-first with legacy registration token compatibility | Stable | FCM acceptance and explicit failures |
| Mobile push (Apple) | APNs HTTP/2 provider API | Stable | APNs acceptance |
| Web Push | RFC 8030, RFC 8291, RFC 8292 | Stable | Push-service acceptance; user-agent acknowledgement only where the service offers receipts |
| In-app inbox | Own database | Optional stable | `PERSISTED`, `SEEN`, `READ` |
| Webhook | HTTP client platform | Extension | Whatever the receiving HTTP contract states |
## Evidence levels
`NONE``PLATFORM_QUEUED``PROVIDER_ACCEPTED``NETWORK_OR_CARRIER_ACCEPTED`
`DEVICE_DELIVERED``USER_AGENT_DISPLAYED``USER_READ`
| Provider signal | Highest evidence it may produce |
|---|---|
| Internal queue commit | `PLATFORM_QUEUED` |
| SES `MessageId` | `PROVIDER_ACCEPTED` |
| SES `Delivery` | `NETWORK_OR_CARRIER_ACCEPTED` |
| Twilio `accepted` / `queued` | `PROVIDER_ACCEPTED` |
| Twilio `sent` | `NETWORK_OR_CARRIER_ACCEPTED` |
| Twilio `delivered` | `DEVICE_DELIVERED` |
| FCM send success | `PROVIDER_ACCEPTED` |
| APNs 2xx | `PROVIDER_ACCEPTED` |
| Web Push `201` | `PROVIDER_ACCEPTED` |
| Web Push receipt capability | `DEVICE_DELIVERED` |
| In-app row commit | `PROVIDER_ACCEPTED` |
| In-app `seen` endpoint | `USER_AGENT_DISPLAYED` |
| In-app `read` endpoint, authenticated app receipt | `USER_READ` |
Promotions the platform will not make, in code or in configuration:
- FCM send success is not `DEVICE_DELIVERED`.
- An APNs 2xx is not `DELIVERED`.
- An SES `MessageId` is not `DELIVERED`.
- An SMTP `250` is not inbox delivery.
## Submission outcomes
`NOT_SUBMITTED`, `CONFIRMED_ACCEPTED`, `CONFIRMED_REJECTED`, `AMBIGUOUS`.
`AMBIGUOUS` is a first-class stored state, not an error path. It means the request body was committed
to the provider and the outcome could not be read. While an ambiguous attempt exists on a recipient
delivery, automatic retry and automatic cross-channel fallback are both blocked.
## Not supported
The platform will not claim any of the following, because no channel above can support them:
- guaranteed delivery
- guaranteed read
- exactly-once human notification
- unconditional multi-provider failover after an unread response
- provider SDK types in the public API
- audience selection, campaign segmentation or jurisdiction rulings
## Target model
`FCM_FID` is the primary mobile push target. `FCM_REGISTRATION_TOKEN_LEGACY` and
`APNS_DEVICE_TOKEN` are separate types with separate lifecycles; they are never flattened into one
string field.
@@ -0,0 +1,94 @@
# Toxiproxy fault injection for the notification delivery platform.
#
# Scope, stated up front: this is the *nightly and release* fault suite, not the PR gate. The PR
# suite runs against a loopback socket harness in-process — deterministic, no Docker, no provider
# sandbox — because a gate that needs infrastructure is a gate people learn to skip. What lives
# here are the faults that harness cannot produce: real TCP behaviour under latency, bandwidth
# starvation, and connection resets at a point the JVM's own socket layer decides.
#
# Usage:
# docker compose -f infra/notification/toxiproxy/docker-compose.yml up -d
# ./gradlew :adapter:outbound:notification:test -Dnotification.faultProxy=http://127.0.0.1:8474
#
# The proxies below front *stub* upstreams, never a provider's real API. Pointing a toxic proxy at
# a live provider sends real notifications to real people from a test run, and adds a rate-limit
# incident on an account the team shares.
services:
toxiproxy:
image: ghcr.io/shopify/toxiproxy:2.11.0
container_name: notification-toxiproxy
ports:
- "8474:8474" # control API
- "18081:18081" # -> ses-stub
- "18082:18082" # -> twilio-stub
- "18083:18083" # -> push-stub (APNs / FCM / Web Push)
networks: [notification-fault]
healthcheck:
test: ["CMD", "/toxiproxy-cli", "list"]
interval: 5s
timeout: 3s
retries: 10
# Deterministic upstreams. Each returns the provider's success shape and nothing else; the
# interesting behaviour is injected by the proxy in front of it, not by the stub.
ses-stub:
image: mendhak/http-https-echo:35
environment:
HTTP_PORT: "8080"
networks: [notification-fault]
twilio-stub:
image: mendhak/http-https-echo:35
environment:
HTTP_PORT: "8080"
networks: [notification-fault]
push-stub:
image: mendhak/http-https-echo:35
environment:
HTTP_PORT: "8080"
networks: [notification-fault]
# Creates the proxies and the toxics once the control API is up. Kept as a job rather than a
# README step so the topology is reproducible and reviewable rather than typed from memory.
provision:
image: ghcr.io/shopify/toxiproxy:2.11.0
depends_on:
toxiproxy:
condition: service_healthy
networks: [notification-fault]
entrypoint:
- /bin/sh
- -c
- |
set -e
CLI="/toxiproxy-cli -h toxiproxy:8474"
$$CLI create -l 0.0.0.0:18081 -u ses-stub:8080 ses
$$CLI create -l 0.0.0.0:18082 -u twilio-stub:8080 twilio
$$CLI create -l 0.0.0.0:18083 -u push-stub:8080 push
# Response loss after the request was committed: the provider received and acted on the
# message, and the answer never came back. This is the AMBIGUOUS case, and it is the one
# fault no provider's documentation describes.
$$CLI toxic add -t timeout -a timeout=0 -n response_loss --downstream --toxicity 0 ses
$$CLI toxic add -t timeout -a timeout=0 -n response_loss --downstream --toxicity 0 twilio
$$CLI toxic add -t timeout -a timeout=0 -n response_loss --downstream --toxicity 0 push
# Latency past the adapter's own timeout, to prove the timeout is the adapter's decision
# rather than the socket's.
$$CLI toxic add -t latency -a latency=8000 -n slow --toxicity 0 ses
$$CLI toxic add -t latency -a latency=8000 -n slow --toxicity 0 twilio
$$CLI toxic add -t latency -a latency=8000 -n slow --toxicity 0 push
# Partial write: the connection dies mid-body. Distinct from response loss, because the
# provider never got a complete request and the attempt is genuinely retryable.
$$CLI toxic add -t limit_data -a bytes=64 -n partial_write --upstream --toxicity 0 ses
$$CLI toxic add -t limit_data -a bytes=64 -n partial_write --upstream --toxicity 0 twilio
$$CLI toxic add -t limit_data -a bytes=64 -n partial_write --upstream --toxicity 0 push
echo "proxies ready; toxics are registered at toxicity=0 and enabled per test"
$$CLI list
networks:
notification-fault:
driver: bridge
+141
View File
@@ -0,0 +1,141 @@
#!/usr/bin/env bash
#
# The MongoDB Advanced capability gate (advanced plan Task 15).
#
# Advanced capabilities are opt-in modules. This script verifies the contracts that can be verified
# without provider infrastructure, and then reports -- explicitly -- which promotion evidence it
# could NOT produce.
#
# Required promotion categories (MongoAdvancedPromotionEvidence.REQUIRED):
#
# stable-platform, actual-topology, security, migration, failure, runbook
#
# `actual-topology` is the one that cannot be substituted. A container gives a functional pass for
# sharding, search, vector and encryption while exercising none of the behaviour that makes them
# Advanced rather than Stable: real shard distribution, a real analyzer, a real KMS. Atlas Local is
# a pull-request convenience and is not release evidence -- see
# MongoAtlasCapabilityContractSuite.Environment.
#
# Usage:
# bash scripts/verify-mongodb-advanced.sh
# MONGODB_DOCKER=1 bash scripts/verify-mongodb-advanced.sh
# MONGODB_SHARDED_URI=... MONGODB_ATLAS_URI=... MONGODB_KMS=... bash scripts/verify-mongodb-advanced.sh
#
set -euo pipefail
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
GRADLE_DIR="${REPO_ROOT}/src"
MODULE=':adapter:outbound:persistence-mongo'
GRADLE=(./gradlew --console=plain)
FAILED=()
MISSING_EVIDENCE=()
echo "MongoDB Advanced capability gate"
echo "repository: ${REPO_ROOT}"
# --- stable-platform -------------------------------------------------------------------------
# An Advanced capability cannot be promoted over a Stable platform that does not itself pass.
echo ""
echo "=== [stable-platform] Stable gate"
if bash "${REPO_ROOT}/scripts/verify-mongodb-platform.sh"; then
echo "stable-platform: supplied"
else
status=$?
if (( status == 2 )); then
echo "stable-platform: INCOMPLETE (the Stable gate skipped lanes)"
MISSING_EVIDENCE+=("stable-platform (Stable gate incomplete)")
else
FAILED+=("stable-platform")
fi
fi
# --- failure + runbook (hermetic) -------------------------------------------------------------
# Every Advanced refusal contract: disabled capability refuses construction, CSFLE/QE cannot share a
# collection, QE substring/prefix/suffix unsupported on 8.0, a non-READY search index cannot serve,
# undeclared scatter-gather is rejected, a dimension mismatch is refused.
echo ""
echo "=== [failure] Advanced contract tests"
if (cd "${GRADLE_DIR}" && "${GRADLE[@]}" "${MODULE}:test" --tests '*advanced*'); then
echo "failure: supplied"
else
FAILED+=("failure")
fi
echo ""
echo "=== [runbook] capability documentation"
for doc in sharding time-series encryption search-vector multi-tenancy gridfs-migration; do
path="${REPO_ROOT}/docs/mongodb/advanced/${doc}.md"
if [[ -f "${path}" ]]; then
echo " + ${doc}.md"
else
echo " - ${doc}.md MISSING"
FAILED+=("runbook:${doc}")
fi
done
if [[ ! -f "${REPO_ROOT}/docs/adr/ADR-MONGO-ADV-001-capability-promotion.md" ]]; then
echo " - ADR-MONGO-ADV-001 MISSING"
FAILED+=("runbook:ADR-MONGO-ADV-001")
fi
# --- actual-topology -------------------------------------------------------------------------
echo ""
echo "=== [actual-topology] provider environments"
if [[ -n "${MONGODB_SHARDED_URI:-}" ]]; then
if (cd "${GRADLE_DIR}" && "${GRADLE[@]}" "${MODULE}:test" --tests '*Shard*' \
-Dmongodb.sharded.uri="${MONGODB_SHARDED_URI}"); then
echo "actual-topology(sharded): supplied"
else
FAILED+=("actual-topology:sharded")
fi
else
echo "actual-topology(sharded): no MONGODB_SHARDED_URI"
MISSING_EVIDENCE+=("actual-topology: sharded cluster")
fi
if [[ -n "${MONGODB_ATLAS_URI:-}" ]]; then
echo "actual-topology(search/vector): MONGODB_ATLAS_URI present"
else
echo "actual-topology(search/vector): no MONGODB_ATLAS_URI"
MISSING_EVIDENCE+=("actual-topology: search/vector on the actual target deployment")
fi
if [[ -n "${MONGODB_KMS:-}" ]]; then
echo "actual-topology(encryption): MONGODB_KMS present"
else
echo "actual-topology(encryption): no MONGODB_KMS"
MISSING_EVIDENCE+=("actual-topology: real KMS and key vault")
fi
# --- security + migration ---------------------------------------------------------------------
# These are review artefacts, not test runs: a role review and a documented migration path per
# capability. The gate records that they are outstanding rather than pretending a green test covers
# them.
MISSING_EVIDENCE+=("security: per-capability privilege review sign-off")
MISSING_EVIDENCE+=("migration: per-capability migration path sign-off")
# --- Report ------------------------------------------------------------------------------------
echo ""
echo "---------------------------------------------------------------"
if (( ${#FAILED[@]} > 0 )); then
echo "ADVANCED GATE: FAILED"
for entry in "${FAILED[@]}"; do echo " - ${entry}"; done
echo "---------------------------------------------------------------"
exit 1
fi
echo "verifiable contracts: PASSED"
if (( ${#MISSING_EVIDENCE[@]} > 0 )); then
echo ""
echo "ADVANCED GATE: NOT PROMOTABLE -- missing evidence:"
for entry in "${MISSING_EVIDENCE[@]}"; do echo " ~ ${entry}"; done
echo ""
echo "A capability stays opt-in until every category in"
echo "MongoAdvancedPromotionEvidence.REQUIRED is supplied. See"
echo "docs/adr/ADR-MONGO-ADV-001-capability-promotion.md."
echo "---------------------------------------------------------------"
exit 2
fi
echo "ADVANCED GATE: PASSED"
echo "---------------------------------------------------------------"
+158
View File
@@ -0,0 +1,158 @@
#!/usr/bin/env bash
#
# The MongoDB Stable release gate (design §30, plan Task 50).
#
# Runs every lane that produces one of the Stable evidence categories:
#
# mapping, transaction, migration, change-stream, security,
# failover, performance, compatibility
#
# The gate exists because "the test suite is green" and "every category has evidence" are different
# statements. A suite passes happily with a whole lane skipped -- no Docker, a disabled tag, a
# renamed task -- and a release built on that suite has no failover or compatibility evidence at
# all, silently. Each lane below is therefore run by name, and a skipped lane is reported as skipped
# rather than counted as passed.
#
# Advanced capabilities are NOT promoted or transitively included here. See
# scripts/verify-mongodb-advanced.sh.
#
# Usage:
# bash scripts/verify-mongodb-platform.sh # hermetic lanes only
# MONGODB_DOCKER=1 bash scripts/verify-mongodb-platform.sh # + container lanes
#
set -euo pipefail
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
GRADLE_DIR="${REPO_ROOT}/src"
MODULE=':adapter:outbound:persistence-mongo'
GRADLE=(./gradlew --console=plain)
RESULTS_DIR="${GRADLE_DIR}/adapter/outbound/persistence-mongo/build/test-results"
RAN=()
SKIPPED=()
FAILED=()
# Counts the tests a lane actually executed, from its JUnit XML.
#
# A lane whose filter matches nothing passes: Gradle runs the task, discovers no tests, and reports
# success. That is the failure mode this whole gate exists to prevent -- an empty lane is not
# evidence, it is the absence of evidence wearing a green tick. Any lane that reports zero executed
# tests is treated as a failure.
executed_tests() {
local task="$1"
local dir="${RESULTS_DIR}/${task}"
[[ -d "${dir}" ]] || { echo 0; return; }
local total=0
shopt -s nullglob
for xml in "${dir}"/*.xml; do
local count
count=$(sed -n 's/.*<testsuite[^>]* tests="\([0-9]*\)".*/\1/p' "${xml}" | head -1)
total=$(( total + ${count:-0} ))
done
shopt -u nullglob
echo "${total}"
}
run_lane() {
local category="$1"
local task="$2"
shift 2
echo ""
echo "=== [${category}] ${task}"
if ! (cd "${GRADLE_DIR}" && "${GRADLE[@]}" "${MODULE}:${task}" "$@"); then
FAILED+=("${category}:${task}")
return
fi
# `check` aggregates several tasks and has no results directory of its own.
if [[ "${task}" == "check" ]]; then
RAN+=("${category}:${task}")
return
fi
local executed
executed=$(executed_tests "${task}")
if (( executed == 0 )); then
echo "!!! ${task} passed without executing a single test — the lane's filter matches nothing,"
echo "!!! so the '${category}' evidence category is empty."
FAILED+=("${category}:${task} (0 tests executed)")
else
RAN+=("${category}:${task} (${executed} tests)")
fi
}
skip_lane() {
local category="$1"
local task="$2"
local reason="$3"
echo ""
echo "=== [${category}] ${task} -- SKIPPED (${reason})"
SKIPPED+=("${category}:${task} (${reason})")
}
docker_available() {
[[ "${MONGODB_DOCKER:-0}" == "1" ]] && command -v docker >/dev/null 2>&1 && docker info >/dev/null 2>&1
}
echo "MongoDB Stable release gate"
echo "repository: ${REPO_ROOT}"
# --- Always-on lanes -------------------------------------------------------------------------
# Static analysis, architecture boundaries, unit and hermetic contract tests. These produce the
# mapping, transaction, migration, change-stream and security evidence that does not need a server.
run_lane "static-analysis" "check" -x "mongoStableContractTest"
run_lane "mapping+transaction+migration+change-stream+security" "mongoStableContractTest"
# --- Container lanes -------------------------------------------------------------------------
# A lane that needs Docker inside `check` teaches people to skip `check`, so these are opt-in --
# but opting out is recorded, not silent.
if docker_available; then
run_lane "compatibility" "mongoCompatibilityTest"
run_lane "migration" "mongoMigrationTest"
run_lane "security" "mongoSecurityIntegrationTest"
run_lane "failover" "mongoReplicaSetTest"
run_lane "failover" "mongoFailoverTest"
run_lane "performance" "mongoPerformanceTest"
else
reason="MONGODB_DOCKER!=1 or Docker unavailable"
skip_lane "compatibility" "mongoCompatibilityTest" "${reason}"
skip_lane "migration" "mongoMigrationTest" "${reason}"
skip_lane "security" "mongoSecurityIntegrationTest" "${reason}"
skip_lane "failover" "mongoReplicaSetTest" "${reason}"
skip_lane "failover" "mongoFailoverTest" "${reason}"
skip_lane "performance" "mongoPerformanceTest" "${reason}"
fi
# --- Architecture-wide gates -----------------------------------------------------------------
echo ""
echo "=== [architecture] repository-wide verification"
if (cd "${GRADLE_DIR}" \
&& "${GRADLE[@]}" verifyCleanArchitectureDependencies \
&& "${GRADLE[@]}" :app-bootstrap:test --tests '*CleanArchitectureTest'); then
RAN+=("architecture:repository-wide")
else
FAILED+=("architecture:repository-wide")
fi
# --- Report ------------------------------------------------------------------------------------
echo ""
echo "---------------------------------------------------------------"
echo "ran: ${#RAN[@]}"
for entry in "${RAN[@]:-}"; do [[ -n "${entry}" ]] && echo " + ${entry}"; done
echo "skipped: ${#SKIPPED[@]}"
for entry in "${SKIPPED[@]:-}"; do [[ -n "${entry}" ]] && echo " ~ ${entry}"; done
echo "failed: ${#FAILED[@]}"
for entry in "${FAILED[@]:-}"; do [[ -n "${entry}" ]] && echo " - ${entry}"; done
echo "---------------------------------------------------------------"
if (( ${#FAILED[@]} > 0 )); then
echo "STABLE GATE: FAILED"
exit 1
fi
if (( ${#SKIPPED[@]} > 0 )); then
echo "STABLE GATE: INCOMPLETE -- lanes above were not run, so their evidence categories are absent."
echo "A release requires every category. Re-run with MONGODB_DOCKER=1 on a host with Docker."
exit 2
fi
echo "STABLE GATE: PASSED -- every evidence category produced."
@@ -0,0 +1,38 @@
package dev.caskeleton.adapter.inbound.web.notification.platform.callback;
import org.springframework.boot.autoconfigure.condition.ConditionalOnProperty;
import org.springframework.context.annotation.Bean;
import org.springframework.context.annotation.Configuration;
import org.springframework.core.Ordered;
import org.springframework.core.annotation.Order;
import org.springframework.security.config.annotation.web.builders.HttpSecurity;
import org.springframework.security.config.http.SessionCreationPolicy;
import org.springframework.security.web.SecurityFilterChain;
/**
* Security chain for the provider callback endpoints.
*
* <p>Callbacks authenticate with a provider signature, not with a user session, so they get their
* own chain: CSRF and session creation are off, and the ordinary user chain never sees them.
* Putting them on the user chain would either break every provider or force the user chain to be
* permissive.
*/
@Configuration(proxyBeanMethods = false)
@ConditionalOnProperty(
prefix = "ca-skeleton.notification.platform.callbacks",
name = "enabled",
havingValue = "true")
public class CallbackMvcSecurityConfiguration {
/** Dedicated, ordered-first chain for the callback path. */
@Bean
@Order(Ordered.HIGHEST_PRECEDENCE + 10)
public SecurityFilterChain notificationCallbackFilterChain(HttpSecurity http) throws Exception {
return http.securityMatcher("/internal/notification/callbacks/**")
.csrf(csrf -> csrf.disable())
.sessionManagement(
session -> session.sessionCreationPolicy(SessionCreationPolicy.STATELESS))
.authorizeHttpRequests(requests -> requests.anyRequest().permitAll())
.build();
}
}
@@ -0,0 +1,75 @@
package dev.caskeleton.adapter.inbound.web.notification.platform.callback;
import dev.caskeleton.application.notification.platform.api.ProviderId;
import dev.caskeleton.application.notification.platform.api.ProviderProfileId;
import dev.caskeleton.application.notification.platform.callback.CallbackRequest;
import jakarta.servlet.http.HttpServletRequest;
import java.time.Clock;
import java.util.ArrayList;
import java.util.Collections;
import java.util.LinkedHashMap;
import java.util.List;
import java.util.Map;
import java.util.Objects;
import java.util.Optional;
/**
* Builds the transport-neutral callback request.
*
* <p>Both the servlet and reactive endpoints use this, so signature verification sees exactly the
* same canonical bytes and URL regardless of which stack received the call.
*/
public final class CallbackRequestFactory {
private final ExternalRequestUrlResolver urlResolver;
private final Clock clock;
public CallbackRequestFactory(ExternalRequestUrlResolver urlResolver, Clock clock) {
this.urlResolver = Objects.requireNonNull(urlResolver, "urlResolver");
this.clock = Objects.requireNonNull(clock, "clock");
}
/** Build from a servlet request plus the already-read raw body. */
public CallbackRequest create(
String provider, String profile, HttpServletRequest request, byte[] body) {
Objects.requireNonNull(provider, "provider");
Objects.requireNonNull(profile, "profile");
Objects.requireNonNull(request, "request");
Objects.requireNonNull(body, "body");
Map<String, List<String>> headers = new LinkedHashMap<>();
for (String name : Collections.list(request.getHeaderNames())) {
headers.put(name, new ArrayList<>(Collections.list(request.getHeaders(name))));
}
return new CallbackRequest(
new ProviderId(provider),
new ProviderProfileId(profile),
urlResolver.resolve(request),
request.getMethod(),
Optional.ofNullable(request.getContentType()),
headers,
body,
clock.instant());
}
/** Build from an already-resolved external URL, used by the reactive endpoint. */
public CallbackRequest create(
String provider,
String profile,
String externalUrl,
String method,
Optional<String> contentType,
Map<String, List<String>> headers,
byte[] body) {
return new CallbackRequest(
new ProviderId(provider),
new ProviderProfileId(profile),
externalUrl,
method,
contentType,
headers,
body,
clock.instant());
}
}
@@ -0,0 +1,71 @@
package dev.caskeleton.adapter.inbound.web.notification.platform.callback;
import jakarta.servlet.http.HttpServletRequest;
import java.util.Locale;
import java.util.Objects;
import java.util.Set;
/**
* Reconstructs the URL the provider actually called.
*
* <p>Several providers sign the request URL, so getting this wrong turns every valid webhook into a
* signature failure. Forwarded headers are only honoured when the immediate peer is a configured
* trusted proxy: trusting them unconditionally would let any caller choose the URL that gets
* verified, which defeats the signature entirely.
*/
public final class ExternalRequestUrlResolver {
private final Set<String> trustedProxies;
public ExternalRequestUrlResolver(Set<String> trustedProxies) {
this.trustedProxies = Set.copyOf(Objects.requireNonNull(trustedProxies, "trustedProxies"));
}
/** External URL of a request. */
public String resolve(HttpServletRequest request) {
Objects.requireNonNull(request, "request");
String scheme = request.getScheme();
String host = request.getServerName();
int port = request.getServerPort();
if (trustedProxies.contains(request.getRemoteAddr())) {
String forwarded = request.getHeader("Forwarded");
if (forwarded != null) {
for (String element : forwarded.split(";", -1)) {
String trimmed = element.trim().toLowerCase(Locale.ROOT);
if (trimmed.startsWith("proto=")) {
scheme = trimmed.substring("proto=".length());
} else if (trimmed.startsWith("host=")) {
host = element.trim().substring("host=".length());
port = -1;
}
}
} else {
String protoHeader = request.getHeader("X-Forwarded-Proto");
String hostHeader = request.getHeader("X-Forwarded-Host");
if (protoHeader != null) {
scheme = protoHeader;
}
if (hostHeader != null) {
host = hostHeader;
port = -1;
}
}
}
StringBuilder url = new StringBuilder(scheme).append("://").append(host);
boolean defaultPort =
port < 0
|| ("https".equalsIgnoreCase(scheme) && port == 443)
|| ("http".equalsIgnoreCase(scheme) && port == 80);
if (!defaultPort) {
url.append(':').append(port);
}
url.append(request.getRequestURI());
String query = request.getQueryString();
if (query != null && !query.isBlank()) {
url.append('?').append(query);
}
return url.toString();
}
}
@@ -0,0 +1,75 @@
package dev.caskeleton.adapter.inbound.web.notification.platform.callback;
import dev.caskeleton.application.notification.platform.api.error.CallbackValidationException;
import dev.caskeleton.application.notification.platform.callback.ProviderCallbackIngestionService;
import jakarta.servlet.http.HttpServletRequest;
import java.util.Objects;
import org.springframework.boot.autoconfigure.condition.ConditionalOnProperty;
import org.springframework.boot.autoconfigure.condition.ConditionalOnWebApplication;
import org.springframework.http.HttpStatus;
import org.springframework.http.ResponseEntity;
import org.springframework.web.bind.annotation.ExceptionHandler;
import org.springframework.web.bind.annotation.PathVariable;
import org.springframework.web.bind.annotation.PostMapping;
import org.springframework.web.bind.annotation.RequestBody;
import org.springframework.web.bind.annotation.RequestMapping;
import org.springframework.web.bind.annotation.RestController;
/**
* Servlet callback endpoint.
*
* <p>The body arrives as raw bytes, never as a parsed form. Providers sign the exact octets, and
* letting the container parse and re-encode them is the most common cause of a valid webhook
* failing verification.
*
* <p>The response is a bare {@code 204}: no body, no diagnostics. A provider only needs to know the
* event is recorded, and an error body would be a channel for leaking what the platform knows.
*
* <p>Registered only in a servlet application and only when callbacks are enabled. An annotated
* controller is also honoured by WebFlux, so without the servlet condition a reactive deployment
* would map both this and the functional router onto the same path — and a provider signature would
* then be verified twice against two different canonical URLs.
*/
@RestController
@ConditionalOnWebApplication(type = ConditionalOnWebApplication.Type.SERVLET)
@ConditionalOnProperty(
prefix = "ca-skeleton.notification.platform.callbacks",
name = "enabled",
havingValue = "true")
@RequestMapping("/internal/notification/callbacks")
public final class NotificationCallbackMvcController {
/** Hard body ceiling applied before any provider adapter is consulted. */
public static final int MAX_BODY_BYTES = 65_536;
private final ProviderCallbackIngestionService ingestion;
private final CallbackRequestFactory requestFactory;
public NotificationCallbackMvcController(
ProviderCallbackIngestionService ingestion, CallbackRequestFactory requestFactory) {
this.ingestion = Objects.requireNonNull(ingestion, "ingestion");
this.requestFactory = Objects.requireNonNull(requestFactory, "requestFactory");
}
/** Receive one provider callback. */
@PostMapping(path = "/{provider}/{profile}")
public ResponseEntity<Void> callback(
@PathVariable String provider,
@PathVariable String profile,
HttpServletRequest request,
@RequestBody byte[] body) {
if (body.length > MAX_BODY_BYTES) {
return ResponseEntity.status(HttpStatus.CONTENT_TOO_LARGE).build();
}
// A duplicate answers 204 exactly like a first delivery. The provider did its job either way,
// and any other status would make it retry an event that is already recorded.
ingestion.ingest(requestFactory.create(provider, profile, request, body));
return ResponseEntity.noContent().build();
}
/** A rejected callback never reveals why beyond the status code. */
@ExceptionHandler(CallbackValidationException.class)
public ResponseEntity<Void> onValidationFailure(CallbackValidationException failure) {
return ResponseEntity.status(HttpStatus.BAD_REQUEST).build();
}
}
@@ -0,0 +1,48 @@
package dev.caskeleton.adapter.inbound.web.notification.platform.callback.reactive;
import java.util.Objects;
import org.springframework.core.io.buffer.DataBuffer;
import org.springframework.core.io.buffer.DataBufferUtils;
import org.springframework.web.reactive.function.server.ServerRequest;
import reactor.core.publisher.Mono;
/**
* Reads the raw body with a hard ceiling and no buffer leaks.
*
* <p>Every {@link DataBuffer} is released on success, on error and on cancellation. A reactive
* endpoint that forgets the cancellation path leaks native memory exactly when it is under the load
* that caused the cancellation.
*/
public final class BoundedCallbackBodyReader {
private final int maxBytes;
public BoundedCallbackBodyReader(int maxBytes) {
if (maxBytes < 1) {
throw new IllegalArgumentException("maxBytes");
}
this.maxBytes = maxBytes;
}
/** Read at most the configured number of bytes. */
public Mono<byte[]> read(ServerRequest request) {
Objects.requireNonNull(request, "request");
return DataBufferUtils.join(request.bodyToFlux(DataBuffer.class), maxBytes)
.map(
buffer -> {
try {
byte[] bytes = new byte[buffer.readableByteCount()];
buffer.read(bytes);
return bytes;
} finally {
DataBufferUtils.release(buffer);
}
})
.defaultIfEmpty(new byte[0]);
}
/** Configured ceiling. */
public int maxBytes() {
return maxBytes;
}
}
@@ -0,0 +1,58 @@
package dev.caskeleton.adapter.inbound.web.notification.platform.callback.reactive;
import dev.caskeleton.adapter.inbound.web.notification.platform.callback.CallbackRequestFactory;
import dev.caskeleton.application.notification.platform.callback.ProviderCallbackIngestionService;
import org.springframework.beans.factory.annotation.Value;
import org.springframework.boot.autoconfigure.condition.ConditionalOnMissingBean;
import org.springframework.boot.autoconfigure.condition.ConditionalOnProperty;
import org.springframework.boot.autoconfigure.condition.ConditionalOnWebApplication;
import org.springframework.context.annotation.Bean;
import org.springframework.context.annotation.Configuration;
import org.springframework.web.reactive.function.server.RouterFunction;
import org.springframework.web.reactive.function.server.ServerResponse;
/**
* Registers the reactive callback transport, and only it.
*
* <p>This configuration is {@code REACTIVE}-only and the servlet controller carries the matching
* {@code SERVLET} condition, so exactly one of the two is ever registered — by construction rather
* than by convention. Both on the same path would mean a provider signature is verified twice
* against two different canonical URLs, a failure that shows up only in production and only for
* signed providers, and reads like a credential problem.
*
* <p>The body ceiling is read as a property rather than through the platform settings type: that
* type belongs to the outbound notification adapter, which this inbound adapter must not depend on.
*/
@Configuration(proxyBeanMethods = false)
@ConditionalOnWebApplication(type = ConditionalOnWebApplication.Type.REACTIVE)
@ConditionalOnProperty(
prefix = "ca-skeleton.notification.platform.callbacks",
name = "enabled",
havingValue = "true")
public class CallbackWebFluxConfiguration {
/** Bounded body reader; the ceiling applies before any provider adapter is consulted. */
@Bean
@ConditionalOnMissingBean
public BoundedCallbackBodyReader notificationCallbackBodyReader(
@Value("${ca-skeleton.notification.platform.callbacks.max-body-bytes:65536}") int maxBytes) {
return new BoundedCallbackBodyReader(maxBytes);
}
/** Reactive handler. */
@Bean
@ConditionalOnMissingBean
public NotificationCallbackWebFluxHandler notificationCallbackWebFluxHandler(
ProviderCallbackIngestionService ingestion,
CallbackRequestFactory requestFactory,
BoundedCallbackBodyReader bodyReader) {
return new NotificationCallbackWebFluxHandler(ingestion, requestFactory, bodyReader);
}
/** Functional route for the callback path. */
@Bean
public RouterFunction<ServerResponse> notificationCallbackRoutes(
NotificationCallbackWebFluxHandler handler) {
return new CallbackWebFluxRouter(handler).routes();
}
}
@@ -0,0 +1,30 @@
package dev.caskeleton.adapter.inbound.web.notification.platform.callback.reactive;
import java.util.Objects;
import org.springframework.web.reactive.function.server.RequestPredicates;
import org.springframework.web.reactive.function.server.RouterFunction;
import org.springframework.web.reactive.function.server.RouterFunctions;
import org.springframework.web.reactive.function.server.ServerResponse;
/**
* Routes the reactive callback path.
*
* <p>Kept separate from the servlet controller so that only one of the two is ever registered; two
* endpoints on the same path would mean a provider's signature is verified twice against two
* different canonical URLs.
*/
public final class CallbackWebFluxRouter {
private final NotificationCallbackWebFluxHandler handler;
public CallbackWebFluxRouter(NotificationCallbackWebFluxHandler handler) {
this.handler = Objects.requireNonNull(handler, "handler");
}
/** Router function for the callback path. */
public RouterFunction<ServerResponse> routes() {
return RouterFunctions.route(
RequestPredicates.POST("/internal/notification/callbacks/{provider}/{profile}"),
handler::handle);
}
}
@@ -0,0 +1,85 @@
package dev.caskeleton.adapter.inbound.web.notification.platform.callback.reactive;
import dev.caskeleton.adapter.inbound.web.notification.platform.callback.CallbackRequestFactory;
import dev.caskeleton.application.notification.platform.api.error.CallbackValidationException;
import dev.caskeleton.application.notification.platform.callback.ProviderCallbackIngestionService;
import java.util.List;
import java.util.Map;
import java.util.Objects;
import java.util.Optional;
import org.springframework.core.io.buffer.DataBufferLimitException;
import org.springframework.http.HttpStatus;
import org.springframework.web.reactive.function.server.ServerRequest;
import org.springframework.web.reactive.function.server.ServerResponse;
import reactor.core.publisher.Mono;
import reactor.core.scheduler.Schedulers;
/**
* Reactive callback endpoint.
*
* <p>Ingestion is blocking — it writes to the database — so it runs on {@code boundedElastic} and
* never on the event loop. Running it inline would stall every other connection the loop is
* serving.
*
* <p>It shares the canonicalisation and the ingestion service with the servlet endpoint, so a
* deployment can switch web stacks without changing what a provider signature is checked against.
*/
public final class NotificationCallbackWebFluxHandler {
private final ProviderCallbackIngestionService ingestion;
private final CallbackRequestFactory requestFactory;
private final BoundedCallbackBodyReader bodyReader;
public NotificationCallbackWebFluxHandler(
ProviderCallbackIngestionService ingestion,
CallbackRequestFactory requestFactory,
BoundedCallbackBodyReader bodyReader) {
this.ingestion = Objects.requireNonNull(ingestion, "ingestion");
this.requestFactory = Objects.requireNonNull(requestFactory, "requestFactory");
this.bodyReader = Objects.requireNonNull(bodyReader, "bodyReader");
}
/** Handle one callback. */
public Mono<ServerResponse> handle(ServerRequest request) {
String provider = request.pathVariable("provider");
String profile = request.pathVariable("profile");
return bodyReader
.read(request)
.flatMap(
body ->
Mono.fromCallable(
() ->
ingestion.ingest(
requestFactory.create(
provider,
profile,
request.uri().toString(),
request.method().name(),
request.headers().contentType().map(Object::toString),
headers(request),
body)))
.subscribeOn(Schedulers.boundedElastic()))
.then(ServerResponse.noContent().build())
.onErrorResume(
DataBufferLimitException.class,
failure -> ServerResponse.status(HttpStatus.CONTENT_TOO_LARGE).build())
.onErrorResume(
CallbackValidationException.class,
failure -> ServerResponse.status(HttpStatus.BAD_REQUEST).build());
}
private static Map<String, List<String>> headers(ServerRequest request) {
Map<String, List<String>> headers = new java.util.LinkedHashMap<>();
request
.headers()
.asHttpHeaders()
.forEach((name, values) -> headers.put(name, List.copyOf(values)));
return Map.copyOf(headers);
}
/** Content type of a request, if declared. */
public static Optional<String> contentType(ServerRequest request) {
return request.headers().contentType().map(Object::toString);
}
}
@@ -0,0 +1,396 @@
package dev.caskeleton.adapter.inbound.web.notification.platform.callback;
import static org.assertj.core.api.Assertions.assertThat;
import static org.assertj.core.api.Assertions.assertThatThrownBy;
import dev.caskeleton.application.notification.platform.api.DeliveryAttemptId;
import dev.caskeleton.application.notification.platform.api.ProviderId;
import dev.caskeleton.application.notification.platform.api.ProviderProfileId;
import dev.caskeleton.application.notification.platform.api.error.CallbackValidationException;
import dev.caskeleton.application.notification.platform.api.error.FailureCategory;
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureCode;
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureDescriptor;
import dev.caskeleton.application.notification.platform.callback.AppendEventResult;
import dev.caskeleton.application.notification.platform.callback.CallbackLimits;
import dev.caskeleton.application.notification.platform.callback.CallbackRequest;
import dev.caskeleton.application.notification.platform.callback.CallbackVerificationResult;
import dev.caskeleton.application.notification.platform.callback.NormalizedProviderEvent;
import dev.caskeleton.application.notification.platform.callback.ProjectionResult;
import dev.caskeleton.application.notification.platform.callback.ProviderCallbackAdapter;
import dev.caskeleton.application.notification.platform.callback.ProviderCallbackAdapterRegistry;
import dev.caskeleton.application.notification.platform.callback.ProviderCallbackIngestionService;
import dev.caskeleton.application.notification.platform.callback.ProviderEventLedger;
import dev.caskeleton.application.notification.platform.callback.ProviderEventProjectionService;
import dev.caskeleton.application.notification.platform.callback.ProviderEventRecord;
import dev.caskeleton.application.notification.platform.callback.ProviderEventRecordId;
import dev.caskeleton.application.notification.platform.callback.VerifiedCallback;
import dev.caskeleton.application.notification.platform.callback.VerifiedProviderEvent;
import dev.caskeleton.application.notification.platform.observation.NotificationMetricsPort;
import dev.caskeleton.application.notification.platform.observation.NotificationSecurityAuditPort;
import java.nio.charset.StandardCharsets;
import java.time.Clock;
import java.time.Duration;
import java.time.Instant;
import java.time.ZoneOffset;
import java.util.ArrayList;
import java.util.List;
import java.util.Map;
import java.util.Set;
import org.junit.jupiter.api.Test;
import org.springframework.http.HttpStatus;
import org.springframework.mock.web.MockHttpServletRequest;
/**
* What the servlet transport is responsible for handing the callback pipeline.
*
* <p>The pipeline itself belongs to application-core and is tested there. What is only testable
* here is the translation: the exact received octets, the externally-visible URL, and headers that
* survive the servlet container's own casing. Each is a common cause of a valid webhook failing
* verification, and none is visible from a unit test of the provider adapter.
*
* <p>The capture point is the provider adapter's {@code verify}, which is the first thing in the
* pipeline to see the whole request. It rejects, so the test never needs a ledger.
*/
class NotificationCallbackMvcControllerTest {
private static final Clock CLOCK =
Clock.fixed(Instant.parse("2026-08-14T00:00:00Z"), ZoneOffset.UTC);
private static final String TRUSTED_PROXY = "10.0.0.1";
private final List<CallbackRequest> verified = new ArrayList<>();
private final List<String> rejections = new ArrayList<>();
private final NotificationCallbackMvcController controller =
new NotificationCallbackMvcController(
new ProviderCallbackIngestionService(
new CapturingRegistry(),
new UnusedLedger(),
// Never reached: verification always fails in this fixture, and the pipeline appends
// only after a valid signature.
new ProviderEventProjectionService(
new UnusedLedger(),
providerId -> java.util.Optional.empty(),
new UnusedAttemptResolver(),
new UnusedProjectionStore(),
(attempt, facts) -> {
throw new UnsupportedOperationException();
},
new UnusedTransactions(),
new DiscardingMetrics()),
new UnusedPayloadProtection(),
new RecordingSecurityAudit(),
new DiscardingMetrics(),
CLOCK),
new CallbackRequestFactory(new ExternalRequestUrlResolver(Set.of(TRUSTED_PROXY)), CLOCK));
@Test
void theExactReceivedOctetsReachTheAdapterUnparsed() {
byte[] body =
"MessageSid=SM1&MessageStatus=delivered&Signed=a+b%2Fc".getBytes(StandardCharsets.UTF_8);
assertThatThrownBy(
() ->
controller.callback(
"twilio", "twilio-primary", request("application/x-www-form-urlencoded"), body))
.isInstanceOf(CallbackValidationException.class);
// Byte for byte, including the percent-encoding a form parse would have consumed and re-encoded
// differently — which is the single most common cause of a valid webhook failing its signature.
assertThat(verified).hasSize(1);
assertThat(verified.get(0).body()).isEqualTo(body);
assertThat(verified.get(0).contentType()).contains("application/x-www-form-urlencoded");
assertThat(verified.get(0).httpMethod()).isEqualTo("POST");
}
@Test
void aForwardedHostFromAnUntrustedPeerIsIgnored() {
var request = request("application/json");
request.setRemoteAddr("203.0.113.9");
request.addHeader("X-Forwarded-Proto", "https");
request.addHeader("X-Forwarded-Host", "attacker.example.com");
assertThatThrownBy(
() ->
controller.callback(
"twilio", "twilio-primary", request, "{}".getBytes(StandardCharsets.UTF_8)))
.isInstanceOf(CallbackValidationException.class);
// Honouring the header unconditionally would let any caller choose the URL that gets verified,
// which defeats the signature entirely.
assertThat(verified.get(0).externalUrl()).doesNotContain("attacker.example.com");
}
@Test
void aForwardedHostFromATrustedProxyBecomesTheCanonicalUrl() {
var request = request("application/json");
request.setRemoteAddr(TRUSTED_PROXY);
request.addHeader("X-Forwarded-Proto", "https");
request.addHeader("X-Forwarded-Host", "callback.example.com");
assertThatThrownBy(
() ->
controller.callback(
"twilio", "twilio-primary", request, "{}".getBytes(StandardCharsets.UTF_8)))
.isInstanceOf(CallbackValidationException.class);
assertThat(verified.get(0).externalUrl())
.isEqualTo(
"https://callback.example.com/internal/notification/callbacks/twilio/twilio-primary");
}
@Test
void headersSurviveTheContainersCasingAndStayAddressableEitherWay() {
var request = request("application/json");
request.addHeader("X-Twilio-Signature", "abc123");
assertThatThrownBy(
() ->
controller.callback(
"twilio", "twilio-primary", request, "{}".getBytes(StandardCharsets.UTF_8)))
.isInstanceOf(CallbackValidationException.class);
assertThat(verified.get(0).header("x-twilio-signature")).contains("abc123");
assertThat(verified.get(0).header("X-TWILIO-SIGNATURE")).contains("abc123");
}
@Test
void aBodyOverTheTransportCeilingIsRefusedBeforeAnyAdapterIsConsulted() {
byte[] oversized = new byte[NotificationCallbackMvcController.MAX_BODY_BYTES + 1];
var response =
controller.callback("twilio", "twilio-primary", request("application/json"), oversized);
assertThat(response.getStatusCode()).isEqualTo(HttpStatus.CONTENT_TOO_LARGE);
// Nothing downstream sees it, so no signature check ever runs over an attacker-sized payload.
assertThat(verified).isEmpty();
}
@Test
void aRejectedCallbackRevealsNothingBeyondTheStatusCode() {
var response =
controller.onValidationFailure(
new CallbackValidationException(
NotificationFailureDescriptor.preDispatch(
NotificationFailureCode.CALLBACK_SIGNATURE_INVALID,
FailureCategory.CALLBACK_VALIDATION_FAILURE)));
// The endpoint is unauthenticated by design — the signature is the authentication — so an error
// body is a free oracle for whoever is probing it.
assertThat(response.getStatusCode()).isEqualTo(HttpStatus.BAD_REQUEST);
assertThat(response.getBody()).isNull();
}
@Test
void aRejectedSignatureIsRecordedAsASecurityEventRatherThanADeliveryEvent() {
assertThatThrownBy(
() ->
controller.callback(
"twilio",
"twilio-primary",
request("application/json"),
"{}".getBytes(StandardCharsets.UTF_8)))
.isInstanceOf(CallbackValidationException.class);
// Writing it to the ledger would let anyone who can reach the endpoint fill a recipient's
// delivery history with noise.
assertThat(rejections).containsExactly("SIGNATURE_MISMATCH");
}
private static MockHttpServletRequest request(String contentType) {
var request =
new MockHttpServletRequest(
"POST", "/internal/notification/callbacks/twilio/twilio-primary");
request.setContentType(contentType);
return request;
}
/** Registry whose adapter records the request and then refuses it. */
private final class CapturingRegistry implements ProviderCallbackAdapterRegistry {
@Override
public ProviderCallbackAdapter require(ProviderProfileId profileId) {
return new ProviderCallbackAdapter() {
@Override
public ProviderId providerId() {
return new ProviderId("twilio");
}
@Override
public CallbackVerificationResult verify(CallbackRequest request) {
verified.add(request);
return CallbackVerificationResult.invalid("SIGNATURE_MISMATCH");
}
@Override
public List<NormalizedProviderEvent> normalize(VerifiedCallback callback) {
throw new UnsupportedOperationException("verification always fails in this fixture");
}
};
}
@Override
public CallbackLimits limitsFor(ProviderProfileId profileId) {
return new CallbackLimits(
65_536L, Set.of("application/json", "application/x-www-form-urlencoded"));
}
}
/** Security audit that keeps the rejection reason. */
private final class RecordingSecurityAudit implements NotificationSecurityAuditPort {
@Override
public void callbackSignatureRejected(ProviderProfileId profileId, String reasonCode) {
rejections.add(reasonCode);
}
@Override
public void callbackRejectedByLimit(ProviderProfileId profileId, String reasonCode) {
rejections.add(reasonCode);
}
}
/** Metrics are exercised elsewhere; discarding them keeps this test about the transport. */
private static final class DiscardingMetrics implements NotificationMetricsPort {
@Override
public void increment(String metricName, Map<String, String> tags) {
// Intentionally empty.
}
@Override
public void record(String metricName, Map<String, String> tags, Duration value) {
// Intentionally empty.
}
@Override
public void gauge(String metricName, Map<String, String> tags, double value) {
// Intentionally empty.
}
}
/** Never reached: attempt correlation happens only for an accepted callback. */
private static final class UnusedAttemptResolver
implements dev.caskeleton.application.notification.platform.callback
.DeliveryAttemptResolverPort {
@Override
public java.util.Optional<
dev.caskeleton.application.notification.platform.callback.DeliveryAttemptSnapshot>
byAttemptId(DeliveryAttemptId attemptId) {
return java.util.Optional.empty();
}
@Override
public java.util.Optional<
dev.caskeleton.application.notification.platform.callback.DeliveryAttemptSnapshot>
byProviderRequestId(ProviderProfileId profileId, String providerRequestIdHash) {
return java.util.Optional.empty();
}
}
/** Never reached: projection runs only after a signature has been accepted. */
private static final class UnusedProjectionStore
implements dev.caskeleton.application.notification.platform.callback
.DeliveryProjectionStorePort {
@Override
public dev.caskeleton.application.notification.platform.callback.DeliveryProjection load(
DeliveryAttemptId attemptId) {
throw new UnsupportedOperationException();
}
@Override
public void save(
DeliveryAttemptId attemptId,
dev.caskeleton.application.notification.platform.callback.DeliveryProjection projection) {
throw new UnsupportedOperationException();
}
}
/** Never reached: nothing in this fixture gets as far as a transaction. */
private static final class UnusedTransactions
implements dev.caskeleton.application.transaction.TransactionPort {
@Override
public <T> T inWrite(java.util.function.Supplier<T> action) {
throw new UnsupportedOperationException();
}
@Override
public <T> T inRootWrite(java.util.function.Supplier<T> action) {
throw new UnsupportedOperationException();
}
@Override
public <T> T inRead(java.util.function.Supplier<T> action) {
throw new UnsupportedOperationException();
}
@Override
public <T> T inNew(java.util.function.Supplier<T> action) {
throw new UnsupportedOperationException();
}
}
/** Never reached: every request in this fixture is rejected before the payload is retained. */
private static final class UnusedPayloadProtection
implements dev.caskeleton.application.notification.platform.callback
.CallbackPayloadProtectionPort {
@Override
public byte[] protectRawPayload(byte[] rawBody) {
throw new UnsupportedOperationException();
}
@Override
public String digest(byte[] rawBody) {
throw new UnsupportedOperationException();
}
@Override
public String fingerprint(
ProviderProfileId profileId, NormalizedProviderEvent event, String rawPayloadDigest) {
throw new UnsupportedOperationException();
}
}
/** Never reached: every request in this fixture is rejected before the append. */
private static final class UnusedLedger implements ProviderEventLedger {
@Override
public AppendEventResult append(VerifiedProviderEvent event) {
throw new UnsupportedOperationException();
}
@Override
public AppendEventResult appendAll(List<VerifiedProviderEvent> events) {
throw new UnsupportedOperationException();
}
@Override
public List<ProviderEventRecord> pendingProjection(int limit) {
throw new UnsupportedOperationException();
}
@Override
public void markApplied(ProviderEventRecordId eventId, ProjectionResult result) {
throw new UnsupportedOperationException();
}
@Override
public void markFailed(ProviderEventRecordId eventId, String errorCode) {
throw new UnsupportedOperationException();
}
@Override
public List<ProviderEventRecord> unmatched(int limit) {
throw new UnsupportedOperationException();
}
@Override
public List<ProviderEventRecord> eventsForAttempt(DeliveryAttemptId attemptId) {
throw new UnsupportedOperationException();
}
}
}
@@ -6,6 +6,34 @@ dependencies {
implementation 'org.springframework.boot:spring-boot-autoconfigure'
implementation 'org.springframework:spring-web' // Slack webhook client (RestClient)
implementation 'org.slf4j:slf4j-api'
// Notification Delivery Platform.
// - mail: the SMTP provider adapter is built on JavaMailSender/MimeMessageHelper, which is where
// multipart/alternative, inline resources and header validation already live. Rebuilding MIME
// by hand to avoid one dependency would be the more dangerous choice.
// - jackson-databind: provider payloads, callback bodies and the canonical variables payload are
// JSON. It stays inside this adapter; application-core never sees a JSON type.
// - reactor-core: only the optional Reactor facade uses it. The core async type stays
// CompletionStage, so nothing else on this classpath depends on Reactor.
implementation 'org.springframework.boot:spring-boot-starter-mail'
implementation 'org.springframework.boot:spring-boot-starter-json'
implementation 'io.projectreactor:reactor-core'
// JSON Schema 2020-12 validation of template variables, using the same validator and version the
// messaging adapter already depends on rather than a second implementation of the same spec.
// The YAML dataformat is excluded: schemas are supplied as JSON strings, so pulling a YAML
// parser onto the runtime classpath would add attack surface for a format nothing reads.
// Thymeleaf is the reference HTML renderer, added as the engine only — not the Spring
// starter, which would drag a view resolver and a servlet integration onto an outbound
// adapter that renders strings and never serves a request.
implementation 'org.thymeleaf:thymeleaf'
implementation('com.networknt:json-schema-validator:3.0.2') {
exclude group: 'tools.jackson.dataformat', module: 'jackson-dataformat-yaml'
exclude group: 'com.fasterxml.jackson.dataformat', module: 'jackson-dataformat-yaml'
}
annotationProcessor 'org.springframework.boot:spring-boot-configuration-processor'
testImplementation 'io.projectreactor:reactor-test'
}
tasks.withType(JavaCompile).configureEach { options.encoding = 'UTF-8' }
@@ -1,23 +1,24 @@
# This is a Gradle generated file for dependency locking.
# Manual edits can break the build and are not advised.
# This file is expected to be part of source control.
biz.aQute.bnd:biz.aQute.bnd.annotation:7.1.0=testCompileClasspath
ch.qos.logback:logback-classic:1.5.21=testCompileClasspath,testRuntimeClasspath
ch.qos.logback:logback-core:1.5.21=testCompileClasspath,testRuntimeClasspath
com.fasterxml.jackson.core:jackson-annotations:2.20=testCompileClasspath,testRuntimeClasspath
biz.aQute.bnd:biz.aQute.bnd.annotation:7.1.0=compileClasspath,testCompileClasspath
ch.qos.logback:logback-classic:1.5.21=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
ch.qos.logback:logback-core:1.5.21=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
com.ethlo.time:itu:1.14.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
com.fasterxml.jackson.core:jackson-annotations:2.20=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
com.github.ben-manes.caffeine:caffeine:3.2.3=annotationProcessor,testAnnotationProcessor
com.github.kevinstern:software-and-algorithms:1.0=annotationProcessor,testAnnotationProcessor
com.github.spotbugs:spotbugs-annotations:4.10.2=spotbugs
com.github.spotbugs:spotbugs-annotations:4.8.6=testCompileClasspath
com.github.spotbugs:spotbugs-annotations:4.8.6=compileClasspath,testCompileClasspath
com.github.spotbugs:spotbugs:4.10.2=spotbugs
com.github.stephenc.jcip:jcip-annotations:1.0-1=spotbugs
com.google.auto.service:auto-service-annotations:1.0.1=annotationProcessor,testAnnotationProcessor
com.google.auto.value:auto-value-annotations:1.9=annotationProcessor,testAnnotationProcessor
com.google.auto:auto-common:1.2.2=annotationProcessor,testAnnotationProcessor
com.google.code.findbugs:jsr305:3.0.2=checkstyle,spotbugs,testCompileClasspath
com.google.code.findbugs:jsr305:3.0.2=checkstyle,compileClasspath,spotbugs,testCompileClasspath
com.google.code.gson:gson:2.13.2=spotbugs
com.google.errorprone:error_prone_annotation:2.49.0=annotationProcessor,testAnnotationProcessor
com.google.errorprone:error_prone_annotations:2.38.0=testCompileClasspath
com.google.errorprone:error_prone_annotations:2.38.0=compileClasspath,testCompileClasspath
com.google.errorprone:error_prone_annotations:2.41.0=spotbugs
com.google.errorprone:error_prone_annotations:2.47.0=checkstyle
com.google.errorprone:error_prone_annotations:2.49.0=annotationProcessor,testAnnotationProcessor
@@ -32,6 +33,7 @@ com.google.j2objc:j2objc-annotations:3.1=annotationProcessor,checkstyle,testAnno
com.google.protobuf:protobuf-java:4.33.2=annotationProcessor,testAnnotationProcessor
com.h3xstream.findsecbugs:findsecbugs-plugin:1.14.0=spotbugsPlugins
com.jayway.jsonpath:json-path:2.9.0=testCompileClasspath,testRuntimeClasspath
com.networknt:json-schema-validator:3.0.2=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
com.puppycrawl.tools:checkstyle:13.5.0=checkstyle
com.vaadin.external.google:android-json:0.0.20131108.vaadin1=testCompileClasspath,testRuntimeClasspath
commons-beanutils:commons-beanutils:1.11.0=checkstyle
@@ -43,8 +45,11 @@ io.github.eisop:dataflow-errorprone:3.41.0-eisop1=annotationProcessor,testAnnota
io.github.java-diff-utils:java-diff-utils:4.12=annotationProcessor,testAnnotationProcessor
io.micrometer:micrometer-commons:1.16.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
io.micrometer:micrometer-observation:1.16.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
jakarta.activation:jakarta.activation-api:2.1.4=testCompileClasspath,testRuntimeClasspath
jakarta.annotation:jakarta.annotation-api:3.0.0=testCompileClasspath,testRuntimeClasspath
io.projectreactor:reactor-core:3.8.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
io.projectreactor:reactor-test:3.8.0=testCompileClasspath,testRuntimeClasspath
jakarta.activation:jakarta.activation-api:2.1.4=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
jakarta.annotation:jakarta.annotation-api:3.0.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
jakarta.mail:jakarta.mail-api:2.1.5=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
jakarta.xml.bind:jakarta.xml.bind-api:4.0.4=testCompileClasspath,testRuntimeClasspath
javax.inject:javax.inject:1=annotationProcessor,testAnnotationProcessor
jaxen:jaxen:2.0.0=spotbugs
@@ -53,6 +58,7 @@ net.bytebuddy:byte-buddy:1.17.8=testCompileClasspath,testRuntimeClasspath
net.minidev:accessors-smart:2.6.0=testCompileClasspath,testRuntimeClasspath
net.minidev:json-smart:2.6.0=testCompileClasspath,testRuntimeClasspath
net.sf.saxon:Saxon-HE:12.9=checkstyle,spotbugs
ognl:ognl:3.3.4=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.antlr:antlr4-runtime:4.13.2=checkstyle
org.apache.bcel:bcel:6.12.0=spotbugs
org.apache.commons:commons-lang3:3.20.0=checkstyle,spotbugs
@@ -60,9 +66,9 @@ org.apache.commons:commons-text:1.15.0=spotbugs
org.apache.commons:commons-text:1.3=checkstyle
org.apache.httpcomponents:httpclient:4.5.13=checkstyle
org.apache.httpcomponents:httpcore:4.4.16=checkstyle
org.apache.logging.log4j:log4j-api:2.25.2=spotbugs,testCompileClasspath,testRuntimeClasspath
org.apache.logging.log4j:log4j-api:2.25.2=compileClasspath,runtimeClasspath,spotbugs,testCompileClasspath,testRuntimeClasspath
org.apache.logging.log4j:log4j-core:2.25.2=spotbugs
org.apache.logging.log4j:log4j-to-slf4j:2.25.2=testCompileClasspath,testRuntimeClasspath
org.apache.logging.log4j:log4j-to-slf4j:2.25.2=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.apache.maven.doxia:doxia-core:1.12.0=checkstyle
org.apache.maven.doxia:doxia-logging-api:1.12.0=checkstyle
org.apache.maven.doxia:doxia-module-xdoc:1.12.0=checkstyle
@@ -73,14 +79,18 @@ org.apache.tomcat.embed:tomcat-embed-websocket:11.0.14=testCompileClasspath,test
org.apache.xbean:xbean-reflect:3.7=checkstyle
org.apiguardian:apiguardian-api:1.1.2=testCompileClasspath
org.assertj:assertj-core:3.27.6=testCompileClasspath,testRuntimeClasspath
org.attoparser:attoparser:2.0.7.RELEASE=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.awaitility:awaitility:4.3.0=testCompileClasspath,testRuntimeClasspath
org.codehaus.plexus:plexus-classworlds:2.6.0=checkstyle
org.codehaus.plexus:plexus-component-annotations:2.1.0=checkstyle
org.codehaus.plexus:plexus-container-default:2.1.0=checkstyle
org.codehaus.plexus:plexus-utils:3.3.0=checkstyle
org.dom4j:dom4j:2.2.0=spotbugs
org.eclipse.angus:angus-activation:2.0.3=runtimeClasspath,testRuntimeClasspath
org.eclipse.angus:angus-mail:2.0.5=runtimeClasspath,testRuntimeClasspath
org.hamcrest:hamcrest:3.0=testCompileClasspath,testRuntimeClasspath
org.javassist:javassist:3.28.0-GA=checkstyle
org.javassist:javassist:3.29.0-GA=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.jspecify:jspecify:1.0.0=annotationProcessor,checkstyle,compileClasspath,runtimeClasspath,testAnnotationProcessor,testCompileClasspath,testRuntimeClasspath
org.junit.jupiter:junit-jupiter-api:6.0.1=testCompileClasspath,testRuntimeClasspath
org.junit.jupiter:junit-jupiter-engine:6.0.1=testRuntimeClasspath
@@ -95,10 +105,10 @@ org.mockito:mockito-core:5.20.0=mockitoAgent,testCompileClasspath,testRuntimeCla
org.mockito:mockito-junit-jupiter:5.20.0=testCompileClasspath,testRuntimeClasspath
org.objenesis:objenesis:3.3=testRuntimeClasspath
org.opentest4j:opentest4j:1.3.0=testCompileClasspath,testRuntimeClasspath
org.osgi:org.osgi.annotation.bundle:2.0.0=testCompileClasspath
org.osgi:org.osgi.annotation.versioning:1.1.2=testCompileClasspath
org.osgi:org.osgi.resource:1.0.0=testCompileClasspath
org.osgi:org.osgi.service.serviceloader:1.0.0=testCompileClasspath
org.osgi:org.osgi.annotation.bundle:2.0.0=compileClasspath,testCompileClasspath
org.osgi:org.osgi.annotation.versioning:1.1.2=compileClasspath,testCompileClasspath
org.osgi:org.osgi.resource:1.0.0=compileClasspath,testCompileClasspath
org.osgi:org.osgi.service.serviceloader:1.0.0=compileClasspath,testCompileClasspath
org.ow2.asm:asm-analysis:9.10.1=spotbugs
org.ow2.asm:asm-commons:9.10.1=spotbugs
org.ow2.asm:asm-tree:9.10.1=spotbugs
@@ -106,28 +116,32 @@ org.ow2.asm:asm-util:9.10.1=spotbugs
org.ow2.asm:asm:9.10.1=spotbugs
org.ow2.asm:asm:9.7.1=testCompileClasspath,testRuntimeClasspath
org.pcollections:pcollections:4.0.1=annotationProcessor,testAnnotationProcessor
org.reactivestreams:reactive-streams:1.0.4=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.reflections:reflections:0.10.2=checkstyle
org.skyscreamer:jsonassert:1.5.3=testCompileClasspath,testRuntimeClasspath
org.slf4j:jul-to-slf4j:2.0.17=testCompileClasspath,testRuntimeClasspath
org.slf4j:jul-to-slf4j:2.0.17=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.slf4j:slf4j-api:2.0.17=compileClasspath,runtimeClasspath,spotbugs,spotbugsSlf4j,testCompileClasspath,testRuntimeClasspath
org.slf4j:slf4j-simple:2.0.17=checkstyle,spotbugsSlf4j
org.springframework.boot:spring-boot-autoconfigure:4.0.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-configuration-processor:4.0.0=annotationProcessor
org.springframework.boot:spring-boot-http-client:4.0.0=testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-http-converter:4.0.0=testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-jackson:4.0.0=testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-jackson:4.0.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-mail:4.0.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-restclient:4.0.0=testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-resttestclient:4.0.0=testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-servlet:4.0.0=testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-starter-jackson-test:4.0.0=testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-starter-jackson:4.0.0=testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-starter-logging:4.0.0=testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-starter-json:4.0.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-starter-logging:4.0.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-starter-mail:4.0.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-starter-test:4.0.0=testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-starter-tomcat-runtime:4.0.0=testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-starter-tomcat:4.0.0=testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-starter-webmvc-test:4.0.0=testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-starter-webmvc:4.0.0=testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-starter:4.0.0=testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-starter:4.0.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-test-autoconfigure:4.0.0=testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-test:4.0.0=testCompileClasspath,testRuntimeClasspath
org.springframework.boot:spring-boot-tomcat:4.0.0=testCompileClasspath,testRuntimeClasspath
@@ -137,16 +151,19 @@ org.springframework.boot:spring-boot-webmvc:4.0.0=testCompileClasspath,testRunti
org.springframework.boot:spring-boot:4.0.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.springframework:spring-aop:7.0.1=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.springframework:spring-beans:7.0.1=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.springframework:spring-context-support:7.0.1=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.springframework:spring-context:7.0.1=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.springframework:spring-core:7.0.1=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.springframework:spring-expression:7.0.1=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.springframework:spring-test:7.0.1=testCompileClasspath,testRuntimeClasspath
org.springframework:spring-web:7.0.1=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.springframework:spring-webmvc:7.0.1=testCompileClasspath,testRuntimeClasspath
org.thymeleaf:thymeleaf:3.1.3.RELEASE=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.unbescape:unbescape:1.1.6.RELEASE=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
org.xmlresolver:xmlresolver:5.3.3=checkstyle,spotbugs
org.xmlunit:xmlunit-core:2.10.4=testCompileClasspath,testRuntimeClasspath
org.yaml:snakeyaml:2.5=testCompileClasspath,testRuntimeClasspath
tools.jackson.core:jackson-core:3.0.2=testCompileClasspath,testRuntimeClasspath
tools.jackson.core:jackson-databind:3.0.2=testCompileClasspath,testRuntimeClasspath
tools.jackson:jackson-bom:3.0.2=testCompileClasspath,testRuntimeClasspath
org.yaml:snakeyaml:2.5=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
tools.jackson.core:jackson-core:3.0.2=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
tools.jackson.core:jackson-databind:3.0.2=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
tools.jackson:jackson-bom:3.0.2=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
empty=
@@ -0,0 +1,41 @@
package dev.caskeleton.adapter.outbound.notification.platform.admin;
import dev.caskeleton.application.notification.platform.admin.AdminAccessDeniedException;
import dev.caskeleton.application.notification.platform.admin.AdminActor;
import dev.caskeleton.application.notification.platform.admin.NotificationAdminAuthority;
import dev.caskeleton.application.notification.platform.api.TenantId;
import java.util.Objects;
import java.util.Optional;
/**
* Operator authority check.
*
* <p>Application authority never grants an operator authority. The two planes are separated so that
* a compromised application credential cannot redrive a message or lift a suppression — the actions
* whose whole purpose is to override the platform's own safety decisions.
*/
public final class AdminAuthorizationGuard {
/** Require an authority, or refuse. */
public void require(AdminActor actor, NotificationAdminAuthority authority) {
Objects.requireNonNull(actor, "actor");
Objects.requireNonNull(authority, "authority");
if (!actor.holds(authority)) {
throw new AdminAccessDeniedException(authority);
}
}
/**
* Require that the actor may act on a tenant.
*
* <p>An actor with no tenant is a global operator; one bound to a tenant may only act inside it.
*/
public void requireTenant(AdminActor actor, TenantId tenantId) {
Objects.requireNonNull(actor, "actor");
Objects.requireNonNull(tenantId, "tenantId");
Optional<TenantId> scope = actor.tenantId();
if (scope.isPresent() && !scope.get().equals(tenantId)) {
throw new AdminAccessDeniedException(NotificationAdminAuthority.SUPPRESS);
}
}
}
@@ -0,0 +1,29 @@
package dev.caskeleton.adapter.outbound.notification.platform.admin;
import dev.caskeleton.application.notification.platform.admin.DuplicateRiskApprovalRequiredException;
import dev.caskeleton.application.notification.platform.api.delivery.AttemptConfirmation;
import dev.caskeleton.application.notification.platform.callback.DeliveryAttemptSnapshot;
import java.util.Objects;
/**
* Blocks an unapproved redrive of an ambiguous attempt.
*
* <p>The platform cannot tell whether the first submission reached the user, so re-sending is a
* decision with a real cost that only a human can accept. Requiring the approval flag makes that
* acceptance an explicit, audited act rather than a default.
*/
public final class DuplicateRiskGuard {
/** Verify the operator accepted the duplicate risk when one exists. */
public void verify(DeliveryAttemptSnapshot attempt, boolean approved) {
Objects.requireNonNull(attempt, "attempt");
boolean risky =
attempt.confirmation() == AttemptConfirmation.AMBIGUOUS
|| attempt.submissionOutcome()
== dev.caskeleton.application.notification.platform.api.delivery.SubmissionOutcome
.CONFIRMED_ACCEPTED;
if (risky && !approved) {
throw new DuplicateRiskApprovalRequiredException();
}
}
}
@@ -0,0 +1,325 @@
package dev.caskeleton.adapter.outbound.notification.platform.admin;
import dev.caskeleton.adapter.outbound.notification.platform.dispatch.ProviderRuntimeRegistry;
import dev.caskeleton.application.notification.platform.admin.AdminActor;
import dev.caskeleton.application.notification.platform.admin.AdminOperationResult;
import dev.caskeleton.application.notification.platform.admin.AdminOperationStorePort;
import dev.caskeleton.application.notification.platform.admin.NotificationAdminAuthority;
import dev.caskeleton.application.notification.platform.admin.NotificationAdminService;
import dev.caskeleton.application.notification.platform.admin.ReconcileCommand;
import dev.caskeleton.application.notification.platform.admin.RedriveCommand;
import dev.caskeleton.application.notification.platform.admin.SetProviderStateCommand;
import dev.caskeleton.application.notification.platform.admin.SuppressCommand;
import dev.caskeleton.application.notification.platform.api.DeliveryAttemptId;
import dev.caskeleton.application.notification.platform.api.delivery.RecipientDeliveryState;
import dev.caskeleton.application.notification.platform.callback.DeliveryAttemptSnapshot;
import dev.caskeleton.application.notification.platform.dispatch.DeliveryAttemptStorePort;
import dev.caskeleton.application.notification.platform.dispatch.RecipientDeliveryStorePort;
import dev.caskeleton.application.notification.platform.dispatch.ReconciliationService;
import dev.caskeleton.application.notification.platform.observation.NotificationAuditEvent;
import dev.caskeleton.application.notification.platform.observation.NotificationAuditPort;
import dev.caskeleton.application.notification.platform.policy.SuppressionEntry;
import dev.caskeleton.application.notification.platform.policy.SuppressionId;
import dev.caskeleton.application.notification.platform.policy.SuppressionSource;
import dev.caskeleton.application.notification.platform.policy.SuppressionStorePort;
import dev.caskeleton.application.notification.platform.provider.ProviderRuntimeState;
import dev.caskeleton.application.transaction.TransactionPort;
import java.time.Clock;
import java.util.ArrayList;
import java.util.List;
import java.util.Map;
import java.util.Objects;
import java.util.Optional;
import java.util.UUID;
/**
* N4 operator plane.
*
* <p>Four properties hold for every operation: a separate authority, an idempotent operation id, a
* recorded reason, and an audit row. The idempotency matters more than it looks — an operator
* retrying a redrive after a timeout must not send the message twice, which is exactly the failure
* the operation is trying to repair.
*
* <p>A dry run reads and reports but writes nothing, so an operator can see the blast radius of a
* bulk action before committing to it.
*/
public final class NotificationAdminServiceImpl implements NotificationAdminService {
private final AdminAuthorizationGuard authorization;
private final DuplicateRiskGuard duplicateRiskGuard;
private final DeliveryAttemptStorePort attempts;
private final RecipientDeliveryStorePort recipients;
private final ReconciliationService reconciliation;
private final SuppressionStorePort suppressions;
private final ProviderRuntimeRegistry runtimes;
private final AdminOperationStorePort operations;
private final NotificationAuditPort audit;
private final TransactionPort transactions;
private final Clock clock;
public NotificationAdminServiceImpl(
AdminAuthorizationGuard authorization,
DuplicateRiskGuard duplicateRiskGuard,
DeliveryAttemptStorePort attempts,
RecipientDeliveryStorePort recipients,
ReconciliationService reconciliation,
SuppressionStorePort suppressions,
ProviderRuntimeRegistry runtimes,
AdminOperationStorePort operations,
NotificationAuditPort audit,
TransactionPort transactions,
Clock clock) {
this.authorization = Objects.requireNonNull(authorization, "authorization");
this.duplicateRiskGuard = Objects.requireNonNull(duplicateRiskGuard, "duplicateRiskGuard");
this.attempts = Objects.requireNonNull(attempts, "attempts");
this.recipients = Objects.requireNonNull(recipients, "recipients");
this.reconciliation = Objects.requireNonNull(reconciliation, "reconciliation");
this.suppressions = Objects.requireNonNull(suppressions, "suppressions");
this.runtimes = Objects.requireNonNull(runtimes, "runtimes");
this.operations = Objects.requireNonNull(operations, "operations");
this.audit = Objects.requireNonNull(audit, "audit");
this.transactions = Objects.requireNonNull(transactions, "transactions");
this.clock = Objects.requireNonNull(clock, "clock");
}
@Override
public AdminOperationResult redrive(RedriveCommand command, AdminActor actor) {
Objects.requireNonNull(command, "command");
authorization.require(actor, NotificationAdminAuthority.REDRIVE);
Optional<AdminOperationResult> replayed = operations.findByOperationId(command.operationId());
if (replayed.isPresent()) {
return replayed.get();
}
DeliveryAttemptSnapshot original =
attempts
.snapshot(command.attemptId())
.orElseThrow(() -> new IllegalStateException("delivery attempt is not available"));
authorization.requireTenant(actor, original.tenantId());
duplicateRiskGuard.verify(original, command.approveDuplicateRisk());
if (command.dryRun()) {
return new AdminOperationResult(
command.operationId(),
true,
1,
Optional.of(original.notificationId()),
Optional.of(original.recipientDeliveryId()),
Optional.empty(),
List.of("DRY_RUN"));
}
return transactions.inWrite(
() -> {
// The logical identities are preserved and only the attempt is new, so the history stays
// one story rather than becoming two unrelated notifications.
recipients.transition(
original.recipientDeliveryId(),
RecipientDeliveryState.READY_TO_DISPATCH,
Optional.of(clock.instant()));
AdminOperationResult result =
new AdminOperationResult(
command.operationId(),
false,
1,
Optional.of(original.notificationId()),
Optional.of(original.recipientDeliveryId()),
Optional.empty(),
List.of(command.reason()));
audit.record(
new NotificationAuditEvent(
"ADMIN_REDRIVE",
actor.actorRef(),
Optional.of(command.reason()),
Optional.of(command.operationId()),
clock.instant(),
Map.of(
"provider", original.providerId().value(),
"channel", original.channel().name())));
return operations.save(result, actor, "ADMIN_REDRIVE");
});
}
@Override
public AdminOperationResult reconcile(ReconcileCommand command, AdminActor actor) {
Objects.requireNonNull(command, "command");
authorization.require(actor, NotificationAdminAuthority.RECONCILE);
Optional<AdminOperationResult> replayed = operations.findByOperationId(command.operationId());
if (replayed.isPresent()) {
return replayed.get();
}
if (command.dryRun()) {
return new AdminOperationResult(
command.operationId(),
true,
command.attemptIds().size(),
Optional.empty(),
Optional.empty(),
Optional.empty(),
List.of("DRY_RUN"));
}
List<String> reasons = new ArrayList<>();
int reconciled = 0;
for (DeliveryAttemptId attemptId : command.attemptIds()) {
reconciliation.reconcile(attemptId);
reconciled++;
}
reasons.add(command.reason());
AdminOperationResult result =
new AdminOperationResult(
command.operationId(),
false,
reconciled,
Optional.empty(),
Optional.empty(),
Optional.empty(),
List.copyOf(reasons));
audit.record(
new NotificationAuditEvent(
"ADMIN_RECONCILE",
actor.actorRef(),
Optional.of(command.reason()),
Optional.of(command.operationId()),
clock.instant(),
Map.of()));
return operations.save(result, actor, "ADMIN_RECONCILE");
}
@Override
public AdminOperationResult suppress(SuppressCommand command, AdminActor actor) {
Objects.requireNonNull(command, "command");
authorization.require(actor, NotificationAdminAuthority.SUPPRESS);
authorization.requireTenant(actor, command.tenantId());
Optional<AdminOperationResult> replayed = operations.findByOperationId(command.operationId());
if (replayed.isPresent()) {
return replayed.get();
}
if (command.dryRun()) {
return new AdminOperationResult(
command.operationId(),
true,
1,
Optional.empty(),
Optional.empty(),
Optional.empty(),
List.of("DRY_RUN"));
}
return transactions.inWrite(
() -> {
int affected;
if (command.remove()) {
// Removal is by fingerprint match rather than by id, because an operator lifting a
// suppression knows the target, not the row identifier the platform assigned.
affected =
suppressions
.activeFor(
command.tenantId(), command.targetFingerprint(), clock.instant())
.stream()
.map(entry -> suppressions.remove(command.tenantId(), entry.id()))
.filter(Optional::isPresent)
.count()
> 0
? 1
: 0;
} else {
suppressions.upsert(
new SuppressionEntry(
new SuppressionId(UUID.randomUUID()),
command.tenantId(),
command.scope(),
command.reason(),
command.targetFingerprint(),
Optional.empty(),
clock.instant(),
command.expiresAt(),
SuppressionSource.ADMIN));
affected = 1;
}
AdminOperationResult result =
new AdminOperationResult(
command.operationId(),
false,
affected,
Optional.empty(),
Optional.empty(),
Optional.empty(),
List.of(command.reasonText()));
audit.record(
new NotificationAuditEvent(
command.remove() ? "ADMIN_SUPPRESSION_REMOVED" : "ADMIN_SUPPRESSION_ADDED",
actor.actorRef(),
Optional.of(command.reason().name()),
Optional.of(command.operationId()),
clock.instant(),
Map.of()));
return operations.save(
result, actor, command.remove() ? "ADMIN_SUPPRESS_REMOVE" : "ADMIN_SUPPRESS_ADD");
});
}
@Override
public AdminOperationResult setProviderState(SetProviderStateCommand command, AdminActor actor) {
Objects.requireNonNull(command, "command");
authorization.require(actor, NotificationAdminAuthority.PROVIDER_CONTROL);
Optional<AdminOperationResult> replayed = operations.findByOperationId(command.operationId());
if (replayed.isPresent()) {
return replayed.get();
}
if (command.dryRun()) {
return new AdminOperationResult(
command.operationId(),
true,
1,
Optional.empty(),
Optional.empty(),
Optional.empty(),
List.of("DRY_RUN"));
}
var runtime = runtimes.current(command.profileId());
switch (command.desiredState()) {
case DISABLED -> runtime.markDisabled();
case DRAINING -> runtime.markDraining();
case HEALTHY -> runtime.markHealthy();
case DEGRADED -> runtime.markDegraded(command.reason());
case THROTTLED -> runtime.markThrottled();
case AUTHENTICATION_FAILED -> runtime.markAuthenticationFailed(command.reason());
}
AdminOperationResult result =
new AdminOperationResult(
command.operationId(),
false,
1,
Optional.empty(),
Optional.empty(),
Optional.empty(),
List.of(command.reason()));
audit.record(
new NotificationAuditEvent(
"ADMIN_PROVIDER_STATE",
actor.actorRef(),
Optional.of(command.reason()),
Optional.of(command.operationId()),
clock.instant(),
Map.of(
"providerProfile", command.profileId().value(),
"status", command.desiredState().name())));
return operations.save(result, actor, "ADMIN_PROVIDER_STATE");
}
/** Current state of a provider runtime, for the health endpoint. */
public ProviderRuntimeState providerState(
dev.caskeleton.application.notification.platform.api.ProviderProfileId profileId) {
return runtimes.state(profileId);
}
}
@@ -0,0 +1,105 @@
package dev.caskeleton.adapter.outbound.notification.platform.autoconfigure;
import dev.caskeleton.adapter.outbound.notification.platform.provider.http.JdkNotificationHttpGateway;
import dev.caskeleton.adapter.outbound.notification.platform.provider.http.NotificationHttpGateway;
import dev.caskeleton.adapter.outbound.notification.platform.security.AesGcmContactPointProtector;
import dev.caskeleton.adapter.outbound.notification.platform.template.JacksonNotificationVariablesCodec;
import dev.caskeleton.adapter.outbound.notification.platform.template.JsonSchemaVariableValidator;
import dev.caskeleton.adapter.outbound.notification.platform.template.NotificationTemplateEngine;
import dev.caskeleton.adapter.outbound.notification.platform.template.PlaceholderTemplateEngine;
import dev.caskeleton.adapter.outbound.notification.platform.template.Sha256MessageDigestAdapter;
import dev.caskeleton.adapter.outbound.notification.platform.template.ThymeleafStringTemplateEngine;
import dev.caskeleton.application.notification.platform.dispatch.MessageDigestPort;
import dev.caskeleton.application.notification.platform.dispatch.NotificationVariablesCodecPort;
import dev.caskeleton.application.notification.platform.security.ContactPointProtector;
import dev.caskeleton.application.notification.platform.security.SecretMaterialProvider;
import dev.caskeleton.application.notification.platform.template.TemplateVariableValidator;
import java.time.Duration;
import org.springframework.beans.factory.annotation.Value;
import org.springframework.boot.autoconfigure.condition.ConditionalOnBean;
import org.springframework.boot.autoconfigure.condition.ConditionalOnMissingBean;
import org.springframework.boot.autoconfigure.condition.ConditionalOnProperty;
import org.springframework.boot.context.properties.EnableConfigurationProperties;
import org.springframework.context.annotation.Bean;
import org.springframework.context.annotation.Configuration;
/**
* Notification platform wiring.
*
* <p>Everything is opt-in and conditional. The platform contributes no beans unless it is enabled,
* and the contact point protector only appears once a secret provider exists — because a protector
* without keys would fail on the first delivery instead of at startup.
*/
@Configuration(proxyBeanMethods = false)
@EnableConfigurationProperties(NotificationPlatformSettings.class)
@ConditionalOnProperty(
prefix = "ca-skeleton.notification.platform",
name = "enabled",
havingValue = "true")
public class NotificationPlatformAutoConfiguration {
/** Canonical variables codec. */
@Bean
@ConditionalOnMissingBean
public NotificationVariablesCodecPort notificationVariablesCodec() {
return new JacksonNotificationVariablesCodec();
}
/** Request fingerprint hashing. */
@Bean
@ConditionalOnMissingBean
public MessageDigestPort notificationMessageDigest() {
return new Sha256MessageDigestAdapter();
}
/** JSON Schema 2020-12 variable validation. */
@Bean
@ConditionalOnMissingBean
public TemplateVariableValidator notificationTemplateVariableValidator() {
return new JsonSchemaVariableValidator();
}
/**
* Template engine, defaulting to the deterministic placeholder substitution.
*
* <p>Thymeleaf is the opt-in alternative: it escapes by default, which matters for HTML email
* bodies built from application input. The default stays the placeholder engine because it has no
* expression evaluator at all, and an unknown engine name fails the boot rather than quietly
* falling back — a deployment that thought it had escaping and did not is the worse outcome.
*/
@Bean
@ConditionalOnMissingBean
public NotificationTemplateEngine notificationTemplateEngine(
@Value("${ca-skeleton.notification.platform.template.engine:placeholder}") String engine) {
return switch (engine.toLowerCase(java.util.Locale.ROOT)) {
case "placeholder" -> new PlaceholderTemplateEngine();
case "thymeleaf" -> new ThymeleafStringTemplateEngine();
default ->
throw new IllegalArgumentException(
"ca-skeleton.notification.platform.template.engine must be"
+ " 'placeholder' or 'thymeleaf', not '"
+ engine
+ "'");
};
}
/** Contact point protection, only once key material is available. */
@Bean
@ConditionalOnBean(SecretMaterialProvider.class)
@ConditionalOnMissingBean
public ContactPointProtector notificationContactPointProtector(SecretMaterialProvider secrets) {
return new AesGcmContactPointProtector(secrets);
}
/**
* Default provider transport.
*
* <p>Replaced in the composition root when the HTTP Client Platform is bound, which is the
* supported way to reuse its TLS, circuit-breaker and SSRF policy.
*/
@Bean
@ConditionalOnMissingBean
public NotificationHttpGateway notificationHttpGateway() {
return new JdkNotificationHttpGateway(Duration.ofSeconds(2));
}
}
@@ -0,0 +1,154 @@
package dev.caskeleton.adapter.outbound.notification.platform.autoconfigure;
import java.time.Duration;
import java.util.Map;
import java.util.Objects;
import org.springframework.boot.context.properties.ConfigurationProperties;
/**
* Bound notification platform configuration.
*
* <p>Validation happens in the constructor, so a misconfiguration fails the boot rather than
* surfacing as a delivery incident hours later. Everything is bounded: there is no property whose
* value may be "unlimited", because an unbounded queue or payload is a resource failure waiting for
* the first burst.
*/
@ConfigurationProperties("ca-skeleton.notification.platform")
public record NotificationPlatformSettings(
boolean enabled, Dispatch dispatch, Callbacks callbacks, Map<String, Provider> providers) {
public NotificationPlatformSettings {
dispatch = dispatch == null ? Dispatch.defaults() : dispatch;
callbacks = callbacks == null ? Callbacks.defaults() : callbacks;
providers = providers == null ? Map.of() : Map.copyOf(providers);
providers.forEach((id, provider) -> provider.validate(id));
}
/** Dispatch runtime bounds. */
public record Dispatch(
int claimBatchSize,
Duration leaseDuration,
Duration pollInterval,
int maxGlobalConcurrency,
int maxAdditionalAttempts,
Duration maxQueueAge,
boolean allowAmbiguousFallback) {
private static final int MAX_CLAIM_BATCH = 1000;
public Dispatch {
Objects.requireNonNull(leaseDuration, "leaseDuration");
Objects.requireNonNull(pollInterval, "pollInterval");
Objects.requireNonNull(maxQueueAge, "maxQueueAge");
if (claimBatchSize < 1 || claimBatchSize > MAX_CLAIM_BATCH) {
throw new IllegalArgumentException(
"ca-skeleton.notification.platform.dispatch.claim-batch-size must be 1.."
+ MAX_CLAIM_BATCH);
}
if (maxGlobalConcurrency < 1) {
throw new IllegalArgumentException("max-global-concurrency must be positive");
}
if (maxAdditionalAttempts < 0) {
throw new IllegalArgumentException("max-additional-attempts must not be negative");
}
if (leaseDuration.isNegative() || leaseDuration.isZero()) {
throw new IllegalArgumentException("lease-duration must be positive and finite");
}
if (leaseDuration.compareTo(pollInterval) <= 0) {
throw new IllegalArgumentException("lease-duration must exceed poll-interval");
}
if (allowAmbiguousFallback) {
// Refused outright rather than warned about: automatic fallback after an ambiguous
// submission is the configuration that turns an unknown into a guaranteed duplicate.
throw new IllegalArgumentException(
"allow-ambiguous-fallback is not a supported configuration");
}
}
/** Conservative defaults. */
public static Dispatch defaults() {
return new Dispatch(
100, Duration.ofSeconds(30), Duration.ofMillis(250), 128, 3, Duration.ofHours(24), false);
}
}
/** Callback endpoint bounds. */
public record Callbacks(boolean enabled, long maxBodyBytes, Duration replaySkew) {
private static final long MAX_BODY_CEILING = 1_048_576L;
public Callbacks {
Objects.requireNonNull(replaySkew, "replaySkew");
if (maxBodyBytes < 1 || maxBodyBytes > MAX_BODY_CEILING) {
throw new IllegalArgumentException("max-body-bytes must be 1.." + MAX_BODY_CEILING);
}
if (replaySkew.isNegative()) {
throw new IllegalArgumentException("replay-skew must not be negative");
}
}
/** Conservative defaults. */
public static Callbacks defaults() {
return new Callbacks(false, 65_536L, Duration.ofMinutes(5));
}
}
/** One provider profile. */
public record Provider(
String type,
boolean enabled,
String environment,
String credentialProfile,
String topic,
String vapidPublicKey,
String callbackSigningSecretRef,
Duration timeout,
int maxConcurrency,
int ratePerSecond) {
/** Fail the boot when a profile cannot possibly work. */
public void validate(String profileId) {
Objects.requireNonNull(profileId, "profileId");
if (!enabled) {
return;
}
require(type != null && !type.isBlank(), profileId, "type is required");
require(environment != null && !environment.isBlank(), profileId, "environment is required");
require(
credentialProfile != null && !credentialProfile.isBlank(),
profileId,
"credential-profile is required");
require(
timeout != null && !timeout.isNegative() && !timeout.isZero(),
profileId,
"timeout must be positive and finite");
require(maxConcurrency >= 1, profileId, "max-concurrency must be positive");
require(ratePerSecond >= 1, profileId, "rate-limit-per-second must be positive");
switch (type == null ? "" : type.toUpperCase(java.util.Locale.ROOT)) {
case "APNS" ->
require(topic != null && !topic.isBlank(), profileId, "APNs profiles require a topic");
case "WEB_PUSH" ->
require(
vapidPublicKey != null && !vapidPublicKey.isBlank(),
profileId,
"Web Push profiles require a VAPID key");
case "TWILIO", "SES" ->
require(
callbackSigningSecretRef != null && !callbackSigningSecretRef.isBlank(),
profileId,
"callback-capable profiles require a callback signing secret reference");
default -> {
// Providers without extra requirements are already covered by the common checks.
}
}
}
private static void require(boolean condition, String profileId, String message) {
if (!condition) {
throw new IllegalArgumentException(
"notification provider profile '" + profileId + "': " + message);
}
}
}
}
@@ -0,0 +1,17 @@
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
/**
* A held concurrency slot for one provider attempt.
*
* <p>Closing it is what releases the slot, so every call site uses try-with-resources. The permit
* also carries the credential generation the attempt ran under, which is what makes a rotation
* auditable after the fact.
*/
public interface AttemptPermit extends AutoCloseable {
/** Credential generation this attempt is bound to. */
long generation();
@Override
void close();
}
@@ -0,0 +1,52 @@
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
import dev.caskeleton.application.notification.platform.api.ProviderProfileId;
import dev.caskeleton.application.notification.platform.callback.DeliveryAttemptSnapshot;
import dev.caskeleton.application.notification.platform.dispatch.ReconciliationGatewayPort;
import dev.caskeleton.application.notification.platform.provider.ReconciliationCapability;
import dev.caskeleton.application.notification.platform.provider.ReconciliationResult;
import java.util.Map;
import java.util.Objects;
import java.util.Optional;
/**
* Routes a reconciliation to the capability that owns the provider.
*
* <p>A profile with no registered capability reports {@code Unsupported} rather than falling back
* to a guess. Inventing a final status for a provider that cannot be queried is precisely the
* behaviour the ambiguity model exists to prevent.
*/
public final class CapabilityReconciliationGateway implements ReconciliationGatewayPort {
private final Map<ProviderProfileId, ReconciliationCapability> capabilities;
private final ProviderRuntimeRegistry runtimes;
public CapabilityReconciliationGateway(
Map<ProviderProfileId, ReconciliationCapability> capabilities,
ProviderRuntimeRegistry runtimes) {
this.capabilities = Map.copyOf(Objects.requireNonNull(capabilities, "capabilities"));
this.runtimes = Objects.requireNonNull(runtimes, "runtimes");
}
@Override
public boolean supports(ProviderProfileId profileId) {
Objects.requireNonNull(profileId, "profileId");
return Optional.ofNullable(capabilities.get(profileId))
.map(capability -> capability.supports(runtimes.current(profileId).profile()))
.orElse(false);
}
@Override
public ReconciliationResult reconcile(DeliveryAttemptSnapshot attempt) {
Objects.requireNonNull(attempt, "attempt");
ReconciliationCapability capability = capabilities.get(attempt.providerProfileId());
if (capability == null) {
return new ReconciliationResult.Unsupported();
}
// The permit is taken so a reconciliation backlog cannot become a second load source during the
// incident that produced it.
try (AttemptPermit permit = runtimes.current(attempt.providerProfileId()).acquireAttempt()) {
return capability.reconcile(attempt).toCompletableFuture().join();
}
}
}
@@ -0,0 +1,72 @@
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
import dev.caskeleton.application.notification.platform.api.ProviderProfileId;
import dev.caskeleton.application.notification.platform.api.RecipientSpec;
import dev.caskeleton.application.notification.platform.api.TenantId;
import dev.caskeleton.application.notification.platform.api.routing.Channel;
import dev.caskeleton.application.notification.platform.api.routing.DeliveryStrategy;
import dev.caskeleton.application.notification.platform.api.routing.ExplicitChannel;
import dev.caskeleton.application.notification.platform.api.routing.OrderedFallback;
import dev.caskeleton.application.notification.platform.dispatch.NotificationRoutePlannerPort;
import dev.caskeleton.application.notification.platform.policy.RouteCandidate;
import java.util.ArrayList;
import java.util.List;
import java.util.Map;
import java.util.Objects;
/**
* Turns a strategy into an ordered route plan using the configured channel-to-profile map.
*
* <p>A channel with no configured provider, or a recipient with no contact point for it, simply
* produces no candidate. The routing engine then reports {@code NO_ELIGIBLE_ROUTE} rather than the
* dispatcher failing on a null, which is the difference between a diagnosable state and a stack
* trace.
*/
public final class ConfiguredRoutePlanner implements NotificationRoutePlannerPort {
private final Map<Channel, ProviderProfileId> profilesByChannel;
public ConfiguredRoutePlanner(Map<Channel, ProviderProfileId> profilesByChannel) {
this.profilesByChannel =
Map.copyOf(Objects.requireNonNull(profilesByChannel, "profilesByChannel"));
}
@Override
public List<RouteCandidate> plan(
TenantId tenantId, RecipientSpec recipient, DeliveryStrategy strategy) {
Objects.requireNonNull(tenantId, "tenantId");
Objects.requireNonNull(recipient, "recipient");
Objects.requireNonNull(strategy, "strategy");
List<Channel> ordered =
switch (strategy) {
case ExplicitChannel explicit -> List.of(explicit.channel());
case OrderedFallback fallback -> fallback.channels();
};
List<RouteCandidate> routes = new ArrayList<>(ordered.size());
int index = 0;
for (Channel channel : ordered) {
ProviderProfileId profileId = profilesByChannel.get(channel);
if (profileId == null) {
continue;
}
var selector =
recipient.contactPoints().stream()
.filter(candidate -> candidate.channel() == channel)
.findFirst();
if (selector.isEmpty()) {
continue;
}
boolean blocked =
recipient
.channelOverride()
.map(override -> override.blockedChannels().contains(channel))
.orElse(false);
routes.add(
new RouteCandidate(
index++, channel, selector.get().contactPointId(), profileId, !blocked, true));
}
return List.copyOf(routes);
}
}
@@ -0,0 +1,9 @@
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
/** Verifies a candidate generation before it becomes the current one. */
@FunctionalInterface
public interface CredentialProbe {
/** Return false when the candidate credential is not usable. */
boolean isUsable(ProviderRuntime candidate);
}
@@ -0,0 +1,19 @@
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
import dev.caskeleton.application.notification.platform.api.error.FailureCategory;
import dev.caskeleton.application.notification.platform.api.error.NotificationException;
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureCode;
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureDescriptor;
/** Raised when a candidate credential generation fails its probe before any cutover. */
public class CredentialValidationException extends NotificationException {
private static final long serialVersionUID = 1L;
public CredentialValidationException() {
super(
NotificationFailureDescriptor.preDispatch(
NotificationFailureCode.PROVIDER_CONFIGURATION_INVALID,
FailureCategory.AUTHENTICATION));
}
}
@@ -0,0 +1,61 @@
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
import dev.caskeleton.adapter.outbound.notification.platform.template.NotificationJsonMapper;
import dev.caskeleton.application.notification.platform.api.ContactPointId;
import dev.caskeleton.application.notification.platform.api.ProviderProfileId;
import dev.caskeleton.application.notification.platform.api.routing.Channel;
import dev.caskeleton.application.notification.platform.dispatch.NotificationRoutingPlanCodecPort;
import dev.caskeleton.application.notification.platform.policy.RouteCandidate;
import java.util.ArrayList;
import java.util.LinkedHashMap;
import java.util.List;
import java.util.Map;
import java.util.Objects;
import java.util.UUID;
import tools.jackson.core.type.TypeReference;
/**
* Route plan encoding.
*
* <p>The plan is frozen at submit time, so this is a snapshot format rather than a view: it stores
* exactly what was decided, including which routes were usable then, and never recomputes.
*/
public final class JacksonRoutingPlanCodec implements NotificationRoutingPlanCodecPort {
@Override
public String encode(List<RouteCandidate> routes) {
Objects.requireNonNull(routes, "routes");
List<Map<String, Object>> encoded = new ArrayList<>(routes.size());
for (RouteCandidate route : routes) {
Map<String, Object> entry = new LinkedHashMap<>();
entry.put("routeIndex", route.routeIndex());
entry.put("channel", route.channel().name());
entry.put("contactPointId", route.contactPointId().value().toString());
entry.put("providerProfileId", route.providerProfileId().value());
entry.put("contactPointActive", route.contactPointActive());
entry.put("providerEnabled", route.providerEnabled());
encoded.add(entry);
}
return NotificationJsonMapper.mapper().writeValueAsString(encoded);
}
@Override
public List<RouteCandidate> decode(String payload) {
Objects.requireNonNull(payload, "payload");
List<LinkedHashMap<String, Object>> raw =
NotificationJsonMapper.mapper()
.readValue(payload, new TypeReference<ArrayList<LinkedHashMap<String, Object>>>() {});
List<RouteCandidate> routes = new ArrayList<>(raw.size());
for (Map<String, Object> entry : raw) {
routes.add(
new RouteCandidate(
((Number) entry.get("routeIndex")).intValue(),
Channel.valueOf(String.valueOf(entry.get("channel"))),
new ContactPointId(UUID.fromString(String.valueOf(entry.get("contactPointId")))),
new ProviderProfileId(String.valueOf(entry.get("providerProfileId"))),
Boolean.TRUE.equals(entry.get("contactPointActive")),
Boolean.TRUE.equals(entry.get("providerEnabled"))));
}
return List.copyOf(routes);
}
}
@@ -0,0 +1,61 @@
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
import dev.caskeleton.application.notification.platform.api.DeliveryAttemptId;
import dev.caskeleton.application.notification.platform.dispatch.DeliveryAttemptStorePort;
import dev.caskeleton.application.notification.platform.dispatch.RecipientLeaseStorePort;
import dev.caskeleton.application.notification.platform.dispatch.ReconciliationService;
import java.time.Duration;
import java.util.List;
import java.util.Objects;
/**
* Recovers deliveries a dead worker left in flight.
*
* <p>An expired lease on a {@code DISPATCHING} delivery is the crash case: the attempt row exists,
* so a provider call may have happened. Recovery therefore reconciles rather than re-dispatching —
* re-dispatching would be the platform choosing to duplicate rather than to ask.
*/
public final class LeaseRecoveryService {
private final RecipientLeaseStorePort leases;
private final DeliveryAttemptStorePort attempts;
private final ReconciliationService reconciliation;
private final Duration staleAfter;
private final int batchSize;
public LeaseRecoveryService(
RecipientLeaseStorePort leases,
DeliveryAttemptStorePort attempts,
ReconciliationService reconciliation,
Duration staleAfter,
int batchSize) {
this.leases = Objects.requireNonNull(leases, "leases");
this.attempts = Objects.requireNonNull(attempts, "attempts");
this.reconciliation = Objects.requireNonNull(reconciliation, "reconciliation");
this.staleAfter = Objects.requireNonNull(staleAfter, "staleAfter");
this.batchSize = batchSize;
if (batchSize < 1) {
throw new IllegalArgumentException("batchSize");
}
if (staleAfter.isNegative() || staleAfter.isZero()) {
throw new IllegalArgumentException("staleAfter must be positive and finite");
}
}
/** Recover one batch of abandoned deliveries; returns how many were handled. */
public int recoverOnce() {
List<dev.caskeleton.application.notification.platform.api.RecipientDeliveryId> abandoned =
leases.expiredDispatching(batchSize, staleAfter);
int handled = 0;
for (var recipientDeliveryId : abandoned) {
for (var attempt : attempts.attemptsOf(recipientDeliveryId)) {
if (attempt.completedAt().isEmpty()) {
DeliveryAttemptId attemptId = attempt.id();
reconciliation.reconcile(attemptId);
handled++;
}
}
}
return handled;
}
}
@@ -0,0 +1,29 @@
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
import dev.caskeleton.application.notification.platform.inbox.InboxItemCreated;
import dev.caskeleton.application.notification.platform.inbox.NotificationInboxSignalPort;
import java.util.Objects;
import org.slf4j.Logger;
import org.slf4j.LoggerFactory;
/**
* Default inbox signal sink.
*
* <p>Emits identifiers only, never content. A deployment with a WebSocket or messaging relay
* replaces it; until then the inbox is still complete, because the row — not the signal — is the
* source of truth.
*/
public final class LoggingInboxSignalPublisher implements NotificationInboxSignalPort {
private static final Logger log = LoggerFactory.getLogger("notification.inbox.signal");
@Override
public void publish(InboxItemCreated event) {
Objects.requireNonNull(event, "event");
log.info(
"event=inbox_item_created itemId={} tenant={} category={}",
event.itemId().value(),
event.principal().tenantId().value(),
event.category());
}
}
@@ -0,0 +1,31 @@
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
import dev.caskeleton.application.notification.platform.api.routing.Channel;
import dev.caskeleton.application.notification.platform.dispatch.TemplateRendererRegistry;
import dev.caskeleton.application.notification.platform.template.NotificationTemplateRenderer;
import java.util.EnumMap;
import java.util.List;
import java.util.Map;
import java.util.Objects;
/** Channel-to-renderer lookup built at wiring time. */
public final class MapTemplateRendererRegistry implements TemplateRendererRegistry {
private final Map<Channel, NotificationTemplateRenderer> renderers;
public MapTemplateRendererRegistry(List<NotificationTemplateRenderer> renderers) {
Objects.requireNonNull(renderers, "renderers");
Map<Channel, NotificationTemplateRenderer> byChannel = new EnumMap<>(Channel.class);
renderers.forEach(renderer -> byChannel.put(renderer.channel(), renderer));
this.renderers = Map.copyOf(byChannel);
}
@Override
public NotificationTemplateRenderer rendererFor(Channel channel) {
NotificationTemplateRenderer renderer = renderers.get(channel);
if (renderer == null) {
throw new IllegalStateException("no renderer registered for the channel");
}
return renderer;
}
}
@@ -0,0 +1,55 @@
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
import java.time.Duration;
import java.util.Objects;
/**
* Dispatch runtime bounds.
*
* <p>Every field is bounded and validated at construction. "Unlimited" is never an accepted value:
* an unbounded claim batch or queue is how a burst becomes an out-of-memory failure instead of
* backpressure.
*/
public record NotificationDispatchProperties(
int claimBatchSize,
Duration leaseDuration,
Duration pollInterval,
int maxGlobalConcurrency,
int maxAdditionalAttempts,
Duration estimatedDispatchDuration,
Duration maxQueueAge,
Duration shutdownGrace) {
private static final int MAX_CLAIM_BATCH = 1000;
public NotificationDispatchProperties {
Objects.requireNonNull(leaseDuration, "leaseDuration");
Objects.requireNonNull(pollInterval, "pollInterval");
Objects.requireNonNull(estimatedDispatchDuration, "estimatedDispatchDuration");
Objects.requireNonNull(maxQueueAge, "maxQueueAge");
Objects.requireNonNull(shutdownGrace, "shutdownGrace");
if (claimBatchSize < 1 || claimBatchSize > MAX_CLAIM_BATCH) {
throw new IllegalArgumentException("claimBatchSize must be 1.." + MAX_CLAIM_BATCH);
}
if (maxGlobalConcurrency < 1) {
throw new IllegalArgumentException("maxGlobalConcurrency");
}
if (maxAdditionalAttempts < 0) {
throw new IllegalArgumentException("maxAdditionalAttempts");
}
requirePositive(leaseDuration, "leaseDuration");
requirePositive(pollInterval, "pollInterval");
requirePositive(maxQueueAge, "maxQueueAge");
if (leaseDuration.compareTo(pollInterval) <= 0) {
// A lease shorter than the poll interval expires before the worker can renew it, so two
// workers would routinely claim the same job.
throw new IllegalArgumentException("leaseDuration must exceed pollInterval");
}
}
private static void requirePositive(Duration value, String name) {
if (value.isNegative() || value.isZero()) {
throw new IllegalArgumentException(name + " must be positive and finite");
}
}
}
@@ -0,0 +1,137 @@
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
import dev.caskeleton.application.notification.platform.dispatch.NotificationDispatchService;
import dev.caskeleton.application.notification.platform.dispatch.RecipientLease;
import dev.caskeleton.application.notification.platform.dispatch.RecipientLeaseStorePort;
import dev.caskeleton.application.notification.platform.observation.NotificationMetricName;
import dev.caskeleton.application.notification.platform.observation.NotificationMetricsPort;
import java.util.List;
import java.util.Map;
import java.util.Objects;
import java.util.concurrent.ExecutorService;
import java.util.concurrent.Executors;
import java.util.concurrent.Semaphore;
import java.util.concurrent.TimeUnit;
import java.util.concurrent.atomic.AtomicBoolean;
import org.slf4j.Logger;
import org.slf4j.LoggerFactory;
/**
* Claims due deliveries and hands them to the dispatcher.
*
* <p>The worker never calls a provider itself. It claims, submits to a bounded executor, and stops
* claiming the moment shutdown begins — so a rolling restart drains rather than abandoning leases
* that then have to time out.
*
* <p>Claiming is bounded twice over: by the claim batch size and by a global concurrency permit.
* The second bound matters because a slow provider would otherwise let the queue depth become the
* thread count.
*/
public final class NotificationSchedulerWorker implements AutoCloseable {
private static final Logger log = LoggerFactory.getLogger(NotificationSchedulerWorker.class);
private final RecipientLeaseStorePort leases;
private final NotificationDispatchService dispatcher;
private final NotificationMetricsPort metrics;
private final NotificationDispatchProperties properties;
private final String workerId;
private final ExecutorService dispatchExecutor;
private final Semaphore globalConcurrency;
private final AtomicBoolean running = new AtomicBoolean();
private final AtomicBoolean shuttingDown = new AtomicBoolean();
public NotificationSchedulerWorker(
RecipientLeaseStorePort leases,
NotificationDispatchService dispatcher,
NotificationMetricsPort metrics,
NotificationDispatchProperties properties,
String workerId) {
this.leases = Objects.requireNonNull(leases, "leases");
this.dispatcher = Objects.requireNonNull(dispatcher, "dispatcher");
this.metrics = Objects.requireNonNull(metrics, "metrics");
this.properties = Objects.requireNonNull(properties, "properties");
this.workerId = Objects.requireNonNull(workerId, "workerId");
this.dispatchExecutor = Executors.newVirtualThreadPerTaskExecutor();
this.globalConcurrency = new Semaphore(properties.maxGlobalConcurrency());
}
/** Claim and dispatch one batch. Returns how many deliveries were claimed. */
public int runOnce() {
if (shuttingDown.get()) {
return 0;
}
List<RecipientLease> claimed =
leases.claim(workerId, properties.claimBatchSize(), properties.leaseDuration());
metrics.gauge(NotificationMetricName.QUEUE_DEPTH, Map.of(), claimed.size());
for (RecipientLease lease : claimed) {
globalConcurrency.acquireUninterruptibly();
dispatchExecutor.execute(
() -> {
try {
dispatcher.dispatch(lease);
} catch (RuntimeException failure) {
// The lease is left to expire rather than being released optimistically: a worker
// that
// failed mid-dispatch cannot prove what the provider did.
log.warn(
"notification dispatch failed worker={} reason={}",
workerId,
failure.getClass().getSimpleName());
} finally {
globalConcurrency.release();
}
});
}
return claimed.size();
}
/** Start the polling loop on a dedicated thread. */
public void start() {
if (!running.compareAndSet(false, true)) {
return;
}
Thread.ofVirtual()
.name("notification-scheduler-" + workerId)
.start(
() -> {
while (running.get() && !shuttingDown.get()) {
try {
if (runOnce() == 0) {
Thread.sleep(properties.pollInterval().toMillis());
}
} catch (InterruptedException interrupted) {
Thread.currentThread().interrupt();
return;
} catch (RuntimeException failure) {
log.warn(
"notification scheduler tick failed worker={} reason={}",
workerId,
failure.getClass().getSimpleName());
}
}
});
}
@Override
public void close() {
shuttingDown.set(true);
running.set(false);
dispatchExecutor.shutdown();
try {
if (!dispatchExecutor.awaitTermination(
properties.shutdownGrace().toMillis(), TimeUnit.MILLISECONDS)) {
dispatchExecutor.shutdownNow();
}
} catch (InterruptedException interrupted) {
Thread.currentThread().interrupt();
dispatchExecutor.shutdownNow();
}
}
/** Whether the worker has stopped claiming new work. */
public boolean shuttingDown() {
return shuttingDown.get();
}
}
@@ -0,0 +1,73 @@
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
import dev.caskeleton.application.notification.platform.api.error.FailureCategory;
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureCode;
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureDescriptor;
import dev.caskeleton.application.notification.platform.api.error.ProviderUnavailableException;
import java.time.Clock;
import java.util.Objects;
import java.util.concurrent.Semaphore;
import java.util.concurrent.atomic.AtomicLong;
/**
* Per-provider rate and concurrency guard.
*
* <p>Tokens are spent on real attempts only. A delivery waiting out its backoff holds no permit,
* because a provider outage would otherwise pin the whole concurrency budget on deliveries that are
* not doing anything.
*/
public final class ProviderAttemptLimiter {
private final Semaphore concurrency;
private final int maxConcurrency;
private final int ratePerSecond;
private final Clock clock;
private final AtomicLong windowStartSecond = new AtomicLong();
private final AtomicLong issuedInWindow = new AtomicLong();
public ProviderAttemptLimiter(int maxConcurrency, int ratePerSecond, Clock clock) {
if (maxConcurrency < 1) {
throw new IllegalArgumentException("maxConcurrency");
}
if (ratePerSecond < 1) {
throw new IllegalArgumentException("ratePerSecond");
}
this.concurrency = new Semaphore(maxConcurrency);
this.maxConcurrency = maxConcurrency;
this.ratePerSecond = ratePerSecond;
this.clock = Objects.requireNonNull(clock, "clock");
this.windowStartSecond.set(clock.instant().getEpochSecond());
}
/** Acquire one attempt slot, or fail fast when the provider budget is spent. */
public void acquire() {
long second = clock.instant().getEpochSecond();
long windowStart = windowStartSecond.get();
if (second != windowStart && windowStartSecond.compareAndSet(windowStart, second)) {
issuedInWindow.set(0L);
}
if (issuedInWindow.incrementAndGet() > ratePerSecond) {
throw unavailable();
}
if (!concurrency.tryAcquire()) {
issuedInWindow.decrementAndGet();
throw unavailable();
}
}
/** Release a previously acquired slot. */
public void release() {
concurrency.release();
}
/** Slots currently held. */
public int activeAttempts() {
return maxConcurrency - concurrency.availablePermits();
}
private static ProviderUnavailableException unavailable() {
return new ProviderUnavailableException(
NotificationFailureDescriptor.preDispatch(
NotificationFailureCode.PROVIDER_UNAVAILABLE, FailureCategory.CAPACITY_REJECTED));
}
}
@@ -0,0 +1,136 @@
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
import dev.caskeleton.application.notification.platform.api.error.FailureCategory;
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureCode;
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureDescriptor;
import dev.caskeleton.application.notification.platform.api.error.ProviderUnavailableException;
import dev.caskeleton.application.notification.platform.provider.NotificationProviderAdapter;
import dev.caskeleton.application.notification.platform.provider.ProviderProfileSnapshot;
import dev.caskeleton.application.notification.platform.provider.ProviderRuntimeState;
import java.util.Objects;
import java.util.Optional;
import java.util.concurrent.atomic.AtomicReference;
/**
* One immutable credential generation of a provider.
*
* <p>Generations are replaced, never mutated. Rotating a key by editing a live client would leave
* in-flight calls half-way between two credentials; replacing the whole runtime and letting the old
* one drain keeps every attempt attributable to exactly one generation.
*
* <p>An authentication failure moves the whole runtime, not the message. One expired credential
* multiplied by a queue of notifications is a self-inflicted outage, so the route opens once and
* raises an operational alert instead.
*/
public final class ProviderRuntime {
private final ProviderProfileSnapshot profile;
private final NotificationProviderAdapter adapter;
private final ProviderAttemptLimiter limiter;
private final AtomicReference<ProviderRuntimeState> state;
private final AtomicReference<String> unhealthyReason = new AtomicReference<>();
public ProviderRuntime(
ProviderProfileSnapshot profile,
NotificationProviderAdapter adapter,
ProviderAttemptLimiter limiter) {
this.profile = Objects.requireNonNull(profile, "profile");
this.adapter = Objects.requireNonNull(adapter, "adapter");
this.limiter = Objects.requireNonNull(limiter, "limiter");
this.state = new AtomicReference<>(ProviderRuntimeState.HEALTHY);
}
/** Profile snapshot including the credential generation. */
public ProviderProfileSnapshot profile() {
return profile;
}
/** Credential generation of this runtime. */
public long generation() {
return profile.credentialGeneration();
}
/** Provider adapter bound to this generation. */
public NotificationProviderAdapter adapter() {
return adapter;
}
/** Current health. */
public ProviderRuntimeState state() {
return state.get();
}
/** Why the runtime is unhealthy, if it is. */
public Optional<String> unhealthyReason() {
return Optional.ofNullable(unhealthyReason.get());
}
/** Attempts currently in flight on this generation. */
public int activeAttempts() {
return limiter.activeAttempts();
}
/**
* Acquire a permit for one attempt.
*
* <p>The health check happens before the limiter, so a disabled or failed provider never consumes
* a token it cannot use.
*/
public AttemptPermit acquireAttempt() {
ProviderRuntimeState current = state.get();
if (!current.admitsNewAttempts()) {
throw new ProviderUnavailableException(
NotificationFailureDescriptor.preDispatch(
NotificationFailureCode.PROVIDER_UNAVAILABLE,
current == ProviderRuntimeState.AUTHENTICATION_FAILED
? FailureCategory.AUTHENTICATION
: FailureCategory.CAPACITY_REJECTED));
}
limiter.acquire();
return new LimiterPermit(profile.credentialGeneration(), limiter);
}
/** Mark the credential as rejected by the provider. */
public void markAuthenticationFailed(String reasonCode) {
unhealthyReason.set(Objects.requireNonNull(reasonCode, "reasonCode"));
state.set(ProviderRuntimeState.AUTHENTICATION_FAILED);
}
/** Mark the provider as rate limited. */
public void markThrottled() {
state.compareAndSet(ProviderRuntimeState.HEALTHY, ProviderRuntimeState.THROTTLED);
}
/** Mark the provider as degraded but still usable. */
public void markDegraded(String reasonCode) {
unhealthyReason.set(reasonCode);
state.compareAndSet(ProviderRuntimeState.HEALTHY, ProviderRuntimeState.DEGRADED);
}
/** Return to healthy after a successful attempt. */
public void markHealthy() {
unhealthyReason.set(null);
state.compareAndSet(ProviderRuntimeState.THROTTLED, ProviderRuntimeState.HEALTHY);
state.compareAndSet(ProviderRuntimeState.DEGRADED, ProviderRuntimeState.HEALTHY);
}
/** Stop admitting new attempts; in-flight attempts finish. */
public void markDraining() {
state.set(ProviderRuntimeState.DRAINING);
}
/** Operator disable. */
public void markDisabled() {
state.set(ProviderRuntimeState.DISABLED);
}
/** A permit that releases exactly one limiter slot. */
private record LimiterPermit(long generation, ProviderAttemptLimiter limiter)
implements AttemptPermit {
@Override
public void close() {
limiter.release();
}
}
}
@@ -0,0 +1,78 @@
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
import dev.caskeleton.application.notification.platform.api.ProviderProfileId;
import dev.caskeleton.application.notification.platform.provider.ProviderRuntimeState;
import java.util.List;
import java.util.Map;
import java.util.Objects;
import java.util.Optional;
import java.util.concurrent.ConcurrentHashMap;
import java.util.concurrent.CopyOnWriteArrayList;
/** Holds the current generation of every provider profile plus the generations still draining. */
public final class ProviderRuntimeRegistry {
private final Map<ProviderProfileId, ProviderRuntime> current = new ConcurrentHashMap<>();
private final Map<ProviderProfileId, CopyOnWriteArrayList<ProviderRuntime>> draining =
new ConcurrentHashMap<>();
/** Register the first generation of a profile. */
public void register(ProviderRuntime runtime) {
Objects.requireNonNull(runtime, "runtime");
current.put(runtime.profile().profileId(), runtime);
}
/** Current generation, or a configuration failure when the profile is unknown. */
public ProviderRuntime current(ProviderProfileId profileId) {
ProviderRuntime runtime = current.get(profileId);
if (runtime == null) {
throw new IllegalStateException("no provider runtime registered for the profile");
}
return runtime;
}
/** Current generation if registered. */
public Optional<ProviderRuntime> find(ProviderProfileId profileId) {
return Optional.ofNullable(current.get(profileId));
}
/**
* Swap in a new generation and start draining the old one.
*
* <p>New dispatches immediately use the new generation while the previous one finishes what it
* already started, which is what makes a credential rotation invisible to callers.
*/
public Optional<ProviderRuntime> replace(ProviderRuntime replacement) {
Objects.requireNonNull(replacement, "replacement");
ProviderProfileId profileId = replacement.profile().profileId();
ProviderRuntime previous = current.put(profileId, replacement);
if (previous != null) {
previous.markDraining();
draining.computeIfAbsent(profileId, key -> new CopyOnWriteArrayList<>()).add(previous);
forgetIfDrained(profileId);
}
return Optional.ofNullable(previous);
}
/** Generations that are draining and still have work in flight. */
public List<ProviderRuntime> drainingGenerations(ProviderProfileId profileId) {
forgetIfDrained(profileId);
return List.copyOf(draining.getOrDefault(profileId, new CopyOnWriteArrayList<>()));
}
/** Health of the current generation. */
public ProviderRuntimeState state(ProviderProfileId profileId) {
return current(profileId).state();
}
private void forgetIfDrained(ProviderProfileId profileId) {
CopyOnWriteArrayList<ProviderRuntime> generations = draining.get(profileId);
if (generations == null) {
return;
}
generations.removeIf(runtime -> runtime.activeAttempts() == 0);
if (generations.isEmpty()) {
draining.remove(profileId);
}
}
}
@@ -0,0 +1,69 @@
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
import dev.caskeleton.application.notification.platform.api.ProviderProfileId;
import dev.caskeleton.application.notification.platform.observation.NotificationAuditEvent;
import dev.caskeleton.application.notification.platform.observation.NotificationAuditPort;
import java.time.Clock;
import java.time.Duration;
import java.util.Map;
import java.util.Objects;
import java.util.Optional;
/**
* Credential and certificate rotation.
*
* <p>The candidate is probed <em>before</em> the swap. Validating after cutover would mean a typo
* in a rotated secret takes the provider down and only then tells anyone; validating first makes a
* bad candidate a no-op that leaves the working generation in place.
*
* <p>Only the generation and key id reach the audit trail — never the credential material itself.
*/
public final class ProviderRuntimeRotator {
private final ProviderRuntimeRegistry registry;
private final CredentialProbe probe;
private final RuntimeDrainCoordinator drainCoordinator;
private final NotificationAuditPort audit;
private final Clock clock;
private final Duration drainTimeout;
public ProviderRuntimeRotator(
ProviderRuntimeRegistry registry,
CredentialProbe probe,
RuntimeDrainCoordinator drainCoordinator,
NotificationAuditPort audit,
Clock clock,
Duration drainTimeout) {
this.registry = Objects.requireNonNull(registry, "registry");
this.probe = Objects.requireNonNull(probe, "probe");
this.drainCoordinator = Objects.requireNonNull(drainCoordinator, "drainCoordinator");
this.audit = Objects.requireNonNull(audit, "audit");
this.clock = Objects.requireNonNull(clock, "clock");
this.drainTimeout = Objects.requireNonNull(drainTimeout, "drainTimeout");
}
/** Cut over to a new credential generation. */
public void rotate(ProviderProfileId profileId, ProviderRuntime candidate) {
Objects.requireNonNull(profileId, "profileId");
Objects.requireNonNull(candidate, "candidate");
if (!candidate.profile().profileId().equals(profileId)) {
throw new IllegalArgumentException("candidate belongs to a different profile");
}
if (!probe.isUsable(candidate)) {
throw new CredentialValidationException();
}
Optional<ProviderRuntime> previous = registry.replace(candidate);
audit.record(
new NotificationAuditEvent(
"PROVIDER_CREDENTIAL_ROTATION",
"system",
Optional.of("ROTATION"),
Optional.empty(),
clock.instant(),
Map.of(
"providerProfile", profileId.value(),
"generation", Long.toString(candidate.generation()))));
previous.ifPresent(runtime -> drainCoordinator.drain(runtime, drainTimeout));
}
}
@@ -0,0 +1,76 @@
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
import dev.caskeleton.application.notification.platform.api.ProviderProfileId;
import dev.caskeleton.application.notification.platform.dispatch.ProviderDispatchGatewayPort;
import dev.caskeleton.application.notification.platform.provider.ProviderProfileSnapshot;
import dev.caskeleton.application.notification.platform.provider.ProviderRuntimeState;
import dev.caskeleton.application.notification.platform.provider.ProviderSubmission;
import dev.caskeleton.application.notification.platform.provider.ProviderSubmissionResult;
import java.util.Objects;
import java.util.concurrent.CompletionException;
/**
* The single outbound call, wrapped in a permit.
*
* <p>The permit is acquired before the call and released in a finally, so a provider that hangs
* consumes exactly one slot and a burst queues rather than exhausting the pool.
*
* <p>A credential rejection is promoted to a runtime state change here rather than being left as a
* per-message failure — one expired key must open the route once, not produce one retry per queued
* notification.
*/
public final class RegistryProviderDispatchGateway implements ProviderDispatchGatewayPort {
private final ProviderRuntimeRegistry runtimes;
public RegistryProviderDispatchGateway(ProviderRuntimeRegistry runtimes) {
this.runtimes = Objects.requireNonNull(runtimes, "runtimes");
}
@Override
public ProviderProfileSnapshot profile(ProviderProfileId profileId) {
return runtimes.current(profileId).profile();
}
@Override
public ProviderRuntimeState state(ProviderProfileId profileId) {
return runtimes.state(profileId);
}
@Override
public ProviderSubmissionResult submit(ProviderSubmission submission) {
Objects.requireNonNull(submission, "submission");
ProviderRuntime runtime = runtimes.current(submission.profile().profileId());
try (AttemptPermit permit = runtime.acquireAttempt()) {
ProviderSubmissionResult result =
runtime.adapter().submit(submission).toCompletableFuture().join();
applyHealth(runtime, result);
return result;
} catch (CompletionException failure) {
// Unwrapped so the dispatcher classifies the real cause rather than the future's wrapper.
Throwable cause = failure.getCause() == null ? failure : failure.getCause();
throw cause instanceof RuntimeException runtimeFailure
? runtimeFailure
: new IllegalStateException("provider submission failed", cause);
}
}
private static void applyHealth(ProviderRuntime runtime, ProviderSubmissionResult result) {
result
.failure()
.ifPresentOrElse(
failure -> {
switch (failure.category()) {
case AUTHENTICATION, AUTHORIZATION ->
runtime.markAuthenticationFailed(failure.code());
case THROTTLED -> runtime.markThrottled();
case TRANSIENT_PROVIDER -> runtime.markDegraded(failure.code());
default -> {
// A message-level failure says nothing about the provider's health.
}
}
},
runtime::markHealthy);
}
}
@@ -0,0 +1,46 @@
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
import java.time.Duration;
import java.util.Objects;
/**
* Waits for a replaced generation to finish its in-flight attempts.
*
* <p>The deadline comes from {@link System#nanoTime()}, not from the injectable clock. A drain
* timeout is a real elapsed-time budget: driving it from a test clock that never advances turns the
* loop into a hang, and driving it from a wall clock makes it sensitive to time adjustments.
*
* <p>Draining is bounded on purpose. A provider that never answers must not hold a credential
* rotation open forever, so after the timeout the generation is abandoned and its attempts follow
* the normal ambiguity and reconciliation path rather than being cancelled mid-flight.
*/
public final class RuntimeDrainCoordinator {
private final Duration pollInterval;
public RuntimeDrainCoordinator(Duration pollInterval) {
this.pollInterval = Objects.requireNonNull(pollInterval, "pollInterval");
if (pollInterval.isNegative() || pollInterval.isZero()) {
throw new IllegalArgumentException("pollInterval");
}
}
/** Drain a generation, returning whether it finished within the timeout. */
public boolean drain(ProviderRuntime runtime, Duration timeout) {
Objects.requireNonNull(runtime, "runtime");
Objects.requireNonNull(timeout, "timeout");
long deadlineNanos = System.nanoTime() + timeout.toNanos();
while (runtime.activeAttempts() > 0) {
if (System.nanoTime() - deadlineNanos >= 0) {
return false;
}
try {
Thread.sleep(pollInterval.toMillis());
} catch (InterruptedException interrupted) {
Thread.currentThread().interrupt();
return false;
}
}
return true;
}
}
@@ -0,0 +1,27 @@
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
import dev.caskeleton.application.notification.platform.api.TenantId;
import dev.caskeleton.application.notification.platform.dispatch.TenantContextPort;
import java.util.Objects;
/**
* Tenant context for a single-tenant deployment.
*
* <p>A multi-tenant deployment replaces this with a request-scoped implementation. It exists so
* that a single-tenant application still goes through the tenant boundary rather than around it —
* the store queries take a tenant either way, and a deployment that later becomes multi-tenant does
* not have to find every unscoped query.
*/
public final class SingleTenantContext implements TenantContextPort {
private final TenantId tenantId;
public SingleTenantContext(String tenantId) {
this.tenantId = new TenantId(Objects.requireNonNull(tenantId, "tenantId"));
}
@Override
public TenantId currentTenant() {
return tenantId;
}
}
@@ -0,0 +1,54 @@
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
import dev.caskeleton.application.notification.platform.dispatch.NotificationIdGeneratorPort;
import java.security.SecureRandom;
import java.time.Clock;
import java.util.Objects;
import java.util.UUID;
import java.util.concurrent.atomic.AtomicLong;
/**
* RFC 9562 UUIDv7.
*
* <p>Time-ordered rather than random because these identifiers are primary keys: a random UUID
* scatters inserts across the whole index, and a notification table takes the highest insert rate
* in the platform.
*
* <p>The monotonic counter guards the case two identifiers are requested inside the same
* millisecond, so ordering holds even under a burst.
*/
public final class UuidV7Generator implements NotificationIdGeneratorPort {
private static final long VERSION_7 = 0x7000L;
private static final long VARIANT_RFC = 0x8000000000000000L;
private final Clock clock;
private final SecureRandom random;
private final AtomicLong lastMillis = new AtomicLong();
private final AtomicLong sequence = new AtomicLong();
public UuidV7Generator(Clock clock) {
this(clock, new SecureRandom());
}
UuidV7Generator(Clock clock, SecureRandom random) {
this.clock = Objects.requireNonNull(clock, "clock");
this.random = Objects.requireNonNull(random, "random");
}
@Override
public UUID nextId() {
long millis = clock.millis();
long previous = lastMillis.getAndSet(millis);
long counter = millis == previous ? sequence.incrementAndGet() : sequence.updateAndGet(x -> 0L);
long high = (millis & 0xFFFFFFFFFFFFL) << 16;
high |= VERSION_7;
high |= counter & 0x0FFFL;
long low = random.nextLong();
low &= 0x3FFFFFFFFFFFFFFFL;
low |= VARIANT_RFC;
return new UUID(high, low);
}
}
@@ -0,0 +1,60 @@
package dev.caskeleton.adapter.outbound.notification.platform.observation;
import dev.caskeleton.application.notification.platform.api.ProviderProfileId;
import dev.caskeleton.application.notification.platform.observation.NotificationAuditEvent;
import dev.caskeleton.application.notification.platform.observation.NotificationAuditPort;
import dev.caskeleton.application.notification.platform.observation.NotificationSecurityAuditPort;
import java.util.Objects;
import java.util.TreeMap;
import org.slf4j.Logger;
import org.slf4j.LoggerFactory;
/**
* Audit sink on a dedicated logger.
*
* <p>Separate from the metrics logger because audit has different retention: a metric may be
* sampled away, while "who lifted this suppression, and why" has to survive.
*
* <p>A rejected callback signature is a security event, not a provider event, so it is recorded
* here and never in the ledger — otherwise anyone who can reach the endpoint could fill a delivery
* history with noise.
*/
public final class LoggingNotificationAudit
implements NotificationAuditPort, NotificationSecurityAuditPort {
private static final Logger audit = LoggerFactory.getLogger("notification.audit");
private static final Logger security = LoggerFactory.getLogger("notification.security");
@Override
public void record(NotificationAuditEvent event) {
Objects.requireNonNull(event, "event");
audit.info(
"action={} actor={} reason={} operationId={} occurredAt={} attributes={}",
event.action(),
event.actorRef(),
event.reasonCode().orElse("-"),
event.operationId().orElse("-"),
event.occurredAt(),
new TreeMap<>(event.boundedAttributes()));
}
@Override
public void callbackSignatureRejected(ProviderProfileId profileId, String reasonCode) {
Objects.requireNonNull(profileId, "profileId");
// The payload is deliberately absent: a forged callback must not get its content into the log
// just by being rejected.
security.warn(
"event=callback_signature_rejected providerProfile={} reason={}",
profileId.value(),
reasonCode);
}
@Override
public void callbackRejectedByLimit(ProviderProfileId profileId, String reasonCode) {
Objects.requireNonNull(profileId, "profileId");
security.warn(
"event=callback_rejected_by_limit providerProfile={} reason={}",
profileId.value(),
reasonCode);
}
}
@@ -0,0 +1,52 @@
package dev.caskeleton.adapter.outbound.notification.platform.observation;
import dev.caskeleton.application.notification.platform.observation.CardinalityGuard;
import dev.caskeleton.application.notification.platform.observation.NotificationMetricsPort;
import java.time.Duration;
import java.util.Map;
import java.util.Objects;
import java.util.TreeMap;
import org.slf4j.Logger;
import org.slf4j.LoggerFactory;
/**
* Structured-log metrics sink.
*
* <p>Every tag map passes the cardinality guard before it is emitted, so a stray notification id
* fails here rather than after it has already multiplied a time series into millions of them.
*
* <p>A Micrometer-backed implementation belongs in the composition root, which owns the registry;
* this one keeps the platform usable — and its tag discipline enforced — without one.
*/
public final class LoggingNotificationMetrics implements NotificationMetricsPort {
private static final Logger log = LoggerFactory.getLogger("notification.metrics");
private final CardinalityGuard guard;
public LoggingNotificationMetrics(CardinalityGuard guard) {
this.guard = Objects.requireNonNull(guard, "guard");
}
@Override
public void increment(String metricName, Map<String, String> tags) {
guard.validate(tags);
log.info("metric={} kind=counter tags={}", metricName, ordered(tags));
}
@Override
public void record(String metricName, Map<String, String> tags, Duration value) {
guard.validate(tags);
log.info("metric={} kind=timer millis={} tags={}", metricName, value.toMillis(), ordered(tags));
}
@Override
public void gauge(String metricName, Map<String, String> tags, double value) {
guard.validate(tags);
log.info("metric={} kind=gauge value={} tags={}", metricName, value, ordered(tags));
}
private static Map<String, String> ordered(Map<String, String> tags) {
return new TreeMap<>(tags);
}
}
@@ -0,0 +1,57 @@
package dev.caskeleton.adapter.outbound.notification.platform.observation;
import dev.caskeleton.adapter.outbound.notification.platform.dispatch.ProviderRuntimeRegistry;
import dev.caskeleton.application.notification.platform.api.ProviderProfileId;
import dev.caskeleton.application.notification.platform.provider.ProviderRuntimeState;
import java.util.ArrayList;
import java.util.List;
import java.util.Map;
import java.util.Objects;
/**
* Builds the operational snapshot.
*
* <p>A provider whose credentials were rejected reports unhealthy even though the process is fine:
* that is exactly the condition an operator needs paged on, and it is invisible from process-level
* health.
*/
public final class NotificationHealthReporter {
private final ProviderRuntimeRegistry runtimes;
private final List<ProviderProfileId> monitoredProfiles;
public NotificationHealthReporter(
ProviderRuntimeRegistry runtimes, List<ProviderProfileId> monitoredProfiles) {
this.runtimes = Objects.requireNonNull(runtimes, "runtimes");
this.monitoredProfiles = List.copyOf(Objects.requireNonNull(monitoredProfiles, "profiles"));
}
/** Current snapshot. */
public NotificationHealthSnapshot snapshot() {
List<NotificationHealthSnapshot.ProviderHealth> providers = new ArrayList<>();
boolean healthy = true;
for (ProviderProfileId profileId : monitoredProfiles) {
var runtime = runtimes.find(profileId);
if (runtime.isEmpty()) {
healthy = false;
providers.add(
new NotificationHealthSnapshot.ProviderHealth(profileId.value(), "UNREGISTERED", 0, 0));
continue;
}
ProviderRuntimeState state = runtime.get().state();
if (state == ProviderRuntimeState.AUTHENTICATION_FAILED
|| state == ProviderRuntimeState.DISABLED) {
healthy = false;
}
providers.add(
new NotificationHealthSnapshot.ProviderHealth(
profileId.value(),
state.name(),
runtime.get().generation(),
runtime.get().activeAttempts()));
}
return new NotificationHealthSnapshot(healthy, providers, Map.of());
}
}
@@ -0,0 +1,31 @@
package dev.caskeleton.adapter.outbound.notification.platform.observation;
import java.util.List;
import java.util.Map;
import java.util.Objects;
/**
* Operational view of the platform.
*
* <p>Provider states, credential generations and queue age — nothing else. A health endpoint is one
* of the least protected surfaces an application exposes, so a sender address or a credential
* reference appearing here would be a leak with a wide audience.
*/
public record NotificationHealthSnapshot(
boolean healthy, List<ProviderHealth> providers, Map<String, Long> queue) {
public NotificationHealthSnapshot {
providers = List.copyOf(Objects.requireNonNull(providers, "providers"));
queue = Map.copyOf(Objects.requireNonNull(queue, "queue"));
}
/** One provider runtime's state. */
public record ProviderHealth(
String profileId, String state, long credentialGeneration, int activeAttempts) {
public ProviderHealth {
Objects.requireNonNull(profileId, "profileId");
Objects.requireNonNull(state, "state");
}
}
}
@@ -0,0 +1,100 @@
package dev.caskeleton.adapter.outbound.notification.platform.provider;
import dev.caskeleton.adapter.outbound.notification.platform.provider.http.NotificationHttpTransportException;
import dev.caskeleton.application.notification.platform.api.error.FailureCategory;
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureCode;
import dev.caskeleton.application.notification.platform.provider.ProviderExecutionEvidence;
import dev.caskeleton.application.notification.platform.provider.ProviderFailure;
import dev.caskeleton.application.notification.platform.provider.ProviderSubmissionResult;
import java.time.Duration;
import java.util.Optional;
/**
* Shared translation from a transport failure into an evidence-carrying result.
*
* <p>Every HTTP provider adapter routes its transport failures through here, so the rule that
* "committed body plus no response equals ambiguous" is written once rather than re-derived per
* provider.
*/
public final class ProviderResults {
private ProviderResults() {}
/** Classify a transport failure. */
public static ProviderSubmissionResult fromTransport(
NotificationHttpTransportException failure, Duration elapsed) {
if (failure.requestBodyCommitted()) {
return ProviderSubmissionResult.ambiguous(
new ProviderFailure(
NotificationFailureCode.PROVIDER_RESPONSE_LOST,
FailureCategory.AMBIGUOUS_SUBMISSION,
false,
Optional.empty(),
Optional.of(failure.reasonCode())),
ProviderExecutionEvidence.responseLost(),
elapsed);
}
return ProviderSubmissionResult.notSubmitted(
new ProviderFailure(
NotificationFailureCode.PROVIDER_TRANSIENT_FAILURE,
FailureCategory.TRANSIENT_PROVIDER,
true,
Optional.empty(),
Optional.of(failure.reasonCode())),
elapsed);
}
/** Classify an HTTP status that is not provider-specific. */
public static ProviderFailure fromStatus(int statusCode, Optional<Duration> retryAfter) {
if (statusCode == 429) {
return new ProviderFailure(
NotificationFailureCode.PROVIDER_THROTTLED,
FailureCategory.THROTTLED,
true,
retryAfter,
Optional.of(Integer.toString(statusCode)));
}
if (statusCode == 401) {
return new ProviderFailure(
NotificationFailureCode.PROVIDER_AUTHENTICATION_FAILED,
FailureCategory.AUTHENTICATION,
false,
Optional.empty(),
Optional.of("401"));
}
if (statusCode == 403) {
return new ProviderFailure(
NotificationFailureCode.PROVIDER_AUTHORIZATION_FAILED,
FailureCategory.AUTHORIZATION,
false,
Optional.empty(),
Optional.of("403"));
}
if (statusCode >= 500) {
return new ProviderFailure(
NotificationFailureCode.PROVIDER_TRANSIENT_FAILURE,
FailureCategory.TRANSIENT_PROVIDER,
true,
retryAfter,
Optional.of(Integer.toString(statusCode)));
}
return new ProviderFailure(
NotificationFailureCode.PROVIDER_PERMANENT_FAILURE,
FailureCategory.PERMANENT_PROVIDER,
false,
Optional.empty(),
Optional.of(Integer.toString(statusCode)));
}
/** Parse a {@code Retry-After} header expressed in seconds. */
public static Optional<Duration> retryAfter(Optional<String> headerValue) {
return headerValue.flatMap(
value -> {
try {
return Optional.of(Duration.ofSeconds(Long.parseLong(value.trim())));
} catch (NumberFormatException notSeconds) {
return Optional.empty();
}
});
}
}
@@ -0,0 +1,29 @@
package dev.caskeleton.adapter.outbound.notification.platform.provider;
import dev.caskeleton.application.notification.platform.api.content.AttachmentRef;
import dev.caskeleton.application.notification.platform.api.error.AttachmentUnavailableException;
import dev.caskeleton.application.notification.platform.api.error.FailureCategory;
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureCode;
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureDescriptor;
import dev.caskeleton.application.notification.platform.provider.AttachmentAccessContext;
import dev.caskeleton.application.notification.platform.provider.AttachmentResolver;
import dev.caskeleton.application.notification.platform.provider.ResolvedAttachment;
/**
* The resolver used when no attachment source is wired.
*
* <p>It refuses rather than returning an empty stream. Sending a mail whose attachment is silently
* missing is worse than not sending it: the recipient is told something is attached and it is not.
*
* <p>The composition root replaces this with a file-server or object-storage backed resolver; both
* leaves are visible there, and neither is reachable from this one.
*/
public final class UnconfiguredAttachmentResolver implements AttachmentResolver {
@Override
public ResolvedAttachment resolve(AttachmentRef reference, AttachmentAccessContext context) {
throw new AttachmentUnavailableException(
NotificationFailureDescriptor.preDispatch(
NotificationFailureCode.ATTACHMENT_UNAVAILABLE, FailureCategory.INVALID_PAYLOAD));
}
}
@@ -0,0 +1,67 @@
package dev.caskeleton.adapter.outbound.notification.platform.provider.apns;
import dev.caskeleton.adapter.outbound.notification.platform.provider.ProviderResults;
import dev.caskeleton.adapter.outbound.notification.platform.provider.http.NotificationHttpResponse;
import dev.caskeleton.adapter.outbound.notification.platform.template.NotificationJsonMapper;
import dev.caskeleton.application.notification.platform.api.error.FailureCategory;
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureCode;
import dev.caskeleton.application.notification.platform.provider.ProviderFailure;
import java.util.Optional;
import java.util.Set;
/** Maps APNs reason strings onto the stable failure vocabulary. */
public final class ApnsFailureClassifier {
private static final Set<String> INVALID_TOKEN_REASONS =
Set.of("BadDeviceToken", "Unregistered", "DeviceTokenNotForTopic");
private static final Set<String> CONFIGURATION_REASONS =
Set.of("BadTopic", "TopicDisallowed", "BadCertificateEnvironment", "InvalidPushType");
/** Classify a non-2xx APNs response. */
public ProviderFailure classify(NotificationHttpResponse response) {
Optional<String> reason = reason(response);
if (reason.filter(INVALID_TOKEN_REASONS::contains).isPresent()) {
return new ProviderFailure(
NotificationFailureCode.CONTACT_POINT_INVALID,
FailureCategory.INVALID_RECIPIENT,
false,
Optional.empty(),
reason);
}
if (reason.filter(CONFIGURATION_REASONS::contains).isPresent()) {
return new ProviderFailure(
NotificationFailureCode.PROVIDER_CONFIGURATION_INVALID,
FailureCategory.AUTHORIZATION,
false,
Optional.empty(),
reason);
}
if (reason.filter("ExpiredProviderToken"::equals).isPresent()) {
return new ProviderFailure(
NotificationFailureCode.PROVIDER_AUTHENTICATION_FAILED,
FailureCategory.AUTHENTICATION,
false,
Optional.empty(),
reason);
}
if (reason.filter("TooManyRequests"::equals).isPresent()) {
return new ProviderFailure(
NotificationFailureCode.PROVIDER_THROTTLED,
FailureCategory.THROTTLED,
true,
Optional.empty(),
reason);
}
return ProviderResults.fromStatus(response.statusCode(), Optional.empty());
}
private static Optional<String> reason(NotificationHttpResponse response) {
try {
var node = NotificationJsonMapper.mapper().readTree(response.bodyAsString());
var reason = node.get("reason");
return reason == null || reason.isNull() ? Optional.empty() : Optional.of(reason.asString());
} catch (RuntimeException unparseable) {
return Optional.empty();
}
}
}
@@ -0,0 +1,100 @@
package dev.caskeleton.adapter.outbound.notification.platform.provider.apns;
import dev.caskeleton.adapter.outbound.notification.platform.provider.ProviderResults;
import dev.caskeleton.adapter.outbound.notification.platform.provider.http.NotificationHttpGateway;
import dev.caskeleton.adapter.outbound.notification.platform.provider.http.NotificationHttpResponse;
import dev.caskeleton.adapter.outbound.notification.platform.provider.http.NotificationHttpTransportException;
import dev.caskeleton.application.notification.platform.api.ProviderId;
import dev.caskeleton.application.notification.platform.api.routing.Channel;
import dev.caskeleton.application.notification.platform.provider.NotificationProviderAdapter;
import dev.caskeleton.application.notification.platform.provider.ProviderCapabilities;
import dev.caskeleton.application.notification.platform.provider.ProviderSubmission;
import dev.caskeleton.application.notification.platform.provider.ProviderSubmissionResult;
import dev.caskeleton.application.notification.platform.security.AccessContext;
import dev.caskeleton.application.notification.platform.security.ContactPointProtector;
import java.time.Duration;
import java.util.Objects;
import java.util.Set;
import java.util.concurrent.CompletableFuture;
import java.util.concurrent.CompletionStage;
import java.util.function.Supplier;
/**
* APNs adapter.
*
* <p>A 2xx is acceptance. Apple documents that an accepted notification may be delivered, stored or
* discarded, and that ordering is not guaranteed, so this adapter never produces a delivery outcome
* and the platform never uses APNs as an ordered event transport.
*/
public final class ApnsNotificationProviderAdapter implements NotificationProviderAdapter {
private static final ProviderId PROVIDER_ID = new ProviderId("apns");
private final NotificationHttpGateway gateway;
private final ApnsRequestMapper mapper;
private final ApnsFailureClassifier classifier;
private final ContactPointProtector protector;
private final Supplier<String> authorizationSupplier;
public ApnsNotificationProviderAdapter(
NotificationHttpGateway gateway,
ApnsRequestMapper mapper,
ApnsFailureClassifier classifier,
ContactPointProtector protector,
Supplier<String> authorizationSupplier,
ApnsProviderProperties properties) {
// The profile is required at construction so a missing topic or environment fails at wiring
// time, but it is never exposed: a public accessor would leak an adapter type across the port.
Objects.requireNonNull(properties, "properties");
this.gateway = Objects.requireNonNull(gateway, "gateway");
this.mapper = Objects.requireNonNull(mapper, "mapper");
this.classifier = Objects.requireNonNull(classifier, "classifier");
this.protector = Objects.requireNonNull(protector, "protector");
this.authorizationSupplier =
Objects.requireNonNull(authorizationSupplier, "authorizationSupplier");
}
@Override
public ProviderId providerId() {
return PROVIDER_ID;
}
@Override
public Set<Channel> channels() {
return Set.of(Channel.PUSH);
}
@Override
public ProviderCapabilities capabilities() {
return new ProviderCapabilities(
false, false, false, false, false, false, false, true, 1, 4096L, Duration.ofDays(30));
}
@Override
public CompletionStage<ProviderSubmissionResult> submit(ProviderSubmission submission) {
Objects.requireNonNull(submission, "submission");
return CompletableFuture.completedFuture(send(submission));
}
private ProviderSubmissionResult send(ProviderSubmission submission) {
long startedNanos = System.nanoTime();
var contactPoint =
protector.reveal(
submission.contactPoint(),
AccessContext.dispatch(submission.profile().profileId().value()));
var request = mapper.map(submission, contactPoint, authorizationSupplier.get());
try {
NotificationHttpResponse response = gateway.exchange(request);
Duration elapsed = Duration.ofNanos(System.nanoTime() - startedNanos);
if (response.isSuccessful()) {
return ProviderSubmissionResult.accepted(
response.header("apns-id").orElse(null), "Accepted", elapsed);
}
return ProviderSubmissionResult.rejected(classifier.classify(response), elapsed);
} catch (NotificationHttpTransportException transportFailure) {
return ProviderResults.fromTransport(
transportFailure, Duration.ofNanos(System.nanoTime() - startedNanos));
}
}
}
@@ -0,0 +1,38 @@
package dev.caskeleton.adapter.outbound.notification.platform.provider.apns;
import dev.caskeleton.application.notification.platform.contact.ApnsEnvironment;
import java.net.URI;
import java.time.Duration;
import java.util.Objects;
import java.util.Set;
/**
* APNs profile.
*
* <p>Environment and topic are required. A sandbox token sent to the production host is a silent
* non-delivery, so the pairing is checked before the call rather than diagnosed afterwards.
*/
public record ApnsProviderProperties(
URI endpoint,
String topic,
ApnsEnvironment environment,
Set<String> allowedPushTypes,
Duration timeout) {
public ApnsProviderProperties {
Objects.requireNonNull(endpoint, "endpoint");
Objects.requireNonNull(topic, "topic");
Objects.requireNonNull(environment, "environment");
allowedPushTypes = Set.copyOf(Objects.requireNonNull(allowedPushTypes, "allowedPushTypes"));
Objects.requireNonNull(timeout, "timeout");
if (topic.isBlank()) {
throw new IllegalArgumentException("topic");
}
if (allowedPushTypes.isEmpty()) {
throw new IllegalArgumentException("allowedPushTypes must not be empty");
}
if (timeout.isNegative() || timeout.isZero()) {
throw new IllegalArgumentException("timeout must be positive and finite");
}
}
}
@@ -0,0 +1,96 @@
package dev.caskeleton.adapter.outbound.notification.platform.provider.apns;
import dev.caskeleton.adapter.outbound.notification.platform.provider.http.JdkNotificationHttpGateway;
import dev.caskeleton.adapter.outbound.notification.platform.provider.http.NotificationHttpRequest;
import dev.caskeleton.adapter.outbound.notification.platform.template.NotificationJsonMapper;
import dev.caskeleton.application.notification.platform.api.content.MobilePushContent;
import dev.caskeleton.application.notification.platform.api.error.FailureCategory;
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureCode;
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureDescriptor;
import dev.caskeleton.application.notification.platform.api.error.ProviderConfigurationException;
import dev.caskeleton.application.notification.platform.contact.ApnsDeviceToken;
import dev.caskeleton.application.notification.platform.contact.ContactPointValue;
import dev.caskeleton.application.notification.platform.provider.ProviderSubmission;
import java.net.URI;
import java.nio.charset.StandardCharsets;
import java.time.Clock;
import java.util.LinkedHashMap;
import java.util.Map;
import java.util.Objects;
/** Builds the APNs HTTP/2 request headers and payload. */
public final class ApnsRequestMapper {
private static final String DEFAULT_PUSH_TYPE = "alert";
private final ApnsProviderProperties properties;
private final Clock clock;
public ApnsRequestMapper(ApnsProviderProperties properties, Clock clock) {
this.properties = Objects.requireNonNull(properties, "properties");
this.clock = Objects.requireNonNull(clock, "clock");
}
/** Map one submission, rejecting an environment or push-type mismatch first. */
public NotificationHttpRequest map(
ProviderSubmission submission, ContactPointValue contactPoint, String authorization) {
Objects.requireNonNull(submission, "submission");
Objects.requireNonNull(authorization, "authorization");
if (!(contactPoint instanceof ApnsDeviceToken token)) {
throw new IllegalArgumentException("APNs requires an APNs device token");
}
if (token.environment() != properties.environment()) {
throw configurationFailure();
}
if (!(submission.content().content() instanceof MobilePushContent push)) {
throw new IllegalArgumentException("APNs requires mobile push content");
}
String pushType = DEFAULT_PUSH_TYPE;
if (!properties.allowedPushTypes().contains(pushType)) {
throw configurationFailure();
}
Map<String, Object> aps = new LinkedHashMap<>();
aps.put("alert", Map.of("title", push.title(), "body", push.body()));
push.presentation().sound().ifPresent(sound -> aps.put("sound", sound));
push.presentation().badge().ifPresent(badge -> aps.put("badge", badge));
Map<String, Object> payload = new LinkedHashMap<>();
payload.put("aps", aps);
payload.putAll(push.data());
Map<String, String> headers = new LinkedHashMap<>();
headers.put("authorization", authorization);
headers.put("apns-topic", properties.topic());
headers.put("apns-push-type", pushType);
headers.put("apns-priority", "10");
headers.put("apns-id", submission.attemptId().value().toString());
submission
.expiresAt()
.ifPresent(
expiry -> headers.put("apns-expiration", Long.toString(expiry.getEpochSecond())));
submission.collapse().ifPresent(spec -> headers.put("apns-collapse-id", spec.key()));
byte[] body =
NotificationJsonMapper.mapper()
.writeValueAsString(payload)
.getBytes(StandardCharsets.UTF_8);
return new NotificationHttpRequest(
"POST",
URI.create(properties.endpoint() + "/3/device/" + token.value()),
JdkNotificationHttpGateway.headers(headers),
body,
properties.timeout());
}
/** Current time, exposed so expiry mapping stays testable. */
public java.time.Instant now() {
return clock.instant();
}
private static ProviderConfigurationException configurationFailure() {
return new ProviderConfigurationException(
NotificationFailureDescriptor.preDispatch(
NotificationFailureCode.PROVIDER_CONFIGURATION_INVALID, FailureCategory.AUTHORIZATION));
}
}
@@ -0,0 +1,89 @@
package dev.caskeleton.adapter.outbound.notification.platform.provider.fcm;
import dev.caskeleton.application.notification.platform.api.error.FailureCategory;
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureCode;
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureDescriptor;
import dev.caskeleton.application.notification.platform.api.error.ProviderPayloadLimitException;
import dev.caskeleton.application.notification.platform.contact.ContactPointValue;
import dev.caskeleton.application.notification.platform.provider.ProviderSubmission;
import dev.caskeleton.application.notification.platform.provider.ProviderSubmissionResult;
import dev.caskeleton.application.notification.platform.security.AccessContext;
import dev.caskeleton.application.notification.platform.security.ContactPointProtector;
import java.time.Duration;
import java.util.ArrayList;
import java.util.List;
import java.util.Map;
import java.util.Objects;
import java.util.concurrent.CompletableFuture;
import java.util.concurrent.CompletionStage;
/**
* Batch submission that keeps per-recipient identity.
*
* <p>One transport call, many attempts. FCM returns a positional result per input, so a partial
* failure is decomposed back to the recipient that owns it; collapsing a batch into one shared
* outcome would mark four delivered recipients as failed because the fifth token was stale.
*/
public final class FcmBatchCoordinator {
private final FcmGateway gateway;
private final FcmMessageMapper messageMapper;
private final FcmTargetMapper targetMapper;
private final FcmFailureClassifier classifier;
private final ContactPointProtector protector;
private final FcmProviderProperties properties;
public FcmBatchCoordinator(
FcmGateway gateway,
FcmMessageMapper messageMapper,
FcmTargetMapper targetMapper,
FcmFailureClassifier classifier,
ContactPointProtector protector,
FcmProviderProperties properties) {
this.gateway = Objects.requireNonNull(gateway, "gateway");
this.messageMapper = Objects.requireNonNull(messageMapper, "messageMapper");
this.targetMapper = Objects.requireNonNull(targetMapper, "targetMapper");
this.classifier = Objects.requireNonNull(classifier, "classifier");
this.protector = Objects.requireNonNull(protector, "protector");
this.properties = Objects.requireNonNull(properties, "properties");
}
/** Submit a batch and return one result per input, in input order. */
public CompletionStage<List<ProviderSubmissionResult>> submit(
List<ProviderSubmission> submissions) {
Objects.requireNonNull(submissions, "submissions");
if (submissions.isEmpty()) {
return CompletableFuture.completedFuture(List.of());
}
if (submissions.size() > properties.maxBatchSize()) {
throw new ProviderPayloadLimitException(
NotificationFailureDescriptor.preDispatch(
NotificationFailureCode.PROVIDER_PAYLOAD_LIMIT, FailureCategory.INVALID_PAYLOAD));
}
long startedNanos = System.nanoTime();
List<Map<String, Object>> messages = new ArrayList<>(submissions.size());
for (ProviderSubmission submission : submissions) {
ContactPointValue value =
protector.reveal(
submission.contactPoint(),
AccessContext.dispatch(submission.profile().profileId().value()));
messages.add(messageMapper.map(submission, targetMapper.map(value)));
}
FcmBatchResult batch = gateway.sendBatch(messages);
if (batch.items().size() != submissions.size()) {
throw new IllegalStateException("FCM returned a result count that does not match the input");
}
Duration elapsed = Duration.ofNanos(System.nanoTime() - startedNanos);
List<ProviderSubmissionResult> results = new ArrayList<>(submissions.size());
for (FcmBatchResult.Item item : batch.items()) {
results.add(
item.success()
? ProviderSubmissionResult.accepted(item.messageId().orElse(null), "SUCCESS", elapsed)
: classifier.classify(item.errorCode().orElseThrow(), elapsed));
}
return CompletableFuture.completedFuture(List.copyOf(results));
}
}
@@ -0,0 +1,35 @@
package dev.caskeleton.adapter.outbound.notification.platform.provider.fcm;
import java.util.List;
import java.util.Objects;
import java.util.Optional;
/** Positional result of one FCM multicast call. */
public record FcmBatchResult(List<Item> items) {
public FcmBatchResult {
items = List.copyOf(Objects.requireNonNull(items, "items"));
}
/** One item result, aligned with the input index. */
public record Item(boolean success, Optional<String> messageId, Optional<String> errorCode) {
public Item {
Objects.requireNonNull(messageId, "messageId");
Objects.requireNonNull(errorCode, "errorCode");
if (success == errorCode.isPresent()) {
throw new IllegalArgumentException("an item is either a success or an error, never both");
}
}
/** Successful item. */
public static Item success(String messageId) {
return new Item(true, Optional.ofNullable(messageId), Optional.empty());
}
/** Failed item. */
public static Item failure(String errorCode) {
return new Item(false, Optional.empty(), Optional.of(errorCode));
}
}
}
@@ -0,0 +1,39 @@
package dev.caskeleton.adapter.outbound.notification.platform.provider.fcm;
import dev.caskeleton.application.notification.platform.api.ContactPointId;
import dev.caskeleton.application.notification.platform.api.TenantId;
import dev.caskeleton.application.notification.platform.contact.ContactPointStatus;
import dev.caskeleton.application.notification.platform.dispatch.ContactPointStorePort;
import java.util.Objects;
/**
* Applies FCM target lifecycle changes.
*
* <p>An {@code UNREGISTERED} response is the provider telling us the target no longer exists. Not
* acting on it means every future notification to that user spends a provider call to learn the
* same thing again.
*/
public final class FcmContactPointUpdater {
private final ContactPointStorePort contactPoints;
private final FcmFailureClassifier classifier;
public FcmContactPointUpdater(
ContactPointStorePort contactPoints, FcmFailureClassifier classifier) {
this.contactPoints = Objects.requireNonNull(contactPoints, "contactPoints");
this.classifier = Objects.requireNonNull(classifier, "classifier");
}
/** Invalidate the contact point when the error code says the target is gone. */
public boolean apply(TenantId tenantId, ContactPointId contactPointId, String errorCode) {
Objects.requireNonNull(tenantId, "tenantId");
Objects.requireNonNull(contactPointId, "contactPointId");
Objects.requireNonNull(errorCode, "errorCode");
if (!classifier.invalidatesContactPoint(errorCode)) {
return false;
}
contactPoints.updateStatus(
tenantId, contactPointId, ContactPointStatus.INVALID, "FCM_" + errorCode);
return true;
}
}
@@ -0,0 +1,76 @@
package dev.caskeleton.adapter.outbound.notification.platform.provider.fcm;
import dev.caskeleton.application.notification.platform.api.error.FailureCategory;
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureCode;
import dev.caskeleton.application.notification.platform.provider.ProviderFailure;
import dev.caskeleton.application.notification.platform.provider.ProviderSubmissionResult;
import java.time.Duration;
import java.util.Optional;
/**
* FCM error codes to the stable failure vocabulary.
*
* <p>{@code UNREGISTERED} is the one that must never be retried: the target is gone, and repeating
* the call cannot bring it back. It invalidates the contact point and lets routing fall back.
*/
public final class FcmFailureClassifier {
/** Classify one FCM error code. */
public ProviderSubmissionResult classify(String errorCode, Duration elapsed) {
return ProviderSubmissionResult.rejected(failure(errorCode), elapsed);
}
/** Failure for one FCM error code. */
public ProviderFailure failure(String errorCode) {
return switch (errorCode) {
case "UNREGISTERED", "INVALID_TOKEN" ->
ProviderFailure.of(
NotificationFailureCode.CONTACT_POINT_INVALID,
FailureCategory.INVALID_RECIPIENT,
false);
case "QUOTA_EXCEEDED" ->
ProviderFailure.of(
NotificationFailureCode.PROVIDER_THROTTLED, FailureCategory.THROTTLED, true);
case "UNAVAILABLE", "INTERNAL" ->
ProviderFailure.of(
NotificationFailureCode.PROVIDER_TRANSIENT_FAILURE,
FailureCategory.TRANSIENT_PROVIDER,
true);
case "INVALID_ARGUMENT" ->
ProviderFailure.of(
NotificationFailureCode.VALIDATION_FAILED, FailureCategory.INVALID_PAYLOAD, false);
case "THIRD_PARTY_AUTH_ERROR", "UNAUTHENTICATED" ->
ProviderFailure.of(
NotificationFailureCode.PROVIDER_AUTHENTICATION_FAILED,
FailureCategory.AUTHENTICATION,
false);
case "SENDER_ID_MISMATCH" ->
ProviderFailure.of(
NotificationFailureCode.PROVIDER_AUTHORIZATION_FAILED,
FailureCategory.AUTHORIZATION,
false);
default ->
ProviderFailure.of(
NotificationFailureCode.PROVIDER_PERMANENT_FAILURE,
FailureCategory.PERMANENT_PROVIDER,
false);
};
}
/** Whether an error code means the contact point should be invalidated. */
public boolean invalidatesContactPoint(String errorCode) {
return failure(errorCode).category() == FailureCategory.INVALID_RECIPIENT;
}
/** Retry hint, where FCM supplies one. */
public Optional<Duration> retryAfter(Optional<String> headerValue) {
return headerValue.flatMap(
value -> {
try {
return Optional.of(Duration.ofSeconds(Long.parseLong(value.trim())));
} catch (NumberFormatException notSeconds) {
return Optional.empty();
}
});
}
}
@@ -0,0 +1,12 @@
package dev.caskeleton.adapter.outbound.notification.platform.provider.fcm;
import java.util.List;
import java.util.Map;
/** The FCM transport seam, so batch decomposition can be tested without a live project. */
@FunctionalInterface
public interface FcmGateway {
/** Send a batch and return one positional result per message. */
FcmBatchResult sendBatch(List<Map<String, Object>> messages);
}
@@ -0,0 +1,72 @@
package dev.caskeleton.adapter.outbound.notification.platform.provider.fcm;
import dev.caskeleton.application.notification.platform.api.content.MobilePushContent;
import dev.caskeleton.application.notification.platform.provider.ProviderSubmission;
import java.time.Clock;
import java.time.Duration;
import java.util.LinkedHashMap;
import java.util.Map;
import java.util.Objects;
import java.util.Optional;
/**
* Builds the FCM message body.
*
* <p>TTL is the minimum of the remaining delivery deadline and the provider maximum. Sending the
* provider maximum when the notification expires in ninety seconds would let FCM keep retrying a
* message the platform has already given up on.
*/
public final class FcmMessageMapper {
private static final int MAX_PAYLOAD_BYTES = 4096;
private final FcmProviderProperties properties;
private final Clock clock;
public FcmMessageMapper(FcmProviderProperties properties, Clock clock) {
this.properties = Objects.requireNonNull(properties, "properties");
this.clock = Objects.requireNonNull(clock, "clock");
}
/** Message body for one submission. */
public Map<String, Object> map(ProviderSubmission submission, FcmWireTarget target) {
Objects.requireNonNull(submission, "submission");
Objects.requireNonNull(target, "target");
if (!(submission.content().content() instanceof MobilePushContent push)) {
throw new IllegalArgumentException("FCM requires mobile push content");
}
Map<String, Object> message = new LinkedHashMap<>();
if ("FID".equals(target.kind())) {
message.put("installation_id", target.value());
} else {
message.put("token", target.value());
}
message.put("notification", Map.of("title", push.title(), "body", push.body()));
if (!push.data().isEmpty()) {
message.put("data", push.data());
}
Map<String, Object> android = new LinkedHashMap<>();
android.put("ttl", ttl(submission).toSeconds() + "s");
submission.collapse().ifPresent(spec -> android.put("collapse_key", spec.key()));
message.put("android", android);
return Map.of("message", message);
}
/** Effective TTL for a submission. */
public Duration ttl(ProviderSubmission submission) {
Optional<Duration> remaining =
submission.expiresAt().map(expiry -> Duration.between(clock.instant(), expiry));
return remaining
.filter(value -> value.compareTo(properties.maxTtl()) < 0)
.filter(value -> !value.isNegative())
.orElse(properties.maxTtl());
}
/** Payload ceiling enforced before the provider call. */
public int maxPayloadBytes() {
return MAX_PAYLOAD_BYTES;
}
}
@@ -0,0 +1,73 @@
package dev.caskeleton.adapter.outbound.notification.platform.provider.fcm;
import dev.caskeleton.application.notification.platform.api.ProviderId;
import dev.caskeleton.application.notification.platform.api.routing.Channel;
import dev.caskeleton.application.notification.platform.provider.BatchNotificationProviderAdapter;
import dev.caskeleton.application.notification.platform.provider.ProviderCapabilities;
import dev.caskeleton.application.notification.platform.provider.ProviderSubmission;
import dev.caskeleton.application.notification.platform.provider.ProviderSubmissionResult;
import java.util.List;
import java.util.Objects;
import java.util.Set;
import java.util.concurrent.CompletionStage;
/**
* FCM adapter.
*
* <p>A successful send means FCM took the message. Firebase describes its own failures as handoff
* failures, which is the clearest statement that success is a handoff and not a device delivery, so
* the strongest evidence this adapter ever produces is {@code PROVIDER_ACCEPTED}.
*/
public final class FcmNotificationProviderAdapter implements BatchNotificationProviderAdapter {
private static final ProviderId PROVIDER_ID = new ProviderId("fcm");
private final FcmBatchCoordinator coordinator;
private final FcmProviderProperties properties;
public FcmNotificationProviderAdapter(
FcmBatchCoordinator coordinator, FcmProviderProperties properties) {
this.coordinator = Objects.requireNonNull(coordinator, "coordinator");
this.properties = Objects.requireNonNull(properties, "properties");
}
@Override
public ProviderId providerId() {
return PROVIDER_ID;
}
@Override
public Set<Channel> channels() {
return Set.of(Channel.PUSH);
}
@Override
public ProviderCapabilities capabilities() {
// deliveryReceipt is false: FCM has no server-side delivery receipt for ordinary sends, and
// claiming one would let the runtime plan a reconciliation that can never succeed.
return new ProviderCapabilities(
true,
false,
false,
false,
false,
false,
false,
true,
properties.maxBatchSize(),
4096L,
properties.maxTtl());
}
@Override
public CompletionStage<ProviderSubmissionResult> submit(ProviderSubmission submission) {
Objects.requireNonNull(submission, "submission");
return coordinator.submit(List.of(submission)).thenApply(results -> results.get(0));
}
@Override
public CompletionStage<List<ProviderSubmissionResult>> submitBatch(
List<ProviderSubmission> submissions) {
return coordinator.submit(submissions);
}
}
@@ -0,0 +1,35 @@
package dev.caskeleton.adapter.outbound.notification.platform.provider.fcm;
import java.net.URI;
import java.time.Duration;
import java.util.Objects;
/** FCM profile. Project and application identity are pinned so a target cannot cross projects. */
public record FcmProviderProperties(
URI endpoint,
String projectId,
String applicationId,
int maxBatchSize,
Duration maxTtl,
Duration timeout) {
/** The Admin SDK multicast ceiling. */
public static final int MAX_SUPPORTED_BATCH = 500;
public FcmProviderProperties {
Objects.requireNonNull(endpoint, "endpoint");
Objects.requireNonNull(projectId, "projectId");
Objects.requireNonNull(applicationId, "applicationId");
Objects.requireNonNull(maxTtl, "maxTtl");
Objects.requireNonNull(timeout, "timeout");
if (projectId.isBlank() || applicationId.isBlank()) {
throw new IllegalArgumentException("projectId and applicationId must not be blank");
}
if (maxBatchSize < 1 || maxBatchSize > MAX_SUPPORTED_BATCH) {
throw new IllegalArgumentException("maxBatchSize must be 1.." + MAX_SUPPORTED_BATCH);
}
if (timeout.isNegative() || timeout.isZero()) {
throw new IllegalArgumentException("timeout must be positive and finite");
}
}
}
@@ -0,0 +1,18 @@
package dev.caskeleton.adapter.outbound.notification.platform.provider.fcm;
import dev.caskeleton.application.notification.platform.contact.ContactPointValue;
import dev.caskeleton.application.notification.platform.contact.FcmInstallationId;
import dev.caskeleton.application.notification.platform.contact.LegacyFcmRegistrationToken;
/** Maps typed push targets to their FCM wire representation. */
public final class FcmTargetMapper {
/** Wire target for a contact point value. */
public FcmWireTarget map(ContactPointValue value) {
return switch (value) {
case FcmInstallationId fid -> new FcmWireTarget("FID", fid.value());
case LegacyFcmRegistrationToken token -> new FcmWireTarget("LEGACY_TOKEN", token.value());
default -> throw new IllegalArgumentException("FCM requires an FCM target");
};
}
}
@@ -0,0 +1,20 @@
package dev.caskeleton.adapter.outbound.notification.platform.provider.fcm;
import java.util.Objects;
/**
* A target in its wire form, with the kind kept explicit.
*
* <p>The kind is not cosmetic: an installation id and a legacy registration token go to different
* request fields, and flattening them would make a migration a runtime guess.
*/
public record FcmWireTarget(String kind, String value) {
public FcmWireTarget {
Objects.requireNonNull(kind, "kind");
Objects.requireNonNull(value, "value");
if (value.isBlank()) {
throw new IllegalArgumentException("value");
}
}
}
@@ -0,0 +1,103 @@
package dev.caskeleton.adapter.outbound.notification.platform.provider.http;
import java.io.IOException;
import java.net.http.HttpClient;
import java.net.http.HttpRequest;
import java.net.http.HttpResponse;
import java.net.http.HttpTimeoutException;
import java.time.Duration;
import java.util.List;
import java.util.Map;
import java.util.Objects;
import java.util.Set;
/**
* Default gateway on the JDK HTTP client.
*
* <p>Redirects are never followed. A provider redirect would move a signed, credential-bearing
* request to a host the profile never approved.
*
* <p>Timeout and connection-reset failures are translated into an explicit statement about whether
* the body was committed, because that single bit is what separates a safe retry from a duplicate.
*/
public final class JdkNotificationHttpGateway implements NotificationHttpGateway {
// Restricted headers the JDK client refuses to let a caller set.
private static final Set<String> RESTRICTED =
Set.of("connection", "content-length", "expect", "host", "upgrade");
private final HttpClient client;
public JdkNotificationHttpGateway(Duration connectTimeout) {
this(
HttpClient.newBuilder()
.followRedirects(HttpClient.Redirect.NEVER)
.connectTimeout(Objects.requireNonNull(connectTimeout, "connectTimeout"))
.build());
}
public JdkNotificationHttpGateway(HttpClient client) {
this.client = Objects.requireNonNull(client, "client");
}
@Override
public NotificationHttpResponse exchange(NotificationHttpRequest request) {
Objects.requireNonNull(request, "request");
HttpRequest.Builder builder =
HttpRequest.newBuilder(request.uri())
.timeout(request.timeout())
.method(request.method(), HttpRequest.BodyPublishers.ofByteArray(request.body()));
request
.headers()
.forEach(
(name, values) -> {
if (!RESTRICTED.contains(name)) {
values.forEach(value -> builder.header(name, value));
}
});
try {
HttpResponse<byte[]> response =
client.send(builder.build(), HttpResponse.BodyHandlers.ofByteArray());
return new NotificationHttpResponse(
response.statusCode(), Map.copyOf(response.headers().map()), response.body());
} catch (HttpTimeoutException timeout) {
// The request timed out after the body was published, so the provider may well have it.
throw new NotificationHttpTransportException("RESPONSE_TIMEOUT", true, timeout);
} catch (IOException failure) {
throw new NotificationHttpTransportException(
"TRANSPORT_FAILURE", bodyWasLikelyCommitted(failure), failure);
} catch (InterruptedException interrupted) {
Thread.currentThread().interrupt();
throw new NotificationHttpTransportException("INTERRUPTED", true, interrupted);
}
}
/**
* A connect failure happens before anything is written; anything else may have written the body.
*
* <p>The default is deliberately the pessimistic one: guessing "not committed" would turn an
* unknown into an automatic resend.
*/
private static boolean bodyWasLikelyCommitted(IOException failure) {
String message = failure.getMessage();
if (message == null) {
return true;
}
String normalized = message.toLowerCase(java.util.Locale.ROOT);
boolean beforeSend =
normalized.contains("connection refused")
|| normalized.contains("unresolved")
|| normalized.contains("no route to host")
|| normalized.contains("connect timed out");
return !beforeSend;
}
/** Header map helper for adapters. */
public static Map<String, List<String>> headers(Map<String, String> singleValued) {
return singleValued.entrySet().stream()
.collect(
java.util.stream.Collectors.toUnmodifiableMap(
Map.Entry::getKey, entry -> List.of(entry.getValue())));
}
}
@@ -0,0 +1,43 @@
package dev.caskeleton.adapter.outbound.notification.platform.provider.http;
import java.net.URI;
import java.util.Locale;
import java.util.Objects;
import java.util.Set;
/** Endpoint validation shared by the provider profiles. */
public final class NotificationEndpoints {
private static final Set<String> LOOPBACK_HOSTS = Set.of("127.0.0.1", "::1", "localhost");
private NotificationEndpoints() {}
/**
* Require TLS, except on the loopback interface.
*
* <p>The exception is narrow on purpose. A plaintext provider endpoint on a routable host exposes
* credentials and message bodies to anything on the path, which is why it is refused outright. A
* loopback endpoint never leaves the machine, so the same reasoning does not apply — and without
* this the contract suite could not exercise a real socket at all, which would mean the ambiguity
* behaviour it exists to prove went untested.
*/
public static URI requireSecureOrLoopback(URI endpoint, String name) {
Objects.requireNonNull(endpoint, name);
String scheme =
endpoint.getScheme() == null ? "" : endpoint.getScheme().toLowerCase(Locale.ROOT);
if ("https".equals(scheme)) {
return endpoint;
}
String host = endpoint.getHost() == null ? "" : endpoint.getHost().toLowerCase(Locale.ROOT);
if ("http".equals(scheme) && LOOPBACK_HOSTS.contains(host)) {
return endpoint;
}
throw new IllegalArgumentException(name + " must use https outside the loopback interface");
}
/** Whether an endpoint is on the loopback interface. */
public static boolean isLoopback(URI endpoint) {
String host = endpoint.getHost() == null ? "" : endpoint.getHost().toLowerCase(Locale.ROOT);
return LOOPBACK_HOSTS.contains(host);
}
}
@@ -0,0 +1,20 @@
package dev.caskeleton.adapter.outbound.notification.platform.provider.http;
/**
* The only way a provider adapter in this leaf reaches the network.
*
* <p>It exists as a port because the registry does not permit {@code adapter-outbound-notification
* → adapter-outbound-httpclient}. The composition root sees both leaves and is the supported place
* to substitute an implementation backed by the HTTP Client Platform, which brings its own TLS,
* circuit breaker, SSRF and dynamic-target policy.
*/
public interface NotificationHttpGateway {
/**
* Execute one request.
*
* @throws NotificationHttpTransportException when no response could be read; the exception states
* whether the request body was already committed
*/
NotificationHttpResponse exchange(NotificationHttpRequest request);
}
@@ -0,0 +1,64 @@
package dev.caskeleton.adapter.outbound.notification.platform.provider.http;
import java.net.URI;
import java.time.Duration;
import java.util.Arrays;
import java.util.List;
import java.util.Locale;
import java.util.Map;
import java.util.Objects;
/** One outbound provider HTTP request. */
@SuppressWarnings("ArrayRecordComponent") // defensive copies on construction and on every accessor
public record NotificationHttpRequest(
String method, URI uri, Map<String, List<String>> headers, byte[] body, Duration timeout) {
public NotificationHttpRequest {
Objects.requireNonNull(method, "method");
Objects.requireNonNull(uri, "uri");
Objects.requireNonNull(headers, "headers");
Objects.requireNonNull(body, "body");
Objects.requireNonNull(timeout, "timeout");
if (timeout.isNegative() || timeout.isZero()) {
throw new IllegalArgumentException("timeout must be finite and positive");
}
headers =
headers.entrySet().stream()
.collect(
java.util.stream.Collectors.toUnmodifiableMap(
entry -> entry.getKey().toLowerCase(Locale.ROOT),
entry -> List.copyOf(entry.getValue())));
body = body.clone();
}
@Override
public byte[] body() {
return body.clone();
}
@Override
public boolean equals(Object other) {
return other instanceof NotificationHttpRequest request
&& method.equals(request.method)
&& uri.equals(request.uri)
&& headers.equals(request.headers)
&& Arrays.equals(body, request.body)
&& timeout.equals(request.timeout);
}
@Override
public int hashCode() {
return Objects.hash(method, uri, headers, Arrays.hashCode(body), timeout);
}
@Override
public String toString() {
// The URI is redacted because a Web Push endpoint is a capability URL and the request body may
// be a rendered message.
return "NotificationHttpRequest[method="
+ method
+ ", uri=redacted, bytes="
+ body.length
+ "]";
}
}

Some files were not shown because too many files have changed in this diff Show More