Merge branch 'main' into worktree-jpa-persistence-platform
This commit is contained in:
@@ -32,6 +32,7 @@ readonly EXPECTED_WORKFLOW_LOCK=(
|
||||
'59cb3a0ffc687a15eefe96bc5e3a70d42be78e1cc85d2e7f7880dac6124ca4c7 .github/workflows/jpa-r2-evidence.yml'
|
||||
'ea7f8214a3cc9ec3e7ba3183a2201fd26a05a61f0b0fdcb1f041b71efca3e81c .github/workflows/jpa-release.yml'
|
||||
'5be7e931db749029d89787da042d6d7cf8e683d60698bd8a2993c29db26355fb .github/workflows/link-check.yml'
|
||||
'3d5afcef6bf1c65dcd8cad3d1687f07c2cfbb15d360f41251e46f9eb8950baac .github/workflows/notification-platform.yml'
|
||||
'64245586cd5936f1a5647b57f2cd9acd316f96fd75f713b1890decb812e7d5fe .github/workflows/object-storage-qualification.yml'
|
||||
'cbc104ea486c746229895e804e3be7716e056a02cce0588c537bce9f442f8b38 .github/workflows/redis-sdk-topology.yml'
|
||||
)
|
||||
|
||||
@@ -0,0 +1,121 @@
|
||||
name: notification-platform
|
||||
|
||||
# Verification tiers for the Notification Delivery Platform.
|
||||
#
|
||||
# The PR tier is deliberately free of any external provider. A gate that depends on a third-party
|
||||
# sandbox fails for reasons that have nothing to do with the change under review, and a gate people
|
||||
# learn to re-run is not a gate. Real provider smoke tests live in the secret-protected tier, where
|
||||
# a failure is an environment signal rather than a merge blocker.
|
||||
#
|
||||
# Every job that invokes Gradle validates the wrapper first with the repository's pinned action;
|
||||
# the wrapper JAR is executable code fetched at build time, so validating it is what keeps a
|
||||
# compromised wrapper from turning any workflow run into arbitrary code execution.
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
paths:
|
||||
- 'src/application-core/src/**/notification/platform/**'
|
||||
- 'src/adapter/outbound/notification/**'
|
||||
- 'src/adapter/outbound/persistence-jpa/src/**/notification/**'
|
||||
- 'src/adapter/inbound/web/src/**/notification/**'
|
||||
- 'docs/notification/**'
|
||||
- 'infra/notification/**'
|
||||
- '.github/workflows/notification-platform.yml'
|
||||
push:
|
||||
branches: [ main ]
|
||||
schedule:
|
||||
# Nightly: the chaos tier, which is slower and inherently less deterministic than the PR tier.
|
||||
- cron: '0 17 * * *'
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: notification-platform-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
pr:
|
||||
name: contract (Java 21, no external provider)
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # actions/checkout@v4.2.2
|
||||
- name: Validate Gradle wrapper
|
||||
id: gradle-wrapper-validation
|
||||
uses: gradle/actions/wrapper-validation@3f131e8634966bd73d06cc69884922b02e6faf92 # gradle/actions@v6
|
||||
- uses: actions/setup-java@c5195efecf7bdfc987ee8bae7a71cb8b11521c00 # actions/setup-java@v4.7.1
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: "21.0.11+10"
|
||||
cache: gradle
|
||||
cache-dependency-path: |
|
||||
src/**/*.gradle
|
||||
src/**/gradle-wrapper.properties
|
||||
src/**/gradle.lockfile
|
||||
- name: Compile and format check
|
||||
working-directory: src
|
||||
run: ./gradlew :application-core:compileJava :adapter:outbound:notification:compileJava --console=plain
|
||||
- name: Application contracts
|
||||
working-directory: src
|
||||
run: ./gradlew :application-core:test --console=plain
|
||||
- name: Provider contract suite
|
||||
working-directory: src
|
||||
run: ./gradlew :adapter:outbound:notification:test --console=plain
|
||||
- name: Persistence and web
|
||||
working-directory: src
|
||||
run: ./gradlew :adapter:outbound:persistence-jpa:test :adapter:inbound:web:test --console=plain
|
||||
- name: Architecture gates
|
||||
working-directory: src
|
||||
run: |
|
||||
./gradlew verifyCleanArchitectureDependencies --console=plain
|
||||
./gradlew :app-bootstrap:test --tests '*CleanArchitectureTest' --tests '*NotificationArchitectureTest' --console=plain
|
||||
- name: Configuration surface
|
||||
working-directory: src
|
||||
run: ./gradlew verifyEnvKeys verifyPublicPathSnapshot --console=plain
|
||||
- name: Static analysis
|
||||
working-directory: src
|
||||
run: ./gradlew :adapter:outbound:notification:check -x test --console=plain
|
||||
|
||||
nightly-chaos:
|
||||
name: chaos (ambiguity, restart recovery, callback burst)
|
||||
if: github.event_name == 'schedule' || github.event_name == 'workflow_dispatch'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # actions/checkout@v4.2.2
|
||||
- name: Validate Gradle wrapper
|
||||
id: gradle-wrapper-validation
|
||||
uses: gradle/actions/wrapper-validation@3f131e8634966bd73d06cc69884922b02e6faf92 # gradle/actions@v6
|
||||
- uses: actions/setup-java@c5195efecf7bdfc987ee8bae7a71cb8b11521c00 # actions/setup-java@v4.7.1
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: "21.0.11+10"
|
||||
cache: gradle
|
||||
cache-dependency-path: |
|
||||
src/**/*.gradle
|
||||
src/**/gradle-wrapper.properties
|
||||
src/**/gradle.lockfile
|
||||
- name: Ambiguity and fault harness
|
||||
working-directory: src
|
||||
run: ./gradlew :adapter:outbound:notification:test --tests '*ChaosSecurity*' --tests '*CrossProviderContractSuite*' --console=plain
|
||||
- name: Full suite
|
||||
working-directory: src
|
||||
run: ./gradlew test --console=plain
|
||||
|
||||
provider-sandbox:
|
||||
name: provider sandbox smoke (secret-protected, non-blocking)
|
||||
if: github.event_name == 'workflow_dispatch'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
environment: notification-provider-sandbox
|
||||
continue-on-error: true
|
||||
steps:
|
||||
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # actions/checkout@v4.2.2
|
||||
- name: Smoke test against real provider sandboxes
|
||||
env:
|
||||
NOTIFICATION_SANDBOX_ENABLED: 'true'
|
||||
run: |
|
||||
echo "Runs only where provider sandbox credentials are configured."
|
||||
echo "Never a required check: an external outage must not block a merge."
|
||||
@@ -0,0 +1,63 @@
|
||||
# ADR-MONGO-001 — MongoDB platform boundary
|
||||
|
||||
- **Status:** Accepted
|
||||
- **Date:** 2026-08-13
|
||||
- **Design source:** `mongodb-superpowers-package/.../2026-08-11-mongodb-document-persistence-platform-design.md` §1, §2 (D-01, D-04, D-05), §5, §6
|
||||
|
||||
## Context
|
||||
|
||||
Two failure modes are common when a team wraps MongoDB.
|
||||
|
||||
The first is flattening: a shared `CommonMongoRepository<T, ID>` and a generic CRUD facade, which
|
||||
forces every collection to share an id strategy, a consistency profile and a query surface. MongoDB's
|
||||
single-document atomicity, aggregation model and change streams stop being reachable, and the first
|
||||
collection that needs something different gets a cast or a leaky generic.
|
||||
|
||||
The second is unrestricted exposure: the driver and `runCommand` available everywhere. Then any
|
||||
service can drop a collection, run an unbounded pipeline, or issue an admin command from a request
|
||||
thread, and no review catches it because there is nothing structural to catch.
|
||||
|
||||
## Decision
|
||||
|
||||
The domain owns its documents; the platform owns the cross-cutting decisions. Four exposure planes:
|
||||
|
||||
| Plane | Contents | Client |
|
||||
|---|---|---|
|
||||
| D1 Standard document persistence | Spring Data repositories, typed queries, mapping manifest, atomic update primitives, optimistic revision | Stable API V1, `apiStrict=true` |
|
||||
| D2 Advanced document operations | `MongoTemplate`, transactions/sessions, bulk, aggregation, keyset cursors, change streams | Stable API V1, `apiStrict=true` |
|
||||
| D3 Explicit Mongo capability | Native BSON, time series, search/vector, CSFLE/QE, shard-aware operations | Separate capability client |
|
||||
| D4 Admin plane | Collection, validator, index, migration, shard, repair | Separate admin client and credential |
|
||||
|
||||
Specifically:
|
||||
|
||||
1. **No `CommonMongoRepository<T, ID>`.** Each aggregate declares its own repository.
|
||||
2. **D1/D2 run on Stable API V1 with `apiStrict=true`,** so a command outside the versioned API fails
|
||||
at development time instead of on the next server upgrade.
|
||||
3. **D3 is not a raw-client escape.** Every call passes a fixed admission order: capability registered
|
||||
→ database profile → collection allowlist → operation name → timeout → consistency profile →
|
||||
result limit → trace → redaction → command category → admin-command refusal → execute.
|
||||
4. **D4 is a separate client with a separate credential.** No application-plane path reaches it;
|
||||
`PolicyAwareMongoNativeGateway` refuses admin-category commands regardless of capability.
|
||||
5. **Advanced and Experimental capabilities are opt-in modules**, never transitive dependencies of the
|
||||
Stable surface.
|
||||
|
||||
## Consequences
|
||||
|
||||
**Positive.** MongoDB's semantics stay reachable. Misuse is refused structurally rather than reviewed
|
||||
for. A server upgrade cannot silently change D1/D2 behaviour. Admin operations have their own audit
|
||||
trail and credential.
|
||||
|
||||
**Negative.** Every operation needs a registered name and profile, so a new query is a small amount of
|
||||
configuration rather than zero. A genuinely new capability requires a registration before it can be
|
||||
used. Both are deliberate: the cost is paid once per operation, at review time.
|
||||
|
||||
**Rejected alternative — "expose the driver, rely on code review."** Review does not scale to every
|
||||
query in every service, and the operations that matter (unbounded pipeline, `dropCollection`,
|
||||
unanchored regex on user input) look unremarkable in a diff.
|
||||
|
||||
## Repository adaptation
|
||||
|
||||
The design assumes 19 Gradle modules under `modules/mongodb/`. This repository's fail-closed registry
|
||||
declares exactly 19 leaf identities, so the modules became package boundaries inside
|
||||
`:adapter:outbound:persistence-mongo`, enforced by ArchUnit. See
|
||||
[docs/mongodb/repository-adaptation.md](../mongodb/repository-adaptation.md).
|
||||
@@ -0,0 +1,58 @@
|
||||
# ADR-MONGO-002 — BSON representation is a pinned manifest
|
||||
|
||||
- **Status:** Accepted
|
||||
- **Date:** 2026-08-13
|
||||
- **Design source:** design §10, decision D-06
|
||||
|
||||
## Context
|
||||
|
||||
How a Java value is represented in BSON is a data contract, but nothing in the default toolchain
|
||||
treats it as one. Spring Data and the MongoDB driver both have defaults, and those defaults have
|
||||
changed across versions. A `BigDecimal` can land as a `Double`, a `String` or a `Decimal128`; a `UUID`
|
||||
can land as `Binary` subtype 3 or subtype 4; an `Instant` can land as a `Date` or a `String`.
|
||||
|
||||
The consequences are asymmetric. A representation change is invisible in a value-equality test —
|
||||
`12.30` looks like `12.30` whether it is a double or a `Decimal128` — but once a collection holds
|
||||
production data, changing it is a full migration. And the UUID case is worse than a migration: legacy
|
||||
Java representation byte-swaps two halves of the UUID, so a document written under one representation
|
||||
and read under the other yields a *different, valid-looking* UUID. Nothing errors. You get the wrong
|
||||
record.
|
||||
|
||||
## Decision
|
||||
|
||||
`MongoTypeRepresentationManifest` pins the representation for every type the platform maps, and
|
||||
`MongoMappingConfiguration` builds the Spring Data converters from it. Nothing relies on a library
|
||||
default.
|
||||
|
||||
| Java | BSON | Rationale |
|
||||
|---|---|---|
|
||||
| `UUID` | `Binary` subtype 4 (`STANDARD`) | Subtype 3 byte-swaps; cross-representation reads are silently wrong. |
|
||||
| `BigDecimal` | `Decimal128` | A double cannot represent `12.30`; money compared as a double is eventually wrong by a cent. |
|
||||
| `BigInteger` | `Decimal128`, or declared `String` when out of range | 34 significant digits; out of range fails on write instead of rounding. |
|
||||
| `Instant` / `OffsetDateTime` / `ZonedDateTime` | UTC `Date` | One instant, one representation. |
|
||||
| `LocalDate` | declared per field | A calendar day is not an instant. |
|
||||
| `LocalDateTime` | **refused** | No offset: the stored value depends on the writing JVM's default zone. |
|
||||
| `enum` | `String` name | Ordinals renumber when someone inserts a constant. |
|
||||
|
||||
Type metadata follows `MongoTypeMetadataPolicy` — `NONE`, `ALIAS` or `CLASS_NAME`. A
|
||||
`@LongLivedMongoDocument` type may not use `CLASS_NAME`: writing a FQCN into a million documents makes
|
||||
a package rename a data migration.
|
||||
|
||||
The manifest is enforced by a golden gate. `MongoBsonSnapshot` canonicalises a stored document,
|
||||
preserving BSON types and keeping missing distinct from null, and
|
||||
`MongoBsonSnapshotAssert.hasTypeSignature(...)` fails on any representation change. The registry
|
||||
pins `UuidCodec(STANDARD)` explicitly rather than inheriting a default, since inheriting the default
|
||||
is the exact drift the gate exists to catch.
|
||||
|
||||
## Consequences
|
||||
|
||||
**Positive.** A library upgrade cannot move a representation without failing a test. Money is exact.
|
||||
UUIDs read back as themselves. Class moves stay refactors.
|
||||
|
||||
**Negative.** Every representation-affecting change requires updating a snapshot *and* writing a
|
||||
migration. A new mapped type needs a manifest entry before it can be used. This is the intended
|
||||
friction: the alternative is discovering the change in production.
|
||||
|
||||
**Rejected alternative — "snapshot the JSON."** JSON destroys exactly the distinctions the gate
|
||||
protects: `Decimal128` and `String` both render as text, `Binary` UUID and `ObjectId` both render as
|
||||
hex, and missing and null both disappear.
|
||||
@@ -0,0 +1,68 @@
|
||||
# ADR-MONGO-003 — Transaction body retry and commit retry are separate loops
|
||||
|
||||
- **Status:** Accepted
|
||||
- **Date:** 2026-08-13
|
||||
- **Design source:** design §14–§16, decisions D-07 through D-10
|
||||
|
||||
## Context
|
||||
|
||||
MongoDB reports two transaction failures that look similar and must be handled in opposite ways.
|
||||
|
||||
`TransientTransactionError` means the transaction definitively did not commit. The correct response is
|
||||
to run the whole thing again.
|
||||
|
||||
`UnknownTransactionCommitResult` means the commit **may already have applied** — typically because the
|
||||
primary changed while the commit was in flight. The correct response is to retry *the commit*, which
|
||||
is a no-op if it already succeeded.
|
||||
|
||||
The common implementation wraps everything in one retry loop. That loop replays the body after an
|
||||
unknown commit, and if the commit did apply, the body applies twice. In a payment or notification path
|
||||
that is a duplicate charge or a duplicate message, produced by the error handler.
|
||||
|
||||
The related trap is session reuse: retrying on the same session after an abort carries the aborted
|
||||
transaction's state into the retry.
|
||||
|
||||
## Decision
|
||||
|
||||
`MongoTransactionRetryCoordinator` implements two loops with different scopes.
|
||||
|
||||
```
|
||||
for each body attempt within the budget:
|
||||
open a NEW session
|
||||
run the body
|
||||
TransientTransactionError -> abort, continue to next body attempt
|
||||
commitWithRetry(session):
|
||||
UnknownTransactionCommitResult -> retry the COMMIT ONLY, same session
|
||||
```
|
||||
|
||||
Rules that follow, all of them load-bearing:
|
||||
|
||||
1. **A new session per body attempt.** No aborted state leaks into a retry.
|
||||
2. **The body is never replayed after a commit ambiguity.** `MongoRetryScope.COMMIT_ONLY` is a
|
||||
distinct value from `BODY` precisely so this cannot be collapsed by accident.
|
||||
3. **One budget bounds both loops.** `MongoRetryBudget` limits attempts *and* elapsed time, with
|
||||
jittered backoff, so a struggling primary is not retried into the ground by every instance at once.
|
||||
4. **An exhausted commit retry surfaces `TRANSACTION_COMMIT_UNKNOWN`,** never a generic failure. An
|
||||
ambiguous outcome reported as a failure invites the caller to retry — the one thing that must not
|
||||
happen. See [docs/mongodb/runbooks/unknown-commit.md](../mongodb/runbooks/unknown-commit.md).
|
||||
5. **Transaction bodies write a deterministic marker** so `MongoCommitReconciler` can establish what
|
||||
actually happened. A transaction that cannot be reconciled has no recovery path.
|
||||
6. **Classification reads labels before codes.** Server error labels are the authoritative statement
|
||||
about retryability; error codes vary by version.
|
||||
|
||||
Surrounding decisions that reduce how often this path is reached at all: single-document atomic
|
||||
operations are preferred over transactions (D-09), partial changes use update operators rather than
|
||||
`save()` (D-07), and whole-document replacement requires an optimistic revision (D-08).
|
||||
|
||||
## Consequences
|
||||
|
||||
**Positive.** A commit ambiguity cannot become a duplicate effect. The ambiguity reaches the caller as
|
||||
an ambiguity. The retry budget is bounded in both attempts and time.
|
||||
|
||||
**Negative.** Callers must handle a third outcome beyond success and failure. Transaction bodies must
|
||||
write a marker they would not otherwise need. Both costs are small compared with reconciling
|
||||
duplicated financial effects after the fact.
|
||||
|
||||
**Rejected alternative — "one retry loop, at-least-once everywhere."** It requires every transaction
|
||||
body to be fully idempotent, which is a much stronger and much less checkable property than writing
|
||||
one marker, and it is silently violated the first time someone adds a non-idempotent step.
|
||||
@@ -0,0 +1,62 @@
|
||||
# ADR-MONGO-004 — Index and schema changes belong to the admin plane
|
||||
|
||||
- **Status:** Accepted
|
||||
- **Date:** 2026-08-13
|
||||
- **Design source:** design §21–§25, decisions D-11, D-13
|
||||
|
||||
## Context
|
||||
|
||||
Spring Data can create indexes automatically from annotations. On a laptop this is convenient. On a
|
||||
collection with a hundred million documents, an index build is a capacity event: it consumes CPU, IO
|
||||
and memory on the primary for minutes to hours, and it starts because a pod restarted.
|
||||
|
||||
Worse, it starts N times when N pods restart, and there is no approval step, no ordering relative to
|
||||
the code that needs the index, and no record afterwards of what was created.
|
||||
|
||||
Schema validators have the same shape with a sharper edge: tightening a validator on a collection with
|
||||
existing data rejects writes to documents that were legal when they were written.
|
||||
|
||||
TTL has a third shape. It looks like a scheduler and is not one: the TTL monitor runs about once a
|
||||
minute and deletes in batches, so an expired document routinely remains readable for minutes or hours.
|
||||
|
||||
## Decision
|
||||
|
||||
**Indexes and validators are declared in a manifest and applied by the admin plane (D4).** Automatic
|
||||
index creation in production is disabled.
|
||||
|
||||
1. `MongoManifestRegistry` holds the declared indexes (`MongoIndexManifest`) and validator
|
||||
(`MongoSchemaManifest`) per collection. The manifest is the source of truth, reviewed in a pull
|
||||
request.
|
||||
2. `MongoIndexDiffEngine` compares manifest against observed state and reports missing, extra and
|
||||
*changed* indexes. Changed ones are reported rather than re-issued: MongoDB will not silently
|
||||
rebuild an index whose definition moved.
|
||||
3. `MongoIndexApplyPolicy` sets what an environment may do — `APPLY` (local), `APPLY_WITH_DIFF`
|
||||
(staging), `DIFF_WITH_APPROVED_APPLY` (production), `REPORT_ONLY` (audit).
|
||||
4. **Ownership gates every drop.** `MongoMetadataOwnership` distinguishes `APPLICATION_MANAGED` from
|
||||
`SEARCH_MANAGED`, `ENCRYPTION_MANAGED` and `EXTERNAL`. Only application-managed objects are
|
||||
droppable on drift. A diff engine without ownership eventually proposes dropping
|
||||
`enxcol_.customers.esc`, and "the drift tool cleaned it up" is a very bad incident summary.
|
||||
5. **Retirement is staged.** `MongoIndexRetirementState` moves an index declared → hidden →
|
||||
observed-unused → droppable, one deployment per transition. Hiding is instantly reversible;
|
||||
dropping is a rebuild.
|
||||
6. **Stable validation actions are `error` and `warn` only.** `errorAndLog` is not part of the Stable
|
||||
contract on 7.0 or 8.0 and is refused. Tightening goes `warn`+`MODERATE` → confirm zero warnings →
|
||||
`error`+`STRICT`, in two deployments.
|
||||
7. **TTL is physical cleanup only** (D-13). `MongoExpirationAccessPolicy` states the rule: a
|
||||
document's presence is not authorization and its absence is not a deadline. Access control checks
|
||||
the expiry field; scheduling uses a scheduler.
|
||||
8. **Migrations are checksummed, locked, precondition-checked and resumable.**
|
||||
`MongoMigrationRunner` fails hard when an applied id's checksum changed — two environments running
|
||||
different code under one id is worse than a failed deploy.
|
||||
|
||||
## Consequences
|
||||
|
||||
**Positive.** Index builds are scheduled by people who know the capacity. Rollback is possible at
|
||||
every step. Drift is visible without being dangerous. Nothing drops what it does not own.
|
||||
|
||||
**Negative.** Adding an index is a manifest change plus an apply, not an annotation. Local development
|
||||
uses `APPLY` so the friction is confined to environments where it is warranted.
|
||||
|
||||
**Rejected alternative — "auto-create with a feature flag."** The flag is either on in production,
|
||||
which is the problem, or off, in which case the manifest is the real mechanism and the annotation is a
|
||||
second, divergent source of truth.
|
||||
@@ -0,0 +1,79 @@
|
||||
# ADR-MONGO-ADV-001 — Advanced capability promotion
|
||||
|
||||
- **Status:** Accepted
|
||||
- **Date:** 2026-08-13
|
||||
- **Design source:** design §2 (D-15), §3.2–§3.3; Advanced expansion plan Task 15
|
||||
|
||||
## Context
|
||||
|
||||
Sharding, time series, CSFLE, Queryable Encryption, search, vector search and multi-tenancy each work
|
||||
in a demo within an afternoon. What they do not do is behave the same way in production, and the
|
||||
differences are not discovered by functional tests:
|
||||
|
||||
- Sharding changes which queries are efficient. A query that misses the shard key becomes
|
||||
scatter-gather, which passes every test on a one-shard cluster.
|
||||
- Encryption's failure modes are KMS failure modes — wrong key, revoked permission, mid-rotation —
|
||||
none of which occur against a local key provider.
|
||||
- Search and vector search can be functionally correct and useless: the index returns results, and
|
||||
the results are not relevant. Recall is not visible in a pass/fail assertion.
|
||||
- Database-per-tenant works until the tenant count crosses what the connection and file-handle
|
||||
budget supports, which is an operational property, not a code property.
|
||||
|
||||
The failure mode this ADR prevents is a capability marked "done" on the strength of a green test that
|
||||
never touched the environment where it will run.
|
||||
|
||||
## Decision
|
||||
|
||||
Every Advanced and Experimental capability is an **opt-in module behind its own flag**, and promotion
|
||||
requires evidence, not confidence.
|
||||
|
||||
### Enablement
|
||||
|
||||
`MongoAdvancedCapabilityFlags` gates construction of every Advanced entry point. A disabled capability
|
||||
does not produce a runtime warning — the type refuses to be constructed, naming the property that
|
||||
enables it (`MongoAdvancedCapabilityFlags.propertyFor(capability)`). Being on the classpath is not
|
||||
being enabled, and `stableNeverDependsOnAdvanced` (ArchUnit) keeps the Stable surface free of them.
|
||||
|
||||
### Promotion evidence
|
||||
|
||||
`MongoAdvancedPromotionGate.verify(evidence)` requires every category:
|
||||
|
||||
| Category | Means |
|
||||
|---|---|
|
||||
| `stable-platform` | The Stable release gate passed on the same revision. |
|
||||
| `actual-topology` | The capability ran on the real topology — a real sharded cluster, the real KMS, the actual target deployment. Atlas Local is a pull-request convenience and explicitly not release evidence (`MongoAtlasCapabilityContractSuite.Environment.ATLAS_LOCAL`). |
|
||||
| `security` | Privileges reviewed; the capability's admin role is separate from the application role. |
|
||||
| `migration` | A documented path in and, where the capability is irreversible, an explicit statement that there is no path back. |
|
||||
| `failure` | Negative cases fail closed: wrong key, missing permission, rotation, non-ready index, unrouted query. |
|
||||
| `runbook` | A runbook exists for the capability's characteristic incident. |
|
||||
|
||||
### Additional per-capability requirements
|
||||
|
||||
- **Search / vector search:** relevance and performance evidence, not functional success alone.
|
||||
`MongoVectorSearchBenchmarkGate` requires recall alongside latency and index size; a gate that
|
||||
measures only latency certifies a fast wrong answer.
|
||||
- **Database-per-tenant and reshard orchestration remain Experimental** until operational scale
|
||||
evidence exists. Both are correct in the small and unbounded in the large.
|
||||
- **Reshard requires an explicit `ReshardApproval`** — a named approver and a stated window. It
|
||||
rewrites the collection.
|
||||
|
||||
### Promotion does not change the dependency boundary
|
||||
|
||||
A capability promoted to Stable **remains an opt-in module** unless a later starter ADR changes the
|
||||
dependency boundary. Promotion is a statement about evidence, not an invitation to add a transitive
|
||||
dependency to every service.
|
||||
|
||||
## Consequences
|
||||
|
||||
**Positive.** No capability reaches production on the strength of a container-only test. The evidence
|
||||
list is the same for every capability, so promotion is reviewable rather than negotiated.
|
||||
|
||||
**Negative.** Promotion requires access to real infrastructure — a sharded cluster, a real KMS, the
|
||||
target deployment. That is the cost of the guarantee: the alternative is finding out in production,
|
||||
where encryption and sharding are both expensive to reverse.
|
||||
|
||||
## Verification
|
||||
|
||||
```bash
|
||||
bash scripts/verify-mongodb-advanced.sh
|
||||
```
|
||||
@@ -0,0 +1,91 @@
|
||||
# Advanced — CSFLE and Queryable Encryption
|
||||
|
||||
**Capabilities:** `MongoCapability.CSFLE`, `MongoCapability.QUERYABLE_ENCRYPTION`
|
||||
**Properties:** `ca-skeleton.persistence-mongo.advanced.csfle.enabled`,
|
||||
`ca-skeleton.persistence-mongo.advanced.queryable-encryption.enabled`
|
||||
**Status:** Advanced.
|
||||
|
||||
## Requirements
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| Topology | Replica set or sharded cluster. |
|
||||
| Server | MongoDB 7.0 or 8.0 (see §4 for the 8.0 query-type limits). |
|
||||
| Privilege | `MongoPrincipalRole.ENCRYPTION_ADMIN` for the key vault; the application role never holds it. |
|
||||
| Environment | A real KMS and key vault. A local key provider does not exercise any of the failure modes that matter. |
|
||||
|
||||
## 1. CSFLE
|
||||
|
||||
`MongoCsfleProfile` binds a collection to its `MongoCsfleFieldPolicy` list, a key vault
|
||||
`MongoCredentialReference` and the key vault namespace. `MongoCsfleClientFactory` builds the encrypted
|
||||
client; `MongoDataKeyResolver` resolves data keys.
|
||||
|
||||
`MongoCsfleMode`:
|
||||
|
||||
| Mode | Queryable | Trade-off |
|
||||
|---|---|---|
|
||||
| `RANDOMIZED` | no | Same plaintext encrypts differently each time. The safe default. |
|
||||
| `DETERMINISTIC` | equality only | Same plaintext always yields the same ciphertext, so equality works — and so does frequency analysis. |
|
||||
| `UNINDEXED` | no | Stored encrypted, excluded from any index. |
|
||||
|
||||
`MongoCsfleFieldPolicy.forPii(field, queryable)` defaults to `RANDOMIZED` when the field is not
|
||||
queried. Deterministic encryption requires a written `equalityQueryJustification`; the constructor
|
||||
refuses a blank one, naming frequency analysis. A low-cardinality deterministic field (a status, a
|
||||
country, a boolean) leaks its distribution to anyone who can read the collection, which is the party
|
||||
encryption was protecting against.
|
||||
|
||||
## 2. Queryable Encryption
|
||||
|
||||
`MongoQueryableEncryptionProfile` binds a collection to `MongoEncryptedFieldDescriptor` entries.
|
||||
`MongoQueryableEncryptionQueryType` has exactly two values:
|
||||
|
||||
- `EQUALITY`
|
||||
- `RANGE` — must declare its domain (`min`, `max`). The constructor refuses a range field without one,
|
||||
because changing the domain later means re-encrypting the field.
|
||||
|
||||
`MongoQueryableEncryptionCollectionManager` owns the collection's lifecycle, because a QE collection
|
||||
is not just a collection: it carries metadata collections.
|
||||
|
||||
## 3. Metadata ownership
|
||||
|
||||
`MongoEncryptionMetadataOwnership` maps `customers` to `enxcol_.customers.esc` and
|
||||
`enxcol_.customers.ecoc`, and reports `__safeContent__`-prefixed indexes as
|
||||
`MongoMetadataOwnership.ENCRYPTION_MANAGED`.
|
||||
|
||||
These are never application-owned and never droppable by drift reconciliation. A drift tool that
|
||||
drops `enxcol_.customers.ecoc` corrupts the collection's queryability. This is the single most
|
||||
important integration point between encryption and
|
||||
[ADR-MONGO-004](../../adr/ADR-MONGO-004-index-schema-admin-plane.md).
|
||||
|
||||
## 4. Unsupported combinations
|
||||
|
||||
Refused at declaration, not discovered at runtime:
|
||||
|
||||
| Combination | Why |
|
||||
|---|---|
|
||||
| CSFLE **and** QE on the same collection | Two incompatible encryption schemes over one namespace. Both profile constructors refuse it. |
|
||||
| CSFLE on a time series collection | `requireNotTimeSeries(true)` raises `MongoOperationRejectedException`. |
|
||||
| QE `prefix` / `suffix` / `substring` | Not available on the platform's 8.0 baseline. The factory methods throw `UnsupportedOperationException` rather than returning a profile that fails later. |
|
||||
| Deterministic CSFLE without a justification | `IllegalArgumentException` naming frequency analysis. |
|
||||
| Range QE without a declared domain | `IllegalArgumentException` naming re-encryption. |
|
||||
|
||||
## 5. Failure recovery
|
||||
|
||||
| Symptom | Cause | Action |
|
||||
|---|---|---|
|
||||
| `MongoEncryptionException` on read | Wrong data key, or the key vault is unreachable | Check KMS reachability and the key vault credential. Data is intact; the client cannot decrypt it. |
|
||||
| `MongoEncryptionException` on write | KMS permission revoked mid-operation | Restore the grant. Writes fail closed — nothing was written in plaintext. |
|
||||
| Queries return nothing on a deterministic field | The field was re-keyed | Equality matching is over ciphertext; a new key produces different ciphertext. Re-encrypt the field. |
|
||||
| QE queries fail after a drift reconciliation | A metadata collection was dropped | Restore from backup. This is why ownership gates drops. |
|
||||
|
||||
**Key rotation.** Rotating the customer master key re-wraps the data keys and does not require
|
||||
re-encrypting documents. Rotating a *data* key does require re-encrypting every document that used
|
||||
it. These are different operations with different costs, and confusing them is how a rotation becomes
|
||||
an outage.
|
||||
|
||||
## 6. Promotion evidence
|
||||
|
||||
Per [ADR-MONGO-ADV-001](../../adr/ADR-MONGO-ADV-001-capability-promotion.md), promotion requires the
|
||||
real KMS and key vault, plus negative cases that fail closed: wrong key, missing permission, rotation
|
||||
mid-operation (`MongoAtlasCapabilityContractSuite.kmsFailureModes`). A local key provider certifies
|
||||
none of these — it never rejects anything.
|
||||
@@ -0,0 +1,63 @@
|
||||
# Advanced — GridFS compatibility and migration
|
||||
|
||||
**Capability:** `MongoCapability.GRIDFS_COMPATIBILITY`
|
||||
**Property:** `ca-skeleton.persistence-mongo.advanced.gridfs-compatibility.enabled`
|
||||
**Status:** Advanced, compatibility only. Decision D-14.
|
||||
|
||||
## Position
|
||||
|
||||
GridFS is a **compatibility adapter for files that already exist there**. New files use the existing
|
||||
Fileserver / Object Storage adapter, which is the source of truth for binary content.
|
||||
|
||||
The reason is not preference. GridFS stores file chunks in the same collections, on the same replica
|
||||
set, competing for the same working set as your documents. A large file read evicts document pages
|
||||
from cache, and file storage growth becomes replica-set growth — which means it becomes oplog
|
||||
pressure, backup duration and failover time. Object storage was built for this and MongoDB was not.
|
||||
|
||||
## Reading legacy files
|
||||
|
||||
`MongoGridFsCompatibilityReader` reads existing GridFS content as
|
||||
`GridFsLegacyContent(legacyId, filename, sizeBytes, checksum, stream)`. It reads; it does not write.
|
||||
|
||||
## Migration
|
||||
|
||||
`MongoGridFsMigrationJob` moves a file to object storage in a fixed order:
|
||||
|
||||
```
|
||||
read legacy content
|
||||
→ write to object storage
|
||||
→ verify the target checksum matches the source
|
||||
→ switch the reference
|
||||
→ (later, separately) delete the source
|
||||
```
|
||||
|
||||
Three properties, each of which exists because of a specific way this goes wrong:
|
||||
|
||||
1. **Verify before switching.** `MongoGridFsObjectReference` requires a non-blank checksum, and the
|
||||
job returns empty and writes no reference when the target checksum does not match the source. A
|
||||
migration that switches the reference on a successful *write* rather than a verified *copy*
|
||||
silently points at a truncated object.
|
||||
2. **The source is never deleted here.** Deletion is a separate, later decision after the new
|
||||
location has been serving reads long enough to be trusted. A migration that deletes as it goes has
|
||||
no rollback.
|
||||
3. **The checkpoint separates migrated from failed.** `MongoGridFsMigrationCheckpoint` tracks
|
||||
`migratedCount()`, `failedCount()`, `clean()` and `lastMigratedLegacyId()`, so a restart continues
|
||||
from the last completed file rather than starting over, and a partially failed run is visible as
|
||||
partial rather than as "done".
|
||||
|
||||
## Failure recovery
|
||||
|
||||
| Symptom | Cause | Action |
|
||||
|---|---|---|
|
||||
| `migrate` returns empty | Checksum mismatch | The copy is bad. Investigate before retrying; do not force the reference. |
|
||||
| `IllegalArgumentException` on the reference | Missing checksum | A reference without a checksum cannot be verified and is refused. |
|
||||
| Checkpoint not `clean()` | Some files failed | Re-run for the failed ids only; the checkpoint names the last successful one. |
|
||||
| Reference switched but content missing | Source deleted too early | Restore from backup. This is what rule 2 prevents. |
|
||||
|
||||
## Promotion evidence
|
||||
|
||||
Actual-topology evidence against the real object storage backend, a security review of the storage
|
||||
credential, the migration path above, failure cases (checksum mismatch refused, missing checksum
|
||||
refused, restart resumes), and this document as the runbook.
|
||||
|
||||
New file storage does not go through here at all — see the fileserver adapter.
|
||||
@@ -0,0 +1,90 @@
|
||||
# Advanced — Multi-tenancy
|
||||
|
||||
**Capabilities:** `MongoCapability.SHARED_COLLECTION_TENANCY` (Advanced),
|
||||
`MongoCapability.DATABASE_PER_TENANT` (Experimental)
|
||||
**Properties:** `ca-skeleton.persistence-mongo.advanced.shared-collection-tenancy.enabled`,
|
||||
`ca-skeleton.persistence-mongo.advanced.database-per-tenant.enabled`
|
||||
|
||||
## 1. Shared collection
|
||||
|
||||
Every tenant's documents live in one collection, discriminated by a tenant field.
|
||||
|
||||
`MongoTenantContext` carries the tenant. `MongoTenantPredicateInjector` adds the tenant predicate to
|
||||
every query, every atomic filter and the **first** aggregation stage. `TenantScopedMongoOperations`
|
||||
is the entry point, so a caller cannot construct an unscoped operation by forgetting.
|
||||
|
||||
Three details are load-bearing:
|
||||
|
||||
- **Injection, not convention.** A tenant predicate that each query is expected to add itself is a
|
||||
cross-tenant leak waiting for one missed `where(...)`. The injector adds it structurally.
|
||||
- **First aggregation stage.** `firstStageMatch(...)` places the tenant `$match` before anything else.
|
||||
A `$lookup` or `$group` that runs before the tenant filter has already crossed the boundary, even if
|
||||
a later stage filters the output.
|
||||
- **An absent tenant is not "all tenants".** The injector takes an `Optional<MongoTenantContext>` so
|
||||
the missing case is a decision the policy makes explicitly, not a predicate that quietly disappears.
|
||||
|
||||
`MongoTenantManifestValidator.validate(manifest, tenantScopedUniqueIndexes)` checks that every unique
|
||||
index that should be per-tenant actually includes the tenant field. A unique index on `email` alone in
|
||||
a shared collection makes an email globally unique across tenants — tenant B cannot register an
|
||||
address tenant A already used, which is both a bug and an information leak.
|
||||
|
||||
`requireShardKeyAnalysed(...)` requires a shard-key readiness report before a shared-collection tenant
|
||||
model is sharded: tenant id as a shard key prefix concentrates the largest tenant on one shard.
|
||||
|
||||
### Observability
|
||||
|
||||
`tenantId` and `rawTenantId` are on `MongoObservationConvention`'s forbidden tag list. Cardinality
|
||||
grows with the customer list, and the tag ships tenant identity into the metrics backend.
|
||||
|
||||
## 2. Database per tenant (Experimental)
|
||||
|
||||
`MongoTenantDatabaseResolver` maps a tenant to its database; `MongoTenantClientRegistry` holds the
|
||||
clients.
|
||||
|
||||
Experimental for a specific reason: it is correct in the small and unbounded in the large. Each tenant
|
||||
database costs connections, file handles and monitoring cardinality. It works beautifully at 20
|
||||
tenants and falls over at 2,000, and nothing in a functional test distinguishes the two. Promotion
|
||||
requires operational scale evidence.
|
||||
|
||||
`MongoTenantLifecyclePolicy`:
|
||||
|
||||
- `requireActivationReady(tenantKey, schemaAndIndexesValidated)` — a tenant is not activated until its
|
||||
schema and indexes are validated. Activating first means the first customer request is the migration
|
||||
test.
|
||||
- `requireDeleteAllowed(...)` — deletion requires an explicit retention decision. Dropping a tenant
|
||||
database is irreversible and takes the backup surface with it.
|
||||
|
||||
`MongoTenantMigrationCoordinator` runs a migration across tenant databases with per-tenant results.
|
||||
Partial failure is normal and must be reported per tenant: "migration failed" across 500 databases is
|
||||
not a report anyone can act on.
|
||||
|
||||
## 3. Choosing
|
||||
|
||||
| | Shared collection | Database per tenant |
|
||||
|---|---|---|
|
||||
| Isolation | Logical, enforced by injection | Physical |
|
||||
| Tenant count | Unbounded | Bounded by connections and file handles |
|
||||
| Per-tenant restore | Hard | Natural |
|
||||
| Noisy neighbour | Shared resources | Isolated |
|
||||
| Migration | One collection | N databases, partial failures |
|
||||
| Cross-tenant query | Possible (and must be forbidden) | Structurally impossible |
|
||||
|
||||
Shared collection is the default. Database-per-tenant is for a small number of tenants with a
|
||||
contractual isolation or per-tenant-restore requirement.
|
||||
|
||||
## 4. Failure recovery
|
||||
|
||||
| Symptom | Cause | Action |
|
||||
|---|---|---|
|
||||
| Cross-tenant data visible | An operation bypassed `TenantScopedMongoOperations` | Treat as a security incident. Find the path, close it, audit access. |
|
||||
| Unique constraint fires across tenants | Unique index missing the tenant field | Rebuild the index with the tenant field as prefix; the validator catches this before it ships. |
|
||||
| One shard holds most data | Tenant id as shard-key prefix with a dominant tenant | Refine the shard key with a high-cardinality suffix. |
|
||||
| Connection exhaustion | Database-per-tenant beyond the connection budget | The scale limit. Consolidate or move to shared collections. |
|
||||
| Migration partially applied across tenants | Normal | `MongoTenantMigrationCoordinator` reports per tenant; re-run for the failures only. |
|
||||
|
||||
## 5. Promotion evidence
|
||||
|
||||
Shared-collection tenancy: actual-topology evidence, a security review covering cross-tenant access,
|
||||
a migration path, failure cases (injection proven on query, atomic filter and first aggregation
|
||||
stage), this runbook. Database-per-tenant additionally requires **operational scale evidence** and
|
||||
stays Experimental until it exists.
|
||||
@@ -0,0 +1,81 @@
|
||||
# Advanced — Search and Vector Search
|
||||
|
||||
**Capabilities:** `MongoCapability.SEARCH`, `MongoCapability.VECTOR_SEARCH`
|
||||
**Properties:** `ca-skeleton.persistence-mongo.advanced.search.enabled`,
|
||||
`ca-skeleton.persistence-mongo.advanced.vector-search.enabled`
|
||||
**Status:** Experimental (design §3.3). Hybrid search likewise.
|
||||
|
||||
## Requirements
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| Topology | A deployment with the search service. Atlas Local in a container is a pull-request convenience and is **not** release evidence. |
|
||||
| Privilege | `MongoPrincipalRole.SEARCH_ADMIN` for index management; the application role queries only. |
|
||||
| Gate | `MongoAtlasCapabilityContractSuite` on the actual target deployment. |
|
||||
|
||||
## 1. Created is not ready
|
||||
|
||||
`MongoSearchIndexState`: `CREATED` → `BUILDING` → `READY`, plus `FAILED` and `DELETING`.
|
||||
|
||||
`MongoSearchReadinessGate.requireReady(state)` refuses anything but `READY`. A search index is built
|
||||
asynchronously: the create call returns immediately and the index answers queries with *partial*
|
||||
results while building. Not an error, not empty — partial. A deployment that creates an index and
|
||||
starts querying serves incomplete results for as long as the build takes, and nothing reports it.
|
||||
|
||||
## 2. Search indexes are not application-owned
|
||||
|
||||
`MongoSearchIndexDescriptor.metadataOwnership()` is `MongoMetadataOwnership.SEARCH_MANAGED`, and
|
||||
`droppableByApplicationDrift()` is false. The index reconciliation described in
|
||||
[ADR-MONGO-004](../../adr/ADR-MONGO-004-index-schema-admin-plane.md) must not drop it.
|
||||
|
||||
## 3. Query guardrails
|
||||
|
||||
`MongoSearchQuery` binds an index, an allowlist of paths, the search text and a result limit.
|
||||
|
||||
- `requireAllowedPaths(allowed)` raises `MongoOperationRejectedException` on a path outside the
|
||||
allowlist. Without it, a caller can search any indexed field, including ones indexed for a
|
||||
different purpose.
|
||||
- Search text length and result count are bounded at construction. An unbounded search text is a
|
||||
cost multiplier on someone else's service.
|
||||
|
||||
## 4. Vector search
|
||||
|
||||
`MongoVectorIndexDescriptor.cosine(path, dimensions)` declares the index.
|
||||
`MongoEmbedding.forIndex(index, values)` binds an embedding to it and **rejects a dimension
|
||||
mismatch** — a 1536-dimension embedding against a 768-dimension index is not a runtime degradation,
|
||||
it is a category error, and catching it at construction beats catching it as a confusing server
|
||||
message.
|
||||
|
||||
`MongoEmbedding` copies its backing array in and out. A vector that shares an array with its caller
|
||||
can be mutated after the query is built, which produces a query nobody wrote.
|
||||
|
||||
`MongoVectorQuery` requires `numCandidates > limit` — searching 10 candidates to return 10 results is
|
||||
an exhaustive scan wearing an ANN index's name. `MongoVectorQuery.nearest(embedding, 10)` uses the
|
||||
standard 20× ratio (200 candidates for 10 results).
|
||||
|
||||
## 5. Relevance is the gate, not functionality
|
||||
|
||||
`MongoVectorSearchBenchmarkGate.standard()` requires **recall** alongside latency and index size.
|
||||
`requiredEvidence()` names recall explicitly.
|
||||
|
||||
This is the difference between search and everything else in the platform. A vector index can be
|
||||
functionally perfect — it accepts the index, accepts the query, returns k results, within the latency
|
||||
budget — and return the wrong k. A gate that measures only latency certifies a fast wrong answer.
|
||||
`failures(recall, latencyMs, indexMb)` reports which dimension failed so the finding is actionable.
|
||||
|
||||
## 6. Failure recovery
|
||||
|
||||
| Symptom | Cause | Action |
|
||||
|---|---|---|
|
||||
| Incomplete results after a deploy | Queried a `BUILDING` index | Wait for `READY`. The gate prevents this; if it fired, something bypassed it. |
|
||||
| `MongoOperationRejectedException` on a path | Path not in the allowlist | Add it deliberately, or fix the caller. |
|
||||
| Dimension mismatch | Model changed | A new model means a new index. Build alongside, cut over, then retire. |
|
||||
| Recall dropped without a code change | The index was rebuilt with different parameters, or the data distribution shifted | Re-run the benchmark gate; treat a recall regression like a failing test. |
|
||||
| Index `FAILED` | Build error on the search service | Search-side diagnosis; the application must not fall back to a scan silently. |
|
||||
|
||||
## 7. Promotion evidence
|
||||
|
||||
Per [ADR-MONGO-ADV-001](../../adr/ADR-MONGO-ADV-001-capability-promotion.md): the actual target
|
||||
deployment (not Atlas Local), security review of `SEARCH_ADMIN`, a rebuild path, failure cases
|
||||
(non-ready index refused, disallowed path refused, dimension mismatch refused), this document as the
|
||||
runbook, **and** relevance evidence. Search and vector search do not promote on functional success.
|
||||
@@ -0,0 +1,82 @@
|
||||
# Advanced — Sharding
|
||||
|
||||
**Capability:** `MongoCapability.SHARDING`
|
||||
**Property:** `ca-skeleton.persistence-mongo.advanced.sharding.enabled`
|
||||
**Status:** Advanced. Reshard orchestration remains Experimental.
|
||||
|
||||
## Requirements
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| Topology | A real sharded cluster. A replica set cannot exercise routing. |
|
||||
| Server | MongoDB 7.0 or 8.0. |
|
||||
| Privilege | `MongoPrincipalRole.SHARD_ADMIN` for the admin plane; the application role is unchanged. |
|
||||
| Gate | `mongoShardedTest` lane with `MongoShardingContractSuite`. |
|
||||
|
||||
## Shard key
|
||||
|
||||
`ShardKeyDescriptor` declares the key as an ordered list of `ShardKeyPart` plus a `ShardStrategy`:
|
||||
|
||||
| Strategy | Distributes | Cost |
|
||||
|---|---|---|
|
||||
| `RANGE` | by value ranges | Range queries stay targeted; a monotonic key (a timestamp, an `ObjectId`) sends every insert to one shard. |
|
||||
| `HASHED` | by hash of the key | Inserts spread evenly; every range query becomes scatter-gather. |
|
||||
|
||||
There is no strategy that is good at both, which is why the choice is a declaration rather than a
|
||||
default.
|
||||
|
||||
## Routing classification
|
||||
|
||||
`ShardAwareQueryValidator` classifies each query before execution:
|
||||
|
||||
| `MongoRoutingClassification` | Meaning |
|
||||
|---|---|
|
||||
| `TARGETED` | The full shard key is present. One shard answers. |
|
||||
| `PREFIX_TARGETED` | A prefix of a compound key is present. A subset of shards answers. |
|
||||
| `SCATTER_GATHER` | No shard-key predicate. Every shard answers. |
|
||||
| `REJECTED` | Scatter-gather where the profile forbids it. |
|
||||
|
||||
A scatter-gather query is not an error — some queries legitimately need every shard — but it must be
|
||||
declared. Undeclared scatter-gather raises `MongoShardRoutingException`. The reason is that
|
||||
scatter-gather passes every test on a single-shard development cluster and only degrades once the
|
||||
cluster grows, at which point the query is already in production and the fix is a schema change.
|
||||
|
||||
## Unsupported combinations
|
||||
|
||||
- Unique index on a field that is not a prefix of the shard key. MongoDB cannot enforce it across
|
||||
shards, and it fails at index creation, not at query time.
|
||||
- Transactions that touch documents on multiple shards remain supported but cost a cross-shard
|
||||
two-phase commit. Prefer a shard key that keeps a transaction's documents co-located.
|
||||
- CSFLE on a sharded collection: see [encryption.md](encryption.md) for the combinations that are
|
||||
refused.
|
||||
|
||||
## Admin plane
|
||||
|
||||
`MongoShardingAdminGateway` (D4, `SHARD_ADMIN` credential) covers shard-collection, refine-shard-key
|
||||
and reshard.
|
||||
|
||||
`ShardKeyAnalyzer` produces a `ShardKeyReadinessReport` before sharding a collection: cardinality,
|
||||
frequency skew and monotonicity. A key with low cardinality creates jumbo chunks that cannot be split;
|
||||
a monotonic key creates a hot shard. Both are visible in the report and invisible in a functional
|
||||
test.
|
||||
|
||||
`ReshardApproval` is required for a reshard — a named approver and a stated window. Resharding
|
||||
rewrites the collection: it duplicates the data during the operation and saturates IO. It is not a
|
||||
runtime operation and the type refuses to pretend otherwise.
|
||||
|
||||
## Failure recovery
|
||||
|
||||
| Symptom | Cause | Action |
|
||||
|---|---|---|
|
||||
| `MongoShardRoutingException` | Undeclared scatter-gather | Add the shard key to the predicate, or declare the query as scatter-gather in its profile after review. |
|
||||
| Jumbo chunks | Low-cardinality shard key | Refine the shard key (adds a suffix, non-destructive) before considering a reshard. |
|
||||
| One hot shard | Monotonic range key | Refine with a high-cardinality prefix, or reshard to hashed if range queries are not needed. |
|
||||
| Balancer never converges | Chunk migration blocked by long-running operations | Check for long transactions and cursors; the balancer waits on them. |
|
||||
|
||||
## Promotion evidence
|
||||
|
||||
Per [ADR-MONGO-ADV-001](../../adr/ADR-MONGO-ADV-001-capability-promotion.md): actual sharded-cluster
|
||||
evidence, security review of the `SHARD_ADMIN` role, a migration path for an existing unsharded
|
||||
collection, failure cases (undeclared scatter-gather refused, jumbo chunk detected), and this
|
||||
document as the runbook. Reshard orchestration stays Experimental until operational scale evidence
|
||||
exists.
|
||||
@@ -0,0 +1,73 @@
|
||||
# Advanced — Time Series
|
||||
|
||||
**Capability:** `MongoCapability.TIME_SERIES`
|
||||
**Property:** `ca-skeleton.persistence-mongo.advanced.time-series.enabled`
|
||||
**Status:** Advanced.
|
||||
|
||||
## Requirements
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| Topology | Replica set or sharded cluster. |
|
||||
| Server | MongoDB 7.0 or 8.0. |
|
||||
| Privilege | Standard application role; collection creation goes through the admin plane. |
|
||||
|
||||
## Descriptor
|
||||
|
||||
`MongoTimeSeriesDescriptor` declares:
|
||||
|
||||
- **timeField** — required, a BSON date. This is the bucketing axis.
|
||||
- **metaField** — optional but nearly always wanted: the series identity (device id, tenant, sensor).
|
||||
Documents sharing a `metaField` value bucket together, which is where the compression comes from.
|
||||
- **granularity** — `MongoTimeSeriesGranularity`:
|
||||
|
||||
| Granularity | Bucket span | Use for |
|
||||
|---|---|---|
|
||||
| `SECONDS` | 1 hour | Sub-second to per-second ingest. |
|
||||
| `MINUTES` | 24 hours | Per-minute metrics. |
|
||||
| `HOURS` | 30 days | Hourly rollups. |
|
||||
|
||||
Granularity that is too fine produces many small buckets and loses the compression; too coarse
|
||||
produces oversized buckets that must be read whole to answer a narrow query.
|
||||
|
||||
## What a time series collection is not
|
||||
|
||||
`MongoTimeSeriesCapabilityValidator` refuses the operations the collection type does not support, at
|
||||
declaration time rather than at first use:
|
||||
|
||||
- **No arbitrary updates.** Time series data is append-mostly. Delete and limited update support
|
||||
exists on recent servers but is not part of this platform's contract.
|
||||
- **No unique index on the measurement.** There is no `_id` to be unique on in the usual sense.
|
||||
- **No CSFLE.** Refused — see [encryption.md](encryption.md).
|
||||
- **No change stream on the raw buckets** as a business event source. The bucket documents are a
|
||||
storage representation, not your measurements.
|
||||
|
||||
Converting an existing regular collection to a time series collection is a copy, not an alter. Plan
|
||||
it as a migration with a dual-write window.
|
||||
|
||||
## TTL
|
||||
|
||||
Time series collections use `expireAfterSeconds` on the collection rather than a TTL index on a
|
||||
field. The [TTL rules](../schema-index-migration-guide.md#4-ttl) still apply: expiry is physical
|
||||
cleanup on a bucket boundary, so a measurement can outlive its expiry by up to a bucket span plus the
|
||||
monitor interval. Do not treat absence as a deadline.
|
||||
|
||||
## Operations
|
||||
|
||||
`MongoTimeSeriesOperations` is the port for insert and windowed read. Reads are bounded by the same
|
||||
`MongoOperationBudget` as everything else: an unbounded time-range query on a time series collection
|
||||
is the fastest way to read a year of data into heap.
|
||||
|
||||
## Failure recovery
|
||||
|
||||
| Symptom | Cause | Action |
|
||||
|---|---|---|
|
||||
| Writes rejected with an unsupported-operation error | An update or unique-index expectation | The collection type does not support it; change the access pattern. |
|
||||
| Poor compression / large storage | Missing `metaField`, or granularity too fine | Both require a rebuild; measure on a copy before committing. |
|
||||
| Slow range queries | Granularity too coarse for the query window | Same: rebuild with the granularity matched to the dominant query. |
|
||||
|
||||
## Promotion evidence
|
||||
|
||||
Actual-topology evidence on the target deployment, a migration path from the existing collection,
|
||||
failure cases (unsupported update refused, CSFLE combination refused), and this document as the
|
||||
runbook.
|
||||
@@ -0,0 +1,92 @@
|
||||
# BSON Mapping Guide
|
||||
|
||||
Design §10 and decision D-06. The representation of a value in BSON is a data contract, not an
|
||||
implementation detail: once a collection holds a million documents, changing how a `BigDecimal` is
|
||||
stored is a migration with downtime, not a code change. `MongoTypeRepresentationManifest` pins the
|
||||
representation so a library upgrade or a different default cannot move it.
|
||||
|
||||
## 1. The manifest
|
||||
|
||||
`MongoTypeRepresentationManifest.standard()` fixes:
|
||||
|
||||
| Java type | BSON | Representation type |
|
||||
|---|---|---|
|
||||
| `UUID` | `Binary` subtype 4 | `MongoUuidRepresentation.STANDARD` |
|
||||
| `BigDecimal` | `Decimal128` | `MongoDecimalRepresentation.DECIMAL_128` |
|
||||
| `BigInteger` | `Decimal128` (or `String` when out of range, declared) | `MongoBigIntegerRepresentation` |
|
||||
| `Instant` / `OffsetDateTime` / `ZonedDateTime` | UTC `Date` | `MongoTemporalRepresentation.UTC_DATE` |
|
||||
| `LocalDate` | `String` (ISO-8601) or UTC `Date`, declared per field | `MongoTemporalRepresentation` |
|
||||
| `enum` | `String` name | `MongoEnumRepresentation.NAME` |
|
||||
|
||||
`MongoMappingConfiguration` and `MongoCustomConversionsFactory` build the Spring Data converters from
|
||||
the manifest, so there is one place to read and one place to change.
|
||||
|
||||
## 2. UUID
|
||||
|
||||
`UuidRepresentation.STANDARD` (subtype 4), always. The driver's legacy Java representation
|
||||
(subtype 3) byte-swaps two halves of the UUID, so a document written by one representation and read
|
||||
by the other yields a different — and valid-looking — UUID. Nothing errors; you just get the wrong
|
||||
row. The golden snapshot kit pins the codec explicitly for this reason
|
||||
(`MongoBsonSnapshot.defaultRegistry()`).
|
||||
|
||||
## 3. Decimal
|
||||
|
||||
`BigDecimal` → `Decimal128`, never `Double`. `12.30` stored as a double is `12.299999999999999`, and
|
||||
a monetary comparison written against it will one day be wrong by a cent for a customer who notices.
|
||||
`BigDecimalToDecimal128Converter` / `Decimal128ToBigDecimalConverter` are registered from the
|
||||
manifest.
|
||||
|
||||
`Decimal128` has 34 significant digits; a `BigDecimal` beyond that range fails on write rather than
|
||||
rounding silently.
|
||||
|
||||
## 4. Time
|
||||
|
||||
Store instants, not local times. `LocalDateTimeMappingGuard` refuses `LocalDateTime` fields on a
|
||||
mapped document: a `LocalDateTime` has no offset, so the value that goes in depends on the JVM
|
||||
default zone of whichever instance wrote it, and the two instances in a rolling deploy can disagree.
|
||||
Use `Instant` when the moment matters and `LocalDate` when the calendar day matters.
|
||||
|
||||
## 5. Type metadata
|
||||
|
||||
`MongoTypeMetadataPolicy` decides what goes in `_class`:
|
||||
|
||||
| Policy | Stored | Use when |
|
||||
|---|---|---|
|
||||
| `NONE` | nothing | The collection holds exactly one type and never will hold a subtype. |
|
||||
| `ALIAS` | a registered short alias | A polymorphic hierarchy in a long-lived collection. |
|
||||
| `CLASS_NAME` | the FQCN | Short-lived or internal collections only. |
|
||||
|
||||
`PolicyAwareMongoTypeMapper` enforces it, and `MongoTypeMetadataRegistry` holds alias → class.
|
||||
A `@LongLivedMongoDocument` type with `CLASS_NAME` is refused: writing `com.example.OrderV2` into a
|
||||
million documents means that renaming the package is a data migration.
|
||||
|
||||
## 6. Missing versus null
|
||||
|
||||
The golden kit keeps these apart deliberately. `MongoBsonSnapshotAssert.hasNoField(...)` and
|
||||
`hasExplicitNull(...)` are different assertions, because in MongoDB they are different documents:
|
||||
`{"a": null}` matches `{a: null}` and `{a: {$exists: true}}`, while `{}` matches only the first.
|
||||
A mapper change that starts writing explicit nulls silently changes what your queries return.
|
||||
|
||||
## 7. Golden representation tests
|
||||
|
||||
Every collection with a fixed representation should have a snapshot test:
|
||||
|
||||
```java
|
||||
MongoBsonSnapshot snapshot = MongoBsonSnapshot.of(storedDocument);
|
||||
MongoBsonSnapshotAssert.assertThat(snapshot)
|
||||
.hasBsonType("amount", "DECIMAL128")
|
||||
.hasBsonType("externalId", "BINARY")
|
||||
.hasNoJavaClassName("dev.caskeleton")
|
||||
.hasTypeSignature("_id:OBJECT_ID,amount:DECIMAL128,createdAt:DATE_TIME,externalId:BINARY");
|
||||
```
|
||||
|
||||
`hasTypeSignature` is the regression gate: it fails on *any* representation change, including ones a
|
||||
value-equality assertion would pass. When it fails, the question is whether the change was intended
|
||||
and has a migration — not whether to update the string.
|
||||
|
||||
## 8. Round trips
|
||||
|
||||
`MongoRoundTripContract` asserts that `write → read` returns an equal domain object *and* that
|
||||
`write → read → write` produces an identical BSON document. The second half is what catches an
|
||||
asymmetric converter: a value that reads back equal but re-serialises differently makes every
|
||||
subsequent `save()` a spurious update, and turns change streams into a noise generator.
|
||||
@@ -0,0 +1,105 @@
|
||||
# Change Stream Guide
|
||||
|
||||
Design §20, decision D-12. A change stream is an **at-least-once projector**, not an event bus.
|
||||
|
||||
## 1. What a change stream is not
|
||||
|
||||
D-12 is explicit: a physical change event is not a business integration event. The two differ in
|
||||
ways that matter to every consumer:
|
||||
|
||||
| Change event | Integration event |
|
||||
|---|---|
|
||||
| Emitted per document write | Emitted per business fact |
|
||||
| Shape follows the storage schema | Shape is a published contract |
|
||||
| A refactor of the document changes it | A refactor of the document does not change it |
|
||||
| Replayed on resume, duplicated on retry | Versioned and deliberately evolved |
|
||||
|
||||
Publishing raw change events externally makes your storage schema a public API, and the first time
|
||||
someone renames a field the downstream consumers break. If you need to bridge to messaging, use the
|
||||
Advanced bridge, which maps to an owned envelope
|
||||
([advanced/multi-tenancy.md](advanced/multi-tenancy.md) is separate;
|
||||
the bridge is described in §7 below).
|
||||
|
||||
## 2. Subscription and resume
|
||||
|
||||
`MongoChangeStreamSubscription` declares the collection, pipeline and consistency. `MongoResumePosition`
|
||||
is either a resume token or a cluster time; `MongoResumeCheckpoint` is what gets persisted and
|
||||
`MongoResumeCheckpointStore` persists it.
|
||||
|
||||
The checkpoint stores the token as a Base64 `encodedToken` string rather than a byte array — a record
|
||||
with an array component has broken equality, and a checkpoint that does not compare correctly is a
|
||||
checkpoint that silently fails its own dedup test.
|
||||
|
||||
## 3. Checkpoint after processing, not after receiving
|
||||
|
||||
The ordering rule that makes at-least-once actually hold:
|
||||
|
||||
```
|
||||
receive event
|
||||
→ process it (idempotently)
|
||||
→ persist the checkpoint
|
||||
```
|
||||
|
||||
Checkpointing on receipt turns the delivery guarantee into at-most-once, and the events lost are
|
||||
exactly the ones the process died while handling.
|
||||
|
||||
## 4. Idempotency
|
||||
|
||||
`MongoChangeEventIdentity` is the dedup key: `(resumeToken, documentKey, clusterTime, operationType)`.
|
||||
`MongoChangeDeduplicationStore` records what has been applied. Duplicates are not an edge case — every
|
||||
resume after any interruption replays at least one event, so a projector that is not idempotent is
|
||||
wrong on its first restart, not on some rare day.
|
||||
|
||||
`MongoChangeProjector` returns a `MongoChangeProjectionResult` so the runner can distinguish applied
|
||||
from skipped-as-duplicate, and the skip count is worth a metric: a sudden rise means something is
|
||||
looping.
|
||||
|
||||
## 5. States and recovery
|
||||
|
||||
`MongoChangeStreamState`: `STARTING`, `RUNNING`, `RESUMING`, `STOPPED`, `HISTORY_LOST`.
|
||||
|
||||
`MongoChangeStreamRecoveryPolicy` returns a `MongoChangeStreamRecoveryDecision`, which is either
|
||||
`resume()` (auto-resume from the checkpoint) or `halt(state, runbook)`. A halting decision **must**
|
||||
name a runbook — a decision that only says "stopped" leaves the on-call engineer to work out from
|
||||
scratch whether the projection can be rebuilt and from what.
|
||||
|
||||
| Situation | Decision |
|
||||
|---|---|
|
||||
| Transient network error, token still valid | `resume()` |
|
||||
| Primary failover | `resume()` — the token survives an election |
|
||||
| `invalidate` (collection dropped/renamed) | `halt(STOPPED, …)` → `MongoInvalidateRecovery` |
|
||||
| Token no longer in the oplog | `halt(HISTORY_LOST, "history-lost")` → `MongoChangeHistoryLostException` |
|
||||
|
||||
## 6. History lost
|
||||
|
||||
`MongoChangeHistoryLostException` is raised when the resume token predates the oldest oplog entry.
|
||||
The stream **cannot** be resumed: the events between the checkpoint and now are gone from the server,
|
||||
and no amount of retrying brings them back.
|
||||
|
||||
What the platform will not do is silently restart from "now". That looks like a recovery and is
|
||||
actually a silent gap in the projection — the worst possible outcome, because nothing reports it. The
|
||||
runner halts and requires an operator decision. See
|
||||
[runbooks/history-lost.md](runbooks/history-lost.md).
|
||||
|
||||
## 7. Bridging to messaging (Advanced)
|
||||
|
||||
`MongoChangeMessagingBridge` is opt-in behind `MongoCapability.CHANGE_STREAM` plus the bridge's own
|
||||
flag. It maps a change event to a platform-owned `MongoIntegrationEventEnvelope` through
|
||||
`MongoChangeToIntegrationEventMapper` and hands it to a `MongoIntegrationEventPublisher` port.
|
||||
|
||||
The port is defined in the bridge package rather than imported from the messaging adapter because the
|
||||
architecture registry forbids adapter-to-adapter dependencies; the composition root supplies the
|
||||
implementation.
|
||||
|
||||
`MongoBridgeOutboxPolicy` and `MongoBridgeCheckpointPolicy` state the delivery contract: publish then
|
||||
checkpoint, at-least-once, consumers must dedup on the envelope's event id.
|
||||
|
||||
## 8. Operating notes
|
||||
|
||||
- Change streams require a replica set. `MongoStartupValidator` refuses a change-stream profile on
|
||||
`STANDALONE`.
|
||||
- The change-stream principal is its own role (`MongoPrincipalRole.CHANGE_STREAM`) with
|
||||
`changeStream` and `find` — not the application write credential.
|
||||
- Oplog window is the recovery budget. If the oplog holds four hours, a consumer that is down for five
|
||||
hours needs a rebuild, not a resume. Alert on consumer lag against the oplog window, not against
|
||||
wall-clock.
|
||||
@@ -0,0 +1,126 @@
|
||||
# Consistency and Transaction Guide
|
||||
|
||||
Design §12–§16, decisions D-07 through D-10. This is the part of the platform where the wrong
|
||||
default is most expensive and the least visible in testing, because every failure mode here needs a
|
||||
primary change to reproduce.
|
||||
|
||||
## 1. Prefer a single-document atomic operation
|
||||
|
||||
D-09: a transaction is for a **multi-document invariant**, nothing else. A single document is already
|
||||
atomic in MongoDB, so wrapping a one-document update in a transaction buys nothing and costs a
|
||||
session, a two-phase commit and a new ambiguous outcome.
|
||||
|
||||
D-07: partial change uses update operators, not `save()`. `MongoAtomicOperations` /
|
||||
`MongoAtomicOperationsTemplate` expose the operator set through `MongoUpdateOperator`
|
||||
(`$set`, `$inc`, `$push`, `$pull`, `$addToSet`, `$min`, `$max`, `$currentDate`, …) with an
|
||||
`AtomicFilter` precondition and a `ReturnDocumentMode`. Read-modify-write through `save()` replaces
|
||||
the whole document and silently discards any field another writer changed in between — a lost update
|
||||
with no error.
|
||||
|
||||
## 2. Whole-document replacement needs a revision
|
||||
|
||||
D-08. `VersionedMongoUpdater` requires a `MongoRevision`: either a Spring Data `@Version` field or an
|
||||
explicit expected-revision predicate in `VersionedUpdateCommand`. A replacement whose filter matched
|
||||
zero documents is not "nothing to do" — `MongoOptimisticConflictTranslator` distinguishes:
|
||||
|
||||
- filter matched nothing and the id does not exist → `MongoDocumentNotFoundException`
|
||||
- filter matched nothing and the id exists → `MongoOptimisticConflictException`
|
||||
|
||||
Collapsing these two into one is how a concurrent overwrite becomes a 404.
|
||||
|
||||
## 3. Consistency profiles
|
||||
|
||||
`MongoConsistencyProfile` names the read/write concern pair; `MongoConsistencyRegistry` binds a
|
||||
profile to an operation or collection, and `MongoConsistencyBinder` /
|
||||
`ReactiveMongoConsistencyBinder` apply it at execution.
|
||||
|
||||
| Profile | Meaning | Use for |
|
||||
|---|---|---|
|
||||
| `PRIMARY_LOCAL` | primary read, local concern | Throughput-sensitive reads that tolerate a rollback window. |
|
||||
| `PRIMARY_MAJORITY` | primary read, majority write | The default for anything a user will see again immediately. |
|
||||
| `CAUSAL_MAJORITY` | majority inside a causal session | Read-your-writes across separate operations. |
|
||||
| `STALE_READ_ALLOWED` | secondary reads permitted | Reporting and analytics that state their staleness. |
|
||||
| `SNAPSHOT_TRANSACTION` | snapshot isolation | Multi-document reads inside a transaction. |
|
||||
|
||||
A profile is a declaration, not a hint: the registry is consulted per operation and an operation
|
||||
without a registered profile is rejected rather than defaulting.
|
||||
|
||||
## 4. Causal sessions
|
||||
|
||||
`MongoCausalSessionContext` plus `SpringMongoCausalSessionExecutor` /
|
||||
`ReactiveMongoCausalSessionExecutor` carry the cluster time and operation time between operations, so
|
||||
"write then read" returns the write even when the read lands on a different node. Without a causal
|
||||
session, `PRIMARY_MAJORITY` gives you durability but not read-your-writes across two calls.
|
||||
|
||||
In the reactive path the session travels in the Reactor context (`ReactiveMongoContextKeys`), not in
|
||||
a thread local — a thread local is empty on the next operator in the chain.
|
||||
|
||||
## 5. Transactions
|
||||
|
||||
`MongoTransactionExecutor` / `ReactiveMongoTransactionExecutor` open a session through the session
|
||||
factory, run the body, and commit. `MongoTransactionProfile` carries the consistency profile, the
|
||||
`maxCommitTime` and the retry budget. Topology matters: a transaction requires a replica set, and
|
||||
`MongoStartupValidator` refuses a transaction-declaring profile on `STANDALONE` at startup rather
|
||||
than at the first call.
|
||||
|
||||
## 6. Retry: body and commit are different loops
|
||||
|
||||
D-10, and the single most consequential rule in the design.
|
||||
|
||||
```
|
||||
for each body attempt:
|
||||
open a NEW session
|
||||
run the body
|
||||
TransientTransactionError -> abort, next body attempt
|
||||
commit
|
||||
UnknownTransactionCommitResult -> retry COMMIT ONLY, same session
|
||||
```
|
||||
|
||||
`MongoTransactionRetryCoordinator` implements exactly this:
|
||||
|
||||
- **A new session per body attempt.** Reusing the session after an abort carries the aborted
|
||||
transaction's state into the retry.
|
||||
- **The body is never replayed after a commit ambiguity.** An unknown commit means the commit may
|
||||
already have applied. Re-running the body would apply it a second time. Only the commit is retried,
|
||||
and a commit retry on an already-committed transaction is a no-op by design.
|
||||
- **A budget bounds both loops.** `MongoRetryBudget` limits attempts *and* elapsed time, with jittered
|
||||
backoff (`delayBefore(attempt, random)`), so a struggling primary is not retried into the ground.
|
||||
|
||||
`MongoRetryDecision` and `MongoRetryScope` (in `…api.error`) say what may be retried:
|
||||
`MongoRetryScope.BODY`, `COMMIT_ONLY`, or `NONE`.
|
||||
|
||||
## 7. Ambiguous outcomes
|
||||
|
||||
`MongoExecutionOutcome` has six values, two of which are ambiguous and must not be collapsed:
|
||||
|
||||
| Outcome | Did the write happen? |
|
||||
|---|---|
|
||||
| `NOT_SENT` | No. Safe to retry. |
|
||||
| `NO_WRITE_PERFORMED` | No — the server answered and did nothing. |
|
||||
| `WRITE_CONFIRMED` | Yes. |
|
||||
| `PARTIAL_BULK_WRITE` | Some of it. See `MongoBulkResult`. |
|
||||
| `WRITE_RESULT_UNKNOWN` | **Unknown.** |
|
||||
| `TRANSACTION_COMMIT_UNKNOWN` | **Unknown.** |
|
||||
|
||||
An unknown outcome is not a failure and must not be reported to a caller as one. The caller either
|
||||
reconciles (`MongoCommitReconciler` re-reads a deterministic marker the body wrote) or surfaces the
|
||||
ambiguity. See [runbooks/unknown-commit.md](runbooks/unknown-commit.md).
|
||||
|
||||
`MongoFailureContext` records only the design-permitted fields — outcome, category, operation name,
|
||||
collection profile, retry scope, attempt — never the query, the document, or the values.
|
||||
|
||||
## 8. Failure translation
|
||||
|
||||
`DefaultMongoFailureClassifier` classifies **labels before codes**. The server's error labels
|
||||
(`TransientTransactionError`, `UnknownTransactionCommitResult`, `RetryableWriteError`) are the
|
||||
authoritative statement about retryability; an error code is a secondary signal whose meaning varies
|
||||
by server version. `DefaultMongoFailureTranslator` maps a classification onto the stable exception
|
||||
hierarchy, and anything unmatched becomes `MongoUnclassifiedFailureException` rather than leaking a
|
||||
driver type.
|
||||
|
||||
## 9. Bulk writes
|
||||
|
||||
`MongoBulkExecutor` returns a `MongoBulkResult` with per-item `MongoBulkItemFailure` entries. An
|
||||
unordered bulk write that partially fails is `PARTIAL_BULK_WRITE`, not a failure: some documents were
|
||||
written. `MongoBulkPartialFailureException` carries the succeeded and failed indexes so a caller can
|
||||
resume rather than replay.
|
||||
@@ -0,0 +1,85 @@
|
||||
# Document Modeling Guide
|
||||
|
||||
Design §7–§9. The platform does not own your documents — D-01 is explicit that the domain owns
|
||||
`@Document`, repositories, queries, index requirements and schema version. What the platform owns is
|
||||
the set of modeling decisions that are expensive to reverse once a collection holds production data.
|
||||
|
||||
## 1. There is no `CommonMongoRepository`
|
||||
|
||||
A generic `CommonMongoRepository<T, ID>` is listed under explicitly unsupported (§3.4), and the
|
||||
reason is not purity. A shared supertype forces every collection to share an id strategy, a
|
||||
consistency profile and a query surface, and the first collection that needs a different one either
|
||||
gets a cast or a leaky generic parameter. Declare a Spring Data repository per aggregate.
|
||||
|
||||
## 2. Embed or reference
|
||||
|
||||
`MongoDocumentModelManifest` records the decision per collection so it is reviewable, and
|
||||
`MongoDocumentModelValidator` refuses the combinations that do not survive growth.
|
||||
|
||||
| Descriptor | Use when |
|
||||
|---|---|
|
||||
| `EmbeddedCollectionDescriptor` | The child is read with the parent, is bounded, and has no independent lifecycle. Declare `maxElements`; an unbounded array is the single most common way a document reaches the size limit. |
|
||||
| `MongoReferenceDescriptor` | The child is queried independently, is unbounded, or outlives the parent. Declare `MongoReferenceLifecycle` so the deletion story is written down rather than discovered. |
|
||||
|
||||
The validator rejects an embedded collection without a bound, and a reference whose lifecycle says
|
||||
the child is owned by the parent but which is also referenced from elsewhere.
|
||||
|
||||
## 3. Size budget
|
||||
|
||||
`MongoDocumentSizeBudget`:
|
||||
|
||||
| Constant | Bytes | Meaning |
|
||||
|---|---|---|
|
||||
| `MONGODB_HARD_LIMIT_BYTES` | 16 MiB | MongoDB's own limit. |
|
||||
| `PLATFORM_CEILING_BYTES` | 4 MiB | The largest budget the platform will accept. |
|
||||
| `DEFAULT_BYTES` | 2 MiB | `MongoDocumentSizeBudget.standard()`. |
|
||||
|
||||
A budget above the ceiling is refused at construction. Budgeting to 16 MiB means the failing write
|
||||
is the first symptom, and by then the collection is already full of near-limit documents.
|
||||
|
||||
## 4. Identity
|
||||
|
||||
`DomainDocumentId` and `MongoIdRepresentation` fix how a domain identifier becomes `_id`. Pick the
|
||||
representation once per collection and record it in the manifest:
|
||||
|
||||
- `OBJECT_ID` — server-generated, monotonic, 12 bytes. Good default when the domain has no natural id.
|
||||
- `UUID_BINARY` — a domain UUID stored as `Binary` subtype 4 (`STANDARD`). Never store a UUID as a
|
||||
string "because it is easier to read"; it doubles the index size and loses the type.
|
||||
- `STRING` — a natural key that is genuinely a string (a slug, an external system's id).
|
||||
|
||||
An `_id` choice is effectively permanent: it is the shard key candidate, the resume-token join key
|
||||
and the pagination tie-breaker.
|
||||
|
||||
## 5. Schema version
|
||||
|
||||
Every long-lived collection carries `DocumentSchemaVersion`. `MongoSchemaVersionPolicy` and
|
||||
`MongoSchemaVersionRange` say which versions the running code can read; a document outside the range
|
||||
raises `MongoDataSchemaUnsupportedException` rather than being silently mapped with missing fields.
|
||||
|
||||
Write the range down before the migration, not after: the range is what lets old and new instances
|
||||
run at once during a rolling deploy.
|
||||
|
||||
## 6. Type metadata
|
||||
|
||||
`@LongLivedMongoDocument` marks a document whose stored type alias must not be a Java class name.
|
||||
`MongoTypeMetadataRegistry` maps alias → class. Storing the FQCN means moving or renaming the class
|
||||
becomes a data migration; storing an alias keeps it a refactor. See
|
||||
[bson-mapping-guide.md](bson-mapping-guide.md) §4.
|
||||
|
||||
## 7. Collection profiles
|
||||
|
||||
`MongoCollectionProfileRegistry` binds a `CollectionProfileName` to its consistency profile, budget
|
||||
and allowlist. A collection that is not registered cannot be reached through
|
||||
`MongoImperativeExecutor` or `ReactiveMongoExecutor` — the allowlist is the mechanism that keeps an
|
||||
unreviewed collection from appearing in production by accident.
|
||||
|
||||
## 8. What to write down before the first insert
|
||||
|
||||
1. Embed/reference decision per child collection, with bounds.
|
||||
2. Size budget.
|
||||
3. `_id` representation.
|
||||
4. Schema version range.
|
||||
5. Index manifest (see [schema-index-migration-guide.md](schema-index-migration-guide.md)).
|
||||
6. Consistency profile (see [consistency-transaction-guide.md](consistency-transaction-guide.md)).
|
||||
|
||||
Each of these is cheap now and a migration later.
|
||||
@@ -0,0 +1,140 @@
|
||||
# Query and Aggregation Guide
|
||||
|
||||
Design §17–§19, decision D-11. Every query and every pipeline is a registered, bounded thing. Free-form
|
||||
JSON queries and unbounded pipelines are explicitly unsupported (§3.4).
|
||||
|
||||
## 1. Registered operations
|
||||
|
||||
Every execution carries a `MongoOperationContext`: a `MongoOperationName`, a `DatabaseProfileName`, a
|
||||
`CollectionProfileName`, a `MongoOperationType` and a `MongoOperationScope`.
|
||||
|
||||
`MongoOperationName` matches `[a-z][a-z0-9.-]{2,95}`. It is the join key for the budget registry, the
|
||||
consistency registry, the metric tag and the log line — a free-form or interpolated name breaks all
|
||||
four at once, which is why the pattern is enforced at construction.
|
||||
|
||||
`MongoOperationScope` uses an `UNSPECIFIED` sentinel rather than `null`, so "the caller did not say"
|
||||
is a value the policy layer can reject rather than an NPE further down.
|
||||
|
||||
## 2. Query guardrails
|
||||
|
||||
`PolicyAwareMongoQueryBuilder` builds a query from `MongoFieldDescriptor` + `MongoOperator` pairs
|
||||
against a `MongoQueryPolicy`. The policy refuses:
|
||||
|
||||
- a field not in the collection's allowlist
|
||||
- an operator not allowed for that field
|
||||
- a sort on an unindexed field
|
||||
- `$where`, `$expr` with arbitrary JavaScript, and server-side evaluation generally
|
||||
- an unbounded `$regex`
|
||||
|
||||
`MongoRegexPolicy` requires an anchored prefix pattern and bounds the pattern length. An unanchored
|
||||
regex is a collection scan wearing an index's clothes, and a user-supplied one is a denial-of-service
|
||||
primitive.
|
||||
|
||||
`MongoSortDescriptor` pairs a field with a direction and is validated against the index manifest, so
|
||||
a sort that would spill to disk fails review rather than production.
|
||||
|
||||
## 3. Operation budgets
|
||||
|
||||
`MongoOperationBudget` bounds four things at once:
|
||||
|
||||
| Bound | Why |
|
||||
|---|---|
|
||||
| `maxTimeMS` | The server stops working on a query nobody is waiting for. |
|
||||
| result limit | An unbounded result set is an OOM with extra steps. |
|
||||
| batch size | Bounds the per-round-trip memory. |
|
||||
| examined-document ceiling | Catches an index regression that a time limit alone would hide on a fast day. |
|
||||
|
||||
`MongoBudgetPolicyRegistry` binds a budget to an operation name; `MongoBudgetEnforcer` applies it and
|
||||
raises `MongoOperationRejectedException` before execution when a request exceeds it, and
|
||||
`MongoTimeoutException` when the server enforces it.
|
||||
|
||||
## 4. Keyset pagination
|
||||
|
||||
Unbounded `skip` is unsupported: `skip(1_000_000)` makes the server walk a million documents to throw
|
||||
them away, so page 1000 costs a thousand times page 1.
|
||||
|
||||
`MongoKeysetQueryBuilder` builds the resume predicate lexicographically. For a sort on `(a DESC, _id
|
||||
DESC)` resuming after `(A, I)`:
|
||||
|
||||
```
|
||||
(a < A) OR (a = A AND _id < I)
|
||||
```
|
||||
|
||||
`validate()` rejects a `MongoKeysetSort` without a unique tie-breaker. Without one, two documents with
|
||||
the same sort value straddle the page boundary and one of them is skipped or repeated — invisibly,
|
||||
and only under concurrency.
|
||||
|
||||
`MongoNullSortOrdering` makes null placement explicit, because MongoDB's own ordering of missing
|
||||
versus null versus present is not what most people assume.
|
||||
|
||||
### Cursors are authenticated
|
||||
|
||||
`MongoKeysetCursorCodec` signs the cursor with HMAC-SHA256 and compares with
|
||||
`MessageDigest.isEqual` (constant time). An unsigned cursor is a client-controlled query predicate: a
|
||||
caller can edit it to read a range they were never offered. A tampered or truncated cursor yields
|
||||
`MongoCursorException`, never a partially-decoded resume position.
|
||||
|
||||
## 5. Aggregation guardrails
|
||||
|
||||
`MongoAggregationPlan` is a registered pipeline: an ordered list of `MongoAggregationStageDescriptor`
|
||||
validated against a `MongoAggregationProfile`. `PolicyAwareMongoAggregationExecutor` runs only a
|
||||
registered plan.
|
||||
|
||||
`MongoAggregationRisk` grades each stage, and the profile sets the ceiling:
|
||||
|
||||
| Risk | Stages | Policy |
|
||||
|---|---|---|
|
||||
| low | `$match` on an indexed prefix, `$limit`, `$project` | Always allowed. |
|
||||
| moderate | `$group`, `$sort` with an index, `$unwind` with a bound | Allowed within budget. |
|
||||
| high | `$lookup`, `$graphLookup`, `$facet`, unindexed `$sort` | Requires explicit approval in the profile. |
|
||||
| forbidden | `$out`, `$merge` outside the admin plane, `$function`, `$accumulator` | Refused. |
|
||||
|
||||
`allowDiskUse` is a declared property of the plan, not a runtime flag. A pipeline that needs disk is a
|
||||
pipeline whose shape should be reviewed.
|
||||
|
||||
## 6. Reactive execution and cursors
|
||||
|
||||
`ReactiveMongoExecutor` / `DefaultReactiveMongoExecutor` carry the operation context in the Reactor
|
||||
context. `MongoCursorGuard` and `MongoCursorLease` bound cursor lifetime:
|
||||
|
||||
- a cursor has a lease with a deadline
|
||||
- cancellation closes the server-side cursor (`MongoCursorTermination`)
|
||||
- an abandoned cursor is a server-side resource, so the lease is released on cancel, error *and*
|
||||
completion — `MongoReactiveCursorPublisher` uses `Flux.using` so all three paths run the same
|
||||
release
|
||||
|
||||
A leaked cursor does not fail anything locally; it consumes a connection and a snapshot on the server
|
||||
until the server's own timeout, which is why the guard is not optional.
|
||||
|
||||
## 7. Geospatial
|
||||
|
||||
`MongoGeoQuery` + `MongoGeoPoint` + `MongoGeoDistance` over a `2dsphere` index. Distances are metres
|
||||
on a sphere (`nearSphere` with `maxDistance`), never degrees — a degree of longitude is a different
|
||||
distance in Oslo than in Nairobi, and a radius expressed in degrees is a bug that only shows up away
|
||||
from the equator. `SpringMongoGeospatialOperations` is the Spring Data binding;
|
||||
`MongoGeospatialOperations` is the port.
|
||||
|
||||
## 8. Native capability gateway
|
||||
|
||||
When a registered operation genuinely needs something outside the Stable API, it goes through
|
||||
`MongoNativeCapabilityGateway` (`PolicyAwareMongoNativeGateway`), never through the driver directly.
|
||||
The admission order is fixed:
|
||||
|
||||
```
|
||||
capability registered
|
||||
→ database profile
|
||||
→ collection allowlist
|
||||
→ operation name present
|
||||
→ timeout / maxTimeMS
|
||||
→ consistency profile
|
||||
→ result / batch limit
|
||||
→ trace
|
||||
→ log redaction
|
||||
→ command category (MongoNativeCommandCategory)
|
||||
→ D4 admin command refused
|
||||
→ execute
|
||||
```
|
||||
|
||||
`ApprovedMongoNativeOperation` is the registration record; `MongoNativeOperationPolicy` is the policy.
|
||||
An admin-plane command reaching this gateway is refused regardless of capability — the admin plane has
|
||||
its own credential and its own client (see [security-observability.md](security-observability.md)).
|
||||
@@ -0,0 +1,124 @@
|
||||
# MongoDB Document Persistence Platform — Repository Adaptation Contract
|
||||
|
||||
**Design source:** `mongodb-superpowers-package/docs/superpowers/specs/2026-08-11-mongodb-document-persistence-platform-design.md`
|
||||
**Stable plan:** `mongodb-superpowers-package/docs/superpowers/plans/2026-08-11-mongodb-document-persistence-platform-implementation-plan.md`
|
||||
**Advanced plan:** `mongodb-superpowers-package/docs/superpowers/plans/2026-08-11-mongodb-advanced-capabilities-expansion-plan.md`
|
||||
|
||||
The design package declares its own module root (`modules/mongodb`) and root package
|
||||
(`io.backend.skeleton.mongodb`) as *implementation assumptions*, not as contract. This file is the
|
||||
single record of how that assumed layout was mapped onto this repository. Only paths, build DSL,
|
||||
and composition-root ownership changed. Public contracts, policy order, and error semantics are
|
||||
implemented exactly as specified.
|
||||
|
||||
## 1. Why the module layout differs
|
||||
|
||||
The design assumes 19 Stable Gradle projects under `modules/mongodb/` and 12 Advanced projects
|
||||
under `modules/mongodb-advanced/`. This repository is a Clean Architecture template whose
|
||||
**fail-closed registry** (`src/config/architecture/modules.json`, enforced by `src/settings.gradle`
|
||||
and `verifyCleanArchitectureDependencies`) declares **exactly 19 leaf identities**. Creating 31 more
|
||||
Gradle projects would violate HARD-STOP #5 in `AGENTS.md`.
|
||||
|
||||
Therefore the design's 31 modules become **package boundaries inside the registered leaf**
|
||||
`:adapter:outbound:persistence-mongo`, following the precedent already set by
|
||||
[docs/httpclient/repository-adaptation.md](../httpclient/repository-adaptation.md). The design's
|
||||
module dependency table (§6.3) is reproduced as ten ArchUnit rules in `MongoModuleBoundaryTest`, so
|
||||
a forbidden edge fails the build the same way a missing Gradle dependency would.
|
||||
|
||||
## 2. Package mapping
|
||||
|
||||
Root package: `io.backend.skeleton.mongodb` → `dev.caskeleton.adapter.outbound.mongo`.
|
||||
|
||||
### 2.1 Stable modules
|
||||
|
||||
| Design module | Repository package |
|
||||
|---|---|
|
||||
| `mongodb-core-api` | `…outbound.mongo.api` (+ `.capability`, `.consistency`, `.error`, `.mapping`, `.observation`, `.profile`, `.schema`) |
|
||||
| `mongodb-spring-data` | `…outbound.mongo.mapping` (+ `.type`), `…outbound.mongo.failure` |
|
||||
| `mongodb-imperative` | `…outbound.mongo.imperative` (+ `.atomic`, `.bulk`, `.revision`) |
|
||||
| `mongodb-reactive` | `…outbound.mongo.reactive` (+ `.cursor`) |
|
||||
| `mongodb-query` | `…outbound.mongo.query` (+ `.budget`, `.pagination`) |
|
||||
| `mongodb-aggregation` | `…outbound.mongo.aggregation` |
|
||||
| `mongodb-transaction` | `…outbound.mongo.transaction` (+ `.retry`, `.session`) |
|
||||
| `mongodb-index-schema` | `…outbound.mongo.schema` (+ `.index`, `.manifest`, `.model`, `.ttl`, `.validation`) |
|
||||
| `mongodb-change-stream` | `…outbound.mongo.changestream` (+ `.projector`, `.recovery`) |
|
||||
| `mongodb-geospatial` | `…outbound.mongo.geo` |
|
||||
| `mongodb-migration-core` | `…outbound.mongo.migration` |
|
||||
| `mongodb-migration-flamingock` | `…outbound.mongo.migration.flamingock` |
|
||||
| `mongodb-observability` | `…outbound.mongo.observation` |
|
||||
| `mongodb-security` | `…outbound.mongo.security` (+ `.admin`), `…outbound.mongo.nativecap` |
|
||||
| `mongodb-spring-boot-starter` | `…outbound.mongo.autoconfigure` |
|
||||
| `mongodb-testkit-core` | `…outbound.mongo.testkit.mapping`, `.compat`, `.performance` (`testkit` source set) |
|
||||
| `mongodb-testkit-replicaset` | `…outbound.mongo.testkit.rs` (`testkit` source set) |
|
||||
| `mongodb-testkit-failover` | `…outbound.mongo.testkit.failover` (`testkit` source set) |
|
||||
| `mongodb-testkit-migration` | `…outbound.mongo.testkit.migration` (`testkit` source set) |
|
||||
|
||||
`…outbound.mongo.architecture` has no design counterpart: it holds the `@MongoOperation` marker and
|
||||
the reusable ArchUnit rule set a fork applies to its own document/repository code.
|
||||
|
||||
### 2.2 Advanced modules
|
||||
|
||||
| Design module | Repository package |
|
||||
|---|---|
|
||||
| `mongodb-sharding` | `…outbound.mongo.advanced.sharding` (+ `.admin` for the D4 shard plane) |
|
||||
| `mongodb-timeseries` | `…outbound.mongo.advanced.timeseries` |
|
||||
| `mongodb-csfle` | `…outbound.mongo.advanced.encryption.csfle` |
|
||||
| `mongodb-queryable-encryption` | `…outbound.mongo.advanced.encryption.qe` |
|
||||
| `mongodb-search` | `…outbound.mongo.advanced.search` |
|
||||
| `mongodb-vector-search` | `…outbound.mongo.advanced.vector` |
|
||||
| `mongodb-tenancy-shared` | `…outbound.mongo.advanced.tenancy.shared` |
|
||||
| `mongodb-tenancy-database` | `…outbound.mongo.advanced.tenancy.database` |
|
||||
| `mongodb-change-stream-messaging-bridge` | `…outbound.mongo.advanced.bridge` |
|
||||
| `mongodb-gridfs-compat` | `…outbound.mongo.advanced.gridfs` |
|
||||
| `mongodb-testkit-sharded` | `…outbound.mongo.testkit.sharded` (`testkit` source set) |
|
||||
| `mongodb-testkit-atlas` | `…outbound.mongo.testkit.atlas` (`testkit` source set) |
|
||||
|
||||
The design's rule that a Stable module never depends on an Advanced one survives as an ArchUnit rule
|
||||
(`stableNeverDependsOnAdvanced`) plus the opt-in flag: every Advanced entry point requires
|
||||
`MongoAdvancedCapabilityFlags` to have the matching capability enabled and refuses construction
|
||||
otherwise. Being on the classpath is not being enabled.
|
||||
|
||||
## 3. Other deliberate substitutions
|
||||
|
||||
| Design assumption | Repository reality | Adaptation |
|
||||
|---|---|---|
|
||||
| Gradle Kotlin DSL under `modules/mongodb*` | Groovy DSL, root `build.gradle` conventions, `LockMode.STRICT` locking | Dependencies declared in `src/adapter/outbound/persistence-mongo/build.gradle`; `gradle.lockfile` regenerated. |
|
||||
| `mongodb-spring-boot-starter` is a separate module the app depends on | `modules.json` gives `adapter-outbound-persistence-mongo` `runtime_memberships: []` and does **not** list it among `app-bootstrap`'s allowed dependencies | The `autoconfigure` package stays inside the leaf and registers through the leaf's own `META-INF/spring/…AutoConfiguration.imports`. This differs from the httpclient precedent, where the starter moved to `:app-bootstrap`; here the registry forbids that edge. |
|
||||
| Spring Boot 4.1 / Spring Data MongoDB 5.1 baseline | Repository baseline is Spring Boot 4.0.0 / Spring Data MongoDB 5.0.0 | The platform targets the Spring Data MongoDB **API surface** common to both; no 5.1-only type is referenced. The support matrix records the actual pinned versions. |
|
||||
| `MongoRetryScope` lives in `mongodb-transaction` | The `mongodb-spring-data` failure translator must classify retry scope, and it cannot depend on `mongodb-transaction` | `MongoRetryScope` lives in `…api.error` (core-api), which both packages already depend on. Same values, same meaning, one legal position in the DAG. |
|
||||
| `mongodb-migration-flamingock` depends on Flamingock | Adding an unvetted external dependency is out of scope for this task, and the design itself requires the public contract not to depend on Flamingock types | The adapter is provider-neutral: it consumes a platform-owned `FlamingockChangeUnitView`. Wiring an actual Flamingock distribution is a one-file change behind that view. |
|
||||
| Testkit as its own Gradle module | The design forbids production modules depending on the testkit | A dedicated `testkit` source set whose output is on the test compile/runtime classpaths only. ArchUnit rule `productionNeverDependsOnTestkit` enforces the direction. |
|
||||
| Per-task `git commit` | `AGENTS.md`: commit policy is `human-only` | Implementation is delivered unstaged; commits are the human's action. This is the only plan step intentionally not executed, and it is recorded here. |
|
||||
| `docs/mongodb/**`, `scripts/verify-mongodb-*.sh` | Repository already owns `docs/` and `scripts/` | Created at the same repository-relative paths. |
|
||||
|
||||
## 4. What is unchanged from the design
|
||||
|
||||
- D1 / D2 / D3 / D4 exposure planes and the ordered D3 admission sequence (§5).
|
||||
- Stable API V1 with `apiStrict=true` on the D1/D2 client generation; D3/D4 on separate generations.
|
||||
- `MongoExecutionOutcome`, including both ambiguous outcomes (`WRITE_RESULT_UNKNOWN`,
|
||||
`TRANSACTION_COMMIT_UNKNOWN`), and `MongoFailureContext`'s permitted-field list.
|
||||
- The complete stable exception hierarchy and the label-before-code classification order.
|
||||
- The BSON representation manifest (UUID `STANDARD`, `Decimal128`, UTC instants, alias type metadata)
|
||||
and the document-size budget.
|
||||
- Update-operator-first writes, and optimistic revision as the precondition for whole-document
|
||||
replacement.
|
||||
- Transaction body retry and commit retry as separate loops: a new session per body attempt, and
|
||||
commit-only retry on unknown commit. The body is never replayed after a commit ambiguity.
|
||||
- Registered operation names and manifests for query, aggregation and index; no free-form JSON query
|
||||
and no unbounded pipeline.
|
||||
- Keyset pagination with an authenticated cursor and a unique tie-breaker requirement.
|
||||
- Change stream as an at-least-once projector with resume-token checkpointing and explicit
|
||||
history-lost handling.
|
||||
- TTL as physical cleanup only, never the sole basis for access denial or business scheduling.
|
||||
- Manifest-owned index/validator state with an apply policy that never drops what it does not own.
|
||||
- Low-cardinality observation tags, command redaction, and the credential reference indirection.
|
||||
- The Stable release gate's evidence categories, and the Advanced promotion gate's requirement for
|
||||
actual-topology evidence.
|
||||
|
||||
## 5. Verification
|
||||
|
||||
```bash
|
||||
bash scripts/verify-mongodb-platform.sh # Stable gate
|
||||
bash scripts/verify-mongodb-advanced.sh # Advanced gate (opt-in lanes)
|
||||
```
|
||||
|
||||
Both scripts run from the repository root and delegate to `src/gradlew`.
|
||||
@@ -0,0 +1,98 @@
|
||||
---
|
||||
title: Runbook — MongoDB primary failover
|
||||
category: mongodb
|
||||
severity: P2
|
||||
owner: oncall
|
||||
last_updated: 2026-08-13
|
||||
status: active
|
||||
---
|
||||
|
||||
# Runbook: MongoDB primary failover
|
||||
|
||||
Design §29, scenarios `PRIMARY_KILL`, `NETWORK_PARTITION`, `SERVER_SELECTION_TIMEOUT`,
|
||||
`WRITE_RESPONSE_LOSS`.
|
||||
|
||||
## Symptoms
|
||||
|
||||
- `MongoServerSelectionException` / `MongoConnectionException` spike, then recovery within seconds.
|
||||
- `MongoSdamObservationListener` reports a topology change (primary removed, new primary elected).
|
||||
- `MongoPoolObservationListener` shows checkout wait times rising while server-side command duration
|
||||
stays flat — the wait is topology, not query cost.
|
||||
- Latency spike on writes with no corresponding rise in read latency.
|
||||
|
||||
A failover that resolves in under ~15 s and produces no `WRITE_RESULT_UNKNOWN` is normal replica-set
|
||||
behaviour and needs no action beyond confirming it self-healed.
|
||||
|
||||
## Diagnosis
|
||||
|
||||
1. Confirm an election actually happened. SDAM events distinguish an election from "the database got
|
||||
slow"; without them the two are indistinguishable in application metrics.
|
||||
2. Split the failure categories. Metric tag `failureCategory`:
|
||||
- `SERVER_SELECTION` / `CONNECTION` → the driver could not reach a primary. `NOT_SENT`; safe.
|
||||
- `TIMEOUT` with outcome `WRITE_RESULT_UNKNOWN` → a write may have applied. Not safe; see below.
|
||||
- `TRANSACTION_COMMIT_UNKNOWN` → go to [unknown-commit.md](unknown-commit.md) instead.
|
||||
3. Check the election duration against `MongoRetryBudget`. If the election outlasted the budget, the
|
||||
retries were exhausted before a primary existed and callers saw errors that a longer budget would
|
||||
have absorbed.
|
||||
4. Check whether the new primary is in the expected region/AZ. A failover to a distant node changes
|
||||
write latency permanently, not transiently.
|
||||
|
||||
## Action
|
||||
|
||||
**Self-healed (the common case).**
|
||||
Confirm outcome distribution contains no `WRITE_RESULT_UNKNOWN`, record the election in the incident
|
||||
log, and close. Nothing to replay.
|
||||
|
||||
**Writes with `WRITE_RESULT_UNKNOWN`.**
|
||||
These writes may or may not have applied. Do not blind-retry.
|
||||
- Idempotent operation (registered `MongoUpdateOperator` with an `AtomicFilter` precondition): retry.
|
||||
The precondition makes the second application a no-op.
|
||||
- Non-idempotent operation: reconcile by reading the target document and comparing against the
|
||||
intended post-state. Retry only if it does not reflect the write.
|
||||
|
||||
**Server selection never recovers.**
|
||||
The set has lost quorum — two of three nodes are down or partitioned. No client-side action fixes
|
||||
this; escalate to the database owner to restore a majority. The application should be failing closed,
|
||||
not queueing.
|
||||
|
||||
**Elections are frequent (more than one a day, unprompted).**
|
||||
This is an infrastructure symptom, not an application one: check node resource saturation, disk
|
||||
latency on the primary, and network stability between members. Repeated elections cause repeated
|
||||
unknown-outcome windows.
|
||||
|
||||
## Escalation
|
||||
|
||||
- P2 → P1 if server selection has failed for more than 2 minutes, or if any non-idempotent write
|
||||
returned `WRITE_RESULT_UNKNOWN` and cannot be reconciled.
|
||||
- Page the database owner for quorum loss, and the service owner for reconciliation of ambiguous
|
||||
writes.
|
||||
|
||||
## Verification
|
||||
|
||||
The failover lane reproduces this deliberately:
|
||||
|
||||
```bash
|
||||
cd src
|
||||
./gradlew :adapter:outbound:persistence-mongo:mongoFailoverTest --console=plain
|
||||
```
|
||||
|
||||
It starts a real three-node set (`MongoThreeNodeReplicaSet`), stops the primary
|
||||
(`MongoPrimaryController`), and injects network faults through Toxiproxy
|
||||
(`ToxiproxyMongoNetworkFaultController`). A single-node set is not sufficient for the election: it
|
||||
never holds one, so every guarantee that depends on a primary change goes untested.
|
||||
|
||||
The network faults need their own fixture (`MongoProxiedReplicaSetNode`) because a stopped container
|
||||
cannot produce them. Stopping a node tells the client the write did not happen; cutting the *path*
|
||||
while the server keeps running produces a client that cannot tell. `MongoNetworkFaultLaneTest`
|
||||
asserts the difference by reaching the same server twice — once through the proxy, once directly:
|
||||
|
||||
- **Partition**: the proxied client fails, the direct client finds the server healthy and the earlier
|
||||
write intact. The path was cut, not the server.
|
||||
- **Response loss**: the proxied client fails, and the direct client then finds the document
|
||||
*present*. The write applied and only the acknowledgement was lost —
|
||||
`DefaultMongoFailureClassifier` returns `WRITE_RESULT_UNKNOWN`, and a retry would have inserted a
|
||||
second document.
|
||||
|
||||
One detail the lane depends on: the connection is warmed before the toxic is applied. On a cold
|
||||
connection it is the driver's handshake whose response is dropped, so the write is never transmitted
|
||||
— `NOT_SENT`, the opposite of the ambiguity being tested.
|
||||
@@ -0,0 +1,90 @@
|
||||
---
|
||||
title: Runbook — MongoDB change stream history lost
|
||||
category: mongodb
|
||||
severity: P1
|
||||
owner: oncall
|
||||
last_updated: 2026-08-13
|
||||
status: active
|
||||
---
|
||||
|
||||
# Runbook: change stream history lost
|
||||
|
||||
Design §20.3, scenarios `OPLOG_HISTORY_LOSS`, `RESUME_TOKEN_LOSS`.
|
||||
|
||||
The stored resume token predates the oldest entry in the oplog. The events between the checkpoint and
|
||||
now are gone from the server; no retry recovers them. `MongoChangeStreamRecoveryPolicy` returns
|
||||
`halt(HISTORY_LOST, "history-lost")` and the runner stops.
|
||||
|
||||
**The platform will not silently restart from "now".** That looks like a recovery and is actually a
|
||||
permanent, unreported gap in the projection.
|
||||
|
||||
## Symptoms
|
||||
|
||||
- `MongoChangeHistoryLostException`.
|
||||
- `MongoChangeStreamState.HISTORY_LOST`; the consumer is stopped, not looping.
|
||||
- Precedes it: consumer lag approaching the oplog window, or a consumer that was down for a long
|
||||
period (a deploy that failed, a scaled-to-zero worker, a long outage).
|
||||
|
||||
## Diagnosis
|
||||
|
||||
1. **Determine the gap.** The checkpoint's cluster time is the start; the oldest oplog entry is the
|
||||
end of what is unrecoverable. Everything in between was never processed.
|
||||
2. **Determine the oplog window.** `rs.printReplicationInfo()` on the primary gives the first and last
|
||||
oplog timestamps. If the window is materially smaller than it was, the write rate rose or the
|
||||
oplog was resized — the consumer may be fine and the server changed.
|
||||
3. **Determine what the projection is missing.** Which collections and which operations does this
|
||||
projector consume? The gap is bounded by that, not by everything that happened.
|
||||
4. **Check for a second consumer.** If another projector on the same collection is healthy, its
|
||||
checkpoint tells you whether the problem is this consumer or the oplog.
|
||||
|
||||
## Action
|
||||
|
||||
Resuming is not an option. The choices are:
|
||||
|
||||
**Rebuild from source.** If the projection is derivable from the current state of the source
|
||||
collections, rebuild it: stop the consumer, rebuild the projection, then start the stream from the
|
||||
cluster time at which the rebuild snapshot was taken. This is the correct answer whenever the
|
||||
projection is a materialised view rather than an event log, and it is the reason a projection should
|
||||
be derivable.
|
||||
|
||||
**Backfill the gap.** If the source documents carry a timestamp covering the gap, run a bounded
|
||||
backfill for that window through the migration runner (checkpointed, resumable — see
|
||||
[schema-index-migration-guide.md](../schema-index-migration-guide.md) §5), then resume from the
|
||||
current cluster time.
|
||||
|
||||
**Accept the gap explicitly.** Only when the projection is advisory and the business owner says so.
|
||||
Record the window in the incident log and reset the checkpoint. This is a decision someone signs, not
|
||||
a default.
|
||||
|
||||
Never: reset the checkpoint to "now" and restart quietly. That converts a visible P1 into an
|
||||
invisible data-quality defect that surfaces months later as "the report has been wrong since March".
|
||||
|
||||
## Prevention
|
||||
|
||||
- **Alert on lag against the oplog window, not wall-clock.** "Consumer is 30 minutes behind" is fine
|
||||
with a 24-hour oplog and an emergency with a 45-minute one. The threshold that matters is
|
||||
`lag / oplogWindow`.
|
||||
- **Size the oplog for the longest tolerable consumer outage**, including a failed deploy discovered
|
||||
the next morning.
|
||||
- **Checkpoint after processing, never on receipt** — see
|
||||
[change-stream-guide.md](../change-stream-guide.md) §3.
|
||||
- **Back up the checkpoint store.** `RESUME_TOKEN_LOSS` is the same incident reached from the other
|
||||
direction: the oplog is fine, the checkpoint is gone.
|
||||
- **Make the projection rebuildable.** A projection that can only be built by replaying every event
|
||||
has no recovery path once the oplog rolls.
|
||||
|
||||
## Escalation
|
||||
|
||||
- P1 on detection. The consumer is stopped, so lag grows for as long as this is unresolved.
|
||||
- Page the service owner for the rebuild decision, and the database owner if the oplog window shrank
|
||||
unexpectedly.
|
||||
|
||||
## Verification
|
||||
|
||||
```bash
|
||||
cd src
|
||||
./gradlew :adapter:outbound:persistence-mongo:mongoFailoverTest --console=plain
|
||||
```
|
||||
|
||||
`MongoFailoverScenario.OPLOG_HISTORY_LOSS` and `RESUME_TOKEN_LOSS` assert the runner halts and names
|
||||
this runbook rather than restarting from the current position.
|
||||
@@ -0,0 +1,87 @@
|
||||
---
|
||||
title: Runbook — MongoDB unknown transaction commit result
|
||||
category: mongodb
|
||||
severity: P1
|
||||
owner: oncall
|
||||
last_updated: 2026-08-13
|
||||
status: active
|
||||
---
|
||||
|
||||
# Runbook: unknown transaction commit result
|
||||
|
||||
Design §16, decision D-10, scenario `UNKNOWN_TRANSACTION_COMMIT_RESULT`.
|
||||
|
||||
`MongoExecutionOutcome.TRANSACTION_COMMIT_UNKNOWN` means the commit **may have applied**. It is not a
|
||||
failure and must never be reported to a caller as one. The single worst response is to re-run the
|
||||
transaction body: if the commit did apply, the body applies a second time.
|
||||
|
||||
## Symptoms
|
||||
|
||||
- `MongoTransactionCommitUnknownException` in logs.
|
||||
- Metric `failureCategory=TRANSACTION_COMMIT_UNKNOWN`.
|
||||
- Usually accompanies a primary election — see [failover.md](failover.md).
|
||||
- Downstream reports of duplicated effects (double charge, double increment) are the symptom of this
|
||||
being handled wrongly, not of the condition itself.
|
||||
|
||||
## Diagnosis
|
||||
|
||||
1. **Confirm the platform did the right thing automatically.**
|
||||
`MongoTransactionRetryCoordinator` retries the *commit only*, on the same session, within
|
||||
`MongoRetryBudget`. A commit retry against an already-committed transaction is a no-op by design.
|
||||
Most occurrences resolve here and never reach a human.
|
||||
|
||||
2. **If the budget was exhausted, determine the actual state.** The commit either applied or it did
|
||||
not; you must find out which, not guess.
|
||||
- If the transaction body wrote a deterministic marker (an idempotency key, a business id, a
|
||||
revision), read it back. That is exactly what `MongoCommitReconciler` does, and it is the
|
||||
reason the design requires transactions to write one.
|
||||
- If there is no marker: reconstruct from a downstream artefact — an outbox row, an audit record,
|
||||
an external side effect. If nothing exists to compare against, the transaction was not
|
||||
designed to be reconcilable and that is the finding to record.
|
||||
|
||||
3. **Check whether the body was replayed.** Grep for a second execution with the same operation name
|
||||
and correlation id. If the body ran twice, the effects need reversing, and the code path that
|
||||
replayed it is a defect: an ambiguous commit is `COMMIT_ONLY` scope
|
||||
(`MongoRetryScope.COMMIT_ONLY`), never `BODY`.
|
||||
|
||||
## Action
|
||||
|
||||
**Commit applied.** Nothing to do. Record the reconciliation.
|
||||
|
||||
**Commit did not apply.** Re-run the whole operation from the top — a new session, a new body
|
||||
attempt. This is safe precisely because you established the previous attempt left no trace.
|
||||
|
||||
**Cannot determine.** Do not retry. Escalate. A blind retry here is a coin flip between "no effect"
|
||||
and "duplicate effect", and duplicates in a financial or notification path are worse than a delay.
|
||||
Freeze the affected entity if the domain supports it, and hand off with: operation name, correlation
|
||||
id, document id, the time window, and what you checked.
|
||||
|
||||
**Recurring.** More than one an hour means the commit path is racing something structural — a
|
||||
`maxCommitTime` shorter than the observed election duration, an oversized transaction, or an
|
||||
undersized retry budget. Fix the budget or the transaction shape; do not raise the retry count and
|
||||
call it resolved.
|
||||
|
||||
## Prevention
|
||||
|
||||
- Every transaction body writes a deterministic marker that identifies its own commit.
|
||||
- `maxCommitTime` exceeds the observed p99 election duration.
|
||||
- Callers surface the ambiguity to their own callers rather than mapping it to a generic 500 — an
|
||||
ambiguous outcome reported as a failure invites the caller to retry, which is the one thing that
|
||||
must not happen.
|
||||
- Prefer a single-document atomic operation (D-09). A transaction that exists only to wrap one
|
||||
document write has invented this failure mode for nothing.
|
||||
|
||||
## Escalation
|
||||
|
||||
- Always P1 when the state cannot be determined and the operation has an external effect.
|
||||
- Page the service owner immediately; the database owner only if elections are the trigger.
|
||||
|
||||
## Verification
|
||||
|
||||
```bash
|
||||
cd src
|
||||
./gradlew :adapter:outbound:persistence-mongo:mongoFailoverTest --console=plain
|
||||
```
|
||||
|
||||
`MongoFailoverScenario.UNKNOWN_TRANSACTION_COMMIT_RESULT` runs this path against a real three-node
|
||||
set, and the coordinator test asserts the body is never replayed after a commit ambiguity.
|
||||
@@ -0,0 +1,137 @@
|
||||
# Schema, Index and Migration Guide
|
||||
|
||||
Design §21–§25, decision D-13. Indexes and validators are **declared** in a manifest and **applied**
|
||||
by an explicit plane. Automatic index creation in production is explicitly unsupported (§3.4): an
|
||||
index build on a large collection is a capacity event, and discovering it because a deployment
|
||||
started one is not an operating model.
|
||||
|
||||
## 1. The manifest is the source of truth
|
||||
|
||||
`MongoManifestRegistry` holds one `MongoCollectionManifest` per collection, containing:
|
||||
|
||||
- `MongoIndexManifest` — the declared indexes (`MongoIndexKey`, `MongoIndexDirection`, uniqueness,
|
||||
partial filter, collation)
|
||||
- `MongoSchemaManifest` — the declared `$jsonSchema` validator
|
||||
- `MongoMetadataOwnership` — who owns each observed object
|
||||
|
||||
Ownership is the field that makes drift handling safe:
|
||||
|
||||
| Ownership | Owner | Droppable on drift |
|
||||
|---|---|---|
|
||||
| `APPLICATION_MANAGED` | this manifest | yes |
|
||||
| `SEARCH_MANAGED` | the search service | no |
|
||||
| `ENCRYPTION_MANAGED` | Queryable Encryption | no |
|
||||
| `EXTERNAL` | someone else (a DBA, another service) | no |
|
||||
|
||||
A diff engine that does not know about ownership eventually proposes dropping
|
||||
`enxcol_.customers.esc` or a search index, and "the drift tool cleaned it up" is a very bad incident
|
||||
summary.
|
||||
|
||||
## 2. Index diff and apply
|
||||
|
||||
`MongoIndexDiffEngine` compares the manifest against `MongoIndexDescriptorView` observations and
|
||||
produces a `MongoIndexDiff`: missing, extra, and *changed* (same name, different definition —
|
||||
MongoDB will not silently rebuild these, so they must be reported rather than re-issued).
|
||||
|
||||
`MongoIndexApplyPolicy` decides what happens with a diff:
|
||||
|
||||
| Policy | Behaviour | Environment |
|
||||
|---|---|---|
|
||||
| `APPLY` | create what is missing | local / test |
|
||||
| `APPLY_WITH_DIFF` | create what is missing and report the rest | staging |
|
||||
| `DIFF_WITH_APPROVED_APPLY` | apply only what a human approved | production |
|
||||
| `REPORT_ONLY` | never write | audit |
|
||||
|
||||
Dropping is never implicit. `MongoIndexRetirementPlan` moves an index through
|
||||
`MongoIndexRetirementState` — declared → hidden → observed-unused → droppable — and each transition
|
||||
is a separate deployment. Hiding an index makes the planner ignore it while keeping it maintained, so
|
||||
an unexpected regression is one command to undo. Dropping it is not.
|
||||
|
||||
## 3. Validators
|
||||
|
||||
`MongoValidatorDescriptor` carries the `$jsonSchema`, a `MongoValidationLevel`
|
||||
(`OFF` / `MODERATE` / `STRICT`) and a `MongoValidationAction`.
|
||||
|
||||
**Stable validation actions are `error` and `warn` only.** `errorAndLog` is not part of the Stable
|
||||
contract on MongoDB 7.0 or 8.0 and the descriptor refuses it.
|
||||
|
||||
`MongoValidatorDiffEngine` produces a `MongoValidatorDiff`; `MongoValidatorApplyPolicy` gates the
|
||||
apply. Tightening a validator on a collection with existing data is the dangerous direction: introduce
|
||||
it as `warn` + `MODERATE`, confirm the warning count is zero, then promote to `error` + `STRICT` in a
|
||||
second deployment.
|
||||
|
||||
## 4. TTL
|
||||
|
||||
D-13: TTL is **physical cleanup**, nothing else.
|
||||
|
||||
`MongoTtlIndexDescriptor` declares the field and `expireAfterSeconds`. `MongoTtlPolicyValidator`
|
||||
enforces what `MongoTtlPolicy` allows, and `MongoExpirationAccessPolicy` states the rule that matters:
|
||||
|
||||
> A document's presence is not authorization, and its absence is not a deadline.
|
||||
|
||||
The TTL monitor runs about once a minute and deletes in batches, so a document can outlive its
|
||||
expiry by minutes to hours under load. Consequences:
|
||||
|
||||
- Access control must check the expiry field, not the document's existence. A still-present expired
|
||||
session is a valid document and an invalid session.
|
||||
- Business scheduling must not be built on TTL. If something must happen at a time, schedule it.
|
||||
- A TTL field must be a BSON date. A TTL index on a string silently never deletes anything.
|
||||
|
||||
## 5. Migrations
|
||||
|
||||
`MongoMigrationRunner` executes `MongoMigration` units with:
|
||||
|
||||
- `MongoMigrationId` — ordered, unique
|
||||
- `MongoMigrationChecksum` — content hash; a changed checksum for an applied id is a hard failure, not
|
||||
a re-run. Editing an applied migration means two environments ran different code under the same id.
|
||||
- `MongoMigrationLedger` — what has been applied
|
||||
- `MongoMigrationLock` — one runner at a time; a rolling deploy starts several instances at once
|
||||
- `MongoMigrationPrecondition` / `MongoMigrationPostcondition` — checked before and after; a migration
|
||||
that cannot verify its own result is a migration whose failure is discovered by a customer
|
||||
- `MongoMigrationCheckpoint` — a resumable position for a backfill
|
||||
|
||||
`MongoMigrationResult` reports applied / incomplete / dry-run with the reason. `INCOMPLETE` is not a
|
||||
failure: a rate-limited backfill that ran out of its time budget has done real work and stored a
|
||||
checkpoint, and reporting it as failed would send the next run back to the beginning.
|
||||
|
||||
`MongoCollectionMigrationLedger` and `MongoCollectionMigrationLock` are the MongoDB-backed
|
||||
implementations. Two details are load-bearing and only exist on a server:
|
||||
|
||||
- The ledger's **unique index** on the migration id, created by `ensureIndexes()`. Without it, two
|
||||
runners that both pass the "not applied yet" read both insert, and the ledger then reports one
|
||||
migration applied twice with two checksums — indistinguishable from tampering.
|
||||
- The lease is taken with **one conditional update**, not read-then-write. A filter matching only a
|
||||
free or expired lease lets the server pick the winner; two runners that each read "free" and then
|
||||
write would both believe they hold it.
|
||||
|
||||
The lease expires so a runner killed mid-migration does not block every future deployment, and
|
||||
`refresh` between batches is what proves the holder is still alive.
|
||||
|
||||
### Backfills restart, they do not restart-from-zero
|
||||
|
||||
A long backfill will be interrupted — a deploy, an OOM, a node replacement. The checkpoint records
|
||||
the last completed key so the restart continues rather than re-processing from the beginning.
|
||||
`MongoBackfillRestartFixture` in the testkit asserts exactly this: kill mid-run, restart, and the
|
||||
result is identical to the uninterrupted run and does not re-apply completed work.
|
||||
|
||||
### Flamingock
|
||||
|
||||
`FlamingockMongoMigrationAdapter` bridges to Flamingock through the platform-owned
|
||||
`FlamingockChangeUnitView`, with `FlamingockLedgerAdapter` and `FlamingockLockAdapter` mapping the
|
||||
ledger and lock. The public contract does not reference Flamingock types, so the provider can be
|
||||
replaced without touching a migration.
|
||||
|
||||
## 6. Ordering with deployments
|
||||
|
||||
```
|
||||
1. Add the index (hidden if it is large) -> deployment N
|
||||
2. Unhide / verify usage -> deployment N+1
|
||||
3. Ship code that depends on the index -> deployment N+1
|
||||
4. Backfill data -> migration, resumable
|
||||
5. Tighten the validator from warn to error -> deployment N+2
|
||||
6. Retire the old index through the retirement states -> deployments N+3…
|
||||
```
|
||||
|
||||
Each step is independently reversible. A deployment that adds an index and the code that requires it
|
||||
at the same time has no safe rollback: rolling back the code leaves the index build running, and
|
||||
rolling back the index breaks the code that is still live on half the fleet.
|
||||
@@ -0,0 +1,147 @@
|
||||
# Security and Observability
|
||||
|
||||
Design §26–§28, decision D-05. The application plane and the admin plane are different credentials on
|
||||
different clients, and telemetry never becomes an exfiltration path.
|
||||
|
||||
## 1. Roles
|
||||
|
||||
`MongoPrincipalRole` — one credential per role, least privilege:
|
||||
|
||||
| Role | Grants |
|
||||
|---|---|
|
||||
| `APP_READ` | `find` on allowlisted collections |
|
||||
| `APP_WRITE` | `insert`, `update`, `delete` on allowlisted collections |
|
||||
| `CHANGE_STREAM` | `changeStream`, `find` |
|
||||
| `MIGRATION` | index and validator management on the target collections |
|
||||
| `SEARCH_ADMIN` | search index management |
|
||||
| `SHARD_ADMIN` | shard key operations |
|
||||
| `ENCRYPTION_ADMIN` | key vault access |
|
||||
| `DBA` | the human plane; never used by an application |
|
||||
|
||||
`MongoSecurityProfileValidator` checks the profile at startup. `forbiddenPrivilegesHeld()` names the
|
||||
privileges the profile holds and must not — the validator reports *which* one, because "your
|
||||
credential is over-privileged" without a name is an unactionable finding.
|
||||
|
||||
The privileges that must never appear on an application credential: `dropDatabase`,
|
||||
`dropCollection`, `shutdown`, `killop`, `root`, `__system`, `dbOwner`, `userAdminAnyDatabase`.
|
||||
|
||||
## 2. Credentials are references, not values
|
||||
|
||||
`MongoCredentialReference` holds a `secret://…` reference plus the role. The reference is resolved at
|
||||
connection time by the secret provider; the password is never a property value, a log field, or a
|
||||
constructor argument that could end up in a stack trace.
|
||||
|
||||
`MongoCredentialRotationPolicy` states the rotation contract: overlapping validity, a drain window,
|
||||
and a rotation that never requires a restart. `MongoClientGenerationRegistry` implements the swap —
|
||||
a new `MongoClientGeneration` starts serving new operations while the previous generation is
|
||||
`markDraining()` until its in-flight operations finish. Killing the old client immediately fails every
|
||||
in-flight request, which is why rotation without generations is an outage.
|
||||
|
||||
Rotation is a failover scenario in the release gate (`MongoFailoverScenario.CREDENTIAL_ROTATION`),
|
||||
not a runbook step people hope works.
|
||||
|
||||
## 3. TLS and connection policy
|
||||
|
||||
`MongoSecurityProfile.production(...)` requires TLS and refuses `tlsAllowInvalidCertificates` /
|
||||
`tlsAllowInvalidHostnames`. `MongoSecurityProfile.local(...)` exists so a developer does not have to
|
||||
weaken the production factory to get a container to connect; the startup validator refuses a local
|
||||
profile on a production runtime profile.
|
||||
|
||||
## 4. Admin plane (D4)
|
||||
|
||||
`MongoAdminGateway` is the only path to `MongoAdminOperation`, and it runs on the D4 client with the
|
||||
DBA-scoped credential — not the application's.
|
||||
|
||||
- `MongoAdminAuthorization` checks the caller's role against the operation.
|
||||
- `MongoAdminRuntimeGuard` refuses high-risk operations (`highRisk()`) unless the runtime profile
|
||||
explicitly permits them; a `dropCollection` reachable from a running application is a data-loss
|
||||
vector regardless of how well-reviewed the calling code is.
|
||||
- `MongoAdminAuditRecord` records who ran what, when and against which collection profile — before
|
||||
execution, so a failed attempt is recorded too.
|
||||
|
||||
The native capability gateway (D3) refuses any admin-category command, so there is no path from the
|
||||
application plane into the admin plane.
|
||||
|
||||
## 5. Observability tags
|
||||
|
||||
`MongoObservationConvention` allowlists exactly eight tag names:
|
||||
|
||||
```
|
||||
mongoProfile, databaseProfile, collectionProfile, operationName,
|
||||
operationType, result, failureCategory, consistencyProfile
|
||||
```
|
||||
|
||||
and explicitly forbids:
|
||||
|
||||
```
|
||||
documentId, rawTenantId, tenantId, dynamicCollectionName, queryParameter,
|
||||
query, fullBson, resumeToken, shardKeyValue, plaintextPII, credential
|
||||
```
|
||||
|
||||
Two reasons, and both matter. Cardinality: a tag whose values are document ids produces one time
|
||||
series per document, which is how a metrics backend falls over. Confidentiality: a metric label is
|
||||
stored, shipped and retained by systems with a different access model than the database.
|
||||
`requireAllowed(tagName)` throws on anything outside the list, so a new tag is a deliberate change to
|
||||
the convention rather than a line in a service.
|
||||
|
||||
`MicrometerMongoOperationObserver` implements the `MongoOperationObserver` port;
|
||||
`NoOpMongoOperationObserver` is the default so observation is opt-in and never a hard dependency.
|
||||
|
||||
## 6. Driver-native listeners
|
||||
|
||||
`MongoDriverObservabilityConfiguration` registers three driver listeners, because they answer
|
||||
questions the application-level timer cannot:
|
||||
|
||||
| Listener | Answers |
|
||||
|---|---|
|
||||
| `MongoCommandObservationListener` | How long did the *server* take, versus how long the caller waited? |
|
||||
| `MongoPoolObservationListener` | Was the wait time connection checkout rather than query execution? |
|
||||
| `MongoSdamObservationListener` | Did the topology change — an election, a node removed — during the window? |
|
||||
|
||||
Without pool and SDAM events, every failover looks like "the database got slow", and the difference
|
||||
between "we need a bigger pool" and "we lost a primary" is invisible.
|
||||
|
||||
## 7. Command redaction
|
||||
|
||||
`MongoObservationRedactor.describe(commandName)`:
|
||||
|
||||
- Authentication and user-management commands (`authenticate`, `saslStart`, `saslContinue`,
|
||||
`getnonce`, `createUser`, `updateUser`, `copydb*`) render as `<redacted>` — their arguments carry
|
||||
credentials and key material.
|
||||
- Structural commands (`ping`, `hello`, `buildInfo`, `listCollections`, `listIndexes`, `collStats`)
|
||||
render by name; their arguments are not data-bearing.
|
||||
- Everything else renders as `name(...)`: you get the command, never the filter or the document.
|
||||
|
||||
`isAlwaysRedacted(...)` is the assertion hook so a test can prove no logging path can render an auth
|
||||
command's arguments.
|
||||
|
||||
## 8. Startup validation
|
||||
|
||||
`MongoStartupValidator` runs at context refresh, before the first request:
|
||||
|
||||
1. `MongoTopologyProbe` reports the actual `MongoTopology`.
|
||||
2. Each declared `MongoTopologyRequirement` is checked against it — a transaction, causal-session or
|
||||
change-stream requirement fails closed on `STANDALONE`.
|
||||
3. `MongoSecurityProfileValidator` checks credentials and TLS.
|
||||
4. `MongoCapabilitySupport` checks declared capabilities against the server version, with
|
||||
`MongoSupportLevel` distinguishing `STABLE` / `ADVANCED` / `EXPERIMENTAL` / `UNSUPPORTED`.
|
||||
5. `MongoPlatformHealthIndicator` reports the outcome for the readiness probe.
|
||||
|
||||
A misconfiguration found at startup costs a failed deploy. The same misconfiguration found at runtime
|
||||
costs an incident, and the failing operation is rarely the one that reveals the cause.
|
||||
|
||||
## 9. How the security lane proves any of this
|
||||
|
||||
```bash
|
||||
cd src
|
||||
./gradlew :adapter:outbound:persistence-mongo:mongoSecurityIntegrationTest --console=plain
|
||||
```
|
||||
|
||||
The lane runs against `MongoAuthenticatedReplicaSetContainer`, which starts mongod with `--auth` and a
|
||||
generated keyfile. That detail is the whole lane: Testcontainers' `MongoDBContainer` starts mongod
|
||||
*without* `--auth`, so users created on it all have every privilege and a least-privilege assertion
|
||||
passes no matter how wrong the roles are. A security test that cannot fail is not a security test.
|
||||
|
||||
What the lane asserts is the refusal: the `read` role's insert is rejected, and the application role's
|
||||
`dropDatabase` is rejected. Then it checks that `MongoSecurityProfileValidator` names the same
|
||||
privilege the server just refused.
|
||||
@@ -0,0 +1,91 @@
|
||||
# MongoDB Platform — Support Matrix
|
||||
|
||||
Design §4. This file records what the platform is *certified* on, not what it happens to run on.
|
||||
A configuration absent from this table is unsupported until someone runs the gate against it and
|
||||
adds a row.
|
||||
|
||||
## 1. Runtime baseline
|
||||
|
||||
| Component | Version | Policy |
|
||||
|---|---|---|
|
||||
| Java | 21 | Repository runtime baseline. |
|
||||
| Spring Boot | 4.0.0 | BOM-managed. Individual driver overrides are forbidden. |
|
||||
| Spring Data MongoDB | 5.0.0 | Repository and `MongoTemplate` integration. Version comes from the Boot BOM. |
|
||||
| MongoDB Java Driver | 5.6.1 | BOM-managed. Never pinned directly in the module. |
|
||||
| Reactor | 3.8.0 | Reactive execution path. |
|
||||
| Micrometer | 1.16.0 | Driver-native observability. |
|
||||
| Testcontainers | 2.0.2 | Replica-set, failover, migration and compatibility lanes. |
|
||||
|
||||
The design's baseline is Spring Boot 4.1.x / Spring Data MongoDB 5.1.x. This repository is on
|
||||
4.0.0 / 5.0.0, so the platform targets only the API surface common to both. See
|
||||
[repository-adaptation.md](repository-adaptation.md) §3.
|
||||
|
||||
## 2. Server versions
|
||||
|
||||
| Lane | Version | Pinned image | Gradle task |
|
||||
|---|---|---|---|
|
||||
| Primary certification | MongoDB 8.0 | `mongo:8.0.16` | `mongoReplicaSetTest`, `mongoFailoverTest` |
|
||||
| Compatibility | MongoDB 7.0 | `mongo:7.0.28` | `mongoCompatibilityTest` |
|
||||
| Network fault injection | — | `ghcr.io/shopify/toxiproxy:2.12.0` | `mongoFailoverTest` |
|
||||
|
||||
Images are pinned, never `latest`: a mutable tag means the certification result describes whatever
|
||||
was pulled that morning, not the version in the row. Override with
|
||||
`-PmongoPrimaryImage=…` / `-PmongoCompatibilityImage=…` when testing a new patch level, and update
|
||||
the row once the gate passes.
|
||||
|
||||
`MongoVersionMatrix.standard()` is the machine-readable form of this table; a version outside it
|
||||
fails `certifies()`.
|
||||
|
||||
## 3. Topologies
|
||||
|
||||
| Topology | Status | What is certified | What is not |
|
||||
|---|---|---|---|
|
||||
| Standalone | **Smoke only** | Basic CRUD and mapping. | Not a production profile and never counts as Stable release evidence (D-03). Transactions, retryable writes and change streams are refused at startup by `MongoStartupValidator`. |
|
||||
| Single-node replica set | **Local default** (D-02) | Transactions, retryable writes, change streams — the same semantics as production. | Elections. A single-node set never holds one, so failover behaviour is untested here. |
|
||||
| 3-node replica set | **Stable production gate** | Everything above plus primary failover, unknown-commit handling and change-stream resume across an election. | Shard routing. |
|
||||
| Sharded cluster | **Advanced gate** | Shard-key routing classification, scatter-gather refusal, `admin` plane operations. | Not included in the Stable gate. |
|
||||
| Atlas / provider-managed | **Per-capability gate** | Search, vector search and encryption against the actual target deployment. | Atlas Local in a container is a pull-request convenience, explicitly **not** release evidence (`MongoAtlasCapabilityContractSuite.Environment`). |
|
||||
|
||||
## 4. Stable API and client generations
|
||||
|
||||
| Plane | Stable API | Purpose |
|
||||
|---|---|---|
|
||||
| D1 Standard document persistence | V1, `apiStrict=true` | Repositories, typed queries, atomic updates, optimistic revision. |
|
||||
| D2 Advanced document operations | V1, `apiStrict=true` | `MongoTemplate`, transactions, bulk, aggregation, keyset cursors, change streams. |
|
||||
| D3 Explicit Mongo capability | Not strict | Native BSON, time series, search/vector, CSFLE/QE, shard-aware operations — each behind a registered capability. |
|
||||
| D4 Admin plane | Not strict | Collection, validator, index, migration, shard and repair commands. Separate credential, separate client. |
|
||||
|
||||
D3 is not a raw-client escape. Every call passes capability registration → database profile →
|
||||
collection allowlist → operation name → timeout → consistency profile → result limit → trace →
|
||||
redaction → command category → D4 refusal, in that order.
|
||||
|
||||
## 5. Validation actions
|
||||
|
||||
Stable validation actions are `error` and `warn`. `errorAndLog` is **not** part of the Stable
|
||||
contract on 7.0 or 8.0 and `MongoValidatorDescriptor` refuses it.
|
||||
|
||||
## 6. Explicitly unsupported
|
||||
|
||||
Per design §3.4, none of the following is provided, and adding one is a design change rather than a
|
||||
feature request:
|
||||
|
||||
- A generic `CommonMongoRepository<T, ID>`.
|
||||
- Arbitrary runtime `runCommand`.
|
||||
- Automatic index creation in production.
|
||||
- A Standalone production contract.
|
||||
- TTL as an exact business scheduler or as the only access control.
|
||||
- Publishing raw change events as external business integration events.
|
||||
- GridFS as the source of truth for new files.
|
||||
- Java fully-qualified class names as a long-lived BSON schema.
|
||||
- Unbounded skip pagination, unbounded aggregation pipelines, unbounded regex, unbounded results.
|
||||
|
||||
## 7. Capability tiers
|
||||
|
||||
| Tier | Capabilities | Enablement |
|
||||
|---|---|---|
|
||||
| Stable | Mapping, imperative/reactive execution, atomic update, optimistic lock, transactions, consistency profiles, retry/translation, query and aggregation guardrails, schema/index manifests, keyset pagination, bulk partial results, change streams, TTL contract, GeoJSON, security, observability | On when `ca-skeleton.persistence-mongo.enabled=true`. |
|
||||
| Advanced | Sharding-aware query, time series, CSFLE, Queryable Encryption (equality/range), change-stream→messaging bridge, shared-collection multi-tenancy | Each behind `ca-skeleton.persistence-mongo.advanced.<capability>.enabled`. |
|
||||
| Experimental | Search, vector search, hybrid search, database-per-tenant, collection-per-tenant, reshard orchestration, provider-specific features | Same flag mechanism; promotion additionally requires the evidence in [ADR-MONGO-ADV-001](../adr/ADR-MONGO-ADV-001-capability-promotion.md). |
|
||||
|
||||
`MongoAdvancedCapabilityFlags.propertyFor(capability)` is the authoritative property name for any
|
||||
capability; the table above is its prose form.
|
||||
@@ -0,0 +1,27 @@
|
||||
# NOTIF-ADR-001 — `submit()` means durable acceptance
|
||||
|
||||
## Status
|
||||
|
||||
Accepted.
|
||||
|
||||
## Context
|
||||
|
||||
The obvious API for a notification platform is `send()` returning success or failure. Every channel
|
||||
this platform supports makes that return value a lie:
|
||||
|
||||
- SES accepts a request, returns a `MessageId`, and can still decline to send.
|
||||
- Twilio separates `accepted`, `sent` and `delivered` into distinct, later events.
|
||||
- APNs accepts a notification and may then deliver, store or discard it.
|
||||
- Web Push separates push-service acceptance from user-agent acknowledgement at the protocol level.
|
||||
|
||||
## Decision
|
||||
|
||||
`submit()` and `schedule()` return once the logical request and its recipient jobs are committed to
|
||||
the database. The receipt carries `notificationId`, `RequestStatus` and `acceptedAt`, and has no
|
||||
`delivered`, `sent` or `read` component. No provider is contacted while the transaction is open.
|
||||
|
||||
## Consequences
|
||||
|
||||
Callers cannot mistake acceptance for delivery, because the type does not offer that reading.
|
||||
Delivery state is a separate query against the projection built from the provider event ledger. The
|
||||
cost is that "did it arrive?" is a second question — which is the honest number of questions.
|
||||
@@ -0,0 +1,25 @@
|
||||
# NOTIF-ADR-002 — append-only event ledger with channel projectors
|
||||
|
||||
## Status
|
||||
|
||||
Accepted.
|
||||
|
||||
## Context
|
||||
|
||||
A single linear delivery status has to be updated in place, which forces a rule for deciding whether
|
||||
a new event outranks the stored one. The natural rule — compare ordinals — is wrong for real provider
|
||||
traffic. Twilio does not guarantee callback ordering, so `sent` arrives after `delivered`. Email
|
||||
generates complaints after deliveries. Both cases lose information under an ordinal rule.
|
||||
|
||||
## Decision
|
||||
|
||||
Provider events are appended to an immutable ledger before any projection runs. Channel-specific
|
||||
projectors merge events into `SubmissionOutcome`, `DeliveryOutcome`, `EvidenceLevel`,
|
||||
`EngagementFacts` and `SuppressionFacts` using explicit transition tables. Projection is idempotent
|
||||
and can be replayed from the ledger.
|
||||
|
||||
## Consequences
|
||||
|
||||
Duplicate, out-of-order and late events are normal inputs rather than defects. A projector bug is
|
||||
recoverable, because the events it mis-projected are still stored. Projector versions can be migrated
|
||||
by replay. The cost is a second write per event and a projection that can lag its ledger.
|
||||
@@ -0,0 +1,37 @@
|
||||
# NOTIF-ADR-003 — ambiguous submission is a first-class state
|
||||
|
||||
## Status
|
||||
|
||||
Accepted.
|
||||
|
||||
## Context
|
||||
|
||||
The most common serious failure is not a rejection. It is a request whose body reached the provider
|
||||
and whose response never came back. The platform has no provider request id, and the user may or may
|
||||
not have received the notification.
|
||||
|
||||
Treating that as a failure produces duplicates: a retry sends a second message, and a cross-channel
|
||||
fallback sends the SMS next to the push that already arrived. Treating it as a success loses real
|
||||
failures.
|
||||
|
||||
## Decision
|
||||
|
||||
`AMBIGUOUS` is a stored `SubmissionOutcome` and `AttemptConfirmation`. Attempts record
|
||||
`requestStarted`, `requestBodyCommitted` and `providerResponseReceived`, each with an
|
||||
`EvidenceCertainty` of `PROVEN`, `INFERRED` or `UNKNOWN`, so an adapter that does not know is not
|
||||
forced to answer `false`.
|
||||
|
||||
While an ambiguous attempt exists on a recipient delivery:
|
||||
|
||||
- automatic retry is blocked unless the provider proves per-request idempotency
|
||||
- automatic cross-channel fallback is blocked unconditionally
|
||||
- reconciliation runs where the provider supports a status query
|
||||
- otherwise the delivery stops and waits for an operator
|
||||
|
||||
Operator redrive of an ambiguous attempt requires explicit duplicate-risk approval.
|
||||
|
||||
## Consequences
|
||||
|
||||
Some notifications stop in a state that needs a human or a reconciliation pass. That is the intended
|
||||
trade: an unresolved unknown is cheaper than a guaranteed duplicate, and the state is visible rather
|
||||
than silently resolved in either direction.
|
||||
@@ -0,0 +1,27 @@
|
||||
# NOTIF-ADR-004 — FCM installation id is the primary target
|
||||
|
||||
## Status
|
||||
|
||||
Accepted.
|
||||
|
||||
## Context
|
||||
|
||||
Firebase now recommends the installation id (FID) and treats registration-token multicast paths as
|
||||
legacy. A contact point model built on a single `token` string would encode the older model as the
|
||||
only one, and a later migration would be a runtime interpretation problem: the same string field
|
||||
would mean different things for different rows.
|
||||
|
||||
## Decision
|
||||
|
||||
`MobilePushTarget` is a sealed hierarchy of `FcmInstallationId`, `LegacyFcmRegistrationToken` and
|
||||
`ApnsDeviceToken`. The kinds are separate types, never a discriminator on one string field, and each
|
||||
carries its own `ContactPointType` so the uniqueness scope and the encryption associated data differ.
|
||||
|
||||
APNs tokens additionally carry their environment, because sandbox and production are separate
|
||||
namespaces rather than a flag.
|
||||
|
||||
## Consequences
|
||||
|
||||
Migrating a target kind is a compile-time change with an exhaustive `switch`, not a runtime guess.
|
||||
The adapter maps each kind to its own wire representation, so a provider changing one path cannot
|
||||
silently change the other. The cost is one more type than a string field would need.
|
||||
@@ -0,0 +1,52 @@
|
||||
# Callbacks and reconciliation
|
||||
|
||||
## Ingestion order
|
||||
|
||||
```text
|
||||
body size limit
|
||||
→ content type
|
||||
→ profile lookup
|
||||
→ signature verification
|
||||
→ append to the ledger
|
||||
→ duplicate detection
|
||||
→ normalization
|
||||
→ attempt resolution
|
||||
→ projection
|
||||
→ side effects
|
||||
→ 2xx
|
||||
```
|
||||
|
||||
Appending before projecting is what makes a fast 2xx honest. The provider is told the event is
|
||||
recorded, and a projector defect becomes a replay problem rather than a lost event.
|
||||
|
||||
A rejected signature is recorded in the security audit, never in the provider event ledger. Writing
|
||||
it to the ledger would let anyone who can reach the endpoint fill a delivery history with noise.
|
||||
|
||||
## Duplicates and ordering
|
||||
|
||||
Duplicate suppression uses `(providerProfileId, providerEventId)` where the provider supplies an
|
||||
event id, and a deterministic fingerprint over profile, request id, event type, occurrence time and
|
||||
payload digest where it does not. A duplicate is acknowledged and projected exactly once.
|
||||
|
||||
Out-of-order callbacks are normal. Ordering is resolved by event semantics, not by arrival time.
|
||||
|
||||
## Unknown fields
|
||||
|
||||
Callback parsers tolerate unknown JSON fields. Normalization only rejects a payload when a field
|
||||
required to identify the attempt is missing. Providers add fields; that must not stop ingestion.
|
||||
|
||||
## Reconciliation
|
||||
|
||||
Reconciliation targets:
|
||||
|
||||
- attempts stuck in `DISPATCHING` past their lease
|
||||
- ambiguous submissions
|
||||
- accepted attempts whose callback SLA has expired
|
||||
- unmatched provider events
|
||||
|
||||
A confirmed query result is appended to the same ledger with `source = RECONCILIATION` and projected
|
||||
by the same projector, so projection replay stays possible: there is no privileged second path that
|
||||
writes projections directly.
|
||||
|
||||
Where a provider has no status-query capability, the platform records `Unsupported` and leaves the
|
||||
attempt ambiguous. It does not infer a final status.
|
||||
@@ -0,0 +1,41 @@
|
||||
# Configuration reference
|
||||
|
||||
## Dispatch
|
||||
|
||||
| Property | Meaning | Bound |
|
||||
|---|---|---|
|
||||
| `claim-batch-size` | Rows claimed per scheduler tick | 1..1000 |
|
||||
| `lease-duration` | How long a claimed job stays owned | positive, finite |
|
||||
| `max-global-concurrency` | Ceiling across all providers | positive |
|
||||
| `max-queue-age` | Age at which a job is escalated | positive |
|
||||
| `max-retry-concurrency` | Ceiling for retry work | positive |
|
||||
| `scheduler-poll-interval` | Queue poll cadence | positive |
|
||||
| `callback-worker-concurrency` | Callback projection workers | positive |
|
||||
|
||||
Every value is bounded. "Unlimited" is not an accepted configuration.
|
||||
|
||||
## Provider profiles
|
||||
|
||||
A profile pins provider type, environment, credential profile, timeouts, concurrency, rate limit,
|
||||
retry policy and callback profile. Sender identity and credential profile are separate concerns.
|
||||
|
||||
## Startup failures
|
||||
|
||||
Startup fails rather than degrading when:
|
||||
|
||||
- a payload or queue setting is unbounded
|
||||
- a timeout is negative
|
||||
- a TTL-required profile has no expiry source
|
||||
- a callback signing secret is missing
|
||||
- a production profile enables trust-all
|
||||
- an APNs profile is missing its environment or topic
|
||||
- a Web Push profile is missing its VAPID key
|
||||
- two provider profiles share an id
|
||||
- a route points only at disabled providers
|
||||
- ambiguous fallback is enabled by default
|
||||
|
||||
## Secrets
|
||||
|
||||
All key material arrives through `SecretMaterialProvider`. Nothing is read from source, from a
|
||||
committed file, or from a plaintext log. Contact point encryption and lookup HMAC keys must be
|
||||
distinct, and the encryption key must be exactly 256 bits.
|
||||
@@ -0,0 +1,62 @@
|
||||
# Delivery evidence model
|
||||
|
||||
## The shape
|
||||
|
||||
```text
|
||||
NotificationRequest
|
||||
└─ RecipientDelivery
|
||||
└─ DeliveryAttempt
|
||||
└─ ProviderEvent (append-only)
|
||||
└─ channel projector
|
||||
└─ SubmissionOutcome / DeliveryOutcome / EvidenceLevel
|
||||
+ EngagementFacts + SuppressionFacts
|
||||
```
|
||||
|
||||
Four identities, four lifecycles. A logical request is not a recipient job, a recipient job is not a
|
||||
provider attempt, and a provider attempt is not the event stream that describes it.
|
||||
|
||||
## Why not one status enum
|
||||
|
||||
A single linear status would have to answer "what happened?" with one value, and the real answers do
|
||||
not fit on one line:
|
||||
|
||||
- An email can be `DELIVERED` and then generate a complaint. Both facts are true and both matter:
|
||||
one for reporting, the other for suppression.
|
||||
- Twilio does not guarantee callback ordering, so `sent` routinely arrives after `delivered`. Under
|
||||
an ordinal rule the later, weaker event silently overwrites the stronger one.
|
||||
- APNs may accept a notification and then store, replace or discard it.
|
||||
|
||||
So the ledger stores events and a channel projector merges them through an explicit transition table.
|
||||
`StandardDeliveryProjector` holds the shared rules; provider projectors add only their own event
|
||||
vocabulary.
|
||||
|
||||
## Merge rules
|
||||
|
||||
| Transition | Result |
|
||||
|---|---|
|
||||
| `sent` → `delivered` | applied |
|
||||
| `delivered` → `sent` | ignored, event still stored |
|
||||
| `delivered` → `complaint` | complaint fact added, delivery preserved |
|
||||
| `complaint` → `delivered` | delivery applied, complaint preserved |
|
||||
| `accepted` → `bounced` | applied |
|
||||
| `read` → `displayed` | ignored |
|
||||
| hard bounce → `delivered` | ignored, hard bounce is terminal |
|
||||
|
||||
Engagement (`opened`, `clicked`) is stored beside the delivery outcome and never changes it.
|
||||
|
||||
## Ambiguity
|
||||
|
||||
```text
|
||||
platform ──── send ────▶ provider
|
||||
│
|
||||
└── accepted
|
||||
✗ connection reset
|
||||
```
|
||||
|
||||
The platform may hold no provider request id while the notification really was sent. The attempt
|
||||
records `requestStarted`, `requestBodyCommitted`, `providerResponseReceived` and an
|
||||
`EvidenceCertainty` for each, so a later decision can tell "we know nothing was sent" apart from "we
|
||||
could not read the answer".
|
||||
|
||||
`ProviderSubmissionResult` enforces this: an ambiguous result may not claim `PROVIDER_ACCEPTED`, and
|
||||
no submission result of any kind may carry a delivery outcome.
|
||||
@@ -0,0 +1,31 @@
|
||||
# Migration guide
|
||||
|
||||
## From the R0 routing seam
|
||||
|
||||
The pre-existing `dev.caskeleton.adapter.outbound.notification` router (`RoutingNotifier`,
|
||||
`FailOpenNotificationProvider`, the Google email and Slack webhook seams) stays untouched. The
|
||||
delivery platform lives beside it under `…notification.platform` and does not modify or delete any
|
||||
R0 class.
|
||||
|
||||
Migration order per capability:
|
||||
|
||||
1. Register the contact points behind `ContactPointStorePort` so the platform owns protected values.
|
||||
2. Publish the template version, and pin the template id, version and locale at every call site.
|
||||
3. Move the call site from the router to the N1 typed facade for the channel.
|
||||
4. Verify evidence in the snapshot rather than in the caller's return value: `submit()` is durable
|
||||
acceptance and nothing more.
|
||||
5. Remove the R0 route only after the platform route has produced provider evidence in the target
|
||||
environment.
|
||||
|
||||
## Return-value semantics change
|
||||
|
||||
The R0 seam returned a send-shaped result. `NotificationReceipt` returns `notificationId`, a request
|
||||
status and an acceptance time. Callers that treated the old return value as proof of delivery must be
|
||||
changed; there is no compatibility shim, because a shim would have to invent the delivery claim this
|
||||
platform exists to avoid.
|
||||
|
||||
## FCM target migration
|
||||
|
||||
Registration tokens keep working through `LegacyFcmRegistrationToken`. New registrations should use
|
||||
`FcmInstallationId`. The two are distinct types, so a migration is a compile-time task rather than a
|
||||
runtime guess.
|
||||
@@ -0,0 +1,94 @@
|
||||
# Notification Delivery Platform — module mapping
|
||||
|
||||
> Source design: `notification-superpowers-package/docs/superpowers/specs/2026-08-10-notification-platform-design.md`
|
||||
>
|
||||
> Source plan: `notification-superpowers-package/docs/superpowers/plans/2026-08-10-notification-platform-implementation-plan.md`
|
||||
|
||||
## Why a mapping exists
|
||||
|
||||
The plan was written against a hypothetical repository (`modules/notification/**`, root package
|
||||
`io.backend.skeleton.notification`, 31 Gradle projects). This repository is a fail-closed
|
||||
19-leaf Clean Architecture template: `src/settings.gradle` rejects any registry that does not
|
||||
contain exactly the 19 modules in `src/config/architecture/modules.json`, and
|
||||
`verifyCleanArchitectureDependencies` rejects any project edge outside `allowed_dependencies`.
|
||||
|
||||
Creating 31 new Gradle projects would violate HARD-STOP #5 of `AGENTS.md`. The package README
|
||||
anticipates this and instructs the implementer to map dependency catalog and package/file paths onto
|
||||
the host repository's rules while preserving the public contracts and reliability semantics.
|
||||
|
||||
Every logical module of the plan is therefore implemented as a **package** inside the registered leaf
|
||||
that owns its responsibility. No public contract, evidence rule, or reliability semantic is dropped.
|
||||
|
||||
## Logical module → registered leaf
|
||||
|
||||
| Plan module | Registered leaf | Package |
|
||||
|---|---|---|
|
||||
| `notification-core-api` | `application-core` | `dev.caskeleton.application.notification.platform.api` |
|
||||
| `notification-content-api` | `application-core` | `…platform.api.content` |
|
||||
| `notification-contact-api` | `application-core` | `…platform.contact` |
|
||||
| `notification-template-api` | `application-core` | `…platform.template` |
|
||||
| `notification-policy` | `application-core` | `…platform.policy` |
|
||||
| `notification-provider-spi` | `application-core` | `…platform.provider` |
|
||||
| `notification-callback-api` | `application-core` | `…platform.callback` |
|
||||
| `notification-email-api` | `application-core` | `…platform.email` |
|
||||
| `notification-sms-api` | `application-core` | `…platform.sms` |
|
||||
| `notification-push-api` | `application-core` | `…platform.push` |
|
||||
| `notification-webpush` (API half) | `application-core` | `…platform.webpush` |
|
||||
| `notification-inbox-api` | `application-core` | `…platform.inbox` |
|
||||
| `notification-admin-api` | `application-core` | `…platform.admin` |
|
||||
| `notification-security` (ports + redaction) | `application-core` | `…platform.security` |
|
||||
| `notification-observability` (ports) | `application-core` | `…platform.observation` |
|
||||
| `notification-dispatch-runtime` | `adapter:outbound:notification` | `dev.caskeleton.adapter.outbound.notification.platform.dispatch` |
|
||||
| `notification-security` (AES-GCM/HMAC impl) | `adapter:outbound:notification` | `…platform.security` |
|
||||
| `notification-template-thymeleaf` (reference renderer) | `adapter:outbound:notification` | `…platform.template` |
|
||||
| `notification-email-smtp` | `adapter:outbound:notification` | `…platform.provider.smtp` |
|
||||
| `notification-email-ses` | `adapter:outbound:notification` | `…platform.provider.ses` |
|
||||
| `notification-sms-twilio` | `adapter:outbound:notification` | `…platform.provider.twilio` |
|
||||
| `notification-push-fcm` | `adapter:outbound:notification` | `…platform.provider.fcm` |
|
||||
| `notification-push-apns` | `adapter:outbound:notification` | `…platform.provider.apns` |
|
||||
| `notification-webpush` (transport + crypto) | `adapter:outbound:notification` | `…platform.provider.webpush` |
|
||||
| `notification-webhook-extension` | `adapter:outbound:notification` | `…platform.provider.webhook` |
|
||||
| `notification-observability` (Micrometer impl) | `adapter:outbound:notification` | `…platform.observation` |
|
||||
| `notification-admin-runtime` | `adapter:outbound:notification` | `…platform.admin` |
|
||||
| `notification-reactor` | `adapter:outbound:notification` | `…platform.reactor` |
|
||||
| `notification-spring-boot-starter` | `adapter:outbound:notification` (+ `app-bootstrap` wiring) | `…platform.autoconfigure` |
|
||||
| `notification-persistence-jpa` | `adapter:outbound:persistence-jpa` | `dev.caskeleton.adapter.outbound.persistence.notification.platform` |
|
||||
| `notification-inbox-jpa` | `adapter:outbound:persistence-jpa` | `…persistence.notification.platform.inbox` |
|
||||
| `notification-callback-mvc` | `adapter:inbound:web` | `dev.caskeleton.adapter.inbound.web.notification.platform.callback` |
|
||||
| `notification-callback-webflux` | `adapter:inbound:web` | `…callback.reactive` |
|
||||
| `notification-testkit` | test source sets of the owning leaves | `…platform.testkit` |
|
||||
|
||||
## Dependency-direction consequences
|
||||
|
||||
The plan's module DAG (`*-api` → `provider-spi`/`policy` → runtime/adapters → starter) is preserved
|
||||
by the leaf DAG that the registry already enforces:
|
||||
|
||||
```text
|
||||
application-core (all *-api, provider SPI, policy, callback contracts)
|
||||
↑ ↑ ↑
|
||||
adapter:outbound:notification adapter:outbound:persistence-jpa adapter:inbound:web
|
||||
↑ ↑ ↑
|
||||
app-bootstrap
|
||||
```
|
||||
|
||||
Two plan edges cannot be expressed as project edges in this repository, and are replaced by ports:
|
||||
|
||||
1. `notification-email-ses`, `notification-sms-twilio`, `notification-push-fcm`,
|
||||
`notification-push-apns`, `notification-webpush`, `notification-webhook-extension`
|
||||
→ `httpclient platform`.
|
||||
`adapter-outbound-notification` is not allowed to depend on `adapter-outbound-httpclient`.
|
||||
The provider adapters therefore call
|
||||
`dev.caskeleton.adapter.outbound.notification.platform.provider.http.NotificationHttpGateway`,
|
||||
an adapter-local port with a JDK `java.net.http.HttpClient` default implementation.
|
||||
`app-bootstrap` sees both leaves and is the supported place to substitute an implementation backed
|
||||
by the HTTP Client Platform (TLS/timeout/circuit-breaker/SSRF/dynamic-target policy reuse).
|
||||
2. `notification-inbox-jpa` → `optional messaging outbox integration`.
|
||||
`adapter-outbound-persistence-jpa` may not depend on `adapter-outbound-messaging`; the inbox
|
||||
publishes through the existing persistence outbox tables plus the
|
||||
`NotificationInboxSignalPort` application port, and `app-bootstrap` binds the relay.
|
||||
|
||||
## Commit policy
|
||||
|
||||
`AGENTS.md` pins commit policy to `human-only`. Step 5 (`git add` / `git commit`) of every plan task
|
||||
is therefore intentionally **not** executed by the agent; the working tree carries the change and the
|
||||
human owner commits.
|
||||
@@ -0,0 +1,47 @@
|
||||
# Operations
|
||||
|
||||
## Runtime shape
|
||||
|
||||
```text
|
||||
durable queue (PostgreSQL, FOR UPDATE SKIP LOCKED)
|
||||
→ expiry check
|
||||
→ suppression and eligibility re-check
|
||||
→ provider health gate
|
||||
→ rate limiter
|
||||
→ concurrency limiter
|
||||
→ provider adapter
|
||||
```
|
||||
|
||||
Provider calls run outside every database transaction. The attempt row is committed first, so after a
|
||||
crash the row is either absent (nothing was sent) or present in `DISPATCHING` (reconciliation has
|
||||
something to ask about).
|
||||
|
||||
## Guards that exist for specific incidents
|
||||
|
||||
| Guard | The incident it prevents |
|
||||
|---|---|
|
||||
| Credential failure opens the provider route | One expired key multiplied by a queue becomes a self-inflicted outage |
|
||||
| Retry budget per provider profile | A provider outage turning every queued notification into its own retry loop |
|
||||
| Ambiguous attempts block automatic fallback | A push whose response was lost arriving alongside the "just in case" SMS |
|
||||
| Permits released during backoff | A slow provider pinning the whole concurrency budget on work that is only waiting |
|
||||
| Bounded drain on rotation | A provider that never answers holding a credential rotation open forever |
|
||||
| Fail-fast intake on capacity | An unbounded in-memory queue absorbing a burst it cannot survive |
|
||||
|
||||
## Scheduling
|
||||
|
||||
`scheduleAt` activates the job, `notBefore` is the earliest permitted provider submission, and
|
||||
`expiresAt` blocks new attempts, retries and fallbacks. Suppression and expiry are re-checked
|
||||
immediately before dispatch, because a scheduled notification can sit in the queue for hours and the
|
||||
user may have opted out in the meantime.
|
||||
|
||||
## Redrive
|
||||
|
||||
A redrive preserves `NotificationId` and `RecipientDeliveryId`, creates a new `DeliveryAttemptId`, and
|
||||
reuses the pinned template version and rendered digest. Sending different content is a new
|
||||
notification, not a redrive. Redriving an ambiguous attempt requires explicit duplicate-risk approval,
|
||||
because the platform genuinely cannot tell whether the first submission reached the user.
|
||||
|
||||
## Actuator surface
|
||||
|
||||
Provider runtime states and generations, queue depth and age, callback and reconciliation health.
|
||||
Never addresses, never credentials.
|
||||
@@ -0,0 +1,65 @@
|
||||
# Provider runbooks
|
||||
|
||||
## SMTP
|
||||
|
||||
| Symptom | Classification | Action |
|
||||
|---|---|---|
|
||||
| Final `2xx` after `DATA` | `CONFIRMED_ACCEPTED` / `PROVIDER_ACCEPTED` | None; this is acceptance, not inbox delivery |
|
||||
| `4yz` | `TRANSIENT_PROVIDER` | Retry under budget and deadline |
|
||||
| `5yz` | `PERMANENT_PROVIDER` or `INVALID_RECIPIENT` | Stop, or invalidate the contact point |
|
||||
| Connection lost after `DATA` | `AMBIGUOUS_SUBMISSION` | Reconcile or escalate; do not resend automatically |
|
||||
|
||||
Connection, read, write and pool-acquire timeouts are all finite. There is no unbounded timeout.
|
||||
|
||||
## Amazon SES
|
||||
|
||||
`MessageId` is acceptance evidence. SES itself documents that it can accept a request and then not
|
||||
send, so `MessageId` is never mapped to `DELIVERED`.
|
||||
|
||||
| Event | Normalized |
|
||||
|---|---|
|
||||
| `Send` | reinforces `PROVIDER_ACCEPTED` |
|
||||
| `Delivery` | `DELIVERY_CONFIRMED` / `NETWORK_OR_CARRIER_ACCEPTED` |
|
||||
| `DeliveryDelay` | delay fact |
|
||||
| `Bounce` (permanent) | `BOUNCED_HARD` plus hard-bounce suppression |
|
||||
| `Bounce` (transient) | `BOUNCED_SOFT`; retry policy input, not a suppression reason |
|
||||
| `Complaint` | complaint fact plus suppression |
|
||||
| `Reject` | `PROVIDER_REJECTED` |
|
||||
| `RenderingFailure` | `TEMPLATE_FAILURE` |
|
||||
|
||||
## Twilio
|
||||
|
||||
`accepted`/`queued` is acceptance only. `sent` is carrier acceptance. `delivered` is device delivery.
|
||||
|
||||
Callbacks are not ordered. A `sent` arriving after `delivered` is stored and ignored by the
|
||||
projection. Missing callbacks are corrected by status polling under the provider rate limit.
|
||||
|
||||
Signature verification uses the canonical external URL from the profile, not the URL the servlet
|
||||
container reconstructed behind a proxy.
|
||||
|
||||
## FCM
|
||||
|
||||
| Error | Classification |
|
||||
|---|---|
|
||||
| `UNREGISTERED` | `INVALID_RECIPIENT`; invalidate the contact point, never retry |
|
||||
| `INVALID_ARGUMENT` | `INVALID_PAYLOAD` |
|
||||
| `QUOTA_EXCEEDED` | `THROTTLED`, exponential backoff |
|
||||
| `UNAVAILABLE` | `TRANSIENT_PROVIDER`, honour `Retry-After`, add jitter |
|
||||
| Credential failure | `AUTHENTICATION`; opens the provider route |
|
||||
|
||||
A batch is one transport call and many attempts. Partial results map back by input index; one
|
||||
transport failure does not become one shared outcome unless the adapter can prove it.
|
||||
|
||||
## APNs
|
||||
|
||||
2xx is acceptance. Environment and topic mismatches are configuration failures, not delivery
|
||||
failures. Sandbox and production tokens are separate namespaces.
|
||||
|
||||
## Web Push
|
||||
|
||||
`TTL` is mandatory by protocol. `201` is acceptance. `404` is an expired subscription per RFC 8030;
|
||||
provider-documented `410` maps the same way. Payloads use `aes128gcm` per RFC 8291 and VAPID JWTs are
|
||||
signed per RFC 8292 with the audience taken from the endpoint origin.
|
||||
|
||||
VAPID key rotation is not ordinary credential rotation: a restricted subscription may need to be
|
||||
re-created, so it is a migration operation.
|
||||
@@ -0,0 +1,56 @@
|
||||
# Security and privacy
|
||||
|
||||
## Protected values
|
||||
|
||||
Email addresses, phone numbers, FCM installation ids and legacy tokens, APNs device tokens, Web Push
|
||||
endpoints and keys, VAPID private keys, provider credentials, callback signing secrets, template
|
||||
variables, rendered bodies, attachment references and unsubscribe tokens.
|
||||
|
||||
## At rest
|
||||
|
||||
Contact points are encrypted with AES-256-GCM. Equality lookup uses a separate HMAC-SHA-256
|
||||
fingerprint.
|
||||
|
||||
Two keys, not one, because the requirements are opposite: the ciphertext must be non-deterministic so
|
||||
two records of the same address are not visibly identical, while equality lookup must be
|
||||
deterministic. The fingerprint is keyed rather than a plain digest because phone numbers and email
|
||||
addresses come from a small, enumerable space — an unkeyed hash of a phone number is recoverable in
|
||||
seconds.
|
||||
|
||||
The contact point kind is bound into the GCM associated data, so a ciphertext cannot be moved between
|
||||
contact kinds without failing the authentication tag.
|
||||
|
||||
An unknown key id is refused rather than silently falling back to the current key: a silent fallback
|
||||
would turn every historical row into a tag failure at read time.
|
||||
|
||||
## Never logged, never a metric tag
|
||||
|
||||
Addresses, tokens, Web Push endpoints and keys, message bodies, template variables, provider
|
||||
credentials, unsubscribe tokens, attachment URLs, raw callback payloads and raw provider request ids.
|
||||
|
||||
Two mechanisms enforce this rather than convention:
|
||||
|
||||
- `CardinalityGuard` validates every metric tag against a closed allowlist.
|
||||
- `SafeDiagnosticContext` rejects any structured-diagnostic field outside its allowlist.
|
||||
|
||||
An allowlist rather than a denylist, because the failure mode of a denylist is that the one field
|
||||
nobody thought of is the one that leaks.
|
||||
|
||||
Every contact point value type overrides `toString()` to print `[redacted]`. That covers the case a
|
||||
central redactor cannot: a value interpolated into a log line by accident.
|
||||
|
||||
## Web Push endpoints
|
||||
|
||||
RFC 8030 defines the push URI as a capability URL — knowing it is sufficient to push to the
|
||||
subscriber. It is handled as a secret, not as a URL.
|
||||
|
||||
## Callbacks
|
||||
|
||||
TLS, provider signature verification over the exact received bytes and external URL, replay defence
|
||||
where a timestamp or nonce is available, body-size and content-type limits, profile binding, rate
|
||||
limiting, idempotent ingestion and a security audit trail for rejections.
|
||||
|
||||
## Tenant isolation
|
||||
|
||||
Every store port carries the tenant boundary in its signature. Administrative operations require an
|
||||
explicit tenant or a global authority.
|
||||
@@ -0,0 +1,68 @@
|
||||
# Notification support matrix
|
||||
|
||||
What each channel can actually prove, and what the platform refuses to claim.
|
||||
|
||||
## Channels
|
||||
|
||||
| Channel | Reference implementation | Grade | Strongest evidence the platform records by default |
|
||||
|---|---|---|---|
|
||||
| Email | SMTP, Amazon SES API | Stable | Provider acceptance; recipient mail-server delivery, bounce and complaint when the provider publishes events |
|
||||
| SMS | Twilio Programmable Messaging | Stable | `accepted`/`queued`, `sent`, and carrier-DLR `delivered`/`undelivered` |
|
||||
| Mobile push (Android and cross-platform) | FCM, FID-first with legacy registration token compatibility | Stable | FCM acceptance and explicit failures |
|
||||
| Mobile push (Apple) | APNs HTTP/2 provider API | Stable | APNs acceptance |
|
||||
| Web Push | RFC 8030, RFC 8291, RFC 8292 | Stable | Push-service acceptance; user-agent acknowledgement only where the service offers receipts |
|
||||
| In-app inbox | Own database | Optional stable | `PERSISTED`, `SEEN`, `READ` |
|
||||
| Webhook | HTTP client platform | Extension | Whatever the receiving HTTP contract states |
|
||||
|
||||
## Evidence levels
|
||||
|
||||
`NONE` → `PLATFORM_QUEUED` → `PROVIDER_ACCEPTED` → `NETWORK_OR_CARRIER_ACCEPTED` →
|
||||
`DEVICE_DELIVERED` → `USER_AGENT_DISPLAYED` → `USER_READ`
|
||||
|
||||
| Provider signal | Highest evidence it may produce |
|
||||
|---|---|
|
||||
| Internal queue commit | `PLATFORM_QUEUED` |
|
||||
| SES `MessageId` | `PROVIDER_ACCEPTED` |
|
||||
| SES `Delivery` | `NETWORK_OR_CARRIER_ACCEPTED` |
|
||||
| Twilio `accepted` / `queued` | `PROVIDER_ACCEPTED` |
|
||||
| Twilio `sent` | `NETWORK_OR_CARRIER_ACCEPTED` |
|
||||
| Twilio `delivered` | `DEVICE_DELIVERED` |
|
||||
| FCM send success | `PROVIDER_ACCEPTED` |
|
||||
| APNs 2xx | `PROVIDER_ACCEPTED` |
|
||||
| Web Push `201` | `PROVIDER_ACCEPTED` |
|
||||
| Web Push receipt capability | `DEVICE_DELIVERED` |
|
||||
| In-app row commit | `PROVIDER_ACCEPTED` |
|
||||
| In-app `seen` endpoint | `USER_AGENT_DISPLAYED` |
|
||||
| In-app `read` endpoint, authenticated app receipt | `USER_READ` |
|
||||
|
||||
Promotions the platform will not make, in code or in configuration:
|
||||
|
||||
- FCM send success is not `DEVICE_DELIVERED`.
|
||||
- An APNs 2xx is not `DELIVERED`.
|
||||
- An SES `MessageId` is not `DELIVERED`.
|
||||
- An SMTP `250` is not inbox delivery.
|
||||
|
||||
## Submission outcomes
|
||||
|
||||
`NOT_SUBMITTED`, `CONFIRMED_ACCEPTED`, `CONFIRMED_REJECTED`, `AMBIGUOUS`.
|
||||
|
||||
`AMBIGUOUS` is a first-class stored state, not an error path. It means the request body was committed
|
||||
to the provider and the outcome could not be read. While an ambiguous attempt exists on a recipient
|
||||
delivery, automatic retry and automatic cross-channel fallback are both blocked.
|
||||
|
||||
## Not supported
|
||||
|
||||
The platform will not claim any of the following, because no channel above can support them:
|
||||
|
||||
- guaranteed delivery
|
||||
- guaranteed read
|
||||
- exactly-once human notification
|
||||
- unconditional multi-provider failover after an unread response
|
||||
- provider SDK types in the public API
|
||||
- audience selection, campaign segmentation or jurisdiction rulings
|
||||
|
||||
## Target model
|
||||
|
||||
`FCM_FID` is the primary mobile push target. `FCM_REGISTRATION_TOKEN_LEGACY` and
|
||||
`APNS_DEVICE_TOKEN` are separate types with separate lifecycles; they are never flattened into one
|
||||
string field.
|
||||
@@ -0,0 +1,94 @@
|
||||
# Toxiproxy fault injection for the notification delivery platform.
|
||||
#
|
||||
# Scope, stated up front: this is the *nightly and release* fault suite, not the PR gate. The PR
|
||||
# suite runs against a loopback socket harness in-process — deterministic, no Docker, no provider
|
||||
# sandbox — because a gate that needs infrastructure is a gate people learn to skip. What lives
|
||||
# here are the faults that harness cannot produce: real TCP behaviour under latency, bandwidth
|
||||
# starvation, and connection resets at a point the JVM's own socket layer decides.
|
||||
#
|
||||
# Usage:
|
||||
# docker compose -f infra/notification/toxiproxy/docker-compose.yml up -d
|
||||
# ./gradlew :adapter:outbound:notification:test -Dnotification.faultProxy=http://127.0.0.1:8474
|
||||
#
|
||||
# The proxies below front *stub* upstreams, never a provider's real API. Pointing a toxic proxy at
|
||||
# a live provider sends real notifications to real people from a test run, and adds a rate-limit
|
||||
# incident on an account the team shares.
|
||||
services:
|
||||
toxiproxy:
|
||||
image: ghcr.io/shopify/toxiproxy:2.11.0
|
||||
container_name: notification-toxiproxy
|
||||
ports:
|
||||
- "8474:8474" # control API
|
||||
- "18081:18081" # -> ses-stub
|
||||
- "18082:18082" # -> twilio-stub
|
||||
- "18083:18083" # -> push-stub (APNs / FCM / Web Push)
|
||||
networks: [notification-fault]
|
||||
healthcheck:
|
||||
test: ["CMD", "/toxiproxy-cli", "list"]
|
||||
interval: 5s
|
||||
timeout: 3s
|
||||
retries: 10
|
||||
|
||||
# Deterministic upstreams. Each returns the provider's success shape and nothing else; the
|
||||
# interesting behaviour is injected by the proxy in front of it, not by the stub.
|
||||
ses-stub:
|
||||
image: mendhak/http-https-echo:35
|
||||
environment:
|
||||
HTTP_PORT: "8080"
|
||||
networks: [notification-fault]
|
||||
|
||||
twilio-stub:
|
||||
image: mendhak/http-https-echo:35
|
||||
environment:
|
||||
HTTP_PORT: "8080"
|
||||
networks: [notification-fault]
|
||||
|
||||
push-stub:
|
||||
image: mendhak/http-https-echo:35
|
||||
environment:
|
||||
HTTP_PORT: "8080"
|
||||
networks: [notification-fault]
|
||||
|
||||
# Creates the proxies and the toxics once the control API is up. Kept as a job rather than a
|
||||
# README step so the topology is reproducible and reviewable rather than typed from memory.
|
||||
provision:
|
||||
image: ghcr.io/shopify/toxiproxy:2.11.0
|
||||
depends_on:
|
||||
toxiproxy:
|
||||
condition: service_healthy
|
||||
networks: [notification-fault]
|
||||
entrypoint:
|
||||
- /bin/sh
|
||||
- -c
|
||||
- |
|
||||
set -e
|
||||
CLI="/toxiproxy-cli -h toxiproxy:8474"
|
||||
$$CLI create -l 0.0.0.0:18081 -u ses-stub:8080 ses
|
||||
$$CLI create -l 0.0.0.0:18082 -u twilio-stub:8080 twilio
|
||||
$$CLI create -l 0.0.0.0:18083 -u push-stub:8080 push
|
||||
|
||||
# Response loss after the request was committed: the provider received and acted on the
|
||||
# message, and the answer never came back. This is the AMBIGUOUS case, and it is the one
|
||||
# fault no provider's documentation describes.
|
||||
$$CLI toxic add -t timeout -a timeout=0 -n response_loss --downstream --toxicity 0 ses
|
||||
$$CLI toxic add -t timeout -a timeout=0 -n response_loss --downstream --toxicity 0 twilio
|
||||
$$CLI toxic add -t timeout -a timeout=0 -n response_loss --downstream --toxicity 0 push
|
||||
|
||||
# Latency past the adapter's own timeout, to prove the timeout is the adapter's decision
|
||||
# rather than the socket's.
|
||||
$$CLI toxic add -t latency -a latency=8000 -n slow --toxicity 0 ses
|
||||
$$CLI toxic add -t latency -a latency=8000 -n slow --toxicity 0 twilio
|
||||
$$CLI toxic add -t latency -a latency=8000 -n slow --toxicity 0 push
|
||||
|
||||
# Partial write: the connection dies mid-body. Distinct from response loss, because the
|
||||
# provider never got a complete request and the attempt is genuinely retryable.
|
||||
$$CLI toxic add -t limit_data -a bytes=64 -n partial_write --upstream --toxicity 0 ses
|
||||
$$CLI toxic add -t limit_data -a bytes=64 -n partial_write --upstream --toxicity 0 twilio
|
||||
$$CLI toxic add -t limit_data -a bytes=64 -n partial_write --upstream --toxicity 0 push
|
||||
|
||||
echo "proxies ready; toxics are registered at toxicity=0 and enabled per test"
|
||||
$$CLI list
|
||||
|
||||
networks:
|
||||
notification-fault:
|
||||
driver: bridge
|
||||
Executable
+141
@@ -0,0 +1,141 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# The MongoDB Advanced capability gate (advanced plan Task 15).
|
||||
#
|
||||
# Advanced capabilities are opt-in modules. This script verifies the contracts that can be verified
|
||||
# without provider infrastructure, and then reports -- explicitly -- which promotion evidence it
|
||||
# could NOT produce.
|
||||
#
|
||||
# Required promotion categories (MongoAdvancedPromotionEvidence.REQUIRED):
|
||||
#
|
||||
# stable-platform, actual-topology, security, migration, failure, runbook
|
||||
#
|
||||
# `actual-topology` is the one that cannot be substituted. A container gives a functional pass for
|
||||
# sharding, search, vector and encryption while exercising none of the behaviour that makes them
|
||||
# Advanced rather than Stable: real shard distribution, a real analyzer, a real KMS. Atlas Local is
|
||||
# a pull-request convenience and is not release evidence -- see
|
||||
# MongoAtlasCapabilityContractSuite.Environment.
|
||||
#
|
||||
# Usage:
|
||||
# bash scripts/verify-mongodb-advanced.sh
|
||||
# MONGODB_DOCKER=1 bash scripts/verify-mongodb-advanced.sh
|
||||
# MONGODB_SHARDED_URI=... MONGODB_ATLAS_URI=... MONGODB_KMS=... bash scripts/verify-mongodb-advanced.sh
|
||||
#
|
||||
set -euo pipefail
|
||||
|
||||
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
GRADLE_DIR="${REPO_ROOT}/src"
|
||||
MODULE=':adapter:outbound:persistence-mongo'
|
||||
GRADLE=(./gradlew --console=plain)
|
||||
|
||||
FAILED=()
|
||||
MISSING_EVIDENCE=()
|
||||
|
||||
echo "MongoDB Advanced capability gate"
|
||||
echo "repository: ${REPO_ROOT}"
|
||||
|
||||
# --- stable-platform -------------------------------------------------------------------------
|
||||
# An Advanced capability cannot be promoted over a Stable platform that does not itself pass.
|
||||
echo ""
|
||||
echo "=== [stable-platform] Stable gate"
|
||||
if bash "${REPO_ROOT}/scripts/verify-mongodb-platform.sh"; then
|
||||
echo "stable-platform: supplied"
|
||||
else
|
||||
status=$?
|
||||
if (( status == 2 )); then
|
||||
echo "stable-platform: INCOMPLETE (the Stable gate skipped lanes)"
|
||||
MISSING_EVIDENCE+=("stable-platform (Stable gate incomplete)")
|
||||
else
|
||||
FAILED+=("stable-platform")
|
||||
fi
|
||||
fi
|
||||
|
||||
# --- failure + runbook (hermetic) -------------------------------------------------------------
|
||||
# Every Advanced refusal contract: disabled capability refuses construction, CSFLE/QE cannot share a
|
||||
# collection, QE substring/prefix/suffix unsupported on 8.0, a non-READY search index cannot serve,
|
||||
# undeclared scatter-gather is rejected, a dimension mismatch is refused.
|
||||
echo ""
|
||||
echo "=== [failure] Advanced contract tests"
|
||||
if (cd "${GRADLE_DIR}" && "${GRADLE[@]}" "${MODULE}:test" --tests '*advanced*'); then
|
||||
echo "failure: supplied"
|
||||
else
|
||||
FAILED+=("failure")
|
||||
fi
|
||||
|
||||
echo ""
|
||||
echo "=== [runbook] capability documentation"
|
||||
for doc in sharding time-series encryption search-vector multi-tenancy gridfs-migration; do
|
||||
path="${REPO_ROOT}/docs/mongodb/advanced/${doc}.md"
|
||||
if [[ -f "${path}" ]]; then
|
||||
echo " + ${doc}.md"
|
||||
else
|
||||
echo " - ${doc}.md MISSING"
|
||||
FAILED+=("runbook:${doc}")
|
||||
fi
|
||||
done
|
||||
if [[ ! -f "${REPO_ROOT}/docs/adr/ADR-MONGO-ADV-001-capability-promotion.md" ]]; then
|
||||
echo " - ADR-MONGO-ADV-001 MISSING"
|
||||
FAILED+=("runbook:ADR-MONGO-ADV-001")
|
||||
fi
|
||||
|
||||
# --- actual-topology -------------------------------------------------------------------------
|
||||
echo ""
|
||||
echo "=== [actual-topology] provider environments"
|
||||
if [[ -n "${MONGODB_SHARDED_URI:-}" ]]; then
|
||||
if (cd "${GRADLE_DIR}" && "${GRADLE[@]}" "${MODULE}:test" --tests '*Shard*' \
|
||||
-Dmongodb.sharded.uri="${MONGODB_SHARDED_URI}"); then
|
||||
echo "actual-topology(sharded): supplied"
|
||||
else
|
||||
FAILED+=("actual-topology:sharded")
|
||||
fi
|
||||
else
|
||||
echo "actual-topology(sharded): no MONGODB_SHARDED_URI"
|
||||
MISSING_EVIDENCE+=("actual-topology: sharded cluster")
|
||||
fi
|
||||
|
||||
if [[ -n "${MONGODB_ATLAS_URI:-}" ]]; then
|
||||
echo "actual-topology(search/vector): MONGODB_ATLAS_URI present"
|
||||
else
|
||||
echo "actual-topology(search/vector): no MONGODB_ATLAS_URI"
|
||||
MISSING_EVIDENCE+=("actual-topology: search/vector on the actual target deployment")
|
||||
fi
|
||||
|
||||
if [[ -n "${MONGODB_KMS:-}" ]]; then
|
||||
echo "actual-topology(encryption): MONGODB_KMS present"
|
||||
else
|
||||
echo "actual-topology(encryption): no MONGODB_KMS"
|
||||
MISSING_EVIDENCE+=("actual-topology: real KMS and key vault")
|
||||
fi
|
||||
|
||||
# --- security + migration ---------------------------------------------------------------------
|
||||
# These are review artefacts, not test runs: a role review and a documented migration path per
|
||||
# capability. The gate records that they are outstanding rather than pretending a green test covers
|
||||
# them.
|
||||
MISSING_EVIDENCE+=("security: per-capability privilege review sign-off")
|
||||
MISSING_EVIDENCE+=("migration: per-capability migration path sign-off")
|
||||
|
||||
# --- Report ------------------------------------------------------------------------------------
|
||||
echo ""
|
||||
echo "---------------------------------------------------------------"
|
||||
if (( ${#FAILED[@]} > 0 )); then
|
||||
echo "ADVANCED GATE: FAILED"
|
||||
for entry in "${FAILED[@]}"; do echo " - ${entry}"; done
|
||||
echo "---------------------------------------------------------------"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "verifiable contracts: PASSED"
|
||||
if (( ${#MISSING_EVIDENCE[@]} > 0 )); then
|
||||
echo ""
|
||||
echo "ADVANCED GATE: NOT PROMOTABLE -- missing evidence:"
|
||||
for entry in "${MISSING_EVIDENCE[@]}"; do echo " ~ ${entry}"; done
|
||||
echo ""
|
||||
echo "A capability stays opt-in until every category in"
|
||||
echo "MongoAdvancedPromotionEvidence.REQUIRED is supplied. See"
|
||||
echo "docs/adr/ADR-MONGO-ADV-001-capability-promotion.md."
|
||||
echo "---------------------------------------------------------------"
|
||||
exit 2
|
||||
fi
|
||||
|
||||
echo "ADVANCED GATE: PASSED"
|
||||
echo "---------------------------------------------------------------"
|
||||
Executable
+158
@@ -0,0 +1,158 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# The MongoDB Stable release gate (design §30, plan Task 50).
|
||||
#
|
||||
# Runs every lane that produces one of the Stable evidence categories:
|
||||
#
|
||||
# mapping, transaction, migration, change-stream, security,
|
||||
# failover, performance, compatibility
|
||||
#
|
||||
# The gate exists because "the test suite is green" and "every category has evidence" are different
|
||||
# statements. A suite passes happily with a whole lane skipped -- no Docker, a disabled tag, a
|
||||
# renamed task -- and a release built on that suite has no failover or compatibility evidence at
|
||||
# all, silently. Each lane below is therefore run by name, and a skipped lane is reported as skipped
|
||||
# rather than counted as passed.
|
||||
#
|
||||
# Advanced capabilities are NOT promoted or transitively included here. See
|
||||
# scripts/verify-mongodb-advanced.sh.
|
||||
#
|
||||
# Usage:
|
||||
# bash scripts/verify-mongodb-platform.sh # hermetic lanes only
|
||||
# MONGODB_DOCKER=1 bash scripts/verify-mongodb-platform.sh # + container lanes
|
||||
#
|
||||
set -euo pipefail
|
||||
|
||||
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
GRADLE_DIR="${REPO_ROOT}/src"
|
||||
MODULE=':adapter:outbound:persistence-mongo'
|
||||
GRADLE=(./gradlew --console=plain)
|
||||
|
||||
RESULTS_DIR="${GRADLE_DIR}/adapter/outbound/persistence-mongo/build/test-results"
|
||||
|
||||
RAN=()
|
||||
SKIPPED=()
|
||||
FAILED=()
|
||||
|
||||
# Counts the tests a lane actually executed, from its JUnit XML.
|
||||
#
|
||||
# A lane whose filter matches nothing passes: Gradle runs the task, discovers no tests, and reports
|
||||
# success. That is the failure mode this whole gate exists to prevent -- an empty lane is not
|
||||
# evidence, it is the absence of evidence wearing a green tick. Any lane that reports zero executed
|
||||
# tests is treated as a failure.
|
||||
executed_tests() {
|
||||
local task="$1"
|
||||
local dir="${RESULTS_DIR}/${task}"
|
||||
[[ -d "${dir}" ]] || { echo 0; return; }
|
||||
local total=0
|
||||
shopt -s nullglob
|
||||
for xml in "${dir}"/*.xml; do
|
||||
local count
|
||||
count=$(sed -n 's/.*<testsuite[^>]* tests="\([0-9]*\)".*/\1/p' "${xml}" | head -1)
|
||||
total=$(( total + ${count:-0} ))
|
||||
done
|
||||
shopt -u nullglob
|
||||
echo "${total}"
|
||||
}
|
||||
|
||||
run_lane() {
|
||||
local category="$1"
|
||||
local task="$2"
|
||||
shift 2
|
||||
echo ""
|
||||
echo "=== [${category}] ${task}"
|
||||
if ! (cd "${GRADLE_DIR}" && "${GRADLE[@]}" "${MODULE}:${task}" "$@"); then
|
||||
FAILED+=("${category}:${task}")
|
||||
return
|
||||
fi
|
||||
# `check` aggregates several tasks and has no results directory of its own.
|
||||
if [[ "${task}" == "check" ]]; then
|
||||
RAN+=("${category}:${task}")
|
||||
return
|
||||
fi
|
||||
local executed
|
||||
executed=$(executed_tests "${task}")
|
||||
if (( executed == 0 )); then
|
||||
echo "!!! ${task} passed without executing a single test — the lane's filter matches nothing,"
|
||||
echo "!!! so the '${category}' evidence category is empty."
|
||||
FAILED+=("${category}:${task} (0 tests executed)")
|
||||
else
|
||||
RAN+=("${category}:${task} (${executed} tests)")
|
||||
fi
|
||||
}
|
||||
|
||||
skip_lane() {
|
||||
local category="$1"
|
||||
local task="$2"
|
||||
local reason="$3"
|
||||
echo ""
|
||||
echo "=== [${category}] ${task} -- SKIPPED (${reason})"
|
||||
SKIPPED+=("${category}:${task} (${reason})")
|
||||
}
|
||||
|
||||
docker_available() {
|
||||
[[ "${MONGODB_DOCKER:-0}" == "1" ]] && command -v docker >/dev/null 2>&1 && docker info >/dev/null 2>&1
|
||||
}
|
||||
|
||||
echo "MongoDB Stable release gate"
|
||||
echo "repository: ${REPO_ROOT}"
|
||||
|
||||
# --- Always-on lanes -------------------------------------------------------------------------
|
||||
# Static analysis, architecture boundaries, unit and hermetic contract tests. These produce the
|
||||
# mapping, transaction, migration, change-stream and security evidence that does not need a server.
|
||||
run_lane "static-analysis" "check" -x "mongoStableContractTest"
|
||||
run_lane "mapping+transaction+migration+change-stream+security" "mongoStableContractTest"
|
||||
|
||||
# --- Container lanes -------------------------------------------------------------------------
|
||||
# A lane that needs Docker inside `check` teaches people to skip `check`, so these are opt-in --
|
||||
# but opting out is recorded, not silent.
|
||||
if docker_available; then
|
||||
run_lane "compatibility" "mongoCompatibilityTest"
|
||||
run_lane "migration" "mongoMigrationTest"
|
||||
run_lane "security" "mongoSecurityIntegrationTest"
|
||||
run_lane "failover" "mongoReplicaSetTest"
|
||||
run_lane "failover" "mongoFailoverTest"
|
||||
run_lane "performance" "mongoPerformanceTest"
|
||||
else
|
||||
reason="MONGODB_DOCKER!=1 or Docker unavailable"
|
||||
skip_lane "compatibility" "mongoCompatibilityTest" "${reason}"
|
||||
skip_lane "migration" "mongoMigrationTest" "${reason}"
|
||||
skip_lane "security" "mongoSecurityIntegrationTest" "${reason}"
|
||||
skip_lane "failover" "mongoReplicaSetTest" "${reason}"
|
||||
skip_lane "failover" "mongoFailoverTest" "${reason}"
|
||||
skip_lane "performance" "mongoPerformanceTest" "${reason}"
|
||||
fi
|
||||
|
||||
# --- Architecture-wide gates -----------------------------------------------------------------
|
||||
echo ""
|
||||
echo "=== [architecture] repository-wide verification"
|
||||
if (cd "${GRADLE_DIR}" \
|
||||
&& "${GRADLE[@]}" verifyCleanArchitectureDependencies \
|
||||
&& "${GRADLE[@]}" :app-bootstrap:test --tests '*CleanArchitectureTest'); then
|
||||
RAN+=("architecture:repository-wide")
|
||||
else
|
||||
FAILED+=("architecture:repository-wide")
|
||||
fi
|
||||
|
||||
# --- Report ------------------------------------------------------------------------------------
|
||||
echo ""
|
||||
echo "---------------------------------------------------------------"
|
||||
echo "ran: ${#RAN[@]}"
|
||||
for entry in "${RAN[@]:-}"; do [[ -n "${entry}" ]] && echo " + ${entry}"; done
|
||||
echo "skipped: ${#SKIPPED[@]}"
|
||||
for entry in "${SKIPPED[@]:-}"; do [[ -n "${entry}" ]] && echo " ~ ${entry}"; done
|
||||
echo "failed: ${#FAILED[@]}"
|
||||
for entry in "${FAILED[@]:-}"; do [[ -n "${entry}" ]] && echo " - ${entry}"; done
|
||||
echo "---------------------------------------------------------------"
|
||||
|
||||
if (( ${#FAILED[@]} > 0 )); then
|
||||
echo "STABLE GATE: FAILED"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if (( ${#SKIPPED[@]} > 0 )); then
|
||||
echo "STABLE GATE: INCOMPLETE -- lanes above were not run, so their evidence categories are absent."
|
||||
echo "A release requires every category. Re-run with MONGODB_DOCKER=1 on a host with Docker."
|
||||
exit 2
|
||||
fi
|
||||
|
||||
echo "STABLE GATE: PASSED -- every evidence category produced."
|
||||
+38
@@ -0,0 +1,38 @@
|
||||
package dev.caskeleton.adapter.inbound.web.notification.platform.callback;
|
||||
|
||||
import org.springframework.boot.autoconfigure.condition.ConditionalOnProperty;
|
||||
import org.springframework.context.annotation.Bean;
|
||||
import org.springframework.context.annotation.Configuration;
|
||||
import org.springframework.core.Ordered;
|
||||
import org.springframework.core.annotation.Order;
|
||||
import org.springframework.security.config.annotation.web.builders.HttpSecurity;
|
||||
import org.springframework.security.config.http.SessionCreationPolicy;
|
||||
import org.springframework.security.web.SecurityFilterChain;
|
||||
|
||||
/**
|
||||
* Security chain for the provider callback endpoints.
|
||||
*
|
||||
* <p>Callbacks authenticate with a provider signature, not with a user session, so they get their
|
||||
* own chain: CSRF and session creation are off, and the ordinary user chain never sees them.
|
||||
* Putting them on the user chain would either break every provider or force the user chain to be
|
||||
* permissive.
|
||||
*/
|
||||
@Configuration(proxyBeanMethods = false)
|
||||
@ConditionalOnProperty(
|
||||
prefix = "ca-skeleton.notification.platform.callbacks",
|
||||
name = "enabled",
|
||||
havingValue = "true")
|
||||
public class CallbackMvcSecurityConfiguration {
|
||||
|
||||
/** Dedicated, ordered-first chain for the callback path. */
|
||||
@Bean
|
||||
@Order(Ordered.HIGHEST_PRECEDENCE + 10)
|
||||
public SecurityFilterChain notificationCallbackFilterChain(HttpSecurity http) throws Exception {
|
||||
return http.securityMatcher("/internal/notification/callbacks/**")
|
||||
.csrf(csrf -> csrf.disable())
|
||||
.sessionManagement(
|
||||
session -> session.sessionCreationPolicy(SessionCreationPolicy.STATELESS))
|
||||
.authorizeHttpRequests(requests -> requests.anyRequest().permitAll())
|
||||
.build();
|
||||
}
|
||||
}
|
||||
+75
@@ -0,0 +1,75 @@
|
||||
package dev.caskeleton.adapter.inbound.web.notification.platform.callback;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.api.ProviderId;
|
||||
import dev.caskeleton.application.notification.platform.api.ProviderProfileId;
|
||||
import dev.caskeleton.application.notification.platform.callback.CallbackRequest;
|
||||
import jakarta.servlet.http.HttpServletRequest;
|
||||
import java.time.Clock;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Optional;
|
||||
|
||||
/**
|
||||
* Builds the transport-neutral callback request.
|
||||
*
|
||||
* <p>Both the servlet and reactive endpoints use this, so signature verification sees exactly the
|
||||
* same canonical bytes and URL regardless of which stack received the call.
|
||||
*/
|
||||
public final class CallbackRequestFactory {
|
||||
|
||||
private final ExternalRequestUrlResolver urlResolver;
|
||||
private final Clock clock;
|
||||
|
||||
public CallbackRequestFactory(ExternalRequestUrlResolver urlResolver, Clock clock) {
|
||||
this.urlResolver = Objects.requireNonNull(urlResolver, "urlResolver");
|
||||
this.clock = Objects.requireNonNull(clock, "clock");
|
||||
}
|
||||
|
||||
/** Build from a servlet request plus the already-read raw body. */
|
||||
public CallbackRequest create(
|
||||
String provider, String profile, HttpServletRequest request, byte[] body) {
|
||||
Objects.requireNonNull(provider, "provider");
|
||||
Objects.requireNonNull(profile, "profile");
|
||||
Objects.requireNonNull(request, "request");
|
||||
Objects.requireNonNull(body, "body");
|
||||
|
||||
Map<String, List<String>> headers = new LinkedHashMap<>();
|
||||
for (String name : Collections.list(request.getHeaderNames())) {
|
||||
headers.put(name, new ArrayList<>(Collections.list(request.getHeaders(name))));
|
||||
}
|
||||
|
||||
return new CallbackRequest(
|
||||
new ProviderId(provider),
|
||||
new ProviderProfileId(profile),
|
||||
urlResolver.resolve(request),
|
||||
request.getMethod(),
|
||||
Optional.ofNullable(request.getContentType()),
|
||||
headers,
|
||||
body,
|
||||
clock.instant());
|
||||
}
|
||||
|
||||
/** Build from an already-resolved external URL, used by the reactive endpoint. */
|
||||
public CallbackRequest create(
|
||||
String provider,
|
||||
String profile,
|
||||
String externalUrl,
|
||||
String method,
|
||||
Optional<String> contentType,
|
||||
Map<String, List<String>> headers,
|
||||
byte[] body) {
|
||||
return new CallbackRequest(
|
||||
new ProviderId(provider),
|
||||
new ProviderProfileId(profile),
|
||||
externalUrl,
|
||||
method,
|
||||
contentType,
|
||||
headers,
|
||||
body,
|
||||
clock.instant());
|
||||
}
|
||||
}
|
||||
+71
@@ -0,0 +1,71 @@
|
||||
package dev.caskeleton.adapter.inbound.web.notification.platform.callback;
|
||||
|
||||
import jakarta.servlet.http.HttpServletRequest;
|
||||
import java.util.Locale;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
|
||||
/**
|
||||
* Reconstructs the URL the provider actually called.
|
||||
*
|
||||
* <p>Several providers sign the request URL, so getting this wrong turns every valid webhook into a
|
||||
* signature failure. Forwarded headers are only honoured when the immediate peer is a configured
|
||||
* trusted proxy: trusting them unconditionally would let any caller choose the URL that gets
|
||||
* verified, which defeats the signature entirely.
|
||||
*/
|
||||
public final class ExternalRequestUrlResolver {
|
||||
|
||||
private final Set<String> trustedProxies;
|
||||
|
||||
public ExternalRequestUrlResolver(Set<String> trustedProxies) {
|
||||
this.trustedProxies = Set.copyOf(Objects.requireNonNull(trustedProxies, "trustedProxies"));
|
||||
}
|
||||
|
||||
/** External URL of a request. */
|
||||
public String resolve(HttpServletRequest request) {
|
||||
Objects.requireNonNull(request, "request");
|
||||
String scheme = request.getScheme();
|
||||
String host = request.getServerName();
|
||||
int port = request.getServerPort();
|
||||
|
||||
if (trustedProxies.contains(request.getRemoteAddr())) {
|
||||
String forwarded = request.getHeader("Forwarded");
|
||||
if (forwarded != null) {
|
||||
for (String element : forwarded.split(";", -1)) {
|
||||
String trimmed = element.trim().toLowerCase(Locale.ROOT);
|
||||
if (trimmed.startsWith("proto=")) {
|
||||
scheme = trimmed.substring("proto=".length());
|
||||
} else if (trimmed.startsWith("host=")) {
|
||||
host = element.trim().substring("host=".length());
|
||||
port = -1;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
String protoHeader = request.getHeader("X-Forwarded-Proto");
|
||||
String hostHeader = request.getHeader("X-Forwarded-Host");
|
||||
if (protoHeader != null) {
|
||||
scheme = protoHeader;
|
||||
}
|
||||
if (hostHeader != null) {
|
||||
host = hostHeader;
|
||||
port = -1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
StringBuilder url = new StringBuilder(scheme).append("://").append(host);
|
||||
boolean defaultPort =
|
||||
port < 0
|
||||
|| ("https".equalsIgnoreCase(scheme) && port == 443)
|
||||
|| ("http".equalsIgnoreCase(scheme) && port == 80);
|
||||
if (!defaultPort) {
|
||||
url.append(':').append(port);
|
||||
}
|
||||
url.append(request.getRequestURI());
|
||||
String query = request.getQueryString();
|
||||
if (query != null && !query.isBlank()) {
|
||||
url.append('?').append(query);
|
||||
}
|
||||
return url.toString();
|
||||
}
|
||||
}
|
||||
+75
@@ -0,0 +1,75 @@
|
||||
package dev.caskeleton.adapter.inbound.web.notification.platform.callback;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.api.error.CallbackValidationException;
|
||||
import dev.caskeleton.application.notification.platform.callback.ProviderCallbackIngestionService;
|
||||
import jakarta.servlet.http.HttpServletRequest;
|
||||
import java.util.Objects;
|
||||
import org.springframework.boot.autoconfigure.condition.ConditionalOnProperty;
|
||||
import org.springframework.boot.autoconfigure.condition.ConditionalOnWebApplication;
|
||||
import org.springframework.http.HttpStatus;
|
||||
import org.springframework.http.ResponseEntity;
|
||||
import org.springframework.web.bind.annotation.ExceptionHandler;
|
||||
import org.springframework.web.bind.annotation.PathVariable;
|
||||
import org.springframework.web.bind.annotation.PostMapping;
|
||||
import org.springframework.web.bind.annotation.RequestBody;
|
||||
import org.springframework.web.bind.annotation.RequestMapping;
|
||||
import org.springframework.web.bind.annotation.RestController;
|
||||
|
||||
/**
|
||||
* Servlet callback endpoint.
|
||||
*
|
||||
* <p>The body arrives as raw bytes, never as a parsed form. Providers sign the exact octets, and
|
||||
* letting the container parse and re-encode them is the most common cause of a valid webhook
|
||||
* failing verification.
|
||||
*
|
||||
* <p>The response is a bare {@code 204}: no body, no diagnostics. A provider only needs to know the
|
||||
* event is recorded, and an error body would be a channel for leaking what the platform knows.
|
||||
*
|
||||
* <p>Registered only in a servlet application and only when callbacks are enabled. An annotated
|
||||
* controller is also honoured by WebFlux, so without the servlet condition a reactive deployment
|
||||
* would map both this and the functional router onto the same path — and a provider signature would
|
||||
* then be verified twice against two different canonical URLs.
|
||||
*/
|
||||
@RestController
|
||||
@ConditionalOnWebApplication(type = ConditionalOnWebApplication.Type.SERVLET)
|
||||
@ConditionalOnProperty(
|
||||
prefix = "ca-skeleton.notification.platform.callbacks",
|
||||
name = "enabled",
|
||||
havingValue = "true")
|
||||
@RequestMapping("/internal/notification/callbacks")
|
||||
public final class NotificationCallbackMvcController {
|
||||
|
||||
/** Hard body ceiling applied before any provider adapter is consulted. */
|
||||
public static final int MAX_BODY_BYTES = 65_536;
|
||||
|
||||
private final ProviderCallbackIngestionService ingestion;
|
||||
private final CallbackRequestFactory requestFactory;
|
||||
|
||||
public NotificationCallbackMvcController(
|
||||
ProviderCallbackIngestionService ingestion, CallbackRequestFactory requestFactory) {
|
||||
this.ingestion = Objects.requireNonNull(ingestion, "ingestion");
|
||||
this.requestFactory = Objects.requireNonNull(requestFactory, "requestFactory");
|
||||
}
|
||||
|
||||
/** Receive one provider callback. */
|
||||
@PostMapping(path = "/{provider}/{profile}")
|
||||
public ResponseEntity<Void> callback(
|
||||
@PathVariable String provider,
|
||||
@PathVariable String profile,
|
||||
HttpServletRequest request,
|
||||
@RequestBody byte[] body) {
|
||||
if (body.length > MAX_BODY_BYTES) {
|
||||
return ResponseEntity.status(HttpStatus.CONTENT_TOO_LARGE).build();
|
||||
}
|
||||
// A duplicate answers 204 exactly like a first delivery. The provider did its job either way,
|
||||
// and any other status would make it retry an event that is already recorded.
|
||||
ingestion.ingest(requestFactory.create(provider, profile, request, body));
|
||||
return ResponseEntity.noContent().build();
|
||||
}
|
||||
|
||||
/** A rejected callback never reveals why beyond the status code. */
|
||||
@ExceptionHandler(CallbackValidationException.class)
|
||||
public ResponseEntity<Void> onValidationFailure(CallbackValidationException failure) {
|
||||
return ResponseEntity.status(HttpStatus.BAD_REQUEST).build();
|
||||
}
|
||||
}
|
||||
+48
@@ -0,0 +1,48 @@
|
||||
package dev.caskeleton.adapter.inbound.web.notification.platform.callback.reactive;
|
||||
|
||||
import java.util.Objects;
|
||||
import org.springframework.core.io.buffer.DataBuffer;
|
||||
import org.springframework.core.io.buffer.DataBufferUtils;
|
||||
import org.springframework.web.reactive.function.server.ServerRequest;
|
||||
import reactor.core.publisher.Mono;
|
||||
|
||||
/**
|
||||
* Reads the raw body with a hard ceiling and no buffer leaks.
|
||||
*
|
||||
* <p>Every {@link DataBuffer} is released on success, on error and on cancellation. A reactive
|
||||
* endpoint that forgets the cancellation path leaks native memory exactly when it is under the load
|
||||
* that caused the cancellation.
|
||||
*/
|
||||
public final class BoundedCallbackBodyReader {
|
||||
|
||||
private final int maxBytes;
|
||||
|
||||
public BoundedCallbackBodyReader(int maxBytes) {
|
||||
if (maxBytes < 1) {
|
||||
throw new IllegalArgumentException("maxBytes");
|
||||
}
|
||||
this.maxBytes = maxBytes;
|
||||
}
|
||||
|
||||
/** Read at most the configured number of bytes. */
|
||||
public Mono<byte[]> read(ServerRequest request) {
|
||||
Objects.requireNonNull(request, "request");
|
||||
return DataBufferUtils.join(request.bodyToFlux(DataBuffer.class), maxBytes)
|
||||
.map(
|
||||
buffer -> {
|
||||
try {
|
||||
byte[] bytes = new byte[buffer.readableByteCount()];
|
||||
buffer.read(bytes);
|
||||
return bytes;
|
||||
} finally {
|
||||
DataBufferUtils.release(buffer);
|
||||
}
|
||||
})
|
||||
.defaultIfEmpty(new byte[0]);
|
||||
}
|
||||
|
||||
/** Configured ceiling. */
|
||||
public int maxBytes() {
|
||||
return maxBytes;
|
||||
}
|
||||
}
|
||||
+58
@@ -0,0 +1,58 @@
|
||||
package dev.caskeleton.adapter.inbound.web.notification.platform.callback.reactive;
|
||||
|
||||
import dev.caskeleton.adapter.inbound.web.notification.platform.callback.CallbackRequestFactory;
|
||||
import dev.caskeleton.application.notification.platform.callback.ProviderCallbackIngestionService;
|
||||
import org.springframework.beans.factory.annotation.Value;
|
||||
import org.springframework.boot.autoconfigure.condition.ConditionalOnMissingBean;
|
||||
import org.springframework.boot.autoconfigure.condition.ConditionalOnProperty;
|
||||
import org.springframework.boot.autoconfigure.condition.ConditionalOnWebApplication;
|
||||
import org.springframework.context.annotation.Bean;
|
||||
import org.springframework.context.annotation.Configuration;
|
||||
import org.springframework.web.reactive.function.server.RouterFunction;
|
||||
import org.springframework.web.reactive.function.server.ServerResponse;
|
||||
|
||||
/**
|
||||
* Registers the reactive callback transport, and only it.
|
||||
*
|
||||
* <p>This configuration is {@code REACTIVE}-only and the servlet controller carries the matching
|
||||
* {@code SERVLET} condition, so exactly one of the two is ever registered — by construction rather
|
||||
* than by convention. Both on the same path would mean a provider signature is verified twice
|
||||
* against two different canonical URLs, a failure that shows up only in production and only for
|
||||
* signed providers, and reads like a credential problem.
|
||||
*
|
||||
* <p>The body ceiling is read as a property rather than through the platform settings type: that
|
||||
* type belongs to the outbound notification adapter, which this inbound adapter must not depend on.
|
||||
*/
|
||||
@Configuration(proxyBeanMethods = false)
|
||||
@ConditionalOnWebApplication(type = ConditionalOnWebApplication.Type.REACTIVE)
|
||||
@ConditionalOnProperty(
|
||||
prefix = "ca-skeleton.notification.platform.callbacks",
|
||||
name = "enabled",
|
||||
havingValue = "true")
|
||||
public class CallbackWebFluxConfiguration {
|
||||
|
||||
/** Bounded body reader; the ceiling applies before any provider adapter is consulted. */
|
||||
@Bean
|
||||
@ConditionalOnMissingBean
|
||||
public BoundedCallbackBodyReader notificationCallbackBodyReader(
|
||||
@Value("${ca-skeleton.notification.platform.callbacks.max-body-bytes:65536}") int maxBytes) {
|
||||
return new BoundedCallbackBodyReader(maxBytes);
|
||||
}
|
||||
|
||||
/** Reactive handler. */
|
||||
@Bean
|
||||
@ConditionalOnMissingBean
|
||||
public NotificationCallbackWebFluxHandler notificationCallbackWebFluxHandler(
|
||||
ProviderCallbackIngestionService ingestion,
|
||||
CallbackRequestFactory requestFactory,
|
||||
BoundedCallbackBodyReader bodyReader) {
|
||||
return new NotificationCallbackWebFluxHandler(ingestion, requestFactory, bodyReader);
|
||||
}
|
||||
|
||||
/** Functional route for the callback path. */
|
||||
@Bean
|
||||
public RouterFunction<ServerResponse> notificationCallbackRoutes(
|
||||
NotificationCallbackWebFluxHandler handler) {
|
||||
return new CallbackWebFluxRouter(handler).routes();
|
||||
}
|
||||
}
|
||||
+30
@@ -0,0 +1,30 @@
|
||||
package dev.caskeleton.adapter.inbound.web.notification.platform.callback.reactive;
|
||||
|
||||
import java.util.Objects;
|
||||
import org.springframework.web.reactive.function.server.RequestPredicates;
|
||||
import org.springframework.web.reactive.function.server.RouterFunction;
|
||||
import org.springframework.web.reactive.function.server.RouterFunctions;
|
||||
import org.springframework.web.reactive.function.server.ServerResponse;
|
||||
|
||||
/**
|
||||
* Routes the reactive callback path.
|
||||
*
|
||||
* <p>Kept separate from the servlet controller so that only one of the two is ever registered; two
|
||||
* endpoints on the same path would mean a provider's signature is verified twice against two
|
||||
* different canonical URLs.
|
||||
*/
|
||||
public final class CallbackWebFluxRouter {
|
||||
|
||||
private final NotificationCallbackWebFluxHandler handler;
|
||||
|
||||
public CallbackWebFluxRouter(NotificationCallbackWebFluxHandler handler) {
|
||||
this.handler = Objects.requireNonNull(handler, "handler");
|
||||
}
|
||||
|
||||
/** Router function for the callback path. */
|
||||
public RouterFunction<ServerResponse> routes() {
|
||||
return RouterFunctions.route(
|
||||
RequestPredicates.POST("/internal/notification/callbacks/{provider}/{profile}"),
|
||||
handler::handle);
|
||||
}
|
||||
}
|
||||
+85
@@ -0,0 +1,85 @@
|
||||
package dev.caskeleton.adapter.inbound.web.notification.platform.callback.reactive;
|
||||
|
||||
import dev.caskeleton.adapter.inbound.web.notification.platform.callback.CallbackRequestFactory;
|
||||
import dev.caskeleton.application.notification.platform.api.error.CallbackValidationException;
|
||||
import dev.caskeleton.application.notification.platform.callback.ProviderCallbackIngestionService;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Optional;
|
||||
import org.springframework.core.io.buffer.DataBufferLimitException;
|
||||
import org.springframework.http.HttpStatus;
|
||||
import org.springframework.web.reactive.function.server.ServerRequest;
|
||||
import org.springframework.web.reactive.function.server.ServerResponse;
|
||||
import reactor.core.publisher.Mono;
|
||||
import reactor.core.scheduler.Schedulers;
|
||||
|
||||
/**
|
||||
* Reactive callback endpoint.
|
||||
*
|
||||
* <p>Ingestion is blocking — it writes to the database — so it runs on {@code boundedElastic} and
|
||||
* never on the event loop. Running it inline would stall every other connection the loop is
|
||||
* serving.
|
||||
*
|
||||
* <p>It shares the canonicalisation and the ingestion service with the servlet endpoint, so a
|
||||
* deployment can switch web stacks without changing what a provider signature is checked against.
|
||||
*/
|
||||
public final class NotificationCallbackWebFluxHandler {
|
||||
|
||||
private final ProviderCallbackIngestionService ingestion;
|
||||
private final CallbackRequestFactory requestFactory;
|
||||
private final BoundedCallbackBodyReader bodyReader;
|
||||
|
||||
public NotificationCallbackWebFluxHandler(
|
||||
ProviderCallbackIngestionService ingestion,
|
||||
CallbackRequestFactory requestFactory,
|
||||
BoundedCallbackBodyReader bodyReader) {
|
||||
this.ingestion = Objects.requireNonNull(ingestion, "ingestion");
|
||||
this.requestFactory = Objects.requireNonNull(requestFactory, "requestFactory");
|
||||
this.bodyReader = Objects.requireNonNull(bodyReader, "bodyReader");
|
||||
}
|
||||
|
||||
/** Handle one callback. */
|
||||
public Mono<ServerResponse> handle(ServerRequest request) {
|
||||
String provider = request.pathVariable("provider");
|
||||
String profile = request.pathVariable("profile");
|
||||
|
||||
return bodyReader
|
||||
.read(request)
|
||||
.flatMap(
|
||||
body ->
|
||||
Mono.fromCallable(
|
||||
() ->
|
||||
ingestion.ingest(
|
||||
requestFactory.create(
|
||||
provider,
|
||||
profile,
|
||||
request.uri().toString(),
|
||||
request.method().name(),
|
||||
request.headers().contentType().map(Object::toString),
|
||||
headers(request),
|
||||
body)))
|
||||
.subscribeOn(Schedulers.boundedElastic()))
|
||||
.then(ServerResponse.noContent().build())
|
||||
.onErrorResume(
|
||||
DataBufferLimitException.class,
|
||||
failure -> ServerResponse.status(HttpStatus.CONTENT_TOO_LARGE).build())
|
||||
.onErrorResume(
|
||||
CallbackValidationException.class,
|
||||
failure -> ServerResponse.status(HttpStatus.BAD_REQUEST).build());
|
||||
}
|
||||
|
||||
private static Map<String, List<String>> headers(ServerRequest request) {
|
||||
Map<String, List<String>> headers = new java.util.LinkedHashMap<>();
|
||||
request
|
||||
.headers()
|
||||
.asHttpHeaders()
|
||||
.forEach((name, values) -> headers.put(name, List.copyOf(values)));
|
||||
return Map.copyOf(headers);
|
||||
}
|
||||
|
||||
/** Content type of a request, if declared. */
|
||||
public static Optional<String> contentType(ServerRequest request) {
|
||||
return request.headers().contentType().map(Object::toString);
|
||||
}
|
||||
}
|
||||
+396
@@ -0,0 +1,396 @@
|
||||
package dev.caskeleton.adapter.inbound.web.notification.platform.callback;
|
||||
|
||||
import static org.assertj.core.api.Assertions.assertThat;
|
||||
import static org.assertj.core.api.Assertions.assertThatThrownBy;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.api.DeliveryAttemptId;
|
||||
import dev.caskeleton.application.notification.platform.api.ProviderId;
|
||||
import dev.caskeleton.application.notification.platform.api.ProviderProfileId;
|
||||
import dev.caskeleton.application.notification.platform.api.error.CallbackValidationException;
|
||||
import dev.caskeleton.application.notification.platform.api.error.FailureCategory;
|
||||
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureCode;
|
||||
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureDescriptor;
|
||||
import dev.caskeleton.application.notification.platform.callback.AppendEventResult;
|
||||
import dev.caskeleton.application.notification.platform.callback.CallbackLimits;
|
||||
import dev.caskeleton.application.notification.platform.callback.CallbackRequest;
|
||||
import dev.caskeleton.application.notification.platform.callback.CallbackVerificationResult;
|
||||
import dev.caskeleton.application.notification.platform.callback.NormalizedProviderEvent;
|
||||
import dev.caskeleton.application.notification.platform.callback.ProjectionResult;
|
||||
import dev.caskeleton.application.notification.platform.callback.ProviderCallbackAdapter;
|
||||
import dev.caskeleton.application.notification.platform.callback.ProviderCallbackAdapterRegistry;
|
||||
import dev.caskeleton.application.notification.platform.callback.ProviderCallbackIngestionService;
|
||||
import dev.caskeleton.application.notification.platform.callback.ProviderEventLedger;
|
||||
import dev.caskeleton.application.notification.platform.callback.ProviderEventProjectionService;
|
||||
import dev.caskeleton.application.notification.platform.callback.ProviderEventRecord;
|
||||
import dev.caskeleton.application.notification.platform.callback.ProviderEventRecordId;
|
||||
import dev.caskeleton.application.notification.platform.callback.VerifiedCallback;
|
||||
import dev.caskeleton.application.notification.platform.callback.VerifiedProviderEvent;
|
||||
import dev.caskeleton.application.notification.platform.observation.NotificationMetricsPort;
|
||||
import dev.caskeleton.application.notification.platform.observation.NotificationSecurityAuditPort;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.time.Clock;
|
||||
import java.time.Duration;
|
||||
import java.time.Instant;
|
||||
import java.time.ZoneOffset;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.springframework.http.HttpStatus;
|
||||
import org.springframework.mock.web.MockHttpServletRequest;
|
||||
|
||||
/**
|
||||
* What the servlet transport is responsible for handing the callback pipeline.
|
||||
*
|
||||
* <p>The pipeline itself belongs to application-core and is tested there. What is only testable
|
||||
* here is the translation: the exact received octets, the externally-visible URL, and headers that
|
||||
* survive the servlet container's own casing. Each is a common cause of a valid webhook failing
|
||||
* verification, and none is visible from a unit test of the provider adapter.
|
||||
*
|
||||
* <p>The capture point is the provider adapter's {@code verify}, which is the first thing in the
|
||||
* pipeline to see the whole request. It rejects, so the test never needs a ledger.
|
||||
*/
|
||||
class NotificationCallbackMvcControllerTest {
|
||||
|
||||
private static final Clock CLOCK =
|
||||
Clock.fixed(Instant.parse("2026-08-14T00:00:00Z"), ZoneOffset.UTC);
|
||||
private static final String TRUSTED_PROXY = "10.0.0.1";
|
||||
|
||||
private final List<CallbackRequest> verified = new ArrayList<>();
|
||||
private final List<String> rejections = new ArrayList<>();
|
||||
|
||||
private final NotificationCallbackMvcController controller =
|
||||
new NotificationCallbackMvcController(
|
||||
new ProviderCallbackIngestionService(
|
||||
new CapturingRegistry(),
|
||||
new UnusedLedger(),
|
||||
// Never reached: verification always fails in this fixture, and the pipeline appends
|
||||
// only after a valid signature.
|
||||
new ProviderEventProjectionService(
|
||||
new UnusedLedger(),
|
||||
providerId -> java.util.Optional.empty(),
|
||||
new UnusedAttemptResolver(),
|
||||
new UnusedProjectionStore(),
|
||||
(attempt, facts) -> {
|
||||
throw new UnsupportedOperationException();
|
||||
},
|
||||
new UnusedTransactions(),
|
||||
new DiscardingMetrics()),
|
||||
new UnusedPayloadProtection(),
|
||||
new RecordingSecurityAudit(),
|
||||
new DiscardingMetrics(),
|
||||
CLOCK),
|
||||
new CallbackRequestFactory(new ExternalRequestUrlResolver(Set.of(TRUSTED_PROXY)), CLOCK));
|
||||
|
||||
@Test
|
||||
void theExactReceivedOctetsReachTheAdapterUnparsed() {
|
||||
byte[] body =
|
||||
"MessageSid=SM1&MessageStatus=delivered&Signed=a+b%2Fc".getBytes(StandardCharsets.UTF_8);
|
||||
|
||||
assertThatThrownBy(
|
||||
() ->
|
||||
controller.callback(
|
||||
"twilio", "twilio-primary", request("application/x-www-form-urlencoded"), body))
|
||||
.isInstanceOf(CallbackValidationException.class);
|
||||
|
||||
// Byte for byte, including the percent-encoding a form parse would have consumed and re-encoded
|
||||
// differently — which is the single most common cause of a valid webhook failing its signature.
|
||||
assertThat(verified).hasSize(1);
|
||||
assertThat(verified.get(0).body()).isEqualTo(body);
|
||||
assertThat(verified.get(0).contentType()).contains("application/x-www-form-urlencoded");
|
||||
assertThat(verified.get(0).httpMethod()).isEqualTo("POST");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aForwardedHostFromAnUntrustedPeerIsIgnored() {
|
||||
var request = request("application/json");
|
||||
request.setRemoteAddr("203.0.113.9");
|
||||
request.addHeader("X-Forwarded-Proto", "https");
|
||||
request.addHeader("X-Forwarded-Host", "attacker.example.com");
|
||||
|
||||
assertThatThrownBy(
|
||||
() ->
|
||||
controller.callback(
|
||||
"twilio", "twilio-primary", request, "{}".getBytes(StandardCharsets.UTF_8)))
|
||||
.isInstanceOf(CallbackValidationException.class);
|
||||
|
||||
// Honouring the header unconditionally would let any caller choose the URL that gets verified,
|
||||
// which defeats the signature entirely.
|
||||
assertThat(verified.get(0).externalUrl()).doesNotContain("attacker.example.com");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aForwardedHostFromATrustedProxyBecomesTheCanonicalUrl() {
|
||||
var request = request("application/json");
|
||||
request.setRemoteAddr(TRUSTED_PROXY);
|
||||
request.addHeader("X-Forwarded-Proto", "https");
|
||||
request.addHeader("X-Forwarded-Host", "callback.example.com");
|
||||
|
||||
assertThatThrownBy(
|
||||
() ->
|
||||
controller.callback(
|
||||
"twilio", "twilio-primary", request, "{}".getBytes(StandardCharsets.UTF_8)))
|
||||
.isInstanceOf(CallbackValidationException.class);
|
||||
|
||||
assertThat(verified.get(0).externalUrl())
|
||||
.isEqualTo(
|
||||
"https://callback.example.com/internal/notification/callbacks/twilio/twilio-primary");
|
||||
}
|
||||
|
||||
@Test
|
||||
void headersSurviveTheContainersCasingAndStayAddressableEitherWay() {
|
||||
var request = request("application/json");
|
||||
request.addHeader("X-Twilio-Signature", "abc123");
|
||||
|
||||
assertThatThrownBy(
|
||||
() ->
|
||||
controller.callback(
|
||||
"twilio", "twilio-primary", request, "{}".getBytes(StandardCharsets.UTF_8)))
|
||||
.isInstanceOf(CallbackValidationException.class);
|
||||
|
||||
assertThat(verified.get(0).header("x-twilio-signature")).contains("abc123");
|
||||
assertThat(verified.get(0).header("X-TWILIO-SIGNATURE")).contains("abc123");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBodyOverTheTransportCeilingIsRefusedBeforeAnyAdapterIsConsulted() {
|
||||
byte[] oversized = new byte[NotificationCallbackMvcController.MAX_BODY_BYTES + 1];
|
||||
|
||||
var response =
|
||||
controller.callback("twilio", "twilio-primary", request("application/json"), oversized);
|
||||
|
||||
assertThat(response.getStatusCode()).isEqualTo(HttpStatus.CONTENT_TOO_LARGE);
|
||||
// Nothing downstream sees it, so no signature check ever runs over an attacker-sized payload.
|
||||
assertThat(verified).isEmpty();
|
||||
}
|
||||
|
||||
@Test
|
||||
void aRejectedCallbackRevealsNothingBeyondTheStatusCode() {
|
||||
var response =
|
||||
controller.onValidationFailure(
|
||||
new CallbackValidationException(
|
||||
NotificationFailureDescriptor.preDispatch(
|
||||
NotificationFailureCode.CALLBACK_SIGNATURE_INVALID,
|
||||
FailureCategory.CALLBACK_VALIDATION_FAILURE)));
|
||||
|
||||
// The endpoint is unauthenticated by design — the signature is the authentication — so an error
|
||||
// body is a free oracle for whoever is probing it.
|
||||
assertThat(response.getStatusCode()).isEqualTo(HttpStatus.BAD_REQUEST);
|
||||
assertThat(response.getBody()).isNull();
|
||||
}
|
||||
|
||||
@Test
|
||||
void aRejectedSignatureIsRecordedAsASecurityEventRatherThanADeliveryEvent() {
|
||||
assertThatThrownBy(
|
||||
() ->
|
||||
controller.callback(
|
||||
"twilio",
|
||||
"twilio-primary",
|
||||
request("application/json"),
|
||||
"{}".getBytes(StandardCharsets.UTF_8)))
|
||||
.isInstanceOf(CallbackValidationException.class);
|
||||
|
||||
// Writing it to the ledger would let anyone who can reach the endpoint fill a recipient's
|
||||
// delivery history with noise.
|
||||
assertThat(rejections).containsExactly("SIGNATURE_MISMATCH");
|
||||
}
|
||||
|
||||
private static MockHttpServletRequest request(String contentType) {
|
||||
var request =
|
||||
new MockHttpServletRequest(
|
||||
"POST", "/internal/notification/callbacks/twilio/twilio-primary");
|
||||
request.setContentType(contentType);
|
||||
return request;
|
||||
}
|
||||
|
||||
/** Registry whose adapter records the request and then refuses it. */
|
||||
private final class CapturingRegistry implements ProviderCallbackAdapterRegistry {
|
||||
|
||||
@Override
|
||||
public ProviderCallbackAdapter require(ProviderProfileId profileId) {
|
||||
return new ProviderCallbackAdapter() {
|
||||
|
||||
@Override
|
||||
public ProviderId providerId() {
|
||||
return new ProviderId("twilio");
|
||||
}
|
||||
|
||||
@Override
|
||||
public CallbackVerificationResult verify(CallbackRequest request) {
|
||||
verified.add(request);
|
||||
return CallbackVerificationResult.invalid("SIGNATURE_MISMATCH");
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<NormalizedProviderEvent> normalize(VerifiedCallback callback) {
|
||||
throw new UnsupportedOperationException("verification always fails in this fixture");
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public CallbackLimits limitsFor(ProviderProfileId profileId) {
|
||||
return new CallbackLimits(
|
||||
65_536L, Set.of("application/json", "application/x-www-form-urlencoded"));
|
||||
}
|
||||
}
|
||||
|
||||
/** Security audit that keeps the rejection reason. */
|
||||
private final class RecordingSecurityAudit implements NotificationSecurityAuditPort {
|
||||
|
||||
@Override
|
||||
public void callbackSignatureRejected(ProviderProfileId profileId, String reasonCode) {
|
||||
rejections.add(reasonCode);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void callbackRejectedByLimit(ProviderProfileId profileId, String reasonCode) {
|
||||
rejections.add(reasonCode);
|
||||
}
|
||||
}
|
||||
|
||||
/** Metrics are exercised elsewhere; discarding them keeps this test about the transport. */
|
||||
private static final class DiscardingMetrics implements NotificationMetricsPort {
|
||||
|
||||
@Override
|
||||
public void increment(String metricName, Map<String, String> tags) {
|
||||
// Intentionally empty.
|
||||
}
|
||||
|
||||
@Override
|
||||
public void record(String metricName, Map<String, String> tags, Duration value) {
|
||||
// Intentionally empty.
|
||||
}
|
||||
|
||||
@Override
|
||||
public void gauge(String metricName, Map<String, String> tags, double value) {
|
||||
// Intentionally empty.
|
||||
}
|
||||
}
|
||||
|
||||
/** Never reached: attempt correlation happens only for an accepted callback. */
|
||||
private static final class UnusedAttemptResolver
|
||||
implements dev.caskeleton.application.notification.platform.callback
|
||||
.DeliveryAttemptResolverPort {
|
||||
|
||||
@Override
|
||||
public java.util.Optional<
|
||||
dev.caskeleton.application.notification.platform.callback.DeliveryAttemptSnapshot>
|
||||
byAttemptId(DeliveryAttemptId attemptId) {
|
||||
return java.util.Optional.empty();
|
||||
}
|
||||
|
||||
@Override
|
||||
public java.util.Optional<
|
||||
dev.caskeleton.application.notification.platform.callback.DeliveryAttemptSnapshot>
|
||||
byProviderRequestId(ProviderProfileId profileId, String providerRequestIdHash) {
|
||||
return java.util.Optional.empty();
|
||||
}
|
||||
}
|
||||
|
||||
/** Never reached: projection runs only after a signature has been accepted. */
|
||||
private static final class UnusedProjectionStore
|
||||
implements dev.caskeleton.application.notification.platform.callback
|
||||
.DeliveryProjectionStorePort {
|
||||
|
||||
@Override
|
||||
public dev.caskeleton.application.notification.platform.callback.DeliveryProjection load(
|
||||
DeliveryAttemptId attemptId) {
|
||||
throw new UnsupportedOperationException();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void save(
|
||||
DeliveryAttemptId attemptId,
|
||||
dev.caskeleton.application.notification.platform.callback.DeliveryProjection projection) {
|
||||
throw new UnsupportedOperationException();
|
||||
}
|
||||
}
|
||||
|
||||
/** Never reached: nothing in this fixture gets as far as a transaction. */
|
||||
private static final class UnusedTransactions
|
||||
implements dev.caskeleton.application.transaction.TransactionPort {
|
||||
|
||||
@Override
|
||||
public <T> T inWrite(java.util.function.Supplier<T> action) {
|
||||
throw new UnsupportedOperationException();
|
||||
}
|
||||
|
||||
@Override
|
||||
public <T> T inRootWrite(java.util.function.Supplier<T> action) {
|
||||
throw new UnsupportedOperationException();
|
||||
}
|
||||
|
||||
@Override
|
||||
public <T> T inRead(java.util.function.Supplier<T> action) {
|
||||
throw new UnsupportedOperationException();
|
||||
}
|
||||
|
||||
@Override
|
||||
public <T> T inNew(java.util.function.Supplier<T> action) {
|
||||
throw new UnsupportedOperationException();
|
||||
}
|
||||
}
|
||||
|
||||
/** Never reached: every request in this fixture is rejected before the payload is retained. */
|
||||
private static final class UnusedPayloadProtection
|
||||
implements dev.caskeleton.application.notification.platform.callback
|
||||
.CallbackPayloadProtectionPort {
|
||||
|
||||
@Override
|
||||
public byte[] protectRawPayload(byte[] rawBody) {
|
||||
throw new UnsupportedOperationException();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String digest(byte[] rawBody) {
|
||||
throw new UnsupportedOperationException();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String fingerprint(
|
||||
ProviderProfileId profileId, NormalizedProviderEvent event, String rawPayloadDigest) {
|
||||
throw new UnsupportedOperationException();
|
||||
}
|
||||
}
|
||||
|
||||
/** Never reached: every request in this fixture is rejected before the append. */
|
||||
private static final class UnusedLedger implements ProviderEventLedger {
|
||||
|
||||
@Override
|
||||
public AppendEventResult append(VerifiedProviderEvent event) {
|
||||
throw new UnsupportedOperationException();
|
||||
}
|
||||
|
||||
@Override
|
||||
public AppendEventResult appendAll(List<VerifiedProviderEvent> events) {
|
||||
throw new UnsupportedOperationException();
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<ProviderEventRecord> pendingProjection(int limit) {
|
||||
throw new UnsupportedOperationException();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void markApplied(ProviderEventRecordId eventId, ProjectionResult result) {
|
||||
throw new UnsupportedOperationException();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void markFailed(ProviderEventRecordId eventId, String errorCode) {
|
||||
throw new UnsupportedOperationException();
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<ProviderEventRecord> unmatched(int limit) {
|
||||
throw new UnsupportedOperationException();
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<ProviderEventRecord> eventsForAttempt(DeliveryAttemptId attemptId) {
|
||||
throw new UnsupportedOperationException();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -6,6 +6,34 @@ dependencies {
|
||||
implementation 'org.springframework.boot:spring-boot-autoconfigure'
|
||||
implementation 'org.springframework:spring-web' // Slack webhook client (RestClient)
|
||||
implementation 'org.slf4j:slf4j-api'
|
||||
|
||||
// Notification Delivery Platform.
|
||||
// - mail: the SMTP provider adapter is built on JavaMailSender/MimeMessageHelper, which is where
|
||||
// multipart/alternative, inline resources and header validation already live. Rebuilding MIME
|
||||
// by hand to avoid one dependency would be the more dangerous choice.
|
||||
// - jackson-databind: provider payloads, callback bodies and the canonical variables payload are
|
||||
// JSON. It stays inside this adapter; application-core never sees a JSON type.
|
||||
// - reactor-core: only the optional Reactor facade uses it. The core async type stays
|
||||
// CompletionStage, so nothing else on this classpath depends on Reactor.
|
||||
implementation 'org.springframework.boot:spring-boot-starter-mail'
|
||||
implementation 'org.springframework.boot:spring-boot-starter-json'
|
||||
implementation 'io.projectreactor:reactor-core'
|
||||
// JSON Schema 2020-12 validation of template variables, using the same validator and version the
|
||||
// messaging adapter already depends on rather than a second implementation of the same spec.
|
||||
// The YAML dataformat is excluded: schemas are supplied as JSON strings, so pulling a YAML
|
||||
// parser onto the runtime classpath would add attack surface for a format nothing reads.
|
||||
// Thymeleaf is the reference HTML renderer, added as the engine only — not the Spring
|
||||
// starter, which would drag a view resolver and a servlet integration onto an outbound
|
||||
// adapter that renders strings and never serves a request.
|
||||
implementation 'org.thymeleaf:thymeleaf'
|
||||
|
||||
implementation('com.networknt:json-schema-validator:3.0.2') {
|
||||
exclude group: 'tools.jackson.dataformat', module: 'jackson-dataformat-yaml'
|
||||
exclude group: 'com.fasterxml.jackson.dataformat', module: 'jackson-dataformat-yaml'
|
||||
}
|
||||
|
||||
annotationProcessor 'org.springframework.boot:spring-boot-configuration-processor'
|
||||
|
||||
testImplementation 'io.projectreactor:reactor-test'
|
||||
}
|
||||
tasks.withType(JavaCompile).configureEach { options.encoding = 'UTF-8' }
|
||||
|
||||
@@ -1,23 +1,24 @@
|
||||
# This is a Gradle generated file for dependency locking.
|
||||
# Manual edits can break the build and are not advised.
|
||||
# This file is expected to be part of source control.
|
||||
biz.aQute.bnd:biz.aQute.bnd.annotation:7.1.0=testCompileClasspath
|
||||
ch.qos.logback:logback-classic:1.5.21=testCompileClasspath,testRuntimeClasspath
|
||||
ch.qos.logback:logback-core:1.5.21=testCompileClasspath,testRuntimeClasspath
|
||||
com.fasterxml.jackson.core:jackson-annotations:2.20=testCompileClasspath,testRuntimeClasspath
|
||||
biz.aQute.bnd:biz.aQute.bnd.annotation:7.1.0=compileClasspath,testCompileClasspath
|
||||
ch.qos.logback:logback-classic:1.5.21=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
ch.qos.logback:logback-core:1.5.21=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
com.ethlo.time:itu:1.14.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
com.fasterxml.jackson.core:jackson-annotations:2.20=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
com.github.ben-manes.caffeine:caffeine:3.2.3=annotationProcessor,testAnnotationProcessor
|
||||
com.github.kevinstern:software-and-algorithms:1.0=annotationProcessor,testAnnotationProcessor
|
||||
com.github.spotbugs:spotbugs-annotations:4.10.2=spotbugs
|
||||
com.github.spotbugs:spotbugs-annotations:4.8.6=testCompileClasspath
|
||||
com.github.spotbugs:spotbugs-annotations:4.8.6=compileClasspath,testCompileClasspath
|
||||
com.github.spotbugs:spotbugs:4.10.2=spotbugs
|
||||
com.github.stephenc.jcip:jcip-annotations:1.0-1=spotbugs
|
||||
com.google.auto.service:auto-service-annotations:1.0.1=annotationProcessor,testAnnotationProcessor
|
||||
com.google.auto.value:auto-value-annotations:1.9=annotationProcessor,testAnnotationProcessor
|
||||
com.google.auto:auto-common:1.2.2=annotationProcessor,testAnnotationProcessor
|
||||
com.google.code.findbugs:jsr305:3.0.2=checkstyle,spotbugs,testCompileClasspath
|
||||
com.google.code.findbugs:jsr305:3.0.2=checkstyle,compileClasspath,spotbugs,testCompileClasspath
|
||||
com.google.code.gson:gson:2.13.2=spotbugs
|
||||
com.google.errorprone:error_prone_annotation:2.49.0=annotationProcessor,testAnnotationProcessor
|
||||
com.google.errorprone:error_prone_annotations:2.38.0=testCompileClasspath
|
||||
com.google.errorprone:error_prone_annotations:2.38.0=compileClasspath,testCompileClasspath
|
||||
com.google.errorprone:error_prone_annotations:2.41.0=spotbugs
|
||||
com.google.errorprone:error_prone_annotations:2.47.0=checkstyle
|
||||
com.google.errorprone:error_prone_annotations:2.49.0=annotationProcessor,testAnnotationProcessor
|
||||
@@ -32,6 +33,7 @@ com.google.j2objc:j2objc-annotations:3.1=annotationProcessor,checkstyle,testAnno
|
||||
com.google.protobuf:protobuf-java:4.33.2=annotationProcessor,testAnnotationProcessor
|
||||
com.h3xstream.findsecbugs:findsecbugs-plugin:1.14.0=spotbugsPlugins
|
||||
com.jayway.jsonpath:json-path:2.9.0=testCompileClasspath,testRuntimeClasspath
|
||||
com.networknt:json-schema-validator:3.0.2=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
com.puppycrawl.tools:checkstyle:13.5.0=checkstyle
|
||||
com.vaadin.external.google:android-json:0.0.20131108.vaadin1=testCompileClasspath,testRuntimeClasspath
|
||||
commons-beanutils:commons-beanutils:1.11.0=checkstyle
|
||||
@@ -43,8 +45,11 @@ io.github.eisop:dataflow-errorprone:3.41.0-eisop1=annotationProcessor,testAnnota
|
||||
io.github.java-diff-utils:java-diff-utils:4.12=annotationProcessor,testAnnotationProcessor
|
||||
io.micrometer:micrometer-commons:1.16.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
io.micrometer:micrometer-observation:1.16.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
jakarta.activation:jakarta.activation-api:2.1.4=testCompileClasspath,testRuntimeClasspath
|
||||
jakarta.annotation:jakarta.annotation-api:3.0.0=testCompileClasspath,testRuntimeClasspath
|
||||
io.projectreactor:reactor-core:3.8.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
io.projectreactor:reactor-test:3.8.0=testCompileClasspath,testRuntimeClasspath
|
||||
jakarta.activation:jakarta.activation-api:2.1.4=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
jakarta.annotation:jakarta.annotation-api:3.0.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
jakarta.mail:jakarta.mail-api:2.1.5=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
jakarta.xml.bind:jakarta.xml.bind-api:4.0.4=testCompileClasspath,testRuntimeClasspath
|
||||
javax.inject:javax.inject:1=annotationProcessor,testAnnotationProcessor
|
||||
jaxen:jaxen:2.0.0=spotbugs
|
||||
@@ -53,6 +58,7 @@ net.bytebuddy:byte-buddy:1.17.8=testCompileClasspath,testRuntimeClasspath
|
||||
net.minidev:accessors-smart:2.6.0=testCompileClasspath,testRuntimeClasspath
|
||||
net.minidev:json-smart:2.6.0=testCompileClasspath,testRuntimeClasspath
|
||||
net.sf.saxon:Saxon-HE:12.9=checkstyle,spotbugs
|
||||
ognl:ognl:3.3.4=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.antlr:antlr4-runtime:4.13.2=checkstyle
|
||||
org.apache.bcel:bcel:6.12.0=spotbugs
|
||||
org.apache.commons:commons-lang3:3.20.0=checkstyle,spotbugs
|
||||
@@ -60,9 +66,9 @@ org.apache.commons:commons-text:1.15.0=spotbugs
|
||||
org.apache.commons:commons-text:1.3=checkstyle
|
||||
org.apache.httpcomponents:httpclient:4.5.13=checkstyle
|
||||
org.apache.httpcomponents:httpcore:4.4.16=checkstyle
|
||||
org.apache.logging.log4j:log4j-api:2.25.2=spotbugs,testCompileClasspath,testRuntimeClasspath
|
||||
org.apache.logging.log4j:log4j-api:2.25.2=compileClasspath,runtimeClasspath,spotbugs,testCompileClasspath,testRuntimeClasspath
|
||||
org.apache.logging.log4j:log4j-core:2.25.2=spotbugs
|
||||
org.apache.logging.log4j:log4j-to-slf4j:2.25.2=testCompileClasspath,testRuntimeClasspath
|
||||
org.apache.logging.log4j:log4j-to-slf4j:2.25.2=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.apache.maven.doxia:doxia-core:1.12.0=checkstyle
|
||||
org.apache.maven.doxia:doxia-logging-api:1.12.0=checkstyle
|
||||
org.apache.maven.doxia:doxia-module-xdoc:1.12.0=checkstyle
|
||||
@@ -73,14 +79,18 @@ org.apache.tomcat.embed:tomcat-embed-websocket:11.0.14=testCompileClasspath,test
|
||||
org.apache.xbean:xbean-reflect:3.7=checkstyle
|
||||
org.apiguardian:apiguardian-api:1.1.2=testCompileClasspath
|
||||
org.assertj:assertj-core:3.27.6=testCompileClasspath,testRuntimeClasspath
|
||||
org.attoparser:attoparser:2.0.7.RELEASE=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.awaitility:awaitility:4.3.0=testCompileClasspath,testRuntimeClasspath
|
||||
org.codehaus.plexus:plexus-classworlds:2.6.0=checkstyle
|
||||
org.codehaus.plexus:plexus-component-annotations:2.1.0=checkstyle
|
||||
org.codehaus.plexus:plexus-container-default:2.1.0=checkstyle
|
||||
org.codehaus.plexus:plexus-utils:3.3.0=checkstyle
|
||||
org.dom4j:dom4j:2.2.0=spotbugs
|
||||
org.eclipse.angus:angus-activation:2.0.3=runtimeClasspath,testRuntimeClasspath
|
||||
org.eclipse.angus:angus-mail:2.0.5=runtimeClasspath,testRuntimeClasspath
|
||||
org.hamcrest:hamcrest:3.0=testCompileClasspath,testRuntimeClasspath
|
||||
org.javassist:javassist:3.28.0-GA=checkstyle
|
||||
org.javassist:javassist:3.29.0-GA=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.jspecify:jspecify:1.0.0=annotationProcessor,checkstyle,compileClasspath,runtimeClasspath,testAnnotationProcessor,testCompileClasspath,testRuntimeClasspath
|
||||
org.junit.jupiter:junit-jupiter-api:6.0.1=testCompileClasspath,testRuntimeClasspath
|
||||
org.junit.jupiter:junit-jupiter-engine:6.0.1=testRuntimeClasspath
|
||||
@@ -95,10 +105,10 @@ org.mockito:mockito-core:5.20.0=mockitoAgent,testCompileClasspath,testRuntimeCla
|
||||
org.mockito:mockito-junit-jupiter:5.20.0=testCompileClasspath,testRuntimeClasspath
|
||||
org.objenesis:objenesis:3.3=testRuntimeClasspath
|
||||
org.opentest4j:opentest4j:1.3.0=testCompileClasspath,testRuntimeClasspath
|
||||
org.osgi:org.osgi.annotation.bundle:2.0.0=testCompileClasspath
|
||||
org.osgi:org.osgi.annotation.versioning:1.1.2=testCompileClasspath
|
||||
org.osgi:org.osgi.resource:1.0.0=testCompileClasspath
|
||||
org.osgi:org.osgi.service.serviceloader:1.0.0=testCompileClasspath
|
||||
org.osgi:org.osgi.annotation.bundle:2.0.0=compileClasspath,testCompileClasspath
|
||||
org.osgi:org.osgi.annotation.versioning:1.1.2=compileClasspath,testCompileClasspath
|
||||
org.osgi:org.osgi.resource:1.0.0=compileClasspath,testCompileClasspath
|
||||
org.osgi:org.osgi.service.serviceloader:1.0.0=compileClasspath,testCompileClasspath
|
||||
org.ow2.asm:asm-analysis:9.10.1=spotbugs
|
||||
org.ow2.asm:asm-commons:9.10.1=spotbugs
|
||||
org.ow2.asm:asm-tree:9.10.1=spotbugs
|
||||
@@ -106,28 +116,32 @@ org.ow2.asm:asm-util:9.10.1=spotbugs
|
||||
org.ow2.asm:asm:9.10.1=spotbugs
|
||||
org.ow2.asm:asm:9.7.1=testCompileClasspath,testRuntimeClasspath
|
||||
org.pcollections:pcollections:4.0.1=annotationProcessor,testAnnotationProcessor
|
||||
org.reactivestreams:reactive-streams:1.0.4=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.reflections:reflections:0.10.2=checkstyle
|
||||
org.skyscreamer:jsonassert:1.5.3=testCompileClasspath,testRuntimeClasspath
|
||||
org.slf4j:jul-to-slf4j:2.0.17=testCompileClasspath,testRuntimeClasspath
|
||||
org.slf4j:jul-to-slf4j:2.0.17=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.slf4j:slf4j-api:2.0.17=compileClasspath,runtimeClasspath,spotbugs,spotbugsSlf4j,testCompileClasspath,testRuntimeClasspath
|
||||
org.slf4j:slf4j-simple:2.0.17=checkstyle,spotbugsSlf4j
|
||||
org.springframework.boot:spring-boot-autoconfigure:4.0.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-configuration-processor:4.0.0=annotationProcessor
|
||||
org.springframework.boot:spring-boot-http-client:4.0.0=testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-http-converter:4.0.0=testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-jackson:4.0.0=testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-jackson:4.0.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-mail:4.0.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-restclient:4.0.0=testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-resttestclient:4.0.0=testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-servlet:4.0.0=testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-starter-jackson-test:4.0.0=testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-starter-jackson:4.0.0=testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-starter-logging:4.0.0=testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-starter-json:4.0.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-starter-logging:4.0.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-starter-mail:4.0.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-starter-test:4.0.0=testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-starter-tomcat-runtime:4.0.0=testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-starter-tomcat:4.0.0=testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-starter-webmvc-test:4.0.0=testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-starter-webmvc:4.0.0=testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-starter:4.0.0=testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-starter:4.0.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-test-autoconfigure:4.0.0=testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-test:4.0.0=testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework.boot:spring-boot-tomcat:4.0.0=testCompileClasspath,testRuntimeClasspath
|
||||
@@ -137,16 +151,19 @@ org.springframework.boot:spring-boot-webmvc:4.0.0=testCompileClasspath,testRunti
|
||||
org.springframework.boot:spring-boot:4.0.0=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework:spring-aop:7.0.1=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework:spring-beans:7.0.1=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework:spring-context-support:7.0.1=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework:spring-context:7.0.1=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework:spring-core:7.0.1=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework:spring-expression:7.0.1=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework:spring-test:7.0.1=testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework:spring-web:7.0.1=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.springframework:spring-webmvc:7.0.1=testCompileClasspath,testRuntimeClasspath
|
||||
org.thymeleaf:thymeleaf:3.1.3.RELEASE=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.unbescape:unbescape:1.1.6.RELEASE=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
org.xmlresolver:xmlresolver:5.3.3=checkstyle,spotbugs
|
||||
org.xmlunit:xmlunit-core:2.10.4=testCompileClasspath,testRuntimeClasspath
|
||||
org.yaml:snakeyaml:2.5=testCompileClasspath,testRuntimeClasspath
|
||||
tools.jackson.core:jackson-core:3.0.2=testCompileClasspath,testRuntimeClasspath
|
||||
tools.jackson.core:jackson-databind:3.0.2=testCompileClasspath,testRuntimeClasspath
|
||||
tools.jackson:jackson-bom:3.0.2=testCompileClasspath,testRuntimeClasspath
|
||||
org.yaml:snakeyaml:2.5=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
tools.jackson.core:jackson-core:3.0.2=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
tools.jackson.core:jackson-databind:3.0.2=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
tools.jackson:jackson-bom:3.0.2=compileClasspath,runtimeClasspath,testCompileClasspath,testRuntimeClasspath
|
||||
empty=
|
||||
|
||||
+41
@@ -0,0 +1,41 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.admin;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.admin.AdminAccessDeniedException;
|
||||
import dev.caskeleton.application.notification.platform.admin.AdminActor;
|
||||
import dev.caskeleton.application.notification.platform.admin.NotificationAdminAuthority;
|
||||
import dev.caskeleton.application.notification.platform.api.TenantId;
|
||||
import java.util.Objects;
|
||||
import java.util.Optional;
|
||||
|
||||
/**
|
||||
* Operator authority check.
|
||||
*
|
||||
* <p>Application authority never grants an operator authority. The two planes are separated so that
|
||||
* a compromised application credential cannot redrive a message or lift a suppression — the actions
|
||||
* whose whole purpose is to override the platform's own safety decisions.
|
||||
*/
|
||||
public final class AdminAuthorizationGuard {
|
||||
|
||||
/** Require an authority, or refuse. */
|
||||
public void require(AdminActor actor, NotificationAdminAuthority authority) {
|
||||
Objects.requireNonNull(actor, "actor");
|
||||
Objects.requireNonNull(authority, "authority");
|
||||
if (!actor.holds(authority)) {
|
||||
throw new AdminAccessDeniedException(authority);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Require that the actor may act on a tenant.
|
||||
*
|
||||
* <p>An actor with no tenant is a global operator; one bound to a tenant may only act inside it.
|
||||
*/
|
||||
public void requireTenant(AdminActor actor, TenantId tenantId) {
|
||||
Objects.requireNonNull(actor, "actor");
|
||||
Objects.requireNonNull(tenantId, "tenantId");
|
||||
Optional<TenantId> scope = actor.tenantId();
|
||||
if (scope.isPresent() && !scope.get().equals(tenantId)) {
|
||||
throw new AdminAccessDeniedException(NotificationAdminAuthority.SUPPRESS);
|
||||
}
|
||||
}
|
||||
}
|
||||
+29
@@ -0,0 +1,29 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.admin;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.admin.DuplicateRiskApprovalRequiredException;
|
||||
import dev.caskeleton.application.notification.platform.api.delivery.AttemptConfirmation;
|
||||
import dev.caskeleton.application.notification.platform.callback.DeliveryAttemptSnapshot;
|
||||
import java.util.Objects;
|
||||
|
||||
/**
|
||||
* Blocks an unapproved redrive of an ambiguous attempt.
|
||||
*
|
||||
* <p>The platform cannot tell whether the first submission reached the user, so re-sending is a
|
||||
* decision with a real cost that only a human can accept. Requiring the approval flag makes that
|
||||
* acceptance an explicit, audited act rather than a default.
|
||||
*/
|
||||
public final class DuplicateRiskGuard {
|
||||
|
||||
/** Verify the operator accepted the duplicate risk when one exists. */
|
||||
public void verify(DeliveryAttemptSnapshot attempt, boolean approved) {
|
||||
Objects.requireNonNull(attempt, "attempt");
|
||||
boolean risky =
|
||||
attempt.confirmation() == AttemptConfirmation.AMBIGUOUS
|
||||
|| attempt.submissionOutcome()
|
||||
== dev.caskeleton.application.notification.platform.api.delivery.SubmissionOutcome
|
||||
.CONFIRMED_ACCEPTED;
|
||||
if (risky && !approved) {
|
||||
throw new DuplicateRiskApprovalRequiredException();
|
||||
}
|
||||
}
|
||||
}
|
||||
+325
@@ -0,0 +1,325 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.admin;
|
||||
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.dispatch.ProviderRuntimeRegistry;
|
||||
import dev.caskeleton.application.notification.platform.admin.AdminActor;
|
||||
import dev.caskeleton.application.notification.platform.admin.AdminOperationResult;
|
||||
import dev.caskeleton.application.notification.platform.admin.AdminOperationStorePort;
|
||||
import dev.caskeleton.application.notification.platform.admin.NotificationAdminAuthority;
|
||||
import dev.caskeleton.application.notification.platform.admin.NotificationAdminService;
|
||||
import dev.caskeleton.application.notification.platform.admin.ReconcileCommand;
|
||||
import dev.caskeleton.application.notification.platform.admin.RedriveCommand;
|
||||
import dev.caskeleton.application.notification.platform.admin.SetProviderStateCommand;
|
||||
import dev.caskeleton.application.notification.platform.admin.SuppressCommand;
|
||||
import dev.caskeleton.application.notification.platform.api.DeliveryAttemptId;
|
||||
import dev.caskeleton.application.notification.platform.api.delivery.RecipientDeliveryState;
|
||||
import dev.caskeleton.application.notification.platform.callback.DeliveryAttemptSnapshot;
|
||||
import dev.caskeleton.application.notification.platform.dispatch.DeliveryAttemptStorePort;
|
||||
import dev.caskeleton.application.notification.platform.dispatch.RecipientDeliveryStorePort;
|
||||
import dev.caskeleton.application.notification.platform.dispatch.ReconciliationService;
|
||||
import dev.caskeleton.application.notification.platform.observation.NotificationAuditEvent;
|
||||
import dev.caskeleton.application.notification.platform.observation.NotificationAuditPort;
|
||||
import dev.caskeleton.application.notification.platform.policy.SuppressionEntry;
|
||||
import dev.caskeleton.application.notification.platform.policy.SuppressionId;
|
||||
import dev.caskeleton.application.notification.platform.policy.SuppressionSource;
|
||||
import dev.caskeleton.application.notification.platform.policy.SuppressionStorePort;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderRuntimeState;
|
||||
import dev.caskeleton.application.transaction.TransactionPort;
|
||||
import java.time.Clock;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Optional;
|
||||
import java.util.UUID;
|
||||
|
||||
/**
|
||||
* N4 operator plane.
|
||||
*
|
||||
* <p>Four properties hold for every operation: a separate authority, an idempotent operation id, a
|
||||
* recorded reason, and an audit row. The idempotency matters more than it looks — an operator
|
||||
* retrying a redrive after a timeout must not send the message twice, which is exactly the failure
|
||||
* the operation is trying to repair.
|
||||
*
|
||||
* <p>A dry run reads and reports but writes nothing, so an operator can see the blast radius of a
|
||||
* bulk action before committing to it.
|
||||
*/
|
||||
public final class NotificationAdminServiceImpl implements NotificationAdminService {
|
||||
|
||||
private final AdminAuthorizationGuard authorization;
|
||||
private final DuplicateRiskGuard duplicateRiskGuard;
|
||||
private final DeliveryAttemptStorePort attempts;
|
||||
private final RecipientDeliveryStorePort recipients;
|
||||
private final ReconciliationService reconciliation;
|
||||
private final SuppressionStorePort suppressions;
|
||||
private final ProviderRuntimeRegistry runtimes;
|
||||
private final AdminOperationStorePort operations;
|
||||
private final NotificationAuditPort audit;
|
||||
private final TransactionPort transactions;
|
||||
private final Clock clock;
|
||||
|
||||
public NotificationAdminServiceImpl(
|
||||
AdminAuthorizationGuard authorization,
|
||||
DuplicateRiskGuard duplicateRiskGuard,
|
||||
DeliveryAttemptStorePort attempts,
|
||||
RecipientDeliveryStorePort recipients,
|
||||
ReconciliationService reconciliation,
|
||||
SuppressionStorePort suppressions,
|
||||
ProviderRuntimeRegistry runtimes,
|
||||
AdminOperationStorePort operations,
|
||||
NotificationAuditPort audit,
|
||||
TransactionPort transactions,
|
||||
Clock clock) {
|
||||
this.authorization = Objects.requireNonNull(authorization, "authorization");
|
||||
this.duplicateRiskGuard = Objects.requireNonNull(duplicateRiskGuard, "duplicateRiskGuard");
|
||||
this.attempts = Objects.requireNonNull(attempts, "attempts");
|
||||
this.recipients = Objects.requireNonNull(recipients, "recipients");
|
||||
this.reconciliation = Objects.requireNonNull(reconciliation, "reconciliation");
|
||||
this.suppressions = Objects.requireNonNull(suppressions, "suppressions");
|
||||
this.runtimes = Objects.requireNonNull(runtimes, "runtimes");
|
||||
this.operations = Objects.requireNonNull(operations, "operations");
|
||||
this.audit = Objects.requireNonNull(audit, "audit");
|
||||
this.transactions = Objects.requireNonNull(transactions, "transactions");
|
||||
this.clock = Objects.requireNonNull(clock, "clock");
|
||||
}
|
||||
|
||||
@Override
|
||||
public AdminOperationResult redrive(RedriveCommand command, AdminActor actor) {
|
||||
Objects.requireNonNull(command, "command");
|
||||
authorization.require(actor, NotificationAdminAuthority.REDRIVE);
|
||||
|
||||
Optional<AdminOperationResult> replayed = operations.findByOperationId(command.operationId());
|
||||
if (replayed.isPresent()) {
|
||||
return replayed.get();
|
||||
}
|
||||
|
||||
DeliveryAttemptSnapshot original =
|
||||
attempts
|
||||
.snapshot(command.attemptId())
|
||||
.orElseThrow(() -> new IllegalStateException("delivery attempt is not available"));
|
||||
authorization.requireTenant(actor, original.tenantId());
|
||||
duplicateRiskGuard.verify(original, command.approveDuplicateRisk());
|
||||
|
||||
if (command.dryRun()) {
|
||||
return new AdminOperationResult(
|
||||
command.operationId(),
|
||||
true,
|
||||
1,
|
||||
Optional.of(original.notificationId()),
|
||||
Optional.of(original.recipientDeliveryId()),
|
||||
Optional.empty(),
|
||||
List.of("DRY_RUN"));
|
||||
}
|
||||
|
||||
return transactions.inWrite(
|
||||
() -> {
|
||||
// The logical identities are preserved and only the attempt is new, so the history stays
|
||||
// one story rather than becoming two unrelated notifications.
|
||||
recipients.transition(
|
||||
original.recipientDeliveryId(),
|
||||
RecipientDeliveryState.READY_TO_DISPATCH,
|
||||
Optional.of(clock.instant()));
|
||||
|
||||
AdminOperationResult result =
|
||||
new AdminOperationResult(
|
||||
command.operationId(),
|
||||
false,
|
||||
1,
|
||||
Optional.of(original.notificationId()),
|
||||
Optional.of(original.recipientDeliveryId()),
|
||||
Optional.empty(),
|
||||
List.of(command.reason()));
|
||||
audit.record(
|
||||
new NotificationAuditEvent(
|
||||
"ADMIN_REDRIVE",
|
||||
actor.actorRef(),
|
||||
Optional.of(command.reason()),
|
||||
Optional.of(command.operationId()),
|
||||
clock.instant(),
|
||||
Map.of(
|
||||
"provider", original.providerId().value(),
|
||||
"channel", original.channel().name())));
|
||||
return operations.save(result, actor, "ADMIN_REDRIVE");
|
||||
});
|
||||
}
|
||||
|
||||
@Override
|
||||
public AdminOperationResult reconcile(ReconcileCommand command, AdminActor actor) {
|
||||
Objects.requireNonNull(command, "command");
|
||||
authorization.require(actor, NotificationAdminAuthority.RECONCILE);
|
||||
|
||||
Optional<AdminOperationResult> replayed = operations.findByOperationId(command.operationId());
|
||||
if (replayed.isPresent()) {
|
||||
return replayed.get();
|
||||
}
|
||||
if (command.dryRun()) {
|
||||
return new AdminOperationResult(
|
||||
command.operationId(),
|
||||
true,
|
||||
command.attemptIds().size(),
|
||||
Optional.empty(),
|
||||
Optional.empty(),
|
||||
Optional.empty(),
|
||||
List.of("DRY_RUN"));
|
||||
}
|
||||
|
||||
List<String> reasons = new ArrayList<>();
|
||||
int reconciled = 0;
|
||||
for (DeliveryAttemptId attemptId : command.attemptIds()) {
|
||||
reconciliation.reconcile(attemptId);
|
||||
reconciled++;
|
||||
}
|
||||
reasons.add(command.reason());
|
||||
|
||||
AdminOperationResult result =
|
||||
new AdminOperationResult(
|
||||
command.operationId(),
|
||||
false,
|
||||
reconciled,
|
||||
Optional.empty(),
|
||||
Optional.empty(),
|
||||
Optional.empty(),
|
||||
List.copyOf(reasons));
|
||||
audit.record(
|
||||
new NotificationAuditEvent(
|
||||
"ADMIN_RECONCILE",
|
||||
actor.actorRef(),
|
||||
Optional.of(command.reason()),
|
||||
Optional.of(command.operationId()),
|
||||
clock.instant(),
|
||||
Map.of()));
|
||||
return operations.save(result, actor, "ADMIN_RECONCILE");
|
||||
}
|
||||
|
||||
@Override
|
||||
public AdminOperationResult suppress(SuppressCommand command, AdminActor actor) {
|
||||
Objects.requireNonNull(command, "command");
|
||||
authorization.require(actor, NotificationAdminAuthority.SUPPRESS);
|
||||
authorization.requireTenant(actor, command.tenantId());
|
||||
|
||||
Optional<AdminOperationResult> replayed = operations.findByOperationId(command.operationId());
|
||||
if (replayed.isPresent()) {
|
||||
return replayed.get();
|
||||
}
|
||||
if (command.dryRun()) {
|
||||
return new AdminOperationResult(
|
||||
command.operationId(),
|
||||
true,
|
||||
1,
|
||||
Optional.empty(),
|
||||
Optional.empty(),
|
||||
Optional.empty(),
|
||||
List.of("DRY_RUN"));
|
||||
}
|
||||
|
||||
return transactions.inWrite(
|
||||
() -> {
|
||||
int affected;
|
||||
if (command.remove()) {
|
||||
// Removal is by fingerprint match rather than by id, because an operator lifting a
|
||||
// suppression knows the target, not the row identifier the platform assigned.
|
||||
affected =
|
||||
suppressions
|
||||
.activeFor(
|
||||
command.tenantId(), command.targetFingerprint(), clock.instant())
|
||||
.stream()
|
||||
.map(entry -> suppressions.remove(command.tenantId(), entry.id()))
|
||||
.filter(Optional::isPresent)
|
||||
.count()
|
||||
> 0
|
||||
? 1
|
||||
: 0;
|
||||
} else {
|
||||
suppressions.upsert(
|
||||
new SuppressionEntry(
|
||||
new SuppressionId(UUID.randomUUID()),
|
||||
command.tenantId(),
|
||||
command.scope(),
|
||||
command.reason(),
|
||||
command.targetFingerprint(),
|
||||
Optional.empty(),
|
||||
clock.instant(),
|
||||
command.expiresAt(),
|
||||
SuppressionSource.ADMIN));
|
||||
affected = 1;
|
||||
}
|
||||
|
||||
AdminOperationResult result =
|
||||
new AdminOperationResult(
|
||||
command.operationId(),
|
||||
false,
|
||||
affected,
|
||||
Optional.empty(),
|
||||
Optional.empty(),
|
||||
Optional.empty(),
|
||||
List.of(command.reasonText()));
|
||||
audit.record(
|
||||
new NotificationAuditEvent(
|
||||
command.remove() ? "ADMIN_SUPPRESSION_REMOVED" : "ADMIN_SUPPRESSION_ADDED",
|
||||
actor.actorRef(),
|
||||
Optional.of(command.reason().name()),
|
||||
Optional.of(command.operationId()),
|
||||
clock.instant(),
|
||||
Map.of()));
|
||||
return operations.save(
|
||||
result, actor, command.remove() ? "ADMIN_SUPPRESS_REMOVE" : "ADMIN_SUPPRESS_ADD");
|
||||
});
|
||||
}
|
||||
|
||||
@Override
|
||||
public AdminOperationResult setProviderState(SetProviderStateCommand command, AdminActor actor) {
|
||||
Objects.requireNonNull(command, "command");
|
||||
authorization.require(actor, NotificationAdminAuthority.PROVIDER_CONTROL);
|
||||
|
||||
Optional<AdminOperationResult> replayed = operations.findByOperationId(command.operationId());
|
||||
if (replayed.isPresent()) {
|
||||
return replayed.get();
|
||||
}
|
||||
if (command.dryRun()) {
|
||||
return new AdminOperationResult(
|
||||
command.operationId(),
|
||||
true,
|
||||
1,
|
||||
Optional.empty(),
|
||||
Optional.empty(),
|
||||
Optional.empty(),
|
||||
List.of("DRY_RUN"));
|
||||
}
|
||||
|
||||
var runtime = runtimes.current(command.profileId());
|
||||
switch (command.desiredState()) {
|
||||
case DISABLED -> runtime.markDisabled();
|
||||
case DRAINING -> runtime.markDraining();
|
||||
case HEALTHY -> runtime.markHealthy();
|
||||
case DEGRADED -> runtime.markDegraded(command.reason());
|
||||
case THROTTLED -> runtime.markThrottled();
|
||||
case AUTHENTICATION_FAILED -> runtime.markAuthenticationFailed(command.reason());
|
||||
}
|
||||
|
||||
AdminOperationResult result =
|
||||
new AdminOperationResult(
|
||||
command.operationId(),
|
||||
false,
|
||||
1,
|
||||
Optional.empty(),
|
||||
Optional.empty(),
|
||||
Optional.empty(),
|
||||
List.of(command.reason()));
|
||||
audit.record(
|
||||
new NotificationAuditEvent(
|
||||
"ADMIN_PROVIDER_STATE",
|
||||
actor.actorRef(),
|
||||
Optional.of(command.reason()),
|
||||
Optional.of(command.operationId()),
|
||||
clock.instant(),
|
||||
Map.of(
|
||||
"providerProfile", command.profileId().value(),
|
||||
"status", command.desiredState().name())));
|
||||
return operations.save(result, actor, "ADMIN_PROVIDER_STATE");
|
||||
}
|
||||
|
||||
/** Current state of a provider runtime, for the health endpoint. */
|
||||
public ProviderRuntimeState providerState(
|
||||
dev.caskeleton.application.notification.platform.api.ProviderProfileId profileId) {
|
||||
return runtimes.state(profileId);
|
||||
}
|
||||
}
|
||||
+105
@@ -0,0 +1,105 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.autoconfigure;
|
||||
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.provider.http.JdkNotificationHttpGateway;
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.provider.http.NotificationHttpGateway;
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.security.AesGcmContactPointProtector;
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.template.JacksonNotificationVariablesCodec;
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.template.JsonSchemaVariableValidator;
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.template.NotificationTemplateEngine;
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.template.PlaceholderTemplateEngine;
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.template.Sha256MessageDigestAdapter;
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.template.ThymeleafStringTemplateEngine;
|
||||
import dev.caskeleton.application.notification.platform.dispatch.MessageDigestPort;
|
||||
import dev.caskeleton.application.notification.platform.dispatch.NotificationVariablesCodecPort;
|
||||
import dev.caskeleton.application.notification.platform.security.ContactPointProtector;
|
||||
import dev.caskeleton.application.notification.platform.security.SecretMaterialProvider;
|
||||
import dev.caskeleton.application.notification.platform.template.TemplateVariableValidator;
|
||||
import java.time.Duration;
|
||||
import org.springframework.beans.factory.annotation.Value;
|
||||
import org.springframework.boot.autoconfigure.condition.ConditionalOnBean;
|
||||
import org.springframework.boot.autoconfigure.condition.ConditionalOnMissingBean;
|
||||
import org.springframework.boot.autoconfigure.condition.ConditionalOnProperty;
|
||||
import org.springframework.boot.context.properties.EnableConfigurationProperties;
|
||||
import org.springframework.context.annotation.Bean;
|
||||
import org.springframework.context.annotation.Configuration;
|
||||
|
||||
/**
|
||||
* Notification platform wiring.
|
||||
*
|
||||
* <p>Everything is opt-in and conditional. The platform contributes no beans unless it is enabled,
|
||||
* and the contact point protector only appears once a secret provider exists — because a protector
|
||||
* without keys would fail on the first delivery instead of at startup.
|
||||
*/
|
||||
@Configuration(proxyBeanMethods = false)
|
||||
@EnableConfigurationProperties(NotificationPlatformSettings.class)
|
||||
@ConditionalOnProperty(
|
||||
prefix = "ca-skeleton.notification.platform",
|
||||
name = "enabled",
|
||||
havingValue = "true")
|
||||
public class NotificationPlatformAutoConfiguration {
|
||||
|
||||
/** Canonical variables codec. */
|
||||
@Bean
|
||||
@ConditionalOnMissingBean
|
||||
public NotificationVariablesCodecPort notificationVariablesCodec() {
|
||||
return new JacksonNotificationVariablesCodec();
|
||||
}
|
||||
|
||||
/** Request fingerprint hashing. */
|
||||
@Bean
|
||||
@ConditionalOnMissingBean
|
||||
public MessageDigestPort notificationMessageDigest() {
|
||||
return new Sha256MessageDigestAdapter();
|
||||
}
|
||||
|
||||
/** JSON Schema 2020-12 variable validation. */
|
||||
@Bean
|
||||
@ConditionalOnMissingBean
|
||||
public TemplateVariableValidator notificationTemplateVariableValidator() {
|
||||
return new JsonSchemaVariableValidator();
|
||||
}
|
||||
|
||||
/**
|
||||
* Template engine, defaulting to the deterministic placeholder substitution.
|
||||
*
|
||||
* <p>Thymeleaf is the opt-in alternative: it escapes by default, which matters for HTML email
|
||||
* bodies built from application input. The default stays the placeholder engine because it has no
|
||||
* expression evaluator at all, and an unknown engine name fails the boot rather than quietly
|
||||
* falling back — a deployment that thought it had escaping and did not is the worse outcome.
|
||||
*/
|
||||
@Bean
|
||||
@ConditionalOnMissingBean
|
||||
public NotificationTemplateEngine notificationTemplateEngine(
|
||||
@Value("${ca-skeleton.notification.platform.template.engine:placeholder}") String engine) {
|
||||
return switch (engine.toLowerCase(java.util.Locale.ROOT)) {
|
||||
case "placeholder" -> new PlaceholderTemplateEngine();
|
||||
case "thymeleaf" -> new ThymeleafStringTemplateEngine();
|
||||
default ->
|
||||
throw new IllegalArgumentException(
|
||||
"ca-skeleton.notification.platform.template.engine must be"
|
||||
+ " 'placeholder' or 'thymeleaf', not '"
|
||||
+ engine
|
||||
+ "'");
|
||||
};
|
||||
}
|
||||
|
||||
/** Contact point protection, only once key material is available. */
|
||||
@Bean
|
||||
@ConditionalOnBean(SecretMaterialProvider.class)
|
||||
@ConditionalOnMissingBean
|
||||
public ContactPointProtector notificationContactPointProtector(SecretMaterialProvider secrets) {
|
||||
return new AesGcmContactPointProtector(secrets);
|
||||
}
|
||||
|
||||
/**
|
||||
* Default provider transport.
|
||||
*
|
||||
* <p>Replaced in the composition root when the HTTP Client Platform is bound, which is the
|
||||
* supported way to reuse its TLS, circuit-breaker and SSRF policy.
|
||||
*/
|
||||
@Bean
|
||||
@ConditionalOnMissingBean
|
||||
public NotificationHttpGateway notificationHttpGateway() {
|
||||
return new JdkNotificationHttpGateway(Duration.ofSeconds(2));
|
||||
}
|
||||
}
|
||||
+154
@@ -0,0 +1,154 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.autoconfigure;
|
||||
|
||||
import java.time.Duration;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import org.springframework.boot.context.properties.ConfigurationProperties;
|
||||
|
||||
/**
|
||||
* Bound notification platform configuration.
|
||||
*
|
||||
* <p>Validation happens in the constructor, so a misconfiguration fails the boot rather than
|
||||
* surfacing as a delivery incident hours later. Everything is bounded: there is no property whose
|
||||
* value may be "unlimited", because an unbounded queue or payload is a resource failure waiting for
|
||||
* the first burst.
|
||||
*/
|
||||
@ConfigurationProperties("ca-skeleton.notification.platform")
|
||||
public record NotificationPlatformSettings(
|
||||
boolean enabled, Dispatch dispatch, Callbacks callbacks, Map<String, Provider> providers) {
|
||||
|
||||
public NotificationPlatformSettings {
|
||||
dispatch = dispatch == null ? Dispatch.defaults() : dispatch;
|
||||
callbacks = callbacks == null ? Callbacks.defaults() : callbacks;
|
||||
providers = providers == null ? Map.of() : Map.copyOf(providers);
|
||||
providers.forEach((id, provider) -> provider.validate(id));
|
||||
}
|
||||
|
||||
/** Dispatch runtime bounds. */
|
||||
public record Dispatch(
|
||||
int claimBatchSize,
|
||||
Duration leaseDuration,
|
||||
Duration pollInterval,
|
||||
int maxGlobalConcurrency,
|
||||
int maxAdditionalAttempts,
|
||||
Duration maxQueueAge,
|
||||
boolean allowAmbiguousFallback) {
|
||||
|
||||
private static final int MAX_CLAIM_BATCH = 1000;
|
||||
|
||||
public Dispatch {
|
||||
Objects.requireNonNull(leaseDuration, "leaseDuration");
|
||||
Objects.requireNonNull(pollInterval, "pollInterval");
|
||||
Objects.requireNonNull(maxQueueAge, "maxQueueAge");
|
||||
if (claimBatchSize < 1 || claimBatchSize > MAX_CLAIM_BATCH) {
|
||||
throw new IllegalArgumentException(
|
||||
"ca-skeleton.notification.platform.dispatch.claim-batch-size must be 1.."
|
||||
+ MAX_CLAIM_BATCH);
|
||||
}
|
||||
if (maxGlobalConcurrency < 1) {
|
||||
throw new IllegalArgumentException("max-global-concurrency must be positive");
|
||||
}
|
||||
if (maxAdditionalAttempts < 0) {
|
||||
throw new IllegalArgumentException("max-additional-attempts must not be negative");
|
||||
}
|
||||
if (leaseDuration.isNegative() || leaseDuration.isZero()) {
|
||||
throw new IllegalArgumentException("lease-duration must be positive and finite");
|
||||
}
|
||||
if (leaseDuration.compareTo(pollInterval) <= 0) {
|
||||
throw new IllegalArgumentException("lease-duration must exceed poll-interval");
|
||||
}
|
||||
if (allowAmbiguousFallback) {
|
||||
// Refused outright rather than warned about: automatic fallback after an ambiguous
|
||||
// submission is the configuration that turns an unknown into a guaranteed duplicate.
|
||||
throw new IllegalArgumentException(
|
||||
"allow-ambiguous-fallback is not a supported configuration");
|
||||
}
|
||||
}
|
||||
|
||||
/** Conservative defaults. */
|
||||
public static Dispatch defaults() {
|
||||
return new Dispatch(
|
||||
100, Duration.ofSeconds(30), Duration.ofMillis(250), 128, 3, Duration.ofHours(24), false);
|
||||
}
|
||||
}
|
||||
|
||||
/** Callback endpoint bounds. */
|
||||
public record Callbacks(boolean enabled, long maxBodyBytes, Duration replaySkew) {
|
||||
|
||||
private static final long MAX_BODY_CEILING = 1_048_576L;
|
||||
|
||||
public Callbacks {
|
||||
Objects.requireNonNull(replaySkew, "replaySkew");
|
||||
if (maxBodyBytes < 1 || maxBodyBytes > MAX_BODY_CEILING) {
|
||||
throw new IllegalArgumentException("max-body-bytes must be 1.." + MAX_BODY_CEILING);
|
||||
}
|
||||
if (replaySkew.isNegative()) {
|
||||
throw new IllegalArgumentException("replay-skew must not be negative");
|
||||
}
|
||||
}
|
||||
|
||||
/** Conservative defaults. */
|
||||
public static Callbacks defaults() {
|
||||
return new Callbacks(false, 65_536L, Duration.ofMinutes(5));
|
||||
}
|
||||
}
|
||||
|
||||
/** One provider profile. */
|
||||
public record Provider(
|
||||
String type,
|
||||
boolean enabled,
|
||||
String environment,
|
||||
String credentialProfile,
|
||||
String topic,
|
||||
String vapidPublicKey,
|
||||
String callbackSigningSecretRef,
|
||||
Duration timeout,
|
||||
int maxConcurrency,
|
||||
int ratePerSecond) {
|
||||
|
||||
/** Fail the boot when a profile cannot possibly work. */
|
||||
public void validate(String profileId) {
|
||||
Objects.requireNonNull(profileId, "profileId");
|
||||
if (!enabled) {
|
||||
return;
|
||||
}
|
||||
require(type != null && !type.isBlank(), profileId, "type is required");
|
||||
require(environment != null && !environment.isBlank(), profileId, "environment is required");
|
||||
require(
|
||||
credentialProfile != null && !credentialProfile.isBlank(),
|
||||
profileId,
|
||||
"credential-profile is required");
|
||||
require(
|
||||
timeout != null && !timeout.isNegative() && !timeout.isZero(),
|
||||
profileId,
|
||||
"timeout must be positive and finite");
|
||||
require(maxConcurrency >= 1, profileId, "max-concurrency must be positive");
|
||||
require(ratePerSecond >= 1, profileId, "rate-limit-per-second must be positive");
|
||||
|
||||
switch (type == null ? "" : type.toUpperCase(java.util.Locale.ROOT)) {
|
||||
case "APNS" ->
|
||||
require(topic != null && !topic.isBlank(), profileId, "APNs profiles require a topic");
|
||||
case "WEB_PUSH" ->
|
||||
require(
|
||||
vapidPublicKey != null && !vapidPublicKey.isBlank(),
|
||||
profileId,
|
||||
"Web Push profiles require a VAPID key");
|
||||
case "TWILIO", "SES" ->
|
||||
require(
|
||||
callbackSigningSecretRef != null && !callbackSigningSecretRef.isBlank(),
|
||||
profileId,
|
||||
"callback-capable profiles require a callback signing secret reference");
|
||||
default -> {
|
||||
// Providers without extra requirements are already covered by the common checks.
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private static void require(boolean condition, String profileId, String message) {
|
||||
if (!condition) {
|
||||
throw new IllegalArgumentException(
|
||||
"notification provider profile '" + profileId + "': " + message);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
+17
@@ -0,0 +1,17 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
|
||||
|
||||
/**
|
||||
* A held concurrency slot for one provider attempt.
|
||||
*
|
||||
* <p>Closing it is what releases the slot, so every call site uses try-with-resources. The permit
|
||||
* also carries the credential generation the attempt ran under, which is what makes a rotation
|
||||
* auditable after the fact.
|
||||
*/
|
||||
public interface AttemptPermit extends AutoCloseable {
|
||||
|
||||
/** Credential generation this attempt is bound to. */
|
||||
long generation();
|
||||
|
||||
@Override
|
||||
void close();
|
||||
}
|
||||
+52
@@ -0,0 +1,52 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.api.ProviderProfileId;
|
||||
import dev.caskeleton.application.notification.platform.callback.DeliveryAttemptSnapshot;
|
||||
import dev.caskeleton.application.notification.platform.dispatch.ReconciliationGatewayPort;
|
||||
import dev.caskeleton.application.notification.platform.provider.ReconciliationCapability;
|
||||
import dev.caskeleton.application.notification.platform.provider.ReconciliationResult;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Optional;
|
||||
|
||||
/**
|
||||
* Routes a reconciliation to the capability that owns the provider.
|
||||
*
|
||||
* <p>A profile with no registered capability reports {@code Unsupported} rather than falling back
|
||||
* to a guess. Inventing a final status for a provider that cannot be queried is precisely the
|
||||
* behaviour the ambiguity model exists to prevent.
|
||||
*/
|
||||
public final class CapabilityReconciliationGateway implements ReconciliationGatewayPort {
|
||||
|
||||
private final Map<ProviderProfileId, ReconciliationCapability> capabilities;
|
||||
private final ProviderRuntimeRegistry runtimes;
|
||||
|
||||
public CapabilityReconciliationGateway(
|
||||
Map<ProviderProfileId, ReconciliationCapability> capabilities,
|
||||
ProviderRuntimeRegistry runtimes) {
|
||||
this.capabilities = Map.copyOf(Objects.requireNonNull(capabilities, "capabilities"));
|
||||
this.runtimes = Objects.requireNonNull(runtimes, "runtimes");
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean supports(ProviderProfileId profileId) {
|
||||
Objects.requireNonNull(profileId, "profileId");
|
||||
return Optional.ofNullable(capabilities.get(profileId))
|
||||
.map(capability -> capability.supports(runtimes.current(profileId).profile()))
|
||||
.orElse(false);
|
||||
}
|
||||
|
||||
@Override
|
||||
public ReconciliationResult reconcile(DeliveryAttemptSnapshot attempt) {
|
||||
Objects.requireNonNull(attempt, "attempt");
|
||||
ReconciliationCapability capability = capabilities.get(attempt.providerProfileId());
|
||||
if (capability == null) {
|
||||
return new ReconciliationResult.Unsupported();
|
||||
}
|
||||
// The permit is taken so a reconciliation backlog cannot become a second load source during the
|
||||
// incident that produced it.
|
||||
try (AttemptPermit permit = runtimes.current(attempt.providerProfileId()).acquireAttempt()) {
|
||||
return capability.reconcile(attempt).toCompletableFuture().join();
|
||||
}
|
||||
}
|
||||
}
|
||||
+72
@@ -0,0 +1,72 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.api.ProviderProfileId;
|
||||
import dev.caskeleton.application.notification.platform.api.RecipientSpec;
|
||||
import dev.caskeleton.application.notification.platform.api.TenantId;
|
||||
import dev.caskeleton.application.notification.platform.api.routing.Channel;
|
||||
import dev.caskeleton.application.notification.platform.api.routing.DeliveryStrategy;
|
||||
import dev.caskeleton.application.notification.platform.api.routing.ExplicitChannel;
|
||||
import dev.caskeleton.application.notification.platform.api.routing.OrderedFallback;
|
||||
import dev.caskeleton.application.notification.platform.dispatch.NotificationRoutePlannerPort;
|
||||
import dev.caskeleton.application.notification.platform.policy.RouteCandidate;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
|
||||
/**
|
||||
* Turns a strategy into an ordered route plan using the configured channel-to-profile map.
|
||||
*
|
||||
* <p>A channel with no configured provider, or a recipient with no contact point for it, simply
|
||||
* produces no candidate. The routing engine then reports {@code NO_ELIGIBLE_ROUTE} rather than the
|
||||
* dispatcher failing on a null, which is the difference between a diagnosable state and a stack
|
||||
* trace.
|
||||
*/
|
||||
public final class ConfiguredRoutePlanner implements NotificationRoutePlannerPort {
|
||||
|
||||
private final Map<Channel, ProviderProfileId> profilesByChannel;
|
||||
|
||||
public ConfiguredRoutePlanner(Map<Channel, ProviderProfileId> profilesByChannel) {
|
||||
this.profilesByChannel =
|
||||
Map.copyOf(Objects.requireNonNull(profilesByChannel, "profilesByChannel"));
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<RouteCandidate> plan(
|
||||
TenantId tenantId, RecipientSpec recipient, DeliveryStrategy strategy) {
|
||||
Objects.requireNonNull(tenantId, "tenantId");
|
||||
Objects.requireNonNull(recipient, "recipient");
|
||||
Objects.requireNonNull(strategy, "strategy");
|
||||
|
||||
List<Channel> ordered =
|
||||
switch (strategy) {
|
||||
case ExplicitChannel explicit -> List.of(explicit.channel());
|
||||
case OrderedFallback fallback -> fallback.channels();
|
||||
};
|
||||
|
||||
List<RouteCandidate> routes = new ArrayList<>(ordered.size());
|
||||
int index = 0;
|
||||
for (Channel channel : ordered) {
|
||||
ProviderProfileId profileId = profilesByChannel.get(channel);
|
||||
if (profileId == null) {
|
||||
continue;
|
||||
}
|
||||
var selector =
|
||||
recipient.contactPoints().stream()
|
||||
.filter(candidate -> candidate.channel() == channel)
|
||||
.findFirst();
|
||||
if (selector.isEmpty()) {
|
||||
continue;
|
||||
}
|
||||
boolean blocked =
|
||||
recipient
|
||||
.channelOverride()
|
||||
.map(override -> override.blockedChannels().contains(channel))
|
||||
.orElse(false);
|
||||
routes.add(
|
||||
new RouteCandidate(
|
||||
index++, channel, selector.get().contactPointId(), profileId, !blocked, true));
|
||||
}
|
||||
return List.copyOf(routes);
|
||||
}
|
||||
}
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
|
||||
|
||||
/** Verifies a candidate generation before it becomes the current one. */
|
||||
@FunctionalInterface
|
||||
public interface CredentialProbe {
|
||||
|
||||
/** Return false when the candidate credential is not usable. */
|
||||
boolean isUsable(ProviderRuntime candidate);
|
||||
}
|
||||
+19
@@ -0,0 +1,19 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.api.error.FailureCategory;
|
||||
import dev.caskeleton.application.notification.platform.api.error.NotificationException;
|
||||
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureCode;
|
||||
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureDescriptor;
|
||||
|
||||
/** Raised when a candidate credential generation fails its probe before any cutover. */
|
||||
public class CredentialValidationException extends NotificationException {
|
||||
|
||||
private static final long serialVersionUID = 1L;
|
||||
|
||||
public CredentialValidationException() {
|
||||
super(
|
||||
NotificationFailureDescriptor.preDispatch(
|
||||
NotificationFailureCode.PROVIDER_CONFIGURATION_INVALID,
|
||||
FailureCategory.AUTHENTICATION));
|
||||
}
|
||||
}
|
||||
+61
@@ -0,0 +1,61 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
|
||||
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.template.NotificationJsonMapper;
|
||||
import dev.caskeleton.application.notification.platform.api.ContactPointId;
|
||||
import dev.caskeleton.application.notification.platform.api.ProviderProfileId;
|
||||
import dev.caskeleton.application.notification.platform.api.routing.Channel;
|
||||
import dev.caskeleton.application.notification.platform.dispatch.NotificationRoutingPlanCodecPort;
|
||||
import dev.caskeleton.application.notification.platform.policy.RouteCandidate;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.UUID;
|
||||
import tools.jackson.core.type.TypeReference;
|
||||
|
||||
/**
|
||||
* Route plan encoding.
|
||||
*
|
||||
* <p>The plan is frozen at submit time, so this is a snapshot format rather than a view: it stores
|
||||
* exactly what was decided, including which routes were usable then, and never recomputes.
|
||||
*/
|
||||
public final class JacksonRoutingPlanCodec implements NotificationRoutingPlanCodecPort {
|
||||
|
||||
@Override
|
||||
public String encode(List<RouteCandidate> routes) {
|
||||
Objects.requireNonNull(routes, "routes");
|
||||
List<Map<String, Object>> encoded = new ArrayList<>(routes.size());
|
||||
for (RouteCandidate route : routes) {
|
||||
Map<String, Object> entry = new LinkedHashMap<>();
|
||||
entry.put("routeIndex", route.routeIndex());
|
||||
entry.put("channel", route.channel().name());
|
||||
entry.put("contactPointId", route.contactPointId().value().toString());
|
||||
entry.put("providerProfileId", route.providerProfileId().value());
|
||||
entry.put("contactPointActive", route.contactPointActive());
|
||||
entry.put("providerEnabled", route.providerEnabled());
|
||||
encoded.add(entry);
|
||||
}
|
||||
return NotificationJsonMapper.mapper().writeValueAsString(encoded);
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<RouteCandidate> decode(String payload) {
|
||||
Objects.requireNonNull(payload, "payload");
|
||||
List<LinkedHashMap<String, Object>> raw =
|
||||
NotificationJsonMapper.mapper()
|
||||
.readValue(payload, new TypeReference<ArrayList<LinkedHashMap<String, Object>>>() {});
|
||||
List<RouteCandidate> routes = new ArrayList<>(raw.size());
|
||||
for (Map<String, Object> entry : raw) {
|
||||
routes.add(
|
||||
new RouteCandidate(
|
||||
((Number) entry.get("routeIndex")).intValue(),
|
||||
Channel.valueOf(String.valueOf(entry.get("channel"))),
|
||||
new ContactPointId(UUID.fromString(String.valueOf(entry.get("contactPointId")))),
|
||||
new ProviderProfileId(String.valueOf(entry.get("providerProfileId"))),
|
||||
Boolean.TRUE.equals(entry.get("contactPointActive")),
|
||||
Boolean.TRUE.equals(entry.get("providerEnabled"))));
|
||||
}
|
||||
return List.copyOf(routes);
|
||||
}
|
||||
}
|
||||
+61
@@ -0,0 +1,61 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.api.DeliveryAttemptId;
|
||||
import dev.caskeleton.application.notification.platform.dispatch.DeliveryAttemptStorePort;
|
||||
import dev.caskeleton.application.notification.platform.dispatch.RecipientLeaseStorePort;
|
||||
import dev.caskeleton.application.notification.platform.dispatch.ReconciliationService;
|
||||
import java.time.Duration;
|
||||
import java.util.List;
|
||||
import java.util.Objects;
|
||||
|
||||
/**
|
||||
* Recovers deliveries a dead worker left in flight.
|
||||
*
|
||||
* <p>An expired lease on a {@code DISPATCHING} delivery is the crash case: the attempt row exists,
|
||||
* so a provider call may have happened. Recovery therefore reconciles rather than re-dispatching —
|
||||
* re-dispatching would be the platform choosing to duplicate rather than to ask.
|
||||
*/
|
||||
public final class LeaseRecoveryService {
|
||||
|
||||
private final RecipientLeaseStorePort leases;
|
||||
private final DeliveryAttemptStorePort attempts;
|
||||
private final ReconciliationService reconciliation;
|
||||
private final Duration staleAfter;
|
||||
private final int batchSize;
|
||||
|
||||
public LeaseRecoveryService(
|
||||
RecipientLeaseStorePort leases,
|
||||
DeliveryAttemptStorePort attempts,
|
||||
ReconciliationService reconciliation,
|
||||
Duration staleAfter,
|
||||
int batchSize) {
|
||||
this.leases = Objects.requireNonNull(leases, "leases");
|
||||
this.attempts = Objects.requireNonNull(attempts, "attempts");
|
||||
this.reconciliation = Objects.requireNonNull(reconciliation, "reconciliation");
|
||||
this.staleAfter = Objects.requireNonNull(staleAfter, "staleAfter");
|
||||
this.batchSize = batchSize;
|
||||
if (batchSize < 1) {
|
||||
throw new IllegalArgumentException("batchSize");
|
||||
}
|
||||
if (staleAfter.isNegative() || staleAfter.isZero()) {
|
||||
throw new IllegalArgumentException("staleAfter must be positive and finite");
|
||||
}
|
||||
}
|
||||
|
||||
/** Recover one batch of abandoned deliveries; returns how many were handled. */
|
||||
public int recoverOnce() {
|
||||
List<dev.caskeleton.application.notification.platform.api.RecipientDeliveryId> abandoned =
|
||||
leases.expiredDispatching(batchSize, staleAfter);
|
||||
int handled = 0;
|
||||
for (var recipientDeliveryId : abandoned) {
|
||||
for (var attempt : attempts.attemptsOf(recipientDeliveryId)) {
|
||||
if (attempt.completedAt().isEmpty()) {
|
||||
DeliveryAttemptId attemptId = attempt.id();
|
||||
reconciliation.reconcile(attemptId);
|
||||
handled++;
|
||||
}
|
||||
}
|
||||
}
|
||||
return handled;
|
||||
}
|
||||
}
|
||||
+29
@@ -0,0 +1,29 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.inbox.InboxItemCreated;
|
||||
import dev.caskeleton.application.notification.platform.inbox.NotificationInboxSignalPort;
|
||||
import java.util.Objects;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
/**
|
||||
* Default inbox signal sink.
|
||||
*
|
||||
* <p>Emits identifiers only, never content. A deployment with a WebSocket or messaging relay
|
||||
* replaces it; until then the inbox is still complete, because the row — not the signal — is the
|
||||
* source of truth.
|
||||
*/
|
||||
public final class LoggingInboxSignalPublisher implements NotificationInboxSignalPort {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger("notification.inbox.signal");
|
||||
|
||||
@Override
|
||||
public void publish(InboxItemCreated event) {
|
||||
Objects.requireNonNull(event, "event");
|
||||
log.info(
|
||||
"event=inbox_item_created itemId={} tenant={} category={}",
|
||||
event.itemId().value(),
|
||||
event.principal().tenantId().value(),
|
||||
event.category());
|
||||
}
|
||||
}
|
||||
+31
@@ -0,0 +1,31 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.api.routing.Channel;
|
||||
import dev.caskeleton.application.notification.platform.dispatch.TemplateRendererRegistry;
|
||||
import dev.caskeleton.application.notification.platform.template.NotificationTemplateRenderer;
|
||||
import java.util.EnumMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
|
||||
/** Channel-to-renderer lookup built at wiring time. */
|
||||
public final class MapTemplateRendererRegistry implements TemplateRendererRegistry {
|
||||
|
||||
private final Map<Channel, NotificationTemplateRenderer> renderers;
|
||||
|
||||
public MapTemplateRendererRegistry(List<NotificationTemplateRenderer> renderers) {
|
||||
Objects.requireNonNull(renderers, "renderers");
|
||||
Map<Channel, NotificationTemplateRenderer> byChannel = new EnumMap<>(Channel.class);
|
||||
renderers.forEach(renderer -> byChannel.put(renderer.channel(), renderer));
|
||||
this.renderers = Map.copyOf(byChannel);
|
||||
}
|
||||
|
||||
@Override
|
||||
public NotificationTemplateRenderer rendererFor(Channel channel) {
|
||||
NotificationTemplateRenderer renderer = renderers.get(channel);
|
||||
if (renderer == null) {
|
||||
throw new IllegalStateException("no renderer registered for the channel");
|
||||
}
|
||||
return renderer;
|
||||
}
|
||||
}
|
||||
+55
@@ -0,0 +1,55 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
|
||||
|
||||
import java.time.Duration;
|
||||
import java.util.Objects;
|
||||
|
||||
/**
|
||||
* Dispatch runtime bounds.
|
||||
*
|
||||
* <p>Every field is bounded and validated at construction. "Unlimited" is never an accepted value:
|
||||
* an unbounded claim batch or queue is how a burst becomes an out-of-memory failure instead of
|
||||
* backpressure.
|
||||
*/
|
||||
public record NotificationDispatchProperties(
|
||||
int claimBatchSize,
|
||||
Duration leaseDuration,
|
||||
Duration pollInterval,
|
||||
int maxGlobalConcurrency,
|
||||
int maxAdditionalAttempts,
|
||||
Duration estimatedDispatchDuration,
|
||||
Duration maxQueueAge,
|
||||
Duration shutdownGrace) {
|
||||
|
||||
private static final int MAX_CLAIM_BATCH = 1000;
|
||||
|
||||
public NotificationDispatchProperties {
|
||||
Objects.requireNonNull(leaseDuration, "leaseDuration");
|
||||
Objects.requireNonNull(pollInterval, "pollInterval");
|
||||
Objects.requireNonNull(estimatedDispatchDuration, "estimatedDispatchDuration");
|
||||
Objects.requireNonNull(maxQueueAge, "maxQueueAge");
|
||||
Objects.requireNonNull(shutdownGrace, "shutdownGrace");
|
||||
if (claimBatchSize < 1 || claimBatchSize > MAX_CLAIM_BATCH) {
|
||||
throw new IllegalArgumentException("claimBatchSize must be 1.." + MAX_CLAIM_BATCH);
|
||||
}
|
||||
if (maxGlobalConcurrency < 1) {
|
||||
throw new IllegalArgumentException("maxGlobalConcurrency");
|
||||
}
|
||||
if (maxAdditionalAttempts < 0) {
|
||||
throw new IllegalArgumentException("maxAdditionalAttempts");
|
||||
}
|
||||
requirePositive(leaseDuration, "leaseDuration");
|
||||
requirePositive(pollInterval, "pollInterval");
|
||||
requirePositive(maxQueueAge, "maxQueueAge");
|
||||
if (leaseDuration.compareTo(pollInterval) <= 0) {
|
||||
// A lease shorter than the poll interval expires before the worker can renew it, so two
|
||||
// workers would routinely claim the same job.
|
||||
throw new IllegalArgumentException("leaseDuration must exceed pollInterval");
|
||||
}
|
||||
}
|
||||
|
||||
private static void requirePositive(Duration value, String name) {
|
||||
if (value.isNegative() || value.isZero()) {
|
||||
throw new IllegalArgumentException(name + " must be positive and finite");
|
||||
}
|
||||
}
|
||||
}
|
||||
+137
@@ -0,0 +1,137 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.dispatch.NotificationDispatchService;
|
||||
import dev.caskeleton.application.notification.platform.dispatch.RecipientLease;
|
||||
import dev.caskeleton.application.notification.platform.dispatch.RecipientLeaseStorePort;
|
||||
import dev.caskeleton.application.notification.platform.observation.NotificationMetricName;
|
||||
import dev.caskeleton.application.notification.platform.observation.NotificationMetricsPort;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.concurrent.ExecutorService;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.Semaphore;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
/**
|
||||
* Claims due deliveries and hands them to the dispatcher.
|
||||
*
|
||||
* <p>The worker never calls a provider itself. It claims, submits to a bounded executor, and stops
|
||||
* claiming the moment shutdown begins — so a rolling restart drains rather than abandoning leases
|
||||
* that then have to time out.
|
||||
*
|
||||
* <p>Claiming is bounded twice over: by the claim batch size and by a global concurrency permit.
|
||||
* The second bound matters because a slow provider would otherwise let the queue depth become the
|
||||
* thread count.
|
||||
*/
|
||||
public final class NotificationSchedulerWorker implements AutoCloseable {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(NotificationSchedulerWorker.class);
|
||||
|
||||
private final RecipientLeaseStorePort leases;
|
||||
private final NotificationDispatchService dispatcher;
|
||||
private final NotificationMetricsPort metrics;
|
||||
private final NotificationDispatchProperties properties;
|
||||
private final String workerId;
|
||||
private final ExecutorService dispatchExecutor;
|
||||
private final Semaphore globalConcurrency;
|
||||
private final AtomicBoolean running = new AtomicBoolean();
|
||||
private final AtomicBoolean shuttingDown = new AtomicBoolean();
|
||||
|
||||
public NotificationSchedulerWorker(
|
||||
RecipientLeaseStorePort leases,
|
||||
NotificationDispatchService dispatcher,
|
||||
NotificationMetricsPort metrics,
|
||||
NotificationDispatchProperties properties,
|
||||
String workerId) {
|
||||
this.leases = Objects.requireNonNull(leases, "leases");
|
||||
this.dispatcher = Objects.requireNonNull(dispatcher, "dispatcher");
|
||||
this.metrics = Objects.requireNonNull(metrics, "metrics");
|
||||
this.properties = Objects.requireNonNull(properties, "properties");
|
||||
this.workerId = Objects.requireNonNull(workerId, "workerId");
|
||||
this.dispatchExecutor = Executors.newVirtualThreadPerTaskExecutor();
|
||||
this.globalConcurrency = new Semaphore(properties.maxGlobalConcurrency());
|
||||
}
|
||||
|
||||
/** Claim and dispatch one batch. Returns how many deliveries were claimed. */
|
||||
public int runOnce() {
|
||||
if (shuttingDown.get()) {
|
||||
return 0;
|
||||
}
|
||||
List<RecipientLease> claimed =
|
||||
leases.claim(workerId, properties.claimBatchSize(), properties.leaseDuration());
|
||||
metrics.gauge(NotificationMetricName.QUEUE_DEPTH, Map.of(), claimed.size());
|
||||
|
||||
for (RecipientLease lease : claimed) {
|
||||
globalConcurrency.acquireUninterruptibly();
|
||||
dispatchExecutor.execute(
|
||||
() -> {
|
||||
try {
|
||||
dispatcher.dispatch(lease);
|
||||
} catch (RuntimeException failure) {
|
||||
// The lease is left to expire rather than being released optimistically: a worker
|
||||
// that
|
||||
// failed mid-dispatch cannot prove what the provider did.
|
||||
log.warn(
|
||||
"notification dispatch failed worker={} reason={}",
|
||||
workerId,
|
||||
failure.getClass().getSimpleName());
|
||||
} finally {
|
||||
globalConcurrency.release();
|
||||
}
|
||||
});
|
||||
}
|
||||
return claimed.size();
|
||||
}
|
||||
|
||||
/** Start the polling loop on a dedicated thread. */
|
||||
public void start() {
|
||||
if (!running.compareAndSet(false, true)) {
|
||||
return;
|
||||
}
|
||||
Thread.ofVirtual()
|
||||
.name("notification-scheduler-" + workerId)
|
||||
.start(
|
||||
() -> {
|
||||
while (running.get() && !shuttingDown.get()) {
|
||||
try {
|
||||
if (runOnce() == 0) {
|
||||
Thread.sleep(properties.pollInterval().toMillis());
|
||||
}
|
||||
} catch (InterruptedException interrupted) {
|
||||
Thread.currentThread().interrupt();
|
||||
return;
|
||||
} catch (RuntimeException failure) {
|
||||
log.warn(
|
||||
"notification scheduler tick failed worker={} reason={}",
|
||||
workerId,
|
||||
failure.getClass().getSimpleName());
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
shuttingDown.set(true);
|
||||
running.set(false);
|
||||
dispatchExecutor.shutdown();
|
||||
try {
|
||||
if (!dispatchExecutor.awaitTermination(
|
||||
properties.shutdownGrace().toMillis(), TimeUnit.MILLISECONDS)) {
|
||||
dispatchExecutor.shutdownNow();
|
||||
}
|
||||
} catch (InterruptedException interrupted) {
|
||||
Thread.currentThread().interrupt();
|
||||
dispatchExecutor.shutdownNow();
|
||||
}
|
||||
}
|
||||
|
||||
/** Whether the worker has stopped claiming new work. */
|
||||
public boolean shuttingDown() {
|
||||
return shuttingDown.get();
|
||||
}
|
||||
}
|
||||
+73
@@ -0,0 +1,73 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.api.error.FailureCategory;
|
||||
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureCode;
|
||||
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureDescriptor;
|
||||
import dev.caskeleton.application.notification.platform.api.error.ProviderUnavailableException;
|
||||
import java.time.Clock;
|
||||
import java.util.Objects;
|
||||
import java.util.concurrent.Semaphore;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
|
||||
/**
|
||||
* Per-provider rate and concurrency guard.
|
||||
*
|
||||
* <p>Tokens are spent on real attempts only. A delivery waiting out its backoff holds no permit,
|
||||
* because a provider outage would otherwise pin the whole concurrency budget on deliveries that are
|
||||
* not doing anything.
|
||||
*/
|
||||
public final class ProviderAttemptLimiter {
|
||||
|
||||
private final Semaphore concurrency;
|
||||
private final int maxConcurrency;
|
||||
private final int ratePerSecond;
|
||||
private final Clock clock;
|
||||
private final AtomicLong windowStartSecond = new AtomicLong();
|
||||
private final AtomicLong issuedInWindow = new AtomicLong();
|
||||
|
||||
public ProviderAttemptLimiter(int maxConcurrency, int ratePerSecond, Clock clock) {
|
||||
if (maxConcurrency < 1) {
|
||||
throw new IllegalArgumentException("maxConcurrency");
|
||||
}
|
||||
if (ratePerSecond < 1) {
|
||||
throw new IllegalArgumentException("ratePerSecond");
|
||||
}
|
||||
this.concurrency = new Semaphore(maxConcurrency);
|
||||
this.maxConcurrency = maxConcurrency;
|
||||
this.ratePerSecond = ratePerSecond;
|
||||
this.clock = Objects.requireNonNull(clock, "clock");
|
||||
this.windowStartSecond.set(clock.instant().getEpochSecond());
|
||||
}
|
||||
|
||||
/** Acquire one attempt slot, or fail fast when the provider budget is spent. */
|
||||
public void acquire() {
|
||||
long second = clock.instant().getEpochSecond();
|
||||
long windowStart = windowStartSecond.get();
|
||||
if (second != windowStart && windowStartSecond.compareAndSet(windowStart, second)) {
|
||||
issuedInWindow.set(0L);
|
||||
}
|
||||
if (issuedInWindow.incrementAndGet() > ratePerSecond) {
|
||||
throw unavailable();
|
||||
}
|
||||
if (!concurrency.tryAcquire()) {
|
||||
issuedInWindow.decrementAndGet();
|
||||
throw unavailable();
|
||||
}
|
||||
}
|
||||
|
||||
/** Release a previously acquired slot. */
|
||||
public void release() {
|
||||
concurrency.release();
|
||||
}
|
||||
|
||||
/** Slots currently held. */
|
||||
public int activeAttempts() {
|
||||
return maxConcurrency - concurrency.availablePermits();
|
||||
}
|
||||
|
||||
private static ProviderUnavailableException unavailable() {
|
||||
return new ProviderUnavailableException(
|
||||
NotificationFailureDescriptor.preDispatch(
|
||||
NotificationFailureCode.PROVIDER_UNAVAILABLE, FailureCategory.CAPACITY_REJECTED));
|
||||
}
|
||||
}
|
||||
+136
@@ -0,0 +1,136 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.api.error.FailureCategory;
|
||||
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureCode;
|
||||
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureDescriptor;
|
||||
import dev.caskeleton.application.notification.platform.api.error.ProviderUnavailableException;
|
||||
import dev.caskeleton.application.notification.platform.provider.NotificationProviderAdapter;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderProfileSnapshot;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderRuntimeState;
|
||||
import java.util.Objects;
|
||||
import java.util.Optional;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
|
||||
/**
|
||||
* One immutable credential generation of a provider.
|
||||
*
|
||||
* <p>Generations are replaced, never mutated. Rotating a key by editing a live client would leave
|
||||
* in-flight calls half-way between two credentials; replacing the whole runtime and letting the old
|
||||
* one drain keeps every attempt attributable to exactly one generation.
|
||||
*
|
||||
* <p>An authentication failure moves the whole runtime, not the message. One expired credential
|
||||
* multiplied by a queue of notifications is a self-inflicted outage, so the route opens once and
|
||||
* raises an operational alert instead.
|
||||
*/
|
||||
public final class ProviderRuntime {
|
||||
|
||||
private final ProviderProfileSnapshot profile;
|
||||
private final NotificationProviderAdapter adapter;
|
||||
private final ProviderAttemptLimiter limiter;
|
||||
private final AtomicReference<ProviderRuntimeState> state;
|
||||
private final AtomicReference<String> unhealthyReason = new AtomicReference<>();
|
||||
|
||||
public ProviderRuntime(
|
||||
ProviderProfileSnapshot profile,
|
||||
NotificationProviderAdapter adapter,
|
||||
ProviderAttemptLimiter limiter) {
|
||||
this.profile = Objects.requireNonNull(profile, "profile");
|
||||
this.adapter = Objects.requireNonNull(adapter, "adapter");
|
||||
this.limiter = Objects.requireNonNull(limiter, "limiter");
|
||||
this.state = new AtomicReference<>(ProviderRuntimeState.HEALTHY);
|
||||
}
|
||||
|
||||
/** Profile snapshot including the credential generation. */
|
||||
public ProviderProfileSnapshot profile() {
|
||||
return profile;
|
||||
}
|
||||
|
||||
/** Credential generation of this runtime. */
|
||||
public long generation() {
|
||||
return profile.credentialGeneration();
|
||||
}
|
||||
|
||||
/** Provider adapter bound to this generation. */
|
||||
public NotificationProviderAdapter adapter() {
|
||||
return adapter;
|
||||
}
|
||||
|
||||
/** Current health. */
|
||||
public ProviderRuntimeState state() {
|
||||
return state.get();
|
||||
}
|
||||
|
||||
/** Why the runtime is unhealthy, if it is. */
|
||||
public Optional<String> unhealthyReason() {
|
||||
return Optional.ofNullable(unhealthyReason.get());
|
||||
}
|
||||
|
||||
/** Attempts currently in flight on this generation. */
|
||||
public int activeAttempts() {
|
||||
return limiter.activeAttempts();
|
||||
}
|
||||
|
||||
/**
|
||||
* Acquire a permit for one attempt.
|
||||
*
|
||||
* <p>The health check happens before the limiter, so a disabled or failed provider never consumes
|
||||
* a token it cannot use.
|
||||
*/
|
||||
public AttemptPermit acquireAttempt() {
|
||||
ProviderRuntimeState current = state.get();
|
||||
if (!current.admitsNewAttempts()) {
|
||||
throw new ProviderUnavailableException(
|
||||
NotificationFailureDescriptor.preDispatch(
|
||||
NotificationFailureCode.PROVIDER_UNAVAILABLE,
|
||||
current == ProviderRuntimeState.AUTHENTICATION_FAILED
|
||||
? FailureCategory.AUTHENTICATION
|
||||
: FailureCategory.CAPACITY_REJECTED));
|
||||
}
|
||||
limiter.acquire();
|
||||
return new LimiterPermit(profile.credentialGeneration(), limiter);
|
||||
}
|
||||
|
||||
/** Mark the credential as rejected by the provider. */
|
||||
public void markAuthenticationFailed(String reasonCode) {
|
||||
unhealthyReason.set(Objects.requireNonNull(reasonCode, "reasonCode"));
|
||||
state.set(ProviderRuntimeState.AUTHENTICATION_FAILED);
|
||||
}
|
||||
|
||||
/** Mark the provider as rate limited. */
|
||||
public void markThrottled() {
|
||||
state.compareAndSet(ProviderRuntimeState.HEALTHY, ProviderRuntimeState.THROTTLED);
|
||||
}
|
||||
|
||||
/** Mark the provider as degraded but still usable. */
|
||||
public void markDegraded(String reasonCode) {
|
||||
unhealthyReason.set(reasonCode);
|
||||
state.compareAndSet(ProviderRuntimeState.HEALTHY, ProviderRuntimeState.DEGRADED);
|
||||
}
|
||||
|
||||
/** Return to healthy after a successful attempt. */
|
||||
public void markHealthy() {
|
||||
unhealthyReason.set(null);
|
||||
state.compareAndSet(ProviderRuntimeState.THROTTLED, ProviderRuntimeState.HEALTHY);
|
||||
state.compareAndSet(ProviderRuntimeState.DEGRADED, ProviderRuntimeState.HEALTHY);
|
||||
}
|
||||
|
||||
/** Stop admitting new attempts; in-flight attempts finish. */
|
||||
public void markDraining() {
|
||||
state.set(ProviderRuntimeState.DRAINING);
|
||||
}
|
||||
|
||||
/** Operator disable. */
|
||||
public void markDisabled() {
|
||||
state.set(ProviderRuntimeState.DISABLED);
|
||||
}
|
||||
|
||||
/** A permit that releases exactly one limiter slot. */
|
||||
private record LimiterPermit(long generation, ProviderAttemptLimiter limiter)
|
||||
implements AttemptPermit {
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
limiter.release();
|
||||
}
|
||||
}
|
||||
}
|
||||
+78
@@ -0,0 +1,78 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.api.ProviderProfileId;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderRuntimeState;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Optional;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.CopyOnWriteArrayList;
|
||||
|
||||
/** Holds the current generation of every provider profile plus the generations still draining. */
|
||||
public final class ProviderRuntimeRegistry {
|
||||
|
||||
private final Map<ProviderProfileId, ProviderRuntime> current = new ConcurrentHashMap<>();
|
||||
private final Map<ProviderProfileId, CopyOnWriteArrayList<ProviderRuntime>> draining =
|
||||
new ConcurrentHashMap<>();
|
||||
|
||||
/** Register the first generation of a profile. */
|
||||
public void register(ProviderRuntime runtime) {
|
||||
Objects.requireNonNull(runtime, "runtime");
|
||||
current.put(runtime.profile().profileId(), runtime);
|
||||
}
|
||||
|
||||
/** Current generation, or a configuration failure when the profile is unknown. */
|
||||
public ProviderRuntime current(ProviderProfileId profileId) {
|
||||
ProviderRuntime runtime = current.get(profileId);
|
||||
if (runtime == null) {
|
||||
throw new IllegalStateException("no provider runtime registered for the profile");
|
||||
}
|
||||
return runtime;
|
||||
}
|
||||
|
||||
/** Current generation if registered. */
|
||||
public Optional<ProviderRuntime> find(ProviderProfileId profileId) {
|
||||
return Optional.ofNullable(current.get(profileId));
|
||||
}
|
||||
|
||||
/**
|
||||
* Swap in a new generation and start draining the old one.
|
||||
*
|
||||
* <p>New dispatches immediately use the new generation while the previous one finishes what it
|
||||
* already started, which is what makes a credential rotation invisible to callers.
|
||||
*/
|
||||
public Optional<ProviderRuntime> replace(ProviderRuntime replacement) {
|
||||
Objects.requireNonNull(replacement, "replacement");
|
||||
ProviderProfileId profileId = replacement.profile().profileId();
|
||||
ProviderRuntime previous = current.put(profileId, replacement);
|
||||
if (previous != null) {
|
||||
previous.markDraining();
|
||||
draining.computeIfAbsent(profileId, key -> new CopyOnWriteArrayList<>()).add(previous);
|
||||
forgetIfDrained(profileId);
|
||||
}
|
||||
return Optional.ofNullable(previous);
|
||||
}
|
||||
|
||||
/** Generations that are draining and still have work in flight. */
|
||||
public List<ProviderRuntime> drainingGenerations(ProviderProfileId profileId) {
|
||||
forgetIfDrained(profileId);
|
||||
return List.copyOf(draining.getOrDefault(profileId, new CopyOnWriteArrayList<>()));
|
||||
}
|
||||
|
||||
/** Health of the current generation. */
|
||||
public ProviderRuntimeState state(ProviderProfileId profileId) {
|
||||
return current(profileId).state();
|
||||
}
|
||||
|
||||
private void forgetIfDrained(ProviderProfileId profileId) {
|
||||
CopyOnWriteArrayList<ProviderRuntime> generations = draining.get(profileId);
|
||||
if (generations == null) {
|
||||
return;
|
||||
}
|
||||
generations.removeIf(runtime -> runtime.activeAttempts() == 0);
|
||||
if (generations.isEmpty()) {
|
||||
draining.remove(profileId);
|
||||
}
|
||||
}
|
||||
}
|
||||
+69
@@ -0,0 +1,69 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.api.ProviderProfileId;
|
||||
import dev.caskeleton.application.notification.platform.observation.NotificationAuditEvent;
|
||||
import dev.caskeleton.application.notification.platform.observation.NotificationAuditPort;
|
||||
import java.time.Clock;
|
||||
import java.time.Duration;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Optional;
|
||||
|
||||
/**
|
||||
* Credential and certificate rotation.
|
||||
*
|
||||
* <p>The candidate is probed <em>before</em> the swap. Validating after cutover would mean a typo
|
||||
* in a rotated secret takes the provider down and only then tells anyone; validating first makes a
|
||||
* bad candidate a no-op that leaves the working generation in place.
|
||||
*
|
||||
* <p>Only the generation and key id reach the audit trail — never the credential material itself.
|
||||
*/
|
||||
public final class ProviderRuntimeRotator {
|
||||
|
||||
private final ProviderRuntimeRegistry registry;
|
||||
private final CredentialProbe probe;
|
||||
private final RuntimeDrainCoordinator drainCoordinator;
|
||||
private final NotificationAuditPort audit;
|
||||
private final Clock clock;
|
||||
private final Duration drainTimeout;
|
||||
|
||||
public ProviderRuntimeRotator(
|
||||
ProviderRuntimeRegistry registry,
|
||||
CredentialProbe probe,
|
||||
RuntimeDrainCoordinator drainCoordinator,
|
||||
NotificationAuditPort audit,
|
||||
Clock clock,
|
||||
Duration drainTimeout) {
|
||||
this.registry = Objects.requireNonNull(registry, "registry");
|
||||
this.probe = Objects.requireNonNull(probe, "probe");
|
||||
this.drainCoordinator = Objects.requireNonNull(drainCoordinator, "drainCoordinator");
|
||||
this.audit = Objects.requireNonNull(audit, "audit");
|
||||
this.clock = Objects.requireNonNull(clock, "clock");
|
||||
this.drainTimeout = Objects.requireNonNull(drainTimeout, "drainTimeout");
|
||||
}
|
||||
|
||||
/** Cut over to a new credential generation. */
|
||||
public void rotate(ProviderProfileId profileId, ProviderRuntime candidate) {
|
||||
Objects.requireNonNull(profileId, "profileId");
|
||||
Objects.requireNonNull(candidate, "candidate");
|
||||
if (!candidate.profile().profileId().equals(profileId)) {
|
||||
throw new IllegalArgumentException("candidate belongs to a different profile");
|
||||
}
|
||||
if (!probe.isUsable(candidate)) {
|
||||
throw new CredentialValidationException();
|
||||
}
|
||||
|
||||
Optional<ProviderRuntime> previous = registry.replace(candidate);
|
||||
audit.record(
|
||||
new NotificationAuditEvent(
|
||||
"PROVIDER_CREDENTIAL_ROTATION",
|
||||
"system",
|
||||
Optional.of("ROTATION"),
|
||||
Optional.empty(),
|
||||
clock.instant(),
|
||||
Map.of(
|
||||
"providerProfile", profileId.value(),
|
||||
"generation", Long.toString(candidate.generation()))));
|
||||
previous.ifPresent(runtime -> drainCoordinator.drain(runtime, drainTimeout));
|
||||
}
|
||||
}
|
||||
+76
@@ -0,0 +1,76 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.api.ProviderProfileId;
|
||||
import dev.caskeleton.application.notification.platform.dispatch.ProviderDispatchGatewayPort;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderProfileSnapshot;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderRuntimeState;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderSubmission;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderSubmissionResult;
|
||||
import java.util.Objects;
|
||||
import java.util.concurrent.CompletionException;
|
||||
|
||||
/**
|
||||
* The single outbound call, wrapped in a permit.
|
||||
*
|
||||
* <p>The permit is acquired before the call and released in a finally, so a provider that hangs
|
||||
* consumes exactly one slot and a burst queues rather than exhausting the pool.
|
||||
*
|
||||
* <p>A credential rejection is promoted to a runtime state change here rather than being left as a
|
||||
* per-message failure — one expired key must open the route once, not produce one retry per queued
|
||||
* notification.
|
||||
*/
|
||||
public final class RegistryProviderDispatchGateway implements ProviderDispatchGatewayPort {
|
||||
|
||||
private final ProviderRuntimeRegistry runtimes;
|
||||
|
||||
public RegistryProviderDispatchGateway(ProviderRuntimeRegistry runtimes) {
|
||||
this.runtimes = Objects.requireNonNull(runtimes, "runtimes");
|
||||
}
|
||||
|
||||
@Override
|
||||
public ProviderProfileSnapshot profile(ProviderProfileId profileId) {
|
||||
return runtimes.current(profileId).profile();
|
||||
}
|
||||
|
||||
@Override
|
||||
public ProviderRuntimeState state(ProviderProfileId profileId) {
|
||||
return runtimes.state(profileId);
|
||||
}
|
||||
|
||||
@Override
|
||||
public ProviderSubmissionResult submit(ProviderSubmission submission) {
|
||||
Objects.requireNonNull(submission, "submission");
|
||||
ProviderRuntime runtime = runtimes.current(submission.profile().profileId());
|
||||
|
||||
try (AttemptPermit permit = runtime.acquireAttempt()) {
|
||||
ProviderSubmissionResult result =
|
||||
runtime.adapter().submit(submission).toCompletableFuture().join();
|
||||
applyHealth(runtime, result);
|
||||
return result;
|
||||
} catch (CompletionException failure) {
|
||||
// Unwrapped so the dispatcher classifies the real cause rather than the future's wrapper.
|
||||
Throwable cause = failure.getCause() == null ? failure : failure.getCause();
|
||||
throw cause instanceof RuntimeException runtimeFailure
|
||||
? runtimeFailure
|
||||
: new IllegalStateException("provider submission failed", cause);
|
||||
}
|
||||
}
|
||||
|
||||
private static void applyHealth(ProviderRuntime runtime, ProviderSubmissionResult result) {
|
||||
result
|
||||
.failure()
|
||||
.ifPresentOrElse(
|
||||
failure -> {
|
||||
switch (failure.category()) {
|
||||
case AUTHENTICATION, AUTHORIZATION ->
|
||||
runtime.markAuthenticationFailed(failure.code());
|
||||
case THROTTLED -> runtime.markThrottled();
|
||||
case TRANSIENT_PROVIDER -> runtime.markDegraded(failure.code());
|
||||
default -> {
|
||||
// A message-level failure says nothing about the provider's health.
|
||||
}
|
||||
}
|
||||
},
|
||||
runtime::markHealthy);
|
||||
}
|
||||
}
|
||||
+46
@@ -0,0 +1,46 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
|
||||
|
||||
import java.time.Duration;
|
||||
import java.util.Objects;
|
||||
|
||||
/**
|
||||
* Waits for a replaced generation to finish its in-flight attempts.
|
||||
*
|
||||
* <p>The deadline comes from {@link System#nanoTime()}, not from the injectable clock. A drain
|
||||
* timeout is a real elapsed-time budget: driving it from a test clock that never advances turns the
|
||||
* loop into a hang, and driving it from a wall clock makes it sensitive to time adjustments.
|
||||
*
|
||||
* <p>Draining is bounded on purpose. A provider that never answers must not hold a credential
|
||||
* rotation open forever, so after the timeout the generation is abandoned and its attempts follow
|
||||
* the normal ambiguity and reconciliation path rather than being cancelled mid-flight.
|
||||
*/
|
||||
public final class RuntimeDrainCoordinator {
|
||||
|
||||
private final Duration pollInterval;
|
||||
|
||||
public RuntimeDrainCoordinator(Duration pollInterval) {
|
||||
this.pollInterval = Objects.requireNonNull(pollInterval, "pollInterval");
|
||||
if (pollInterval.isNegative() || pollInterval.isZero()) {
|
||||
throw new IllegalArgumentException("pollInterval");
|
||||
}
|
||||
}
|
||||
|
||||
/** Drain a generation, returning whether it finished within the timeout. */
|
||||
public boolean drain(ProviderRuntime runtime, Duration timeout) {
|
||||
Objects.requireNonNull(runtime, "runtime");
|
||||
Objects.requireNonNull(timeout, "timeout");
|
||||
long deadlineNanos = System.nanoTime() + timeout.toNanos();
|
||||
while (runtime.activeAttempts() > 0) {
|
||||
if (System.nanoTime() - deadlineNanos >= 0) {
|
||||
return false;
|
||||
}
|
||||
try {
|
||||
Thread.sleep(pollInterval.toMillis());
|
||||
} catch (InterruptedException interrupted) {
|
||||
Thread.currentThread().interrupt();
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
}
|
||||
+27
@@ -0,0 +1,27 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.api.TenantId;
|
||||
import dev.caskeleton.application.notification.platform.dispatch.TenantContextPort;
|
||||
import java.util.Objects;
|
||||
|
||||
/**
|
||||
* Tenant context for a single-tenant deployment.
|
||||
*
|
||||
* <p>A multi-tenant deployment replaces this with a request-scoped implementation. It exists so
|
||||
* that a single-tenant application still goes through the tenant boundary rather than around it —
|
||||
* the store queries take a tenant either way, and a deployment that later becomes multi-tenant does
|
||||
* not have to find every unscoped query.
|
||||
*/
|
||||
public final class SingleTenantContext implements TenantContextPort {
|
||||
|
||||
private final TenantId tenantId;
|
||||
|
||||
public SingleTenantContext(String tenantId) {
|
||||
this.tenantId = new TenantId(Objects.requireNonNull(tenantId, "tenantId"));
|
||||
}
|
||||
|
||||
@Override
|
||||
public TenantId currentTenant() {
|
||||
return tenantId;
|
||||
}
|
||||
}
|
||||
+54
@@ -0,0 +1,54 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.dispatch;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.dispatch.NotificationIdGeneratorPort;
|
||||
import java.security.SecureRandom;
|
||||
import java.time.Clock;
|
||||
import java.util.Objects;
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
|
||||
/**
|
||||
* RFC 9562 UUIDv7.
|
||||
*
|
||||
* <p>Time-ordered rather than random because these identifiers are primary keys: a random UUID
|
||||
* scatters inserts across the whole index, and a notification table takes the highest insert rate
|
||||
* in the platform.
|
||||
*
|
||||
* <p>The monotonic counter guards the case two identifiers are requested inside the same
|
||||
* millisecond, so ordering holds even under a burst.
|
||||
*/
|
||||
public final class UuidV7Generator implements NotificationIdGeneratorPort {
|
||||
|
||||
private static final long VERSION_7 = 0x7000L;
|
||||
private static final long VARIANT_RFC = 0x8000000000000000L;
|
||||
|
||||
private final Clock clock;
|
||||
private final SecureRandom random;
|
||||
private final AtomicLong lastMillis = new AtomicLong();
|
||||
private final AtomicLong sequence = new AtomicLong();
|
||||
|
||||
public UuidV7Generator(Clock clock) {
|
||||
this(clock, new SecureRandom());
|
||||
}
|
||||
|
||||
UuidV7Generator(Clock clock, SecureRandom random) {
|
||||
this.clock = Objects.requireNonNull(clock, "clock");
|
||||
this.random = Objects.requireNonNull(random, "random");
|
||||
}
|
||||
|
||||
@Override
|
||||
public UUID nextId() {
|
||||
long millis = clock.millis();
|
||||
long previous = lastMillis.getAndSet(millis);
|
||||
long counter = millis == previous ? sequence.incrementAndGet() : sequence.updateAndGet(x -> 0L);
|
||||
|
||||
long high = (millis & 0xFFFFFFFFFFFFL) << 16;
|
||||
high |= VERSION_7;
|
||||
high |= counter & 0x0FFFL;
|
||||
|
||||
long low = random.nextLong();
|
||||
low &= 0x3FFFFFFFFFFFFFFFL;
|
||||
low |= VARIANT_RFC;
|
||||
return new UUID(high, low);
|
||||
}
|
||||
}
|
||||
+60
@@ -0,0 +1,60 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.observation;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.api.ProviderProfileId;
|
||||
import dev.caskeleton.application.notification.platform.observation.NotificationAuditEvent;
|
||||
import dev.caskeleton.application.notification.platform.observation.NotificationAuditPort;
|
||||
import dev.caskeleton.application.notification.platform.observation.NotificationSecurityAuditPort;
|
||||
import java.util.Objects;
|
||||
import java.util.TreeMap;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
/**
|
||||
* Audit sink on a dedicated logger.
|
||||
*
|
||||
* <p>Separate from the metrics logger because audit has different retention: a metric may be
|
||||
* sampled away, while "who lifted this suppression, and why" has to survive.
|
||||
*
|
||||
* <p>A rejected callback signature is a security event, not a provider event, so it is recorded
|
||||
* here and never in the ledger — otherwise anyone who can reach the endpoint could fill a delivery
|
||||
* history with noise.
|
||||
*/
|
||||
public final class LoggingNotificationAudit
|
||||
implements NotificationAuditPort, NotificationSecurityAuditPort {
|
||||
|
||||
private static final Logger audit = LoggerFactory.getLogger("notification.audit");
|
||||
private static final Logger security = LoggerFactory.getLogger("notification.security");
|
||||
|
||||
@Override
|
||||
public void record(NotificationAuditEvent event) {
|
||||
Objects.requireNonNull(event, "event");
|
||||
audit.info(
|
||||
"action={} actor={} reason={} operationId={} occurredAt={} attributes={}",
|
||||
event.action(),
|
||||
event.actorRef(),
|
||||
event.reasonCode().orElse("-"),
|
||||
event.operationId().orElse("-"),
|
||||
event.occurredAt(),
|
||||
new TreeMap<>(event.boundedAttributes()));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void callbackSignatureRejected(ProviderProfileId profileId, String reasonCode) {
|
||||
Objects.requireNonNull(profileId, "profileId");
|
||||
// The payload is deliberately absent: a forged callback must not get its content into the log
|
||||
// just by being rejected.
|
||||
security.warn(
|
||||
"event=callback_signature_rejected providerProfile={} reason={}",
|
||||
profileId.value(),
|
||||
reasonCode);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void callbackRejectedByLimit(ProviderProfileId profileId, String reasonCode) {
|
||||
Objects.requireNonNull(profileId, "profileId");
|
||||
security.warn(
|
||||
"event=callback_rejected_by_limit providerProfile={} reason={}",
|
||||
profileId.value(),
|
||||
reasonCode);
|
||||
}
|
||||
}
|
||||
+52
@@ -0,0 +1,52 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.observation;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.observation.CardinalityGuard;
|
||||
import dev.caskeleton.application.notification.platform.observation.NotificationMetricsPort;
|
||||
import java.time.Duration;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.TreeMap;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
/**
|
||||
* Structured-log metrics sink.
|
||||
*
|
||||
* <p>Every tag map passes the cardinality guard before it is emitted, so a stray notification id
|
||||
* fails here rather than after it has already multiplied a time series into millions of them.
|
||||
*
|
||||
* <p>A Micrometer-backed implementation belongs in the composition root, which owns the registry;
|
||||
* this one keeps the platform usable — and its tag discipline enforced — without one.
|
||||
*/
|
||||
public final class LoggingNotificationMetrics implements NotificationMetricsPort {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger("notification.metrics");
|
||||
|
||||
private final CardinalityGuard guard;
|
||||
|
||||
public LoggingNotificationMetrics(CardinalityGuard guard) {
|
||||
this.guard = Objects.requireNonNull(guard, "guard");
|
||||
}
|
||||
|
||||
@Override
|
||||
public void increment(String metricName, Map<String, String> tags) {
|
||||
guard.validate(tags);
|
||||
log.info("metric={} kind=counter tags={}", metricName, ordered(tags));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void record(String metricName, Map<String, String> tags, Duration value) {
|
||||
guard.validate(tags);
|
||||
log.info("metric={} kind=timer millis={} tags={}", metricName, value.toMillis(), ordered(tags));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void gauge(String metricName, Map<String, String> tags, double value) {
|
||||
guard.validate(tags);
|
||||
log.info("metric={} kind=gauge value={} tags={}", metricName, value, ordered(tags));
|
||||
}
|
||||
|
||||
private static Map<String, String> ordered(Map<String, String> tags) {
|
||||
return new TreeMap<>(tags);
|
||||
}
|
||||
}
|
||||
+57
@@ -0,0 +1,57 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.observation;
|
||||
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.dispatch.ProviderRuntimeRegistry;
|
||||
import dev.caskeleton.application.notification.platform.api.ProviderProfileId;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderRuntimeState;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
|
||||
/**
|
||||
* Builds the operational snapshot.
|
||||
*
|
||||
* <p>A provider whose credentials were rejected reports unhealthy even though the process is fine:
|
||||
* that is exactly the condition an operator needs paged on, and it is invisible from process-level
|
||||
* health.
|
||||
*/
|
||||
public final class NotificationHealthReporter {
|
||||
|
||||
private final ProviderRuntimeRegistry runtimes;
|
||||
private final List<ProviderProfileId> monitoredProfiles;
|
||||
|
||||
public NotificationHealthReporter(
|
||||
ProviderRuntimeRegistry runtimes, List<ProviderProfileId> monitoredProfiles) {
|
||||
this.runtimes = Objects.requireNonNull(runtimes, "runtimes");
|
||||
this.monitoredProfiles = List.copyOf(Objects.requireNonNull(monitoredProfiles, "profiles"));
|
||||
}
|
||||
|
||||
/** Current snapshot. */
|
||||
public NotificationHealthSnapshot snapshot() {
|
||||
List<NotificationHealthSnapshot.ProviderHealth> providers = new ArrayList<>();
|
||||
boolean healthy = true;
|
||||
|
||||
for (ProviderProfileId profileId : monitoredProfiles) {
|
||||
var runtime = runtimes.find(profileId);
|
||||
if (runtime.isEmpty()) {
|
||||
healthy = false;
|
||||
providers.add(
|
||||
new NotificationHealthSnapshot.ProviderHealth(profileId.value(), "UNREGISTERED", 0, 0));
|
||||
continue;
|
||||
}
|
||||
ProviderRuntimeState state = runtime.get().state();
|
||||
if (state == ProviderRuntimeState.AUTHENTICATION_FAILED
|
||||
|| state == ProviderRuntimeState.DISABLED) {
|
||||
healthy = false;
|
||||
}
|
||||
providers.add(
|
||||
new NotificationHealthSnapshot.ProviderHealth(
|
||||
profileId.value(),
|
||||
state.name(),
|
||||
runtime.get().generation(),
|
||||
runtime.get().activeAttempts()));
|
||||
}
|
||||
|
||||
return new NotificationHealthSnapshot(healthy, providers, Map.of());
|
||||
}
|
||||
}
|
||||
+31
@@ -0,0 +1,31 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.observation;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
|
||||
/**
|
||||
* Operational view of the platform.
|
||||
*
|
||||
* <p>Provider states, credential generations and queue age — nothing else. A health endpoint is one
|
||||
* of the least protected surfaces an application exposes, so a sender address or a credential
|
||||
* reference appearing here would be a leak with a wide audience.
|
||||
*/
|
||||
public record NotificationHealthSnapshot(
|
||||
boolean healthy, List<ProviderHealth> providers, Map<String, Long> queue) {
|
||||
|
||||
public NotificationHealthSnapshot {
|
||||
providers = List.copyOf(Objects.requireNonNull(providers, "providers"));
|
||||
queue = Map.copyOf(Objects.requireNonNull(queue, "queue"));
|
||||
}
|
||||
|
||||
/** One provider runtime's state. */
|
||||
public record ProviderHealth(
|
||||
String profileId, String state, long credentialGeneration, int activeAttempts) {
|
||||
|
||||
public ProviderHealth {
|
||||
Objects.requireNonNull(profileId, "profileId");
|
||||
Objects.requireNonNull(state, "state");
|
||||
}
|
||||
}
|
||||
}
|
||||
+100
@@ -0,0 +1,100 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.provider;
|
||||
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.provider.http.NotificationHttpTransportException;
|
||||
import dev.caskeleton.application.notification.platform.api.error.FailureCategory;
|
||||
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureCode;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderExecutionEvidence;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderFailure;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderSubmissionResult;
|
||||
import java.time.Duration;
|
||||
import java.util.Optional;
|
||||
|
||||
/**
|
||||
* Shared translation from a transport failure into an evidence-carrying result.
|
||||
*
|
||||
* <p>Every HTTP provider adapter routes its transport failures through here, so the rule that
|
||||
* "committed body plus no response equals ambiguous" is written once rather than re-derived per
|
||||
* provider.
|
||||
*/
|
||||
public final class ProviderResults {
|
||||
|
||||
private ProviderResults() {}
|
||||
|
||||
/** Classify a transport failure. */
|
||||
public static ProviderSubmissionResult fromTransport(
|
||||
NotificationHttpTransportException failure, Duration elapsed) {
|
||||
if (failure.requestBodyCommitted()) {
|
||||
return ProviderSubmissionResult.ambiguous(
|
||||
new ProviderFailure(
|
||||
NotificationFailureCode.PROVIDER_RESPONSE_LOST,
|
||||
FailureCategory.AMBIGUOUS_SUBMISSION,
|
||||
false,
|
||||
Optional.empty(),
|
||||
Optional.of(failure.reasonCode())),
|
||||
ProviderExecutionEvidence.responseLost(),
|
||||
elapsed);
|
||||
}
|
||||
return ProviderSubmissionResult.notSubmitted(
|
||||
new ProviderFailure(
|
||||
NotificationFailureCode.PROVIDER_TRANSIENT_FAILURE,
|
||||
FailureCategory.TRANSIENT_PROVIDER,
|
||||
true,
|
||||
Optional.empty(),
|
||||
Optional.of(failure.reasonCode())),
|
||||
elapsed);
|
||||
}
|
||||
|
||||
/** Classify an HTTP status that is not provider-specific. */
|
||||
public static ProviderFailure fromStatus(int statusCode, Optional<Duration> retryAfter) {
|
||||
if (statusCode == 429) {
|
||||
return new ProviderFailure(
|
||||
NotificationFailureCode.PROVIDER_THROTTLED,
|
||||
FailureCategory.THROTTLED,
|
||||
true,
|
||||
retryAfter,
|
||||
Optional.of(Integer.toString(statusCode)));
|
||||
}
|
||||
if (statusCode == 401) {
|
||||
return new ProviderFailure(
|
||||
NotificationFailureCode.PROVIDER_AUTHENTICATION_FAILED,
|
||||
FailureCategory.AUTHENTICATION,
|
||||
false,
|
||||
Optional.empty(),
|
||||
Optional.of("401"));
|
||||
}
|
||||
if (statusCode == 403) {
|
||||
return new ProviderFailure(
|
||||
NotificationFailureCode.PROVIDER_AUTHORIZATION_FAILED,
|
||||
FailureCategory.AUTHORIZATION,
|
||||
false,
|
||||
Optional.empty(),
|
||||
Optional.of("403"));
|
||||
}
|
||||
if (statusCode >= 500) {
|
||||
return new ProviderFailure(
|
||||
NotificationFailureCode.PROVIDER_TRANSIENT_FAILURE,
|
||||
FailureCategory.TRANSIENT_PROVIDER,
|
||||
true,
|
||||
retryAfter,
|
||||
Optional.of(Integer.toString(statusCode)));
|
||||
}
|
||||
return new ProviderFailure(
|
||||
NotificationFailureCode.PROVIDER_PERMANENT_FAILURE,
|
||||
FailureCategory.PERMANENT_PROVIDER,
|
||||
false,
|
||||
Optional.empty(),
|
||||
Optional.of(Integer.toString(statusCode)));
|
||||
}
|
||||
|
||||
/** Parse a {@code Retry-After} header expressed in seconds. */
|
||||
public static Optional<Duration> retryAfter(Optional<String> headerValue) {
|
||||
return headerValue.flatMap(
|
||||
value -> {
|
||||
try {
|
||||
return Optional.of(Duration.ofSeconds(Long.parseLong(value.trim())));
|
||||
} catch (NumberFormatException notSeconds) {
|
||||
return Optional.empty();
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
+29
@@ -0,0 +1,29 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.provider;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.api.content.AttachmentRef;
|
||||
import dev.caskeleton.application.notification.platform.api.error.AttachmentUnavailableException;
|
||||
import dev.caskeleton.application.notification.platform.api.error.FailureCategory;
|
||||
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureCode;
|
||||
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureDescriptor;
|
||||
import dev.caskeleton.application.notification.platform.provider.AttachmentAccessContext;
|
||||
import dev.caskeleton.application.notification.platform.provider.AttachmentResolver;
|
||||
import dev.caskeleton.application.notification.platform.provider.ResolvedAttachment;
|
||||
|
||||
/**
|
||||
* The resolver used when no attachment source is wired.
|
||||
*
|
||||
* <p>It refuses rather than returning an empty stream. Sending a mail whose attachment is silently
|
||||
* missing is worse than not sending it: the recipient is told something is attached and it is not.
|
||||
*
|
||||
* <p>The composition root replaces this with a file-server or object-storage backed resolver; both
|
||||
* leaves are visible there, and neither is reachable from this one.
|
||||
*/
|
||||
public final class UnconfiguredAttachmentResolver implements AttachmentResolver {
|
||||
|
||||
@Override
|
||||
public ResolvedAttachment resolve(AttachmentRef reference, AttachmentAccessContext context) {
|
||||
throw new AttachmentUnavailableException(
|
||||
NotificationFailureDescriptor.preDispatch(
|
||||
NotificationFailureCode.ATTACHMENT_UNAVAILABLE, FailureCategory.INVALID_PAYLOAD));
|
||||
}
|
||||
}
|
||||
+67
@@ -0,0 +1,67 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.provider.apns;
|
||||
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.provider.ProviderResults;
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.provider.http.NotificationHttpResponse;
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.template.NotificationJsonMapper;
|
||||
import dev.caskeleton.application.notification.platform.api.error.FailureCategory;
|
||||
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureCode;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderFailure;
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
|
||||
/** Maps APNs reason strings onto the stable failure vocabulary. */
|
||||
public final class ApnsFailureClassifier {
|
||||
|
||||
private static final Set<String> INVALID_TOKEN_REASONS =
|
||||
Set.of("BadDeviceToken", "Unregistered", "DeviceTokenNotForTopic");
|
||||
private static final Set<String> CONFIGURATION_REASONS =
|
||||
Set.of("BadTopic", "TopicDisallowed", "BadCertificateEnvironment", "InvalidPushType");
|
||||
|
||||
/** Classify a non-2xx APNs response. */
|
||||
public ProviderFailure classify(NotificationHttpResponse response) {
|
||||
Optional<String> reason = reason(response);
|
||||
if (reason.filter(INVALID_TOKEN_REASONS::contains).isPresent()) {
|
||||
return new ProviderFailure(
|
||||
NotificationFailureCode.CONTACT_POINT_INVALID,
|
||||
FailureCategory.INVALID_RECIPIENT,
|
||||
false,
|
||||
Optional.empty(),
|
||||
reason);
|
||||
}
|
||||
if (reason.filter(CONFIGURATION_REASONS::contains).isPresent()) {
|
||||
return new ProviderFailure(
|
||||
NotificationFailureCode.PROVIDER_CONFIGURATION_INVALID,
|
||||
FailureCategory.AUTHORIZATION,
|
||||
false,
|
||||
Optional.empty(),
|
||||
reason);
|
||||
}
|
||||
if (reason.filter("ExpiredProviderToken"::equals).isPresent()) {
|
||||
return new ProviderFailure(
|
||||
NotificationFailureCode.PROVIDER_AUTHENTICATION_FAILED,
|
||||
FailureCategory.AUTHENTICATION,
|
||||
false,
|
||||
Optional.empty(),
|
||||
reason);
|
||||
}
|
||||
if (reason.filter("TooManyRequests"::equals).isPresent()) {
|
||||
return new ProviderFailure(
|
||||
NotificationFailureCode.PROVIDER_THROTTLED,
|
||||
FailureCategory.THROTTLED,
|
||||
true,
|
||||
Optional.empty(),
|
||||
reason);
|
||||
}
|
||||
return ProviderResults.fromStatus(response.statusCode(), Optional.empty());
|
||||
}
|
||||
|
||||
private static Optional<String> reason(NotificationHttpResponse response) {
|
||||
try {
|
||||
var node = NotificationJsonMapper.mapper().readTree(response.bodyAsString());
|
||||
var reason = node.get("reason");
|
||||
return reason == null || reason.isNull() ? Optional.empty() : Optional.of(reason.asString());
|
||||
} catch (RuntimeException unparseable) {
|
||||
return Optional.empty();
|
||||
}
|
||||
}
|
||||
}
|
||||
+100
@@ -0,0 +1,100 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.provider.apns;
|
||||
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.provider.ProviderResults;
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.provider.http.NotificationHttpGateway;
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.provider.http.NotificationHttpResponse;
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.provider.http.NotificationHttpTransportException;
|
||||
import dev.caskeleton.application.notification.platform.api.ProviderId;
|
||||
import dev.caskeleton.application.notification.platform.api.routing.Channel;
|
||||
import dev.caskeleton.application.notification.platform.provider.NotificationProviderAdapter;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderCapabilities;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderSubmission;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderSubmissionResult;
|
||||
import dev.caskeleton.application.notification.platform.security.AccessContext;
|
||||
import dev.caskeleton.application.notification.platform.security.ContactPointProtector;
|
||||
import java.time.Duration;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.CompletionStage;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* APNs adapter.
|
||||
*
|
||||
* <p>A 2xx is acceptance. Apple documents that an accepted notification may be delivered, stored or
|
||||
* discarded, and that ordering is not guaranteed, so this adapter never produces a delivery outcome
|
||||
* and the platform never uses APNs as an ordered event transport.
|
||||
*/
|
||||
public final class ApnsNotificationProviderAdapter implements NotificationProviderAdapter {
|
||||
|
||||
private static final ProviderId PROVIDER_ID = new ProviderId("apns");
|
||||
|
||||
private final NotificationHttpGateway gateway;
|
||||
private final ApnsRequestMapper mapper;
|
||||
private final ApnsFailureClassifier classifier;
|
||||
private final ContactPointProtector protector;
|
||||
private final Supplier<String> authorizationSupplier;
|
||||
|
||||
public ApnsNotificationProviderAdapter(
|
||||
NotificationHttpGateway gateway,
|
||||
ApnsRequestMapper mapper,
|
||||
ApnsFailureClassifier classifier,
|
||||
ContactPointProtector protector,
|
||||
Supplier<String> authorizationSupplier,
|
||||
ApnsProviderProperties properties) {
|
||||
// The profile is required at construction so a missing topic or environment fails at wiring
|
||||
// time, but it is never exposed: a public accessor would leak an adapter type across the port.
|
||||
Objects.requireNonNull(properties, "properties");
|
||||
this.gateway = Objects.requireNonNull(gateway, "gateway");
|
||||
this.mapper = Objects.requireNonNull(mapper, "mapper");
|
||||
this.classifier = Objects.requireNonNull(classifier, "classifier");
|
||||
this.protector = Objects.requireNonNull(protector, "protector");
|
||||
this.authorizationSupplier =
|
||||
Objects.requireNonNull(authorizationSupplier, "authorizationSupplier");
|
||||
}
|
||||
|
||||
@Override
|
||||
public ProviderId providerId() {
|
||||
return PROVIDER_ID;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Set<Channel> channels() {
|
||||
return Set.of(Channel.PUSH);
|
||||
}
|
||||
|
||||
@Override
|
||||
public ProviderCapabilities capabilities() {
|
||||
return new ProviderCapabilities(
|
||||
false, false, false, false, false, false, false, true, 1, 4096L, Duration.ofDays(30));
|
||||
}
|
||||
|
||||
@Override
|
||||
public CompletionStage<ProviderSubmissionResult> submit(ProviderSubmission submission) {
|
||||
Objects.requireNonNull(submission, "submission");
|
||||
return CompletableFuture.completedFuture(send(submission));
|
||||
}
|
||||
|
||||
private ProviderSubmissionResult send(ProviderSubmission submission) {
|
||||
long startedNanos = System.nanoTime();
|
||||
var contactPoint =
|
||||
protector.reveal(
|
||||
submission.contactPoint(),
|
||||
AccessContext.dispatch(submission.profile().profileId().value()));
|
||||
var request = mapper.map(submission, contactPoint, authorizationSupplier.get());
|
||||
|
||||
try {
|
||||
NotificationHttpResponse response = gateway.exchange(request);
|
||||
Duration elapsed = Duration.ofNanos(System.nanoTime() - startedNanos);
|
||||
if (response.isSuccessful()) {
|
||||
return ProviderSubmissionResult.accepted(
|
||||
response.header("apns-id").orElse(null), "Accepted", elapsed);
|
||||
}
|
||||
return ProviderSubmissionResult.rejected(classifier.classify(response), elapsed);
|
||||
} catch (NotificationHttpTransportException transportFailure) {
|
||||
return ProviderResults.fromTransport(
|
||||
transportFailure, Duration.ofNanos(System.nanoTime() - startedNanos));
|
||||
}
|
||||
}
|
||||
}
|
||||
+38
@@ -0,0 +1,38 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.provider.apns;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.contact.ApnsEnvironment;
|
||||
import java.net.URI;
|
||||
import java.time.Duration;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
|
||||
/**
|
||||
* APNs profile.
|
||||
*
|
||||
* <p>Environment and topic are required. A sandbox token sent to the production host is a silent
|
||||
* non-delivery, so the pairing is checked before the call rather than diagnosed afterwards.
|
||||
*/
|
||||
public record ApnsProviderProperties(
|
||||
URI endpoint,
|
||||
String topic,
|
||||
ApnsEnvironment environment,
|
||||
Set<String> allowedPushTypes,
|
||||
Duration timeout) {
|
||||
|
||||
public ApnsProviderProperties {
|
||||
Objects.requireNonNull(endpoint, "endpoint");
|
||||
Objects.requireNonNull(topic, "topic");
|
||||
Objects.requireNonNull(environment, "environment");
|
||||
allowedPushTypes = Set.copyOf(Objects.requireNonNull(allowedPushTypes, "allowedPushTypes"));
|
||||
Objects.requireNonNull(timeout, "timeout");
|
||||
if (topic.isBlank()) {
|
||||
throw new IllegalArgumentException("topic");
|
||||
}
|
||||
if (allowedPushTypes.isEmpty()) {
|
||||
throw new IllegalArgumentException("allowedPushTypes must not be empty");
|
||||
}
|
||||
if (timeout.isNegative() || timeout.isZero()) {
|
||||
throw new IllegalArgumentException("timeout must be positive and finite");
|
||||
}
|
||||
}
|
||||
}
|
||||
+96
@@ -0,0 +1,96 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.provider.apns;
|
||||
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.provider.http.JdkNotificationHttpGateway;
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.provider.http.NotificationHttpRequest;
|
||||
import dev.caskeleton.adapter.outbound.notification.platform.template.NotificationJsonMapper;
|
||||
import dev.caskeleton.application.notification.platform.api.content.MobilePushContent;
|
||||
import dev.caskeleton.application.notification.platform.api.error.FailureCategory;
|
||||
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureCode;
|
||||
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureDescriptor;
|
||||
import dev.caskeleton.application.notification.platform.api.error.ProviderConfigurationException;
|
||||
import dev.caskeleton.application.notification.platform.contact.ApnsDeviceToken;
|
||||
import dev.caskeleton.application.notification.platform.contact.ContactPointValue;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderSubmission;
|
||||
import java.net.URI;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.time.Clock;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
|
||||
/** Builds the APNs HTTP/2 request headers and payload. */
|
||||
public final class ApnsRequestMapper {
|
||||
|
||||
private static final String DEFAULT_PUSH_TYPE = "alert";
|
||||
|
||||
private final ApnsProviderProperties properties;
|
||||
private final Clock clock;
|
||||
|
||||
public ApnsRequestMapper(ApnsProviderProperties properties, Clock clock) {
|
||||
this.properties = Objects.requireNonNull(properties, "properties");
|
||||
this.clock = Objects.requireNonNull(clock, "clock");
|
||||
}
|
||||
|
||||
/** Map one submission, rejecting an environment or push-type mismatch first. */
|
||||
public NotificationHttpRequest map(
|
||||
ProviderSubmission submission, ContactPointValue contactPoint, String authorization) {
|
||||
Objects.requireNonNull(submission, "submission");
|
||||
Objects.requireNonNull(authorization, "authorization");
|
||||
if (!(contactPoint instanceof ApnsDeviceToken token)) {
|
||||
throw new IllegalArgumentException("APNs requires an APNs device token");
|
||||
}
|
||||
if (token.environment() != properties.environment()) {
|
||||
throw configurationFailure();
|
||||
}
|
||||
if (!(submission.content().content() instanceof MobilePushContent push)) {
|
||||
throw new IllegalArgumentException("APNs requires mobile push content");
|
||||
}
|
||||
String pushType = DEFAULT_PUSH_TYPE;
|
||||
if (!properties.allowedPushTypes().contains(pushType)) {
|
||||
throw configurationFailure();
|
||||
}
|
||||
|
||||
Map<String, Object> aps = new LinkedHashMap<>();
|
||||
aps.put("alert", Map.of("title", push.title(), "body", push.body()));
|
||||
push.presentation().sound().ifPresent(sound -> aps.put("sound", sound));
|
||||
push.presentation().badge().ifPresent(badge -> aps.put("badge", badge));
|
||||
|
||||
Map<String, Object> payload = new LinkedHashMap<>();
|
||||
payload.put("aps", aps);
|
||||
payload.putAll(push.data());
|
||||
|
||||
Map<String, String> headers = new LinkedHashMap<>();
|
||||
headers.put("authorization", authorization);
|
||||
headers.put("apns-topic", properties.topic());
|
||||
headers.put("apns-push-type", pushType);
|
||||
headers.put("apns-priority", "10");
|
||||
headers.put("apns-id", submission.attemptId().value().toString());
|
||||
submission
|
||||
.expiresAt()
|
||||
.ifPresent(
|
||||
expiry -> headers.put("apns-expiration", Long.toString(expiry.getEpochSecond())));
|
||||
submission.collapse().ifPresent(spec -> headers.put("apns-collapse-id", spec.key()));
|
||||
|
||||
byte[] body =
|
||||
NotificationJsonMapper.mapper()
|
||||
.writeValueAsString(payload)
|
||||
.getBytes(StandardCharsets.UTF_8);
|
||||
return new NotificationHttpRequest(
|
||||
"POST",
|
||||
URI.create(properties.endpoint() + "/3/device/" + token.value()),
|
||||
JdkNotificationHttpGateway.headers(headers),
|
||||
body,
|
||||
properties.timeout());
|
||||
}
|
||||
|
||||
/** Current time, exposed so expiry mapping stays testable. */
|
||||
public java.time.Instant now() {
|
||||
return clock.instant();
|
||||
}
|
||||
|
||||
private static ProviderConfigurationException configurationFailure() {
|
||||
return new ProviderConfigurationException(
|
||||
NotificationFailureDescriptor.preDispatch(
|
||||
NotificationFailureCode.PROVIDER_CONFIGURATION_INVALID, FailureCategory.AUTHORIZATION));
|
||||
}
|
||||
}
|
||||
+89
@@ -0,0 +1,89 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.provider.fcm;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.api.error.FailureCategory;
|
||||
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureCode;
|
||||
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureDescriptor;
|
||||
import dev.caskeleton.application.notification.platform.api.error.ProviderPayloadLimitException;
|
||||
import dev.caskeleton.application.notification.platform.contact.ContactPointValue;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderSubmission;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderSubmissionResult;
|
||||
import dev.caskeleton.application.notification.platform.security.AccessContext;
|
||||
import dev.caskeleton.application.notification.platform.security.ContactPointProtector;
|
||||
import java.time.Duration;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.CompletionStage;
|
||||
|
||||
/**
|
||||
* Batch submission that keeps per-recipient identity.
|
||||
*
|
||||
* <p>One transport call, many attempts. FCM returns a positional result per input, so a partial
|
||||
* failure is decomposed back to the recipient that owns it; collapsing a batch into one shared
|
||||
* outcome would mark four delivered recipients as failed because the fifth token was stale.
|
||||
*/
|
||||
public final class FcmBatchCoordinator {
|
||||
|
||||
private final FcmGateway gateway;
|
||||
private final FcmMessageMapper messageMapper;
|
||||
private final FcmTargetMapper targetMapper;
|
||||
private final FcmFailureClassifier classifier;
|
||||
private final ContactPointProtector protector;
|
||||
private final FcmProviderProperties properties;
|
||||
|
||||
public FcmBatchCoordinator(
|
||||
FcmGateway gateway,
|
||||
FcmMessageMapper messageMapper,
|
||||
FcmTargetMapper targetMapper,
|
||||
FcmFailureClassifier classifier,
|
||||
ContactPointProtector protector,
|
||||
FcmProviderProperties properties) {
|
||||
this.gateway = Objects.requireNonNull(gateway, "gateway");
|
||||
this.messageMapper = Objects.requireNonNull(messageMapper, "messageMapper");
|
||||
this.targetMapper = Objects.requireNonNull(targetMapper, "targetMapper");
|
||||
this.classifier = Objects.requireNonNull(classifier, "classifier");
|
||||
this.protector = Objects.requireNonNull(protector, "protector");
|
||||
this.properties = Objects.requireNonNull(properties, "properties");
|
||||
}
|
||||
|
||||
/** Submit a batch and return one result per input, in input order. */
|
||||
public CompletionStage<List<ProviderSubmissionResult>> submit(
|
||||
List<ProviderSubmission> submissions) {
|
||||
Objects.requireNonNull(submissions, "submissions");
|
||||
if (submissions.isEmpty()) {
|
||||
return CompletableFuture.completedFuture(List.of());
|
||||
}
|
||||
if (submissions.size() > properties.maxBatchSize()) {
|
||||
throw new ProviderPayloadLimitException(
|
||||
NotificationFailureDescriptor.preDispatch(
|
||||
NotificationFailureCode.PROVIDER_PAYLOAD_LIMIT, FailureCategory.INVALID_PAYLOAD));
|
||||
}
|
||||
|
||||
long startedNanos = System.nanoTime();
|
||||
List<Map<String, Object>> messages = new ArrayList<>(submissions.size());
|
||||
for (ProviderSubmission submission : submissions) {
|
||||
ContactPointValue value =
|
||||
protector.reveal(
|
||||
submission.contactPoint(),
|
||||
AccessContext.dispatch(submission.profile().profileId().value()));
|
||||
messages.add(messageMapper.map(submission, targetMapper.map(value)));
|
||||
}
|
||||
|
||||
FcmBatchResult batch = gateway.sendBatch(messages);
|
||||
if (batch.items().size() != submissions.size()) {
|
||||
throw new IllegalStateException("FCM returned a result count that does not match the input");
|
||||
}
|
||||
|
||||
Duration elapsed = Duration.ofNanos(System.nanoTime() - startedNanos);
|
||||
List<ProviderSubmissionResult> results = new ArrayList<>(submissions.size());
|
||||
for (FcmBatchResult.Item item : batch.items()) {
|
||||
results.add(
|
||||
item.success()
|
||||
? ProviderSubmissionResult.accepted(item.messageId().orElse(null), "SUCCESS", elapsed)
|
||||
: classifier.classify(item.errorCode().orElseThrow(), elapsed));
|
||||
}
|
||||
return CompletableFuture.completedFuture(List.copyOf(results));
|
||||
}
|
||||
}
|
||||
+35
@@ -0,0 +1,35 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.provider.fcm;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Objects;
|
||||
import java.util.Optional;
|
||||
|
||||
/** Positional result of one FCM multicast call. */
|
||||
public record FcmBatchResult(List<Item> items) {
|
||||
|
||||
public FcmBatchResult {
|
||||
items = List.copyOf(Objects.requireNonNull(items, "items"));
|
||||
}
|
||||
|
||||
/** One item result, aligned with the input index. */
|
||||
public record Item(boolean success, Optional<String> messageId, Optional<String> errorCode) {
|
||||
|
||||
public Item {
|
||||
Objects.requireNonNull(messageId, "messageId");
|
||||
Objects.requireNonNull(errorCode, "errorCode");
|
||||
if (success == errorCode.isPresent()) {
|
||||
throw new IllegalArgumentException("an item is either a success or an error, never both");
|
||||
}
|
||||
}
|
||||
|
||||
/** Successful item. */
|
||||
public static Item success(String messageId) {
|
||||
return new Item(true, Optional.ofNullable(messageId), Optional.empty());
|
||||
}
|
||||
|
||||
/** Failed item. */
|
||||
public static Item failure(String errorCode) {
|
||||
return new Item(false, Optional.empty(), Optional.of(errorCode));
|
||||
}
|
||||
}
|
||||
}
|
||||
+39
@@ -0,0 +1,39 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.provider.fcm;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.api.ContactPointId;
|
||||
import dev.caskeleton.application.notification.platform.api.TenantId;
|
||||
import dev.caskeleton.application.notification.platform.contact.ContactPointStatus;
|
||||
import dev.caskeleton.application.notification.platform.dispatch.ContactPointStorePort;
|
||||
import java.util.Objects;
|
||||
|
||||
/**
|
||||
* Applies FCM target lifecycle changes.
|
||||
*
|
||||
* <p>An {@code UNREGISTERED} response is the provider telling us the target no longer exists. Not
|
||||
* acting on it means every future notification to that user spends a provider call to learn the
|
||||
* same thing again.
|
||||
*/
|
||||
public final class FcmContactPointUpdater {
|
||||
|
||||
private final ContactPointStorePort contactPoints;
|
||||
private final FcmFailureClassifier classifier;
|
||||
|
||||
public FcmContactPointUpdater(
|
||||
ContactPointStorePort contactPoints, FcmFailureClassifier classifier) {
|
||||
this.contactPoints = Objects.requireNonNull(contactPoints, "contactPoints");
|
||||
this.classifier = Objects.requireNonNull(classifier, "classifier");
|
||||
}
|
||||
|
||||
/** Invalidate the contact point when the error code says the target is gone. */
|
||||
public boolean apply(TenantId tenantId, ContactPointId contactPointId, String errorCode) {
|
||||
Objects.requireNonNull(tenantId, "tenantId");
|
||||
Objects.requireNonNull(contactPointId, "contactPointId");
|
||||
Objects.requireNonNull(errorCode, "errorCode");
|
||||
if (!classifier.invalidatesContactPoint(errorCode)) {
|
||||
return false;
|
||||
}
|
||||
contactPoints.updateStatus(
|
||||
tenantId, contactPointId, ContactPointStatus.INVALID, "FCM_" + errorCode);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
+76
@@ -0,0 +1,76 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.provider.fcm;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.api.error.FailureCategory;
|
||||
import dev.caskeleton.application.notification.platform.api.error.NotificationFailureCode;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderFailure;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderSubmissionResult;
|
||||
import java.time.Duration;
|
||||
import java.util.Optional;
|
||||
|
||||
/**
|
||||
* FCM error codes to the stable failure vocabulary.
|
||||
*
|
||||
* <p>{@code UNREGISTERED} is the one that must never be retried: the target is gone, and repeating
|
||||
* the call cannot bring it back. It invalidates the contact point and lets routing fall back.
|
||||
*/
|
||||
public final class FcmFailureClassifier {
|
||||
|
||||
/** Classify one FCM error code. */
|
||||
public ProviderSubmissionResult classify(String errorCode, Duration elapsed) {
|
||||
return ProviderSubmissionResult.rejected(failure(errorCode), elapsed);
|
||||
}
|
||||
|
||||
/** Failure for one FCM error code. */
|
||||
public ProviderFailure failure(String errorCode) {
|
||||
return switch (errorCode) {
|
||||
case "UNREGISTERED", "INVALID_TOKEN" ->
|
||||
ProviderFailure.of(
|
||||
NotificationFailureCode.CONTACT_POINT_INVALID,
|
||||
FailureCategory.INVALID_RECIPIENT,
|
||||
false);
|
||||
case "QUOTA_EXCEEDED" ->
|
||||
ProviderFailure.of(
|
||||
NotificationFailureCode.PROVIDER_THROTTLED, FailureCategory.THROTTLED, true);
|
||||
case "UNAVAILABLE", "INTERNAL" ->
|
||||
ProviderFailure.of(
|
||||
NotificationFailureCode.PROVIDER_TRANSIENT_FAILURE,
|
||||
FailureCategory.TRANSIENT_PROVIDER,
|
||||
true);
|
||||
case "INVALID_ARGUMENT" ->
|
||||
ProviderFailure.of(
|
||||
NotificationFailureCode.VALIDATION_FAILED, FailureCategory.INVALID_PAYLOAD, false);
|
||||
case "THIRD_PARTY_AUTH_ERROR", "UNAUTHENTICATED" ->
|
||||
ProviderFailure.of(
|
||||
NotificationFailureCode.PROVIDER_AUTHENTICATION_FAILED,
|
||||
FailureCategory.AUTHENTICATION,
|
||||
false);
|
||||
case "SENDER_ID_MISMATCH" ->
|
||||
ProviderFailure.of(
|
||||
NotificationFailureCode.PROVIDER_AUTHORIZATION_FAILED,
|
||||
FailureCategory.AUTHORIZATION,
|
||||
false);
|
||||
default ->
|
||||
ProviderFailure.of(
|
||||
NotificationFailureCode.PROVIDER_PERMANENT_FAILURE,
|
||||
FailureCategory.PERMANENT_PROVIDER,
|
||||
false);
|
||||
};
|
||||
}
|
||||
|
||||
/** Whether an error code means the contact point should be invalidated. */
|
||||
public boolean invalidatesContactPoint(String errorCode) {
|
||||
return failure(errorCode).category() == FailureCategory.INVALID_RECIPIENT;
|
||||
}
|
||||
|
||||
/** Retry hint, where FCM supplies one. */
|
||||
public Optional<Duration> retryAfter(Optional<String> headerValue) {
|
||||
return headerValue.flatMap(
|
||||
value -> {
|
||||
try {
|
||||
return Optional.of(Duration.ofSeconds(Long.parseLong(value.trim())));
|
||||
} catch (NumberFormatException notSeconds) {
|
||||
return Optional.empty();
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.provider.fcm;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
/** The FCM transport seam, so batch decomposition can be tested without a live project. */
|
||||
@FunctionalInterface
|
||||
public interface FcmGateway {
|
||||
|
||||
/** Send a batch and return one positional result per message. */
|
||||
FcmBatchResult sendBatch(List<Map<String, Object>> messages);
|
||||
}
|
||||
+72
@@ -0,0 +1,72 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.provider.fcm;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.api.content.MobilePushContent;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderSubmission;
|
||||
import java.time.Clock;
|
||||
import java.time.Duration;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Optional;
|
||||
|
||||
/**
|
||||
* Builds the FCM message body.
|
||||
*
|
||||
* <p>TTL is the minimum of the remaining delivery deadline and the provider maximum. Sending the
|
||||
* provider maximum when the notification expires in ninety seconds would let FCM keep retrying a
|
||||
* message the platform has already given up on.
|
||||
*/
|
||||
public final class FcmMessageMapper {
|
||||
|
||||
private static final int MAX_PAYLOAD_BYTES = 4096;
|
||||
|
||||
private final FcmProviderProperties properties;
|
||||
private final Clock clock;
|
||||
|
||||
public FcmMessageMapper(FcmProviderProperties properties, Clock clock) {
|
||||
this.properties = Objects.requireNonNull(properties, "properties");
|
||||
this.clock = Objects.requireNonNull(clock, "clock");
|
||||
}
|
||||
|
||||
/** Message body for one submission. */
|
||||
public Map<String, Object> map(ProviderSubmission submission, FcmWireTarget target) {
|
||||
Objects.requireNonNull(submission, "submission");
|
||||
Objects.requireNonNull(target, "target");
|
||||
if (!(submission.content().content() instanceof MobilePushContent push)) {
|
||||
throw new IllegalArgumentException("FCM requires mobile push content");
|
||||
}
|
||||
|
||||
Map<String, Object> message = new LinkedHashMap<>();
|
||||
if ("FID".equals(target.kind())) {
|
||||
message.put("installation_id", target.value());
|
||||
} else {
|
||||
message.put("token", target.value());
|
||||
}
|
||||
message.put("notification", Map.of("title", push.title(), "body", push.body()));
|
||||
if (!push.data().isEmpty()) {
|
||||
message.put("data", push.data());
|
||||
}
|
||||
|
||||
Map<String, Object> android = new LinkedHashMap<>();
|
||||
android.put("ttl", ttl(submission).toSeconds() + "s");
|
||||
submission.collapse().ifPresent(spec -> android.put("collapse_key", spec.key()));
|
||||
message.put("android", android);
|
||||
|
||||
return Map.of("message", message);
|
||||
}
|
||||
|
||||
/** Effective TTL for a submission. */
|
||||
public Duration ttl(ProviderSubmission submission) {
|
||||
Optional<Duration> remaining =
|
||||
submission.expiresAt().map(expiry -> Duration.between(clock.instant(), expiry));
|
||||
return remaining
|
||||
.filter(value -> value.compareTo(properties.maxTtl()) < 0)
|
||||
.filter(value -> !value.isNegative())
|
||||
.orElse(properties.maxTtl());
|
||||
}
|
||||
|
||||
/** Payload ceiling enforced before the provider call. */
|
||||
public int maxPayloadBytes() {
|
||||
return MAX_PAYLOAD_BYTES;
|
||||
}
|
||||
}
|
||||
+73
@@ -0,0 +1,73 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.provider.fcm;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.api.ProviderId;
|
||||
import dev.caskeleton.application.notification.platform.api.routing.Channel;
|
||||
import dev.caskeleton.application.notification.platform.provider.BatchNotificationProviderAdapter;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderCapabilities;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderSubmission;
|
||||
import dev.caskeleton.application.notification.platform.provider.ProviderSubmissionResult;
|
||||
import java.util.List;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.CompletionStage;
|
||||
|
||||
/**
|
||||
* FCM adapter.
|
||||
*
|
||||
* <p>A successful send means FCM took the message. Firebase describes its own failures as handoff
|
||||
* failures, which is the clearest statement that success is a handoff and not a device delivery, so
|
||||
* the strongest evidence this adapter ever produces is {@code PROVIDER_ACCEPTED}.
|
||||
*/
|
||||
public final class FcmNotificationProviderAdapter implements BatchNotificationProviderAdapter {
|
||||
|
||||
private static final ProviderId PROVIDER_ID = new ProviderId("fcm");
|
||||
|
||||
private final FcmBatchCoordinator coordinator;
|
||||
private final FcmProviderProperties properties;
|
||||
|
||||
public FcmNotificationProviderAdapter(
|
||||
FcmBatchCoordinator coordinator, FcmProviderProperties properties) {
|
||||
this.coordinator = Objects.requireNonNull(coordinator, "coordinator");
|
||||
this.properties = Objects.requireNonNull(properties, "properties");
|
||||
}
|
||||
|
||||
@Override
|
||||
public ProviderId providerId() {
|
||||
return PROVIDER_ID;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Set<Channel> channels() {
|
||||
return Set.of(Channel.PUSH);
|
||||
}
|
||||
|
||||
@Override
|
||||
public ProviderCapabilities capabilities() {
|
||||
// deliveryReceipt is false: FCM has no server-side delivery receipt for ordinary sends, and
|
||||
// claiming one would let the runtime plan a reconciliation that can never succeed.
|
||||
return new ProviderCapabilities(
|
||||
true,
|
||||
false,
|
||||
false,
|
||||
false,
|
||||
false,
|
||||
false,
|
||||
false,
|
||||
true,
|
||||
properties.maxBatchSize(),
|
||||
4096L,
|
||||
properties.maxTtl());
|
||||
}
|
||||
|
||||
@Override
|
||||
public CompletionStage<ProviderSubmissionResult> submit(ProviderSubmission submission) {
|
||||
Objects.requireNonNull(submission, "submission");
|
||||
return coordinator.submit(List.of(submission)).thenApply(results -> results.get(0));
|
||||
}
|
||||
|
||||
@Override
|
||||
public CompletionStage<List<ProviderSubmissionResult>> submitBatch(
|
||||
List<ProviderSubmission> submissions) {
|
||||
return coordinator.submit(submissions);
|
||||
}
|
||||
}
|
||||
+35
@@ -0,0 +1,35 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.provider.fcm;
|
||||
|
||||
import java.net.URI;
|
||||
import java.time.Duration;
|
||||
import java.util.Objects;
|
||||
|
||||
/** FCM profile. Project and application identity are pinned so a target cannot cross projects. */
|
||||
public record FcmProviderProperties(
|
||||
URI endpoint,
|
||||
String projectId,
|
||||
String applicationId,
|
||||
int maxBatchSize,
|
||||
Duration maxTtl,
|
||||
Duration timeout) {
|
||||
|
||||
/** The Admin SDK multicast ceiling. */
|
||||
public static final int MAX_SUPPORTED_BATCH = 500;
|
||||
|
||||
public FcmProviderProperties {
|
||||
Objects.requireNonNull(endpoint, "endpoint");
|
||||
Objects.requireNonNull(projectId, "projectId");
|
||||
Objects.requireNonNull(applicationId, "applicationId");
|
||||
Objects.requireNonNull(maxTtl, "maxTtl");
|
||||
Objects.requireNonNull(timeout, "timeout");
|
||||
if (projectId.isBlank() || applicationId.isBlank()) {
|
||||
throw new IllegalArgumentException("projectId and applicationId must not be blank");
|
||||
}
|
||||
if (maxBatchSize < 1 || maxBatchSize > MAX_SUPPORTED_BATCH) {
|
||||
throw new IllegalArgumentException("maxBatchSize must be 1.." + MAX_SUPPORTED_BATCH);
|
||||
}
|
||||
if (timeout.isNegative() || timeout.isZero()) {
|
||||
throw new IllegalArgumentException("timeout must be positive and finite");
|
||||
}
|
||||
}
|
||||
}
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.provider.fcm;
|
||||
|
||||
import dev.caskeleton.application.notification.platform.contact.ContactPointValue;
|
||||
import dev.caskeleton.application.notification.platform.contact.FcmInstallationId;
|
||||
import dev.caskeleton.application.notification.platform.contact.LegacyFcmRegistrationToken;
|
||||
|
||||
/** Maps typed push targets to their FCM wire representation. */
|
||||
public final class FcmTargetMapper {
|
||||
|
||||
/** Wire target for a contact point value. */
|
||||
public FcmWireTarget map(ContactPointValue value) {
|
||||
return switch (value) {
|
||||
case FcmInstallationId fid -> new FcmWireTarget("FID", fid.value());
|
||||
case LegacyFcmRegistrationToken token -> new FcmWireTarget("LEGACY_TOKEN", token.value());
|
||||
default -> throw new IllegalArgumentException("FCM requires an FCM target");
|
||||
};
|
||||
}
|
||||
}
|
||||
+20
@@ -0,0 +1,20 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.provider.fcm;
|
||||
|
||||
import java.util.Objects;
|
||||
|
||||
/**
|
||||
* A target in its wire form, with the kind kept explicit.
|
||||
*
|
||||
* <p>The kind is not cosmetic: an installation id and a legacy registration token go to different
|
||||
* request fields, and flattening them would make a migration a runtime guess.
|
||||
*/
|
||||
public record FcmWireTarget(String kind, String value) {
|
||||
|
||||
public FcmWireTarget {
|
||||
Objects.requireNonNull(kind, "kind");
|
||||
Objects.requireNonNull(value, "value");
|
||||
if (value.isBlank()) {
|
||||
throw new IllegalArgumentException("value");
|
||||
}
|
||||
}
|
||||
}
|
||||
+103
@@ -0,0 +1,103 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.provider.http;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.net.http.HttpClient;
|
||||
import java.net.http.HttpRequest;
|
||||
import java.net.http.HttpResponse;
|
||||
import java.net.http.HttpTimeoutException;
|
||||
import java.time.Duration;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
|
||||
/**
|
||||
* Default gateway on the JDK HTTP client.
|
||||
*
|
||||
* <p>Redirects are never followed. A provider redirect would move a signed, credential-bearing
|
||||
* request to a host the profile never approved.
|
||||
*
|
||||
* <p>Timeout and connection-reset failures are translated into an explicit statement about whether
|
||||
* the body was committed, because that single bit is what separates a safe retry from a duplicate.
|
||||
*/
|
||||
public final class JdkNotificationHttpGateway implements NotificationHttpGateway {
|
||||
|
||||
// Restricted headers the JDK client refuses to let a caller set.
|
||||
private static final Set<String> RESTRICTED =
|
||||
Set.of("connection", "content-length", "expect", "host", "upgrade");
|
||||
|
||||
private final HttpClient client;
|
||||
|
||||
public JdkNotificationHttpGateway(Duration connectTimeout) {
|
||||
this(
|
||||
HttpClient.newBuilder()
|
||||
.followRedirects(HttpClient.Redirect.NEVER)
|
||||
.connectTimeout(Objects.requireNonNull(connectTimeout, "connectTimeout"))
|
||||
.build());
|
||||
}
|
||||
|
||||
public JdkNotificationHttpGateway(HttpClient client) {
|
||||
this.client = Objects.requireNonNull(client, "client");
|
||||
}
|
||||
|
||||
@Override
|
||||
public NotificationHttpResponse exchange(NotificationHttpRequest request) {
|
||||
Objects.requireNonNull(request, "request");
|
||||
HttpRequest.Builder builder =
|
||||
HttpRequest.newBuilder(request.uri())
|
||||
.timeout(request.timeout())
|
||||
.method(request.method(), HttpRequest.BodyPublishers.ofByteArray(request.body()));
|
||||
request
|
||||
.headers()
|
||||
.forEach(
|
||||
(name, values) -> {
|
||||
if (!RESTRICTED.contains(name)) {
|
||||
values.forEach(value -> builder.header(name, value));
|
||||
}
|
||||
});
|
||||
|
||||
try {
|
||||
HttpResponse<byte[]> response =
|
||||
client.send(builder.build(), HttpResponse.BodyHandlers.ofByteArray());
|
||||
return new NotificationHttpResponse(
|
||||
response.statusCode(), Map.copyOf(response.headers().map()), response.body());
|
||||
} catch (HttpTimeoutException timeout) {
|
||||
// The request timed out after the body was published, so the provider may well have it.
|
||||
throw new NotificationHttpTransportException("RESPONSE_TIMEOUT", true, timeout);
|
||||
} catch (IOException failure) {
|
||||
throw new NotificationHttpTransportException(
|
||||
"TRANSPORT_FAILURE", bodyWasLikelyCommitted(failure), failure);
|
||||
} catch (InterruptedException interrupted) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new NotificationHttpTransportException("INTERRUPTED", true, interrupted);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A connect failure happens before anything is written; anything else may have written the body.
|
||||
*
|
||||
* <p>The default is deliberately the pessimistic one: guessing "not committed" would turn an
|
||||
* unknown into an automatic resend.
|
||||
*/
|
||||
private static boolean bodyWasLikelyCommitted(IOException failure) {
|
||||
String message = failure.getMessage();
|
||||
if (message == null) {
|
||||
return true;
|
||||
}
|
||||
String normalized = message.toLowerCase(java.util.Locale.ROOT);
|
||||
boolean beforeSend =
|
||||
normalized.contains("connection refused")
|
||||
|| normalized.contains("unresolved")
|
||||
|| normalized.contains("no route to host")
|
||||
|| normalized.contains("connect timed out");
|
||||
return !beforeSend;
|
||||
}
|
||||
|
||||
/** Header map helper for adapters. */
|
||||
public static Map<String, List<String>> headers(Map<String, String> singleValued) {
|
||||
return singleValued.entrySet().stream()
|
||||
.collect(
|
||||
java.util.stream.Collectors.toUnmodifiableMap(
|
||||
Map.Entry::getKey, entry -> List.of(entry.getValue())));
|
||||
}
|
||||
}
|
||||
+43
@@ -0,0 +1,43 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.provider.http;
|
||||
|
||||
import java.net.URI;
|
||||
import java.util.Locale;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
|
||||
/** Endpoint validation shared by the provider profiles. */
|
||||
public final class NotificationEndpoints {
|
||||
|
||||
private static final Set<String> LOOPBACK_HOSTS = Set.of("127.0.0.1", "::1", "localhost");
|
||||
|
||||
private NotificationEndpoints() {}
|
||||
|
||||
/**
|
||||
* Require TLS, except on the loopback interface.
|
||||
*
|
||||
* <p>The exception is narrow on purpose. A plaintext provider endpoint on a routable host exposes
|
||||
* credentials and message bodies to anything on the path, which is why it is refused outright. A
|
||||
* loopback endpoint never leaves the machine, so the same reasoning does not apply — and without
|
||||
* this the contract suite could not exercise a real socket at all, which would mean the ambiguity
|
||||
* behaviour it exists to prove went untested.
|
||||
*/
|
||||
public static URI requireSecureOrLoopback(URI endpoint, String name) {
|
||||
Objects.requireNonNull(endpoint, name);
|
||||
String scheme =
|
||||
endpoint.getScheme() == null ? "" : endpoint.getScheme().toLowerCase(Locale.ROOT);
|
||||
if ("https".equals(scheme)) {
|
||||
return endpoint;
|
||||
}
|
||||
String host = endpoint.getHost() == null ? "" : endpoint.getHost().toLowerCase(Locale.ROOT);
|
||||
if ("http".equals(scheme) && LOOPBACK_HOSTS.contains(host)) {
|
||||
return endpoint;
|
||||
}
|
||||
throw new IllegalArgumentException(name + " must use https outside the loopback interface");
|
||||
}
|
||||
|
||||
/** Whether an endpoint is on the loopback interface. */
|
||||
public static boolean isLoopback(URI endpoint) {
|
||||
String host = endpoint.getHost() == null ? "" : endpoint.getHost().toLowerCase(Locale.ROOT);
|
||||
return LOOPBACK_HOSTS.contains(host);
|
||||
}
|
||||
}
|
||||
+20
@@ -0,0 +1,20 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.provider.http;
|
||||
|
||||
/**
|
||||
* The only way a provider adapter in this leaf reaches the network.
|
||||
*
|
||||
* <p>It exists as a port because the registry does not permit {@code adapter-outbound-notification
|
||||
* → adapter-outbound-httpclient}. The composition root sees both leaves and is the supported place
|
||||
* to substitute an implementation backed by the HTTP Client Platform, which brings its own TLS,
|
||||
* circuit breaker, SSRF and dynamic-target policy.
|
||||
*/
|
||||
public interface NotificationHttpGateway {
|
||||
|
||||
/**
|
||||
* Execute one request.
|
||||
*
|
||||
* @throws NotificationHttpTransportException when no response could be read; the exception states
|
||||
* whether the request body was already committed
|
||||
*/
|
||||
NotificationHttpResponse exchange(NotificationHttpRequest request);
|
||||
}
|
||||
+64
@@ -0,0 +1,64 @@
|
||||
package dev.caskeleton.adapter.outbound.notification.platform.provider.http;
|
||||
|
||||
import java.net.URI;
|
||||
import java.time.Duration;
|
||||
import java.util.Arrays;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
|
||||
/** One outbound provider HTTP request. */
|
||||
@SuppressWarnings("ArrayRecordComponent") // defensive copies on construction and on every accessor
|
||||
public record NotificationHttpRequest(
|
||||
String method, URI uri, Map<String, List<String>> headers, byte[] body, Duration timeout) {
|
||||
|
||||
public NotificationHttpRequest {
|
||||
Objects.requireNonNull(method, "method");
|
||||
Objects.requireNonNull(uri, "uri");
|
||||
Objects.requireNonNull(headers, "headers");
|
||||
Objects.requireNonNull(body, "body");
|
||||
Objects.requireNonNull(timeout, "timeout");
|
||||
if (timeout.isNegative() || timeout.isZero()) {
|
||||
throw new IllegalArgumentException("timeout must be finite and positive");
|
||||
}
|
||||
headers =
|
||||
headers.entrySet().stream()
|
||||
.collect(
|
||||
java.util.stream.Collectors.toUnmodifiableMap(
|
||||
entry -> entry.getKey().toLowerCase(Locale.ROOT),
|
||||
entry -> List.copyOf(entry.getValue())));
|
||||
body = body.clone();
|
||||
}
|
||||
|
||||
@Override
|
||||
public byte[] body() {
|
||||
return body.clone();
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean equals(Object other) {
|
||||
return other instanceof NotificationHttpRequest request
|
||||
&& method.equals(request.method)
|
||||
&& uri.equals(request.uri)
|
||||
&& headers.equals(request.headers)
|
||||
&& Arrays.equals(body, request.body)
|
||||
&& timeout.equals(request.timeout);
|
||||
}
|
||||
|
||||
@Override
|
||||
public int hashCode() {
|
||||
return Objects.hash(method, uri, headers, Arrays.hashCode(body), timeout);
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
// The URI is redacted because a Web Push endpoint is a capability URL and the request body may
|
||||
// be a rendered message.
|
||||
return "NotificationHttpRequest[method="
|
||||
+ method
|
||||
+ ", uri=redacted, bytes="
|
||||
+ body.length
|
||||
+ "]";
|
||||
}
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user