feat: jpa, messaging, notification, mongo, graphql 어댑터터 리펙토링

This commit is contained in:
DongHyeonka
2026-08-18 10:59:56 +09:00
parent 2f5d2fc219
commit e98b56eb03
372 changed files with 25131 additions and 20357 deletions
+574
View File
@@ -0,0 +1,574 @@
#!/usr/bin/env bash
# The dynamic half of the Compose contract: the lanes actually run.
#
# One entry point, because the order below is what makes a lane's result mean anything and every
# step in it exists because skipping it produced a false green somewhere:
#
# 1. a unique project and a fresh evidence directory, so two runs cannot read each other's results
# and a stale directory cannot be mistaken for this run's;
# 2. the static contract, then `config`, then `create` — a stack that cannot render has no partial
# failure mode, and finding that out after `up` costs a teardown;
# 3. `up --wait` for long-running services only, then health and the resolved activation report,
# which is the application's answer rather than the flags this script passed in;
# 4. every one-shot the lane declares, each of which must exit zero — a missing or skipped
# required client is a lane failure, not a lane that had nothing to check;
# 5. sanitized evidence;
# 6. teardown scoped to this project alone, on success and failure alike, after the logs are
# collected rather than before.
#
# CI calls this script. Inlining a subset of these commands into a workflow is how a lane ends up
# running without its one-shots and reporting green.
set -euo pipefail
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
CONTRACTS="${REPO_ROOT}/src/config/runtime/compose-profile-contracts.json"
EVIDENCE_ROOT="${REPO_ROOT}/src/app-bootstrap/build/evidence/runtime-smoke"
RUN_ID="$(date -u +%Y%m%dT%H%M%SZ)-$$"
# Compose project names accept lowercase alphanumerics, hyphens and underscores only, so the run id
# is lowercased for the project while the evidence directory keeps the readable timestamp.
PROJECT_RUN_ID="$(tr '[:upper:]' '[:lower:]' <<<"${RUN_ID}")"
MODE=""
SELECTED_LANE=""
usage() {
cat >&2 <<'USAGE'
usage:
run-compose-runtime-smoke.sh --matrix <contracts.json> run every blocking lane, zero-skip
run-compose-runtime-smoke.sh --lane <id> reproduce one lane; not a matrix run
--lane exists to reproduce a failure, and a green single lane is not evidence that the matrix
passes. Only --matrix is.
USAGE
exit 64
}
while [[ $# -gt 0 ]]; do
case "$1" in
--matrix) MODE="matrix"; CONTRACTS="$2"; shift 2 ;;
--lane) MODE="lane"; SELECTED_LANE="$2"; shift 2 ;;
*) usage ;;
esac
done
[[ -n "${MODE}" ]] || usage
for tool in jq docker; do
command -v "${tool}" >/dev/null 2>&1 || { echo "${tool} is required" >&2; exit 78; }
done
[[ -r "${CONTRACTS}" ]] || { echo "missing ${CONTRACTS}" >&2; exit 78; }
# ---- per-lane state, cleaned up by the trap ---------------------------------
PROJECT=""
LANE_TMP=""
LANE_EVIDENCE=""
LANE_ARGS=()
cleanup_lane() {
local status=$?
if [[ -n "${PROJECT}" ]]; then
# Logs first. A teardown that runs before the logs are collected destroys the only description
# of why the lane failed.
if [[ -n "${LANE_EVIDENCE}" ]]; then
docker compose -p "${PROJECT}" "${LANE_ARGS[@]}" logs --no-color >"${LANE_EVIDENCE}/compose.log" 2>&1 || true
docker compose -p "${PROJECT}" "${LANE_ARGS[@]}" ps --format json >"${LANE_EVIDENCE}/containers.json" 2>&1 || true
redact_in_place "${LANE_EVIDENCE}/compose.log"
# The warning summary belongs here, after the log it summarises exists.
#
# It used to run at the end of run_lane, which is before this trap collects compose.log — so it
# grepped a file that had not been written yet and produced an empty warnings.log for every
# lane. Fifteen lanes reported zero warnings while their logs held up to three each, and the
# one artifact Wave 4's gate reads was the one that could not see them.
grep -E ' (WARN|ERROR) ' "${LANE_EVIDENCE}/compose.log" 2>/dev/null \
>"${LANE_EVIDENCE}/warnings.log" || true
fi
# This project only. Never a bare `down`, which would take out whatever else the developer has
# running, and never a volume outside it.
docker compose -p "${PROJECT}" "${LANE_ARGS[@]}" down --volumes --remove-orphans >/dev/null 2>&1 || true
fi
[[ -n "${LANE_TMP}" && -d "${LANE_TMP}" ]] && rm -rf "${LANE_TMP}"
# .env.lane is gone — the lane's values live in the generated overlay now — but a file left by an
# older revision of this script would still be read by the local overlay's env_file list.
rm -f "${REPO_ROOT}/src/.env.lane" "${REPO_ROOT}/infra/keycloak/secrets/graphql-smoke-client-secret"
# The lane keypair, including the CA's private half. A certificate that outlives the run it was
# made for is a credential nobody is tracking.
rm -f "${REPO_ROOT}"/infra/postgres/tls/ca.key "${REPO_ROOT}"/infra/postgres/tls/ca.crt \
"${REPO_ROOT}"/infra/postgres/tls/ca.srl "${REPO_ROOT}"/infra/postgres/tls/server.key \
"${REPO_ROOT}"/infra/postgres/tls/server.crt "${REPO_ROOT}"/infra/postgres/tls/server.csr \
"${REPO_ROOT}"/infra/mailpit/tls/server.key "${REPO_ROOT}"/infra/mailpit/tls/server.crt
return "${status}"
}
# Secrets must not survive into an artifact. The generated values are known here, so they are
# replaced by name rather than by guessing at what a secret looks like.
redact_in_place() {
local file="$1"
[[ -r "${file}" ]] || return 0
local value
for value in "${GENERATED_SECRETS[@]:-}"; do
[[ -n "${value}" ]] && sed -i "s|${value}|<redacted>|g" "${file}"
done
}
random_secret() { head -c 32 /dev/urandom | base64 | tr -d '=+/' | cut -c1-32; }
# Key material, which is not the same thing as a password. The notification platform base64-decodes
# each of its eight keys and requires at least 32 bytes, so random_secret's alphanumeric 32
# characters decode to 24 and are rejected. This keeps the padding and the full alphabet.
random_key() { head -c 32 /dev/urandom | base64 | tr -d '\n'; }
run_lane() {
local lane_json="$1"
local id profile runtime
id="$(jq -r '.id' <<<"${lane_json}")"
profile="$(jq -r '.composeProfile // empty' <<<"${lane_json}")"
runtime="$(jq -r '.springRuntime' <<<"${lane_json}")"
PROJECT="casmoke-${id}-${PROJECT_RUN_ID}"
LANE_EVIDENCE="${EVIDENCE_ROOT}/${id}/${RUN_ID}"
if [[ -e "${LANE_EVIDENCE}" ]]; then
echo "${id}: ${LANE_EVIDENCE} already exists; refusing to write into a previous run's evidence" >&2
return 1
fi
mkdir -p "${LANE_EVIDENCE}"
LANE_TMP="$(mktemp -d)"
chmod 700 "${LANE_TMP}"
LANE_ARGS=()
local key file
while read -r key; do
file="$(jq -r --arg k "${key}" '.composeFiles[$k]' "${CONTRACTS}")"
LANE_ARGS+=(-f "${REPO_ROOT}/${file}")
done < <(jq -r '.files[]' <<<"${lane_json}")
[[ -n "${profile}" ]] && LANE_ARGS+=(--profile "${profile}")
# A fresh client secret per run, mode 0600, removed on teardown. The realm references it by name.
local secret_dir="${REPO_ROOT}/infra/keycloak/secrets"
local secret_file="${secret_dir}/graphql-smoke-client-secret"
mkdir -p "${secret_dir}"
local client_secret; client_secret="$(random_secret)"
GENERATED_SECRETS+=("${client_secret}")
(umask 077 && printf '%s' "${client_secret}" >"${secret_file}")
# A CA and a server certificate for the host name `db`, when the lane brings the TLS overlay.
#
# Generated per run and removed on teardown, like the realm secret: a committed test certificate is
# a private key in Git. The host name matters — the prod runtime connects with sslmode=verify-full,
# which checks the certificate against the name it dialled, so a certificate for anything but `db`
# fails exactly as a redirected connection would. That is the check, working.
local tls_dir="${REPO_ROOT}/infra/postgres/tls"
if jq -e '.files | index("tls")' <<<"${lane_json}" >/dev/null; then
mkdir -p "${tls_dir}"
(
umask 077
openssl req -x509 -newkey rsa:2048 -sha256 -days 1 -nodes \
-keyout "${tls_dir}/ca.key" -out "${tls_dir}/ca.crt" \
-subj "/CN=ca-skeleton-lane-ca" >/dev/null 2>&1
openssl req -newkey rsa:2048 -nodes \
-keyout "${tls_dir}/server.key" -out "${tls_dir}/server.csr" \
-subj "/CN=db" >/dev/null 2>&1
openssl x509 -req -in "${tls_dir}/server.csr" -sha256 -days 1 \
-CA "${tls_dir}/ca.crt" -CAkey "${tls_dir}/ca.key" -CAcreateserial \
-extfile <(printf 'subjectAltName=DNS:db\nextendedKeyUsage=serverAuth\n') \
-out "${tls_dir}/server.crt" >/dev/null 2>&1
)
# The CA certificate is public and is mounted as a Compose secret, which preserves the source
# file's mode; 0600 would be unreadable to the application's non-root user.
chmod 0644 "${tls_dir}/ca.crt"
rm -f "${tls_dir}/server.csr"
[[ -s "${tls_dir}/server.crt" ]] || { echo "${id}: could not generate the lane certificate" >&2; return 1; }
fi
# A certificate for the host name `mailpit`, when the lane brings the reference SMTP relay.
#
# Same rule as the PostgreSQL one above and for the same reason: the notification platform's
# transport type has no plaintext member, so a relay without a certificate is one the platform
# cannot be configured to talk to. Generated per run, removed on teardown.
local mail_tls="${REPO_ROOT}/infra/mailpit/tls"
if jq -e '.services | index("mailpit")' <<<"${lane_json}" >/dev/null; then
mkdir -p "${mail_tls}"
(
umask 077
openssl req -x509 -newkey rsa:2048 -sha256 -days 1 -nodes \
-keyout "${mail_tls}/server.key" -out "${mail_tls}/server.crt" \
-subj "/CN=mailpit" \
-addext "subjectAltName=DNS:mailpit" \
-addext "extendedKeyUsage=serverAuth" >/dev/null 2>&1
)
# Mailpit runs as a non-root user and reads both halves; 0600 would be unreadable to it.
chmod 0644 "${mail_tls}/server.crt" "${mail_tls}/server.key"
[[ -s "${mail_tls}/server.crt" ]] || { echo "${id}: could not generate the Mailpit certificate" >&2; return 1; }
fi
# The values a deployment must supply: the seven with no inline default, and the secrets the prod
# env validator requires to be non-blank. A lane supplies them explicitly rather than inheriting a
# developer's src/.env, which is the difference between a lane that reproduces anywhere and one
# that reproduces on the machine it was written on. All generated per run and removed on teardown.
#
# The database credentials are additionally *exported*, because two different mechanisms have to
# agree on them. The application reads this file; the `db` service takes its POSTGRES_USER and
# POSTGRES_PASSWORD from Compose interpolation — `${APP_DATASOURCE_PASSWORD:-ca_skeleton}` — and
# interpolation reads the process environment and the project .env file, never a service's
# env_file. So the application got the generated password and the database got the literal
# default, and no lane whose overlay did not happen to restate the value could ever authenticate.
# shared-infra-local passed only because the local overlay restates it, which made the dev lane's
# failure look like a dev-specific problem rather than the general one it is.
export POSTGRES_DB=ca_skeleton
export APP_DATASOURCE_USERNAME=ca_skeleton
export APP_DATASOURCE_PASSWORD="${client_secret}"
# Which phase the notification client runs. Exported rather than defaulted in Compose, because a
# default made local-notification-serving run the ingest assertions and pass without ever checking
# a delivery.
export NOTIFICATION_SMOKE_PHASE
NOTIFICATION_SMOKE_PHASE="$(jq -r '.notificationSmokePhase // ""' <<<"${lane_json}")"
# The capability schema streams this lane installs, exported for the same reason as the credentials
# above: the migration containers take them through Compose interpolation, which reads the process
# environment and never a lane overlay. They are deliberately not part of activationEnv — the
# application does not read this value, the two one-shots that install and promote the schema do,
# and putting an operator input in the application's environment block is how it would come to look
# like a switch the application honours.
export CAPABILITY_STREAMS
CAPABILITY_STREAMS="$(jq -r '.capabilityStreams // [] | join(" ")' <<<"${lane_json}")"
# The public health address, which is not the same in every runtime: application-local.yml pins
# presentation.api-base-path to /api while the shipped default is /v1. The smoke client used to
# hardcode /api, so it verified the local runtime and 404'd against a dev application that had
# started perfectly.
if [[ "${runtime}" == "local" ]]; then
export APP_HEALTH_PATH=/api/healthcheck
else
export APP_HEALTH_PATH=/v1/healthcheck
fi
# The notification platform's eight purpose-scoped keys.
#
# All eight, always, for every lane — not only the notification ones. They cost nothing when the
# capability is off, because the platform binds nothing at all then, and a lane that supplies only
# the keys it currently needs is a lane that breaks the moment a switch is added to it.
#
# Distinct by construction: the platform refuses to start if two purposes carry the same material,
# which is the check that stops one leaked key from being all eight. Each is redacted from the
# evidence like every other generated value.
local -a notification_keys=()
local purpose
for purpose in CONTACT_ENCRYPTION CONTACT_LOOKUP_HMAC CALLBACK_SIGNING PROVIDER_CREDENTIAL \
PAYLOAD_ENCRYPTION VAPID_SIGNING PROVIDER_REQUEST_LOOKUP_HMAC \
CALLBACK_FINGERPRINT_HMAC; do
local value; value="$(random_key)"
GENERATED_SECRETS+=("${value}")
notification_keys+=("APP_NOTIFICATION_PLATFORM_${purpose}_KEY=${value}")
# The id, which is not secret and is deliberately tied to this run. A lane that reused a fixed
# id across runs would be asserting the one thing the platform refuses to assume: that material
# and id change together.
notification_keys+=("APP_NOTIFICATION_PLATFORM_${purpose}_KEY_ID=lane-${PROJECT_RUN_ID}-$(echo "${purpose}" | tr '[:upper:]_' '[:lower:]-')")
done
# Every deployment-supplied value the lane owns, collected here and written into the generated
# overlay's `environment:` block below rather than into an env file.
#
# These lived in src/.env.lane, and an env file cannot win. env_file lists merge across overlays
# and the later one takes precedence, so the developer's optional src/.env — declared by the local
# overlay, after the base — silently replaced them. That is not hypothetical: it pinned the JWT
# issuer to http://localhost:8081, which inside the app container is the application itself, so the
# application fetched JWKS from its own port and answered AUTH_JWKS_UNAVAILABLE to every
# authenticated request while the lane had supplied the correct issuer all along.
#
# An `environment:` block outranks every env_file regardless of order, which is the same fix the
# lane's activation switches already use. The cost is that these values appear in `docker inspect`
# for this run's own throwaway project; they are generated per run, removed on teardown, and
# redacted from every evidence file by name. A lane whose settings are silently discarded is worse.
local -a lane_environment=(
"APP_NAME=ca-skeleton-${id}"
# APP_DATASOURCE_URL is deliberately absent. The environment overlays own it and they do not
# agree by accident: the local one is a plain jdbc:postgresql URL and the prod-smoke one carries
# sslmode=verify-full and a CA path, because the prod profile refuses to start without them.
# Setting it here overrode both — this overlay is appended last — and off-prod, shared-infra-dev
# and prod-smoke failed the profile check on a TLS requirement the stack had satisfied.
"APP_DATASOURCE_USERNAME=${APP_DATASOURCE_USERNAME}"
"APP_DATASOURCE_PASSWORD=${APP_DATASOURCE_PASSWORD}"
"APP_SECURITY_JWT_ISSUER=http://keycloak:8080/realms/ca-skeleton"
"APP_SECURITY_JWT_AUDIENCE=ca-skeleton-api"
"SPRING_PROFILES_ACTIVE=${runtime}"
"${notification_keys[@]}"
)
local generated
for generated in APP_SECURITY_JWT_SIGNING_KEY APP_SECURITY_OAUTH_CLIENT_SECRET \
APP_EXTERNAL_API_KEY APP_PRIVACY_PSEUDONYMIZATION_SALT; do
local secret; secret="$(random_secret)"
GENERATED_SECRETS+=("${secret}")
lane_environment+=("${generated}=${secret}")
done
# The lane's own settings go into a generated overlay rather than into that file.
#
# env_file lists merge across overlays and the later file wins, so the developer's optional
# src/.env — declared by the local overlay, after the base — silently overrode the lane's values.
# local-messaging started Kafka, set APP_MESSAGING_BROKER=kafka, and was refused by the dependency
# validator for a broker it had supplied, because an empty value from a file nobody mentioned won
# the merge. An `environment:` block beats every env_file regardless of order, so the lane's
# settings stop depending on where a file happens to sit in the stack.
#
# The overlay adds no service, so the set the static verifier checks is unchanged.
# Written as a function, because a handoff lane rewrites it between phases: the same project and
# the same database, with the application recreated under a different activation environment.
local lane_overlay="${LANE_TMP}/lane-overrides.yml"
write_lane_overlay() {
local phase_override="${1:-{\}}"
(
umask 077
{
echo "services:"
echo " app:"
echo " environment:"
local entry
for entry in "${lane_environment[@]}"; do
printf ' %s: "%s"\n' "${entry%%=*}" "${entry#*=}"
done
jq -r --argjson override "${phase_override}" \
'.activationEnv + $override | to_entries[] | " \(.key): \"\(.value)\""' \
<<<"${lane_json}"
} >"${lane_overlay}"
)
}
write_lane_overlay "$(jq -c '.handoff.firstPhaseEnv // {}' <<<"${lane_json}")"
LANE_ARGS+=(-f "${lane_overlay}")
GENERATED_SECRETS+=("${client_secret}")
echo "== ${id} (project ${PROJECT}, runtime ${runtime})"
# 2. static contract, then render, then create
"${REPO_ROOT}/scripts/verify-compose-profile-contracts.sh" >"${LANE_EVIDENCE}/contract.log" 2>&1
( cd "${REPO_ROOT}" && docker compose -p "${PROJECT}" "${LANE_ARGS[@]}" config >"${LANE_EVIDENCE}/rendered.yml" )
redact_in_place "${LANE_EVIDENCE}/rendered.yml"
# Build before create. The app service declares both `build:` and `image:`, so Compose reuses a
# matching tag if one is lying around — and the first run of this lane did exactly that, starting
# a jar built from a different state of the repository and failing on a class that no longer
# exists in the tree. A lane that runs a stale image produces evidence about code nobody changed.
( cd "${REPO_ROOT}" && docker compose -p "${PROJECT}" "${LANE_ARGS[@]}" build app \
>"${LANE_EVIDENCE}/build.log" 2>&1 )
( cd "${REPO_ROOT}" && docker compose -p "${PROJECT}" "${LANE_ARGS[@]}" create >/dev/null )
# 3a. Anything that has to happen before the application starts, in the order the lane lists it.
#
# A capability schema stream is the case this exists for. Each stream keeps its own Flyway history
# table and is promoted to ACTIVE deliberately after installation, and the application refuses to
# start until that promotion has happened — so running it with the other one-shots, after `up`,
# would run it after the thing it is a precondition for. The distinction is the lane's, not this
# script's guess: preStartServices run here, oneShotServices run after the readiness poll.
#
# Read on fd 4 with stdin closed, for the reason the lane loop at the bottom of this file already
# records: `docker compose run` consumes stdin, so a plain `while read` loop over several services
# runs the first one and then finds its input exhausted. That is not hypothetical here — it is how
# this lane ran its migration and silently skipped its promotion, and how shared-infra-local,
# shared-infra-dev and prod-smoke each ran auth-smoke and skipped minio-init and
# object-storage-smoke while reporting green. The evidence directories showed one log where the
# contract named three.
local pre_start
while read -r -u 4 pre_start; do
[[ -z "${pre_start}" ]] && continue
echo "-- ${id}: ${pre_start} (before start)"
if ! ( cd "${REPO_ROOT}" && docker compose -p "${PROJECT}" "${LANE_ARGS[@]}" run --rm "${pre_start}" </dev/null ) \
>"${LANE_EVIDENCE}/${pre_start}.log" 2>&1; then
redact_in_place "${LANE_EVIDENCE}/${pre_start}.log"
echo "${id}: ${pre_start} did not succeed; see ${pre_start}.log" >&2
return 1
fi
redact_in_place "${LANE_EVIDENCE}/${pre_start}.log"
done 4< <(jq -r '.preStartServices[]?' <<<"${lane_json}")
# 3. long-running services only. A one-shot is not a --wait target: it is meant to exit.
local long_running
long_running="$(jq -r '.longRunningServices[]' <<<"${lane_json}")"
# shellcheck disable=SC2086
( cd "${REPO_ROOT}" && docker compose -p "${PROJECT}" "${LANE_ARGS[@]}" up --wait --wait-timeout 300 ${long_running} )
await_application_report() {
# The application's own answer, not this script's input.
#
# The management port, not the application port. Actuator runs on its own connector here so the
# management plane is not published on the public one, and fetching 8080/actuator returned an
# empty file that looked exactly like a failed assertion about the profile.
local management_port
management_port="$(docker compose -p "${PROJECT}" "${LANE_ARGS[@]}" exec -T app \
sh -c 'echo "${MANAGEMENT_SERVER_PORT:-9001}"' 2>/dev/null | tr -d "\r\n")"
management_port="${management_port:-9001}"
# Poll rather than trust `up --wait`. Not every overlay defines a healthcheck — the dev one does
# not — and where there is none `--wait` returns as soon as the container is created, so the first
# fetch landed before the application had finished starting and produced an empty file that read
# exactly like a failed assertion about the profile.
local attempt
for attempt in $(seq 1 60); do
if docker compose -p "${PROJECT}" "${LANE_ARGS[@]}" exec -T app \
wget -qO- "http://localhost:${management_port}/actuator/adapteractivation" \
>"${LANE_EVIDENCE}/activation.json" 2>/dev/null \
&& [[ -s "${LANE_EVIDENCE}/activation.json" ]]; then
break
fi
sleep 5
done
docker compose -p "${PROJECT}" "${LANE_ARGS[@]}" exec -T app \
wget -qO- "http://localhost:${management_port}/actuator/health" \
>"${LANE_EVIDENCE}/health.json" 2>/dev/null || true
local reported
reported="$(jq -r '.activeProfile // empty' "${LANE_EVIDENCE}/activation.json" 2>/dev/null || true)"
if [[ "${reported}" != "${runtime}" ]]; then
echo "${id}: the application reports profile '${reported}', the contract says '${runtime}'" >&2
return 1
fi
# The switches the application says are on must be exactly the ones the lane asked for. Without
# this a lane called local-jpa that ran with JPA off would render the right services, start
# cleanly, and prove nothing — which is the failure mode a green lane is least able to reveal.
local expected_on actual_on
expected_on="$(jq -r '.expectedSwitchesOn | sort | join(",")' <<<"${lane_json}")"
actual_on="$(jq -r '.switches | to_entries | map(select(.value)) | map(.key) | sort | join(",")' \
"${LANE_EVIDENCE}/activation.json")"
if [[ "${expected_on}" != "${actual_on}" ]]; then
echo "${id}: the application reports [${actual_on}] on, the contract expects [${expected_on}]" >&2
return 1
fi
# The vendor the application resolved, not the one the lane started. These differ exactly when a
# profile pins a datasource that outranks the lane's environment, and every other field in the
# report looks correct while it happens.
local expected_vendor actual_vendor
expected_vendor="$(jq -r '.expectedPersistenceVendor' <<<"${lane_json}")"
actual_vendor="$(jq -r '.persistenceVendor // "absent"' "${LANE_EVIDENCE}/activation.json")"
if [[ "${expected_vendor}" != "${actual_vendor}" ]]; then
echo "${id}: the application resolved vendor '${actual_vendor}', the contract expects '${expected_vendor}'" >&2
return 1
fi
}
await_application_report || return 1
# 3b. The handoff: one project, one database, the application recreated under the second phase.
#
# Not two lanes glued together. The evidence a handoff needs is that a request accepted while the
# platform was INGEST_ONLY is delivered exactly once after a restart into SERVING — on the route
# frozen at accept — and that is only evidence if the row is the same row. Tearing the volume down
# between phases, or copying rows into a second project, proves nothing about a handoff.
if jq -e '.handoff' <<<"${lane_json}" >/dev/null; then
local first_smoke second_smoke
first_smoke="$(jq -r '.handoff.firstPhaseSmoke' <<<"${lane_json}")"
second_smoke="$(jq -r '.handoff.secondPhaseSmoke' <<<"${lane_json}")"
echo "-- ${id}: ${first_smoke} (handoff phase 1)"
if ! ( cd "${REPO_ROOT}" && NOTIFICATION_SMOKE_PHASE="${first_smoke}" \
docker compose -p "${PROJECT}" "${LANE_ARGS[@]}" run --rm notification-smoke </dev/null ) \
>"${LANE_EVIDENCE}/notification-smoke-phase1.log" 2>&1; then
redact_in_place "${LANE_EVIDENCE}/notification-smoke-phase1.log"
echo "${id}: the first handoff phase did not succeed" >&2
return 1
fi
redact_in_place "${LANE_EVIDENCE}/notification-smoke-phase1.log"
cp "${LANE_EVIDENCE}/activation.json" "${LANE_EVIDENCE}/activation-phase1.json"
# The application only. The database and its volume stay up, which is the whole point.
echo "-- ${id}: restarting the application into the second phase"
( cd "${REPO_ROOT}" && docker compose -p "${PROJECT}" "${LANE_ARGS[@]}" stop app >/dev/null 2>&1 )
write_lane_overlay '{}'
( cd "${REPO_ROOT}" && docker compose -p "${PROJECT}" "${LANE_ARGS[@]}" \
up -d --force-recreate --wait --wait-timeout 300 app )
await_application_report || return 1
fi
# 3c. The startup log carries no WARN and no ERROR.
#
# Wave 4's condition is about startup specifically, and a summary of the whole lane cannot express
# it: the lanes that exercise authentication end with a deliberate malformed-token probe, and the
# application answering it with a WARN is the security control working. Splitting at the line where
# the application reports it has started is what makes "startup is silent" checkable instead of
# argued — and it is checked here, per lane, rather than read off an artifact afterwards.
#
# StartupWarningZeroTest asserts the same property in-process for the all-off composition. This is
# the same assertion against a real container with the lane's switches actually on.
local startup_log="${LANE_EVIDENCE}/startup.log"
docker compose -p "${PROJECT}" "${LANE_ARGS[@]}" logs --no-color app \
>"${startup_log}" 2>&1 || true
redact_in_place "${startup_log}"
local startup_noise
startup_noise="$(awk '/Started CaSkeletonApplication/{exit} /(WARN|ERROR)/{print}' "${startup_log}")"
if [[ -n "${startup_noise}" ]]; then
echo "${id}: the startup log is not silent:" >&2
echo "${startup_noise}" | head -20 >&2
return 1
fi
# 4. every declared one-shot, each of which must exit zero
#
# fd 4 and a closed stdin, same reason as the pre-start loop above. And the count is checked rather
# than trusted: this loop ran exactly one one-shot per lane for its whole life, so the three lanes
# that declare three clients verified object storage in none of their runs and said so in neither
# their output nor their exit status. A loop that silently does less than the contract asks is the
# one failure a green lane cannot reveal, so the arithmetic is now part of the lane.
local one_shot ran_one_shots=0 declared_one_shots
declared_one_shots="$(jq -r '.oneShotServices // [] | length' <<<"${lane_json}")"
while read -r -u 4 one_shot; do
[[ -z "${one_shot}" ]] && continue
echo "-- ${id}: ${one_shot}"
if ! ( cd "${REPO_ROOT}" && docker compose -p "${PROJECT}" "${LANE_ARGS[@]}" run --rm "${one_shot}" </dev/null ) \
>"${LANE_EVIDENCE}/${one_shot}.log" 2>&1; then
redact_in_place "${LANE_EVIDENCE}/${one_shot}.log"
echo "${id}: ${one_shot} did not exit zero; see ${one_shot}.log" >&2
return 1
fi
redact_in_place "${LANE_EVIDENCE}/${one_shot}.log"
ran_one_shots=$((ran_one_shots + 1))
done 4< <(jq -r '.oneShotServices[]?' <<<"${lane_json}")
if [[ "${ran_one_shots}" -ne "${declared_one_shots}" ]]; then
echo "${id}: ran ${ran_one_shots} of ${declared_one_shots} one-shots; a partial lane is not a pass" >&2
return 1
fi
# 5. the manifest ties the evidence to the exact stack that produced it
jq -n \
--arg lane "${id}" --arg run "${RUN_ID}" --arg project "${PROJECT}" \
--arg runtime "${runtime}" --arg profile "${profile}" \
--argjson services "$(jq '.services' <<<"${lane_json}")" \
'{lane:$lane, runId:$run, project:$project, springRuntime:$runtime,
composeProfile:(if $profile=="" then null else $profile end), services:$services}' \
>"${LANE_EVIDENCE}/manifest.json"
echo "== ${id}: passed"
}
GENERATED_SECRETS=()
FAILED_LANES=()
if [[ "${MODE}" == "lane" ]]; then
lane_json="$(jq -c --arg id "${SELECTED_LANE}" '.lanes[] | select(.id == $id)' "${CONTRACTS}")"
[[ -n "${lane_json}" ]] || { echo "no lane '${SELECTED_LANE}' in ${CONTRACTS}" >&2; exit 64; }
trap cleanup_lane EXIT
run_lane "${lane_json}"
echo "one lane reproduced. A green single lane is not evidence that the matrix passes."
exit 0
fi
blocking="$(jq -c '.lanes[] | select(.blocking == true)' "${CONTRACTS}")"
total="$(wc -l <<<"${blocking}" | tr -d ' ')"
echo "running ${total} blocking lane(s), zero-skip"
# Read on fd 3, not stdin. `docker compose exec` and friends consume stdin, and inside a plain
# `while read` loop they ate the remaining lanes — so the first lane ran, the loop ended, and the
# script reported that all six had passed. A wrapper whose own success message is a false green is
# worse than no wrapper, and nothing else here would have caught it: the count came from the
# contract and the exit status from the one lane that did run.
RAN_LANES=0
while read -r -u 3 lane_json; do
[[ -z "${lane_json}" ]] && continue
RAN_LANES=$((RAN_LANES + 1))
if ( trap cleanup_lane EXIT; run_lane "${lane_json}" ); then :; else
FAILED_LANES+=("$(jq -r '.id' <<<"${lane_json}")")
fi
done 3<<<"${blocking}"
if [[ "${RAN_LANES}" -ne "${total}" ]]; then
echo "run-compose-runtime-smoke: ran ${RAN_LANES} of ${total} lanes; a partial matrix is not a pass" >&2
exit 1
fi
if [[ ${#FAILED_LANES[@]} -gt 0 ]]; then
echo "run-compose-runtime-smoke: ${#FAILED_LANES[@]} lane(s) failed: ${FAILED_LANES[*]}" >&2
exit 1
fi
echo "run-compose-runtime-smoke: all ${total} blocking lanes passed; evidence under ${EVIDENCE_ROOT}"