diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index cc86500c7..e2e5ceb2e 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -5,6 +5,9 @@ on: branches: [main] pull_request: +permissions: + contents: read + jobs: changes: name: Detect changes @@ -13,6 +16,7 @@ jobs: pull-requests: read outputs: go: ${{ steps.filter.outputs.go }} + installer: ${{ steps.filter.outputs.installer }} steps: - uses: actions/checkout@v6 - uses: dorny/paths-filter@v3 @@ -26,6 +30,22 @@ jobs: - '.golangci.yml' - 'Makefile' - '.github/workflows/**' + installer: + - 'install.sh' + - 'tests/install_next_test.sh' + - '.github/workflows/ci.yml' + + installer: + name: Installer + needs: changes + if: needs.changes.outputs.installer == 'true' + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v6 + - name: Check installer scripts + run: | + shellcheck -s sh install.sh tests/install_next_test.sh + sh tests/install_next_test.sh lint: name: Lint @@ -38,14 +58,7 @@ jobs: - uses: actions/setup-go@v5 with: go-version: "1.27.x" - - - name: Cache Go modules - uses: actions/cache@v5 - with: - path: ~/go/pkg/mod - key: ${{ runner.os }}-go-mod-${{ hashFiles('**/go.sum') }} - restore-keys: | - ${{ runner.os }}-go-mod- + cache: true - name: Run golangci-lint uses: golangci/golangci-lint-action@v9 @@ -63,22 +76,7 @@ jobs: - uses: actions/setup-go@v5 with: go-version: "1.27.x" - - - name: Cache Go modules - uses: actions/cache@v5 - with: - path: ~/go/pkg/mod - key: ${{ runner.os }}-go-mod-${{ hashFiles('**/go.sum') }} - restore-keys: | - ${{ runner.os }}-go-mod- - - - name: Cache Go build cache - uses: actions/cache@v5 - with: - path: ~/.cache/go-build - key: ${{ runner.os }}-go-build-${{ hashFiles('**/go.sum') }} - restore-keys: | - ${{ runner.os }}-go-build- + cache: true - name: Run tests run: go test -race -v ./... diff --git a/.goreleaser.yaml b/.goreleaser.yaml index e7ebe8ea8..c5168d61f 100644 --- a/.goreleaser.yaml +++ b/.goreleaser.yaml @@ -23,7 +23,7 @@ builds: env: - CGO_ENABLED=0 ldflags: - - -s -w -X main.version={{.Version}} -X main.commit={{.Commit}} -X main.date={{.Date}} + - -s -w -X main.version={{.Version}} -X main.commit={{.Commit}} -X main.date={{.Date}} -X main.dirty={{.IsGitDirty}} # Docker builds disabled for v2.0.0 release (will add back with proper buildx setup) # dockers: diff --git a/.mockery.yaml b/.mockery.yaml index 32b878d62..8d82fee07 100644 --- a/.mockery.yaml +++ b/.mockery.yaml @@ -17,20 +17,28 @@ packages: EventSubscriber: EventHandler: SecretProvider: - EnvLoader: ContainerLogWriter: - AttachmentConfigProvider: + Metrics: + LogExporter: + ContainerLogStreamer: TokenStore: - DomainSecretStore: + AppState: + AppStateReader: + ImageResolver: + SecretWriter: RateLimiter: BackupStorage: - RouteChecker: HTTPChallengeSink: PublicCertificateIssuer: CertificateStore: SecretResolver: CloudflareZoneResolver: CertificateAuthority: + AppTrafficRefresher: + PruneProtectionStore: + PruneRuntime: + GCBarrier: + GCLease: github.com/bnema/gordon/internal/boundaries/in: interfaces: ContainerService: @@ -42,12 +50,12 @@ packages: AuthService: HealthService: HTTPProber: - SecretService: LogService: VolumeService: PublicTLSService: TrafficStatusService: - StandaloneServiceService: + AppService: + AppReconciler: # Exception: pushImageOps is a CLI-local interface, not a boundary port. # Mocked here because it abstracts Docker SDK calls that require a running # daemon, making unit/integration tests impractical without a test double. diff --git a/AGENTS.md b/AGENTS.md index 25b47c744..ca3a03442 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -111,10 +111,9 @@ CLI commands do NOT use zerowrap — they use `cliWriteLine`/`cliWritef` for out ### ControlPlane Pattern -`ControlPlane` interface (`controlplane.go`) abstracts local vs remote operations. -- `controlplane_remote.go` — delegates to `remote.Client` HTTP methods. -- `controlplane_local.go` — calls service interfaces directly. -- Test fakes in `push_test.go` — update when adding interface methods. +`ControlPlane` interface (`controlplane.go`) is the seam CLI commands depend on. +- `*remote.Client` implements it for both the explicit remote and the local admin socket. +- Tests use the mockery mock in `cli/mocks/` — run `mockery` after adding interface methods. ### HTTP Admin Handlers diff --git a/Dockerfile b/Dockerfile index 05b3ca0da..acf2297ba 100644 --- a/Dockerfile +++ b/Dockerfile @@ -5,6 +5,11 @@ ARG BUILDPLATFORM ARG VERSION=dev ARG COMMIT=unknown ARG BUILD_DATE=unknown +# Whether the source checkout had uncommitted changes. The Makefile and the +# release tooling pass an explicit true/false value. Because .git is excluded +# from the build context (see .dockerignore), an ad-hoc build that passes no +# build-arg reports "unknown" rather than falsely claiming a clean checkout. +ARG DIRTY=unknown FROM --platform=$BUILDPLATFORM golang:1.27-alpine3.22 AS builder @@ -13,6 +18,7 @@ ARG TARGETARCH=amd64 ARG VERSION ARG COMMIT ARG BUILD_DATE +ARG DIRTY RUN apk add --no-cache git ca-certificates tzdata @@ -25,7 +31,7 @@ COPY . . RUN CGO_ENABLED=0 GOOS="${TARGETOS}" GOARCH="${TARGETARCH}" go build \ -trimpath \ - -ldflags="-s -w -X main.version=${VERSION} -X main.commit=${COMMIT} -X main.date=${BUILD_DATE}" \ + -ldflags="-s -w -X main.version=${VERSION} -X main.commit=${COMMIT} -X main.date=${BUILD_DATE} -X main.dirty=${DIRTY}" \ -o /gordon ./main.go FROM alpine:3.22 diff --git a/Makefile b/Makefile index fb358f2c9..5fcfb1e55 100644 --- a/Makefile +++ b/Makefile @@ -10,12 +10,17 @@ ENGINE := podman VERSION := $(shell git describe --tags --always --dirty) COMMIT := $(shell git rev-parse --short HEAD) BUILD_DATE := $(shell date -u '+%Y-%m-%d_%I:%M:%S%p') +# Probe the working tree directly (tracked edits and untracked files) instead +# of parsing VERSION, so an overridden VERSION cannot mask a dirty checkout. +# `git describe --dirty`, which VERSION uses, only reflects tracked edits. +DIRTY := $(if $(shell git status --porcelain --untracked-files=normal 2>/dev/null),true,false) # Build flags LDFLAGS := -s -w \ -X main.version=$(VERSION) \ -X main.commit=$(COMMIT) \ - -X main.date=$(BUILD_DATE) + -X main.date=$(BUILD_DATE) \ + -X main.dirty=$(DIRTY) # Architectures ARCHS := amd64 arm64 @@ -83,7 +88,7 @@ build: ## Build binaries for linux (amd64 and arm64) @echo "Building Go binaries..." @mkdir -p $(DIST_DIR) @rm -f $(DIST_DIR)/* - @echo "Building with version $(VERSION), commit $(COMMIT), date $(BUILD_DATE)" + @echo "Building with version $(VERSION), commit $(COMMIT), date $(BUILD_DATE), dirty $(DIRTY)" @CGO_ENABLED=0 GOOS=linux GOARCH=amd64 go build -ldflags="$(LDFLAGS)" -o $(DIST_DIR)/gordon-linux-amd64 ./main.go @CGO_ENABLED=0 GOOS=linux GOARCH=arm64 go build -ldflags="$(LDFLAGS)" -o $(DIST_DIR)/gordon-linux-arm64 ./main.go @echo "Go binaries built successfully" @@ -105,6 +110,7 @@ build-push: ## Build and push Docker images --build-arg VERSION="$(VERSION)" \ --build-arg COMMIT="$(COMMIT)" \ --build-arg BUILD_DATE="$(BUILD_DATE)" \ + --build-arg DIRTY="$(DIRTY)" \ -t $(REPO):$(TAG)-$$arch .; \ $(ENGINE) push $(REPO):$(TAG)-$$arch; \ done diff --git a/README.md b/README.md index bacbbd85c..20a60be3d 100644 --- a/README.md +++ b/README.md @@ -2,57 +2,79 @@ [![License: GPL-3.0](https://img.shields.io/badge/License-GPL%203.0-blue.svg)](https://www.gnu.org/licenses/gpl-3.0) -Self-hosted container deployment. Push an image, Gordon routes it to the web. +Self-hosted container deployment. Push an image, declare an app, deploy it. - Website: https://bnema.dev/gordon -- Documentation: [Docs](https://gordon.bnema.dev/docs) | [Wiki](https://gordon.bnema.dev/wiki) +- Documentation: [Docs](https://bnema.dev/gordon/docs) | [Wiki](https://bnema.dev/gordon/wiki) - Discuss: [GitHub Discussions](https://github.com/bnema/gordon/discussions) --- ## What is Gordon? -Gordon is a private container registry and HTTP reverse proxy for your VPS. Push a container image that exposes a web port — Gordon deploys it with zero downtime. +Gordon is a private container registry, an app runtime, and a reverse proxy for your VPS. + +The flow is explicit: build and push an image, declare the app in a standalone TOML file, apply it, then deploy. Push transfers OCI content only — it never deploys. ## Quick Start ```bash -# Install -curl -fsSL https://gordon.bnema.dev/install.sh | sh +# Install the latest stable release to ~/.local/bin +curl -fsSL https://bnema.dev/gordon/install | sh -# Start the server +# Start the server (restart your shell first if the installer updated PATH) gordon serve ``` -Config is created at `~/.config/gordon/gordon.toml`. See the [Getting Started guide](https://gordon.bnema.dev/docs/getting-started) for full setup. +The installer uses `~/.local/bin` without `sudo` and can add the effective install directory to Fish, Bash, or Zsh PATH configuration. Set `GORDON_UPDATE_PATH=1` to update PATH without prompting or `GORDON_UPDATE_PATH=0` to leave configuration unchanged. Override the destination with an absolute path such as `GORDON_INSTALL_DIR="$HOME/bin"`, `GORDON_INSTALL_DIR="$HOME/.local/bin"`, or `GORDON_INSTALL_DIR=/usr/local/bin`. + +To build the current `next` branch commit locally, use `GORDON_CHANNEL=next`. This is an unverified development source build, not a checksum-verified release, and requires a compatible Go toolchain. See the [installation guide](https://bnema.dev/gordon/docs/installation#choosing-an-install-channel). + +Config is created at `~/.config/gordon/gordon.toml`. See the [Getting Started guide](https://bnema.dev/gordon/docs/getting-started) for full setup. ## Deploy with the CLI -Build locally, push directly to your Gordon server: +Build locally, push to your Gordon server, declare the app, deploy: ```bash -# Push using an explicit domain override -gordon push myapp:latest --domain app.example.com +# Push the image (OCI transfer only, never deploys) +gordon images push myapp --build --remote prod + +# Declare the app (blog.toml references the pushed tag) +gordon apps apply --file blog.toml --remote prod -# Or add a route manually, then deploy -gordon routes add app.example.com myapp:latest -gordon routes deploy app.example.com +# Activate it +gordon apps deploy blog --remote prod # Check status -gordon status +gordon daemon status +gordon apps status blog --remote prod ``` -Roll back, restart, or manage secrets — all from the command line: +Minimal app file (`blog.toml`): + +```toml +name = "blog" + +[services.web] +image = "gordon.mydomain.com/myapp:v1.2.0" + +[[services.web.http]] +host = "blog.mydomain.com" +port = 3000 +``` + +Manage lifecycle and secrets — all from the command line: ```bash -gordon pin app.example.com # Pin to a specific image tag -gordon restart app.example.com # Restart the container -gordon secrets set app.example.com DB_HOST=db.internal API_KEY=secret123 +gordon apps restart blog --remote prod # Restart from pinned digests +gordon apps stop blog --remote prod # Stop, preserve all data +gordon apps secrets set blog --service web DATABASE_URL=... --remote prod ``` ## Deploy from CI/CD -Push to Gordon's registry from any CI pipeline. Gordon deploys automatically on image push. +Push to Gordon's registry from any CI pipeline, then apply and deploy with the CLI. Push never triggers a deploy by itself. ### GitHub Actions @@ -64,17 +86,19 @@ Push to Gordon's registry from any CI pipeline. Gordon deploys automatically on password: ${{ secrets.GORDON_TOKEN }} ``` -### Docker CLI +See the [Deploy Action README](.github/actions/deploy/README.md) for multi-platform builds, monorepo support, and all available options. + +### Docker CLI + Gordon CLI ```bash -docker login registry.mydomain.com -docker build -t registry.mydomain.com/myapp:v1.0.0 . -docker push registry.mydomain.com/myapp:v1.0.0 -# -> Deployed automatically +docker login gordon.mydomain.com +docker build -t gordon.mydomain.com/myapp:v1.0.0 . +docker push gordon.mydomain.com/myapp:v1.0.0 +# Then, with the Gordon binary: +gordon apps apply --file blog.toml --remote prod +gordon apps deploy blog --remote prod ``` -See the [Deploy Action README](.github/actions/deploy/README.md) for multi-platform builds, monorepo support, and all available options. - ## CLI Commands ### Server @@ -82,36 +106,41 @@ See the [Deploy Action README](.github/actions/deploy/README.md) for multi-platf | Command | Description | |---------|-------------| | `gordon serve` | Start the Gordon server | -| `gordon status` | Show server and route health | -| `gordon config show` | Display server configuration | +| `gordon daemon status` | Show server and app fleet status | +| `gordon daemon logs` | Show Gordon process logs | +| `gordon daemon reload` | Reload installation configuration | +| `gordon daemon config show` | Display installation configuration | -### Deployment +### Applications | Command | Description | |---------|-------------| -| `gordon push [image]` | Tag, push, and optionally deploy an image | -| `gordon routes list` | List all routes | -| `gordon routes add ` | Create or update a route | -| `gordon routes remove ` | Remove a route | -| `gordon routes deploy ` | Redeploy a route | -| `gordon pin ` | Pin a route to a specific image tag | -| `gordon restart ` | Restart a route container | +| `gordon apps apply --file FILE` | Validate and persist an app manifest | +| `gordon apps deploy APP` | Activate an app revision | +| `gordon apps list` | List applications | +| `gordon apps show APP` | Show desired and active state | +| `gordon apps status APP` | Show effective vs observed state | +| `gordon apps logs APP` | Show logs for an app service | +| `gordon apps restart APP` | Restart from pinned digests | +| `gordon apps stop APP` | Stop, preserve all data | +| `gordon apps start APP` | Start a stopped app | +| `gordon apps remove APP` | Remove workloads (volumes/secrets retained) | ### Images & Registry | Command | Description | |---------|-------------| +| `gordon images push [image]` | Tag and push an image (never deploys) | | `gordon images list` | List runtime and registry images | | `gordon images prune` | Clean up dangling images and old tags | | `gordon images tags ` | List registry tags for a repository | -### Secrets & Config +### Secrets | Command | Description | |---------|-------------| -| `gordon secrets list ` | List secrets for a route | -| `gordon secrets set KEY=VAL` | Set secrets | -| `gordon secrets remove ` | Remove a secret | +| `gordon apps secrets list APP` | List app secret names | +| `gordon apps secrets set APP --service SVC KEY=VAL` | Set app secret values | ### Remotes & Auth @@ -125,23 +154,22 @@ See the [Deploy Action README](.github/actions/deploy/README.md) for multi-platf ## Features -- Private Docker registry on your VPS +- Private Docker/Podman registry on your VPS +- Declarative apps: one TOML file per app, explicit apply then deploy - Domain-to-container routing through a smart TCP edge reverse proxy -- Automatic deployment on image push -- Auto-routing from image labels -- Remote CLI management -- Zero downtime updates -- Persistent volumes from Dockerfile VOLUME directives -- Environment variable management with secrets support -- Network isolation per application -- Single binary, ~15MB RAM +- Sequential service replacement (withdraw traffic, replace one generation, verify readiness, republish) +- Remote CLI management (daemon is the sole writer) +- Declarative per-service volumes, retained across lifecycle operations +- Per-service secrets in pass, app-wide public env in the manifest +- Per-app private networks plus opt-in shared networks +- Single binary > [!NOTE] -> Gordon exposes public traffic through entrypoints such as `[entrypoints.edge]` with `protocol = "smart_tcp"`. It can terminate TLS via static certificates, public ACME certificates, or its internal CA. Cloudflare and upstream reverse proxies are optional deployment choices, not requirements. +> Gordon exposes public traffic through entrypoints such as `[entrypoints.edge]` with `protocol = "smart_tcp"`. It can terminate TLS via static certificates, public ACME certificates, or its internal CA. Cloudflare and upstream reverse proxies are optional deployment choices, not requirements. Use Gordon's ownership-aware volume commands; runtime commands such as `docker volume prune` bypass Gordon's retention checks. ## Documentation -Full documentation at **[gordon.bnema.dev](https://bnema.dev/gordon)** +Full documentation at **[bnema.dev/gordon](https://bnema.dev/gordon)** - [Docs](https://bnema.dev/gordon/docs) — Installation, configuration, CLI reference - [Wiki](https://bnema.dev/gordon/wiki) — Tutorials, guides, and examples diff --git a/docs/cli/apps.md b/docs/cli/apps.md new file mode 100644 index 000000000..5fd3ef1f3 --- /dev/null +++ b/docs/cli/apps.md @@ -0,0 +1,390 @@ +# Apps Commands + +Validate, persist, inspect, and operate applications. + +All mutations are executed by the daemon through the admin API. +Without a reachable daemon the commands fail with `daemon-unavailable` +instead of writing locally. Mutations are idempotent: every request carries a +client-generated key, and the daemon binds that key to the exact request. +Repeating a key replays the recorded result; reusing it for a different request +is refused. An interrupted operation is never executed twice, so an ambiguous +outcome is re-queried by key before any retry (never retry under a fresh +key). Unknown apps are reported as `app not found` and create no state. + +## Local and remote targets + +With no `--remote`/`GORDON_REMOTE` selected, commands discover the daemon's +owner-only administration socket. The CLI checks `$XDG_RUNTIME_DIR/gordon/admin.sock` +when set, then `/run/user//gordon/admin.sock`, then `~/.gordon/run/admin.sock`. +It accepts only a safe owner-owned socket. This works with `auth.enabled=false`: +the socket carries no bearer token and grants the `local-owner` principal only +app administration and app log reads. An explicit remote is authoritative and +never falls back to the socket. +See [Local-only Mode](../config/auth.md#local-only-mode). + +## gordon apps + +### Subcommands + +| Subcommand | Description | +|------------|-------------| +| `apply` | Validate and persist an app manifest | +| `operations` | Inspect app operation journals | +| `list` | List applications | +| `show` | Show desired and active state for an app | +| `diff` | Show the normalized desired-vs-active diff | +| `secrets` | Manage app secret values | +| `deploy` | Activate an app revision | +| `restart` | Restart an app from pinned digests | +| `stop` | Stop an app (preserves all data) | +| `start` | Start a stopped app from active state | +| `remove` | Remove app workloads (volumes and secrets are retained) | +| `status` | Show effective vs observed state for an app | +| `logs` | Show logs for an app service | + +--- + +## gordon apps apply + +```bash +gordon apps apply --file blog.toml [--dry-run] [--deploy] [--json] +``` + +Validates a manifest file and persists it as desired state (or dry-runs). +With `--deploy`, chains exactly the accepted revision into a deploy after +persistence succeeds; the two outcomes are reported separately because a +deploy may fail after the apply succeeded. + +When the daemon accepts the deploy asynchronously (HTTP 202), the command +reuses the same by-key watch as `gordon apps deploy`: it polls the +operation journal until terminal, prints progress only when the operation +or step state changes, and never reissues the deploy. A terminal +`partial`/`failed` deploy exits nonzero while still reporting the +successful apply. Ctrl-C stops only local polling: the daemon-side +operation keeps running and the command prints how to resume it. + +With `--json`, stdout carries exactly one final combined document after the +deploy reaches a terminal state: + +```json +{ + "apply": { "app": "blog", "resulting_revision": "rev-b", "pending": true }, + "deploy": { "op": "op-1", "app": "blog", "status": "success", "outcome": "success" } +} +``` + +No initial running document is emitted, and progress, transient warnings, +and Ctrl-C resume guidance go to stderr. + +`--dry-run` and `--deploy` cannot be combined. + +--- + +## gordon apps list + +```bash +gordon apps list [--json] +``` + +Lists applications with one row per app. States are `applied` when no revision is active, `pending` when desired state awaits activation, `stopped` when stopped intent is set, and otherwise `active`. A non-success last operation is appended to `active` in parentheses. + +--- + +## gordon apps show + +```bash +gordon apps show APP [--json] +``` + +Shows desired revision and acceptance status, pending state, per-service +effective revision, container, digest, and restart safety, the resources the +app owns (volumes, secret paths, image references — never secret values), +stopped intent, and the last operation with its outcome. + +--- + +## gordon apps operations show + +```bash +gordon apps operations show APP --key KEY [--json] +``` + +Recovers one operation journal entry by its client-generated request key. +`--key` is required. Use it to re-query an ambiguous mutation outcome before +retrying: repeating the same key replays the recorded result, while a key +reused for a different request is refused. + +### Flags + +| Flag | Description | +|------|-------------| +| `--key` | Request key (required) | +| `--json` | Output as JSON | + +--- + +## gordon apps operations watch + +```bash +gordon apps operations watch APP --key KEY [--json] +``` + +Polls the operation journal by request key until the operation reaches a +terminal state, then renders the terminal journal. It never reissues the +mutation, so it is the safe way to resume after an interrupted deploy or to +follow an operation started elsewhere. Progress is printed only when the +operation or step state changes; unchanged polls are not repeated. + +Exit status is nonzero for terminal `partial`/`failed` outcomes and for an +interrupted watch. Ctrl-C stops only local polling: the daemon-side +operation is not cancelled and keeps running, and the command prints the +command to resume. With `--json`, stdout carries exactly one final document +and progress goes to stderr. + +### Flags + +| Flag | Description | +|------|-------------| +| `--key` | Request key (required) | +| `--json` | Output as JSON | + +--- + +## gordon apps diff + +```bash +gordon apps diff APP [--json] +``` + +Shows the normalized desired-vs-active diff (`added`, `removed`, `changed`). + +--- + +## gordon apps secrets + +Values are accepted via `KEY=VALUE` arguments (discouraged: shell history) or +stdin. Names must already exist in desired or active state. Only key names are +ever echoed back — never values. Secrets are service-scoped: `--service` is +required for `set` and `delete`, and optional for `list` where it filters to +one service. + +### gordon apps secrets list + +```bash +gordon apps secrets list APP [--service SVC] [--json] +``` + +Lists registration metadata for an app's secrets — service, key, name, source, +and presence — never secret values. `--service` filters to one service. + +#### Flags + +| Flag | Description | +|------|-------------| +| `--service` | Filter by service | +| `--json` | Output as JSON | + +#### JSON Output + +```json +[ + {"service": "web", "key": "DATABASE_URL", "name": "app_database_url", "source": "desired", "presence": "unknown"} +] +``` + +### gordon apps secrets set + +```bash +gordon apps secrets set APP --service SVC KEY=VALUE… [--json] +gordon apps secrets set APP --service SVC --stdin [--json] +gordon apps secrets set APP --service SVC --stdin --key KEY [--json] +``` + +Reads values in one of three ways: + +- `KEY=VALUE` arguments: the value must be a single line of 1–65536 bytes. +- `--stdin`: reads `KEY=VALUE` lines; blank and whitespace-only lines are + ignored and every non-blank value byte is preserved. +- `--stdin --key KEY`: reads one raw value for `KEY`; a single trailing newline + is stripped and the value must still be a single non-empty line. + +`--key` requires `--stdin` and cannot be combined with `KEY=VALUE` arguments. +An empty value is rejected, so `KEY=` is invalid. + +Running containers keep the values they were created with. Apply new values +with `gordon apps deploy APP --service SVC`: deploy sees the changed secret and +recreates the service. `restart` does not apply new values. + +#### Flags + +| Flag | Description | +|------|-------------| +| `--service` | Service the secrets belong to (required) | +| `--stdin` | Read `KEY=VALUE` lines, or one raw value with `--key` | +| `--key` | Secret key for single-value stdin mode | +| `--json` | Output as JSON | + +#### JSON Output + +```json +{"app": "blog", "service": "web", "keys": ["DATABASE_URL"]} +``` + +### gordon apps secrets delete + +```bash +gordon apps secrets delete APP KEY --service SVC [--json] +``` + +Deletes the registered value for `KEY`. + +#### Flags + +| Flag | Description | +|------|-------------| +| `--service` | Service the secret belongs to (required) | +| `--json` | Output as JSON | + +--- + +## gordon apps deploy + +```bash +gordon apps deploy APP [--revision REV] [--service SVC | --all] [--json] +``` + +Activates a revision (default: desired head). An app with several services +needs `--service NAME` (one service) or `--all` (every service); without +either, the command fails before any change and lists the services. A +single-service app needs neither. `apps apply --deploy` always deploys every +service. + +| Flag | Description | +|------|-------------| +| `--revision REV` | Revision to activate (default: desired head) | +| `--service NAME` | Deploy one service only | +| `--all` | Deploy every service | +| `--json` | Output as JSON | + +Deploy is the only command that applies changes. It recreates a service +when its image digest, spec, app environment, or secret values changed. A +new image behind the same tag (for example `latest`) has a new digest and is +deployed. A service already running with all of these unchanged keeps its +container and is reported `unchanged`. Services with host binds or devices +are always replaced, so bind and device policy changes apply on deploy. + +Fail-fast across services: +the first failure stops the deploy, successful services are preserved, +later services stay unchanged. + +When the daemon accepts the deploy and runs the two phases in the +background (HTTP 202), the command polls the operation journal by key until +it is terminal and prints concise progress only when the operation or step +state changes. It never reissues the mutation. In `--json` mode stdout +carries one final document and progress goes to stderr. Ctrl-C stops only +local polling (the daemon-side operation keeps running) and prints the +`gordon apps operations watch` command to resume. + +--- + +## gordon apps restart + +```bash +gordon apps restart APP [--service SVC | --all] [--json] +``` + +Restarts the same container. Nothing is applied: new secret values, env, +image, or config need `gordon apps deploy`. Traffic is withdrawn, the +container restarts, readiness is checked, and traffic returns. If the +container no longer exists, restart rebuilds it from its pinned digest. An +app with several services needs `--service NAME` or `--all`. + +--- + +## gordon apps stop + +```bash +gordon apps stop APP [--json] +``` + +Persists the durable stopped intent and stops exact containers. +All data (volumes, secrets) is preserved. Stopped apps stay stopped +across reboot. + +--- + +## gordon apps start + +```bash +gordon apps start APP [--json] +``` + +Clears the stopped intent and ensures running from active state. + +--- + +## gordon apps remove + +```bash +gordon apps remove APP [--json] +``` + +Withdraws workloads. Volumes and secrets are retained as owned orphans +under the old internal UUID; removing frees the name but never implicitly +attaches retained resources to a new app reusing the name. + +There is deliberately no `purge`: destructive volume deletion requires a +separately accepted destructive-action contract. + +--- + +## gordon apps status + +```bash +gordon apps status APP [--json] +``` + +Shows effective vs observed state per service. + +--- + +## gordon apps logs + +```bash +gordon apps logs APP [--service SVC] [--follow] [--tail N] [--json] +``` + +Streams logs for a service in the app's active deployment. The app name and +`--service` are the only accepted identity: domains and raw container IDs are +never accepted. `--service` is required when the app has several services. + +--- + +## Workflow Example + +```bash +# Push the image first (OCI transfer only) +gordon images push myapp --build --remote prod + +# Apply the manifest that references the pushed tag +gordon apps apply --file blog.toml --remote prod + +# Write secret values for registered names +gordon apps secrets set blog --service web DATABASE_URL=... --remote prod + +# Deploy the accepted revision (polls 202 operations to terminal) +gordon apps deploy blog --remote prod + +# Resume watching an interrupted or externally started operation +gordon apps operations watch blog --key --remote prod + +# Inspect +gordon apps show blog --remote prod +gordon apps status blog --remote prod +``` + +## Related + +- [CLI Overview](./index.md) +- [Images Commands](./images.md#gordon-images-push) +- [Deployment](../deployment/index.md) diff --git a/docs/cli/attachments.md b/docs/cli/attachments.md deleted file mode 100644 index 377c04666..000000000 --- a/docs/cli/attachments.md +++ /dev/null @@ -1,349 +0,0 @@ -# Attachments Commands - -Manage attachments on local or remote Gordon instances. - -Remote targeting uses client config or an active remote by default. -When you provide a concrete target (for example `attachments list app.example.com` -or `attachments remove app.example.com postgres:18`) and no remote is selected, -Gordon can also auto-infer a saved remote when exactly one match is found. -Use `--remote` and `--token` to override. See [CLI Overview](./index.md). - -## Requirements - -Attachments require `network_isolation.enabled = true` in your configuration (enabled by default). Without network isolation, containers use Docker's default bridge network which does not provide DNS resolution - your app won't be able to reach attachments by hostname (e.g., `postgres:5432`). - -The CLI will warn you if you try to add an attachment without network isolation enabled. - -## gordon attachments - -### Subcommands - -| Subcommand | Description | -|------------|-------------| -| `list` | List all attachments or attachments for a specific target | -| `add` | Add an attachment to a domain or network group | -| `push` | Build or push attachment images to the Gordon registry | -| `remove` | Remove an attachment from a domain or network group | -| `orphans` | List running attachment containers no longer configured | -| `prune` | Dry-run or stop orphaned attachment containers while preserving volumes | - -### Alias - -`gordon attach` is an alias for `gordon attachments`: - -```bash -gordon attach list -gordon attach add app.example.com postgres:18 -gordon attach remove app.example.com postgres:18 -``` - ---- - -## gordon attachments list - -List all configured attachments. - -```bash -# List all attachments -gordon attachments list -gordon attachments list --json - -# List attachments for a specific domain or group -gordon attachments list app.example.com -gordon attachments list backend - -# Remote (override) -gordon attachments list --remote https://gordon.mydomain.com --token $TOKEN -``` - -### Output - -``` -Target Attachments --------------------------------------------------------------------------------- -app.example.com postgres:18, redis:7-alpine -api.example.com postgres:18 -backend (group) rabbitmq:3-management -``` - -### JSON Output - -```bash -gordon attachments list --json -``` - -```json -[ - { - "target": "app.example.com", - "attachments": ["postgres:18", "redis:7-alpine"] - }, - { - "target": "backend", - "attachments": ["rabbitmq:3-management"] - } -] -``` - -### Options - -| Option | Description | -|--------|-------------| -| `--json` | Output attachments as JSON | -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | - ---- - -## gordon attachments orphans - -List running attachment containers that are no longer configured. - -```bash -gordon attachments orphans -gordon attachments orphans --json -``` - -### Output - -| Container ID | Name | Image | Status | Owner | -|--------------|------|-------|--------|-------| -| `pg123` | `postgres` | `postgres:16` | `running` | `myapp.example.com` | - -### JSON Output - -```json -{ - "attachments": [ - { - "container_id": "pg123", - "name": "postgres", - "image": "postgres:16", - "status": "running", - "owner": "myapp.example.com" - } - ] -} -``` - -## gordon attachments prune - -Lists orphaned attachments by default (dry-run). Use `--stop` to stop and remove the orphaned containers while preserving attachment volumes/data. - -```bash -gordon attachments prune # dry-run/report only -gordon attachments prune --stop # stop/remove orphaned containers, preserve volumes -``` - -### Output - -```text -Orphaned attachments - postgres -> postgres:16 -Hint: orphaned attachments preserved; rerun with --stop to stop/remove containers while preserving volumes -``` - -### Output with `--stop` - -```text -Removed orphaned attachment containers: - postgres -Hint: attachment volumes and data were preserved -``` - -### Options - -| Option | Description | -|--------|-------------| -| `--stop` | Stop and remove orphaned attachment containers while preserving volumes/data | -| `--json` | Output the cleanup report as JSON | -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | - -## gordon attachments add - -Add an attachment to a domain or network group. - -```bash -gordon attachments add -gordon attachments add app.example.com postgres:18 -gordon attachments add backend redis:7-alpine -``` - -### Arguments - -| Argument | Description | -|----------|-------------| -| `` | The domain name or network group name | -| `` | The container image to attach | - -### Options - -| Option | Description | -|--------|-------------| -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | - -### Examples - -```bash -# Add database to a domain -gordon attachments add app.example.com postgres:18 - -# Add cache to a network group (shared by all domains in the group) -gordon attachments add backend redis:7-alpine - -# Remote (override) -gordon attachments add app.example.com postgres:18 --remote https://gordon.mydomain.com --token $TOKEN -``` - ---- - -## gordon attachments push - -Push attachment images to the Gordon registry. - -```bash -gordon attachments push [options] -``` - -### Arguments - -| Argument | Description | -|----------|-------------| -| `` | The attachment image to build/tag and push | - -### Options - -| Option | Description | -|--------|-------------| -| `--build` | Build the image first using `docker buildx` | -| `-f, --file` | Path to Dockerfile (default: `./Dockerfile`, used with `--build`) | -| `--platform` | Target platform for buildx (default: `linux/amd64`) | -| `--build-arg` | Additional build args (repeatable, `KEY=VALUE`) | -| `--tag` | Override version tag | -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | - -### Description - -`gordon attachments push` pushes attachment images such as databases and caches to the Gordon registry so they are available when routes deploy. It does not trigger deployment. The image must already be configured as an attachment first. - -It uses the same native chunked upload transport as `gordon push`, sending image -layers in 50MB chunks so pushes work through Cloudflare-proxied Gordon -instances. Keep the server's `max_blob_chunk_size` larger than the client chunk -size; the default `95MB` is compatible. - -### Examples - -```bash -# Push a pre-built attachment image -gordon attachments push pitlane-pgsql - -# Build and push an attachment image -gordon attachments push pitlane-pgsql --build - -# Push with a specific tag -gordon attachments push pitlane-pgsql --tag v18 -``` - ---- - -## gordon attachments remove - -Remove an attachment. - -```bash -gordon attachments remove -gordon attachments remove app.example.com postgres:18 -``` - -### Arguments - -| Argument | Description | -|----------|-------------| -| `` | The domain name or network group name | -| `` | The container image to remove | - -### Options - -| Option | Description | -|--------|-------------| -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | - -### Examples - -```bash -# Remove database from a domain -gordon attachments remove app.example.com postgres:18 - -# Remove from network group -gordon attachments remove backend redis:7-alpine - -# Remote (override) -gordon attachments remove app.example.com postgres:18 --remote https://gordon.mydomain.com --token $TOKEN -``` - ---- - -## Workflow Examples - -### Add Database and Cache - -```bash -# Add PostgreSQL database -gordon attachments add app.example.com postgres:18 - -# Add Redis cache -gordon attachments add app.example.com redis:7-alpine - -# Verify -gordon attachments list app.example.com -``` - -### Shared Services via Network Groups - -```bash -# First, ensure you have a network group defined in your config: -# [network_groups] -# "backend" = ["app.example.com", "api.example.com"] - -# Add shared cache to the group -gordon attachments add backend redis:7-alpine - -# Both app.example.com and api.example.com can now access the shared Redis -``` - -### First Deploy with Custom Attachment Image - -```bash -# Configure the route and attachment -gordon bootstrap app.example.com myapp:latest --attachment pitlane-pgsql - -# Push the custom attachment image first -gordon attachments push pitlane-pgsql --build - -# Then push and deploy the route image -gordon push myapp:latest --domain app.example.com --build --no-confirm -``` - -### CI/CD Integration - -```bash -# In your CI/CD pipeline -export GORDON_REMOTE=https://gordon.mydomain.com -export GORDON_TOKEN=$GORDON_TOKEN - -# Add new attachment -gordon attachments add app.example.com elasticsearch:8 - -# Trigger redeploy to pick up the new attachment -gordon routes deploy app.example.com -``` - -## Related - -- [CLI Overview](./index.md) -- [Attachments Configuration](../config/attachments.md) -- [Network Groups](../config/network-groups.md) diff --git a/docs/cli/auth.md b/docs/cli/auth.md index 72f2cc1b8..c756f54e6 100644 --- a/docs/cli/auth.md +++ b/docs/cli/auth.md @@ -135,7 +135,7 @@ Standard Go durations also work: `24h`, `30m`, `1h30m` Registry scopes: `push`, `pull`, `push,pull` -Admin scopes: `admin:*:*`, `admin:routes:read`, `admin:routes:write`, `admin:config:read`, `admin:config:write`, `admin:status:read`, `admin:logs:read`, `admin:volumes:read`, `admin:volumes:write`, `admin:secrets:read`, `admin:secrets:write` +Admin scopes: `admin:*:*`, `admin:apps:read`, `admin:apps:write`, `admin:config:read`, `admin:config:write`, `admin:status:read`, `admin:logs:read`, `admin:volumes:read`, `admin:volumes:write` Combine scopes with commas: @@ -271,14 +271,12 @@ Usage: | `pull` | Pull images from registry | | `push,pull` | Both push and pull (default) | | `admin:*:*` | Full admin access | -| `admin:routes:read` | Read-only routes access | -| `admin:routes:write` | Routes write access | +| `admin:apps:read` | Read-only apps access (list, show, diff, status) | +| `admin:apps:write` | App mutations (apply, deploy, lifecycle, secrets) | | `admin:config:read` | Read-only config access | | `admin:config:write` | Config write access | | `admin:status:read` | Read-only status/health | | `admin:logs:read` | Read-only logs access | -| `admin:secrets:read` | List secret keys | -| `admin:secrets:write` | Set/delete secrets | ## Token Expiry Formats diff --git a/docs/cli/autoroute.md b/docs/cli/autoroute.md deleted file mode 100644 index 4d7550aae..000000000 --- a/docs/cli/autoroute.md +++ /dev/null @@ -1,205 +0,0 @@ -# Auto-Route Commands - -Manage the auto-route domain allowlist on local or remote Gordon instances. - -When auto-route is enabled, Gordon handles image pushes in two ways: - -1. **Image-name matching** — when a pushed image matches an existing route's configured image, Gordon auto-deploys that route. Since the route already exists, no allowlist check is needed. -2. **Label-based creation** — when a pushed image carries `gordon.domain` labels, Gordon creates new routes for those domains. The domain allowlist restricts which domains may be created this way, preventing untrusted images from registering arbitrary domains. - -The allowlist gates only the creation of new routes from labels. Routes created manually with `gordon routes add` or already present in configuration are not subject to the allowlist. - -Remote targeting uses client config or an active remote by default. -Use `--remote` and `--token` to override. See [CLI Overview](./index.md). - -## gordon autoroute - -Manage auto-route settings. - -### Subcommands - -| Subcommand | Description | -|------------|-------------| -| `allow` | Manage auto-route allowed domains | - ---- - -## gordon autoroute allow - -Manage auto-route allowed domains. - -### Subcommands - -| Subcommand | Description | -|------------|-------------| -| `add` | Add a domain pattern to the allowlist | -| `list` | List allowed domains | -| `remove` | Remove a domain pattern | - ---- - -## gordon autoroute allow add - -Add a domain pattern to the auto-route allowlist. - -```bash -gordon autoroute allow add -gordon autoroute allow add example.com -gordon autoroute allow add "*.staging.example.com" -gordon autoroute allow add example.com --remote https://gordon.mydomain.com --token $TOKEN -``` - -### Arguments - -| Argument | Description | -|----------|-------------| -| `` | Domain pattern to allow for auto-route | - -### Options - -| Option | Description | -|--------|-------------| -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | - -### Description - -Patterns must be lowercase and must not have trailing dots. -A bare `*` matches all domains. Wildcards in `*.domain.tld` form match exactly one subdomain level, so `*.example.com` matches `app.example.com` but not `api.app.example.com`. - -### Examples - -```bash -# Allow an exact domain -gordon autoroute allow add example.com - -# Allow one subdomain level under staging.example.com -gordon autoroute allow add "*.staging.example.com" - -# Remote (override) -gordon autoroute allow add example.com --remote https://gordon.mydomain.com --token $TOKEN -``` - -### Output - -```text -✓ Allowed domain added -``` - ---- - -## gordon autoroute allow list - -List auto-route allowed domains. - -```bash -gordon autoroute allow list -gordon autoroute allow list --json -gordon autoroute allow list --remote https://gordon.mydomain.com --token $TOKEN -``` - -### Options - -| Option | Description | -|--------|-------------| -| `--json` | Output allowed domains as JSON | -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | - -### Description - -Shows the current allowlist used to restrict which domains auto-route may claim. -When no patterns are configured, the command prints a friendly empty-state message instead of a table. - -### Examples - -```bash -# Local -gordon autoroute allow list - -# JSON -gordon autoroute allow list --json - -# Remote (override) -gordon autoroute allow list --remote https://gordon.mydomain.com --token $TOKEN -``` - -### Output - -```text -example.com -*.staging.example.com -``` - -When the allowlist is empty: - -```text -No allowed domains configured -``` - -### JSON Output - -```bash -gordon autoroute allow list --json -``` - -```json -{ - "domains": ["example.com", "*.staging.example.com"] -} -``` - ---- - -## gordon autoroute allow remove - -Remove a domain pattern from the auto-route allowlist. - -```bash -gordon autoroute allow remove -gordon autoroute allow remove example.com -gordon autoroute allow remove example.com --remote https://gordon.mydomain.com --token $TOKEN -``` - -### Arguments - -| Argument | Description | -|----------|-------------| -| `` | Domain pattern to remove from the allowlist | - -### Options - -| Option | Description | -|--------|-------------| -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | - -### Description - -Removes an exact allowlist entry. -Use the same pattern string you added, including the `*.` prefix for wildcard entries. - -### Examples - -```bash -# Remove an exact domain -gordon autoroute allow remove example.com - -# Remove a wildcard domain -gordon autoroute allow remove "*.staging.example.com" - -# Remote (override) -gordon autoroute allow remove example.com --remote https://gordon.mydomain.com --token $TOKEN -``` - -### Output - -```text -✓ Allowed domain removed -``` - -## Related - -- [CLI Overview](./index.md) -- [Routes Command](./routes.md) -- [Config Command](./config.md) diff --git a/docs/cli/backup.md b/docs/cli/backup.md index ed2b391b7..ab28007e9 100644 --- a/docs/cli/backup.md +++ b/docs/cli/backup.md @@ -1,71 +1,124 @@ # Backup Command -Manage database backups and volume backups. +Manage app database backups and app volume backups. -## gordon backups databases +Backups are identified by app, service, and the declared resource. Domains are +routing addresses and are never a backup identity. + +## Declaring backup targets + +A service declares its databases and volumes in the app manifest, and lists +which of them are backed up: + +```toml +[services.api] +image = "registry.example.com/shop/api:1.4.2" + +[[services.api.database]] +name = "orders" +type = "postgres" +schedule = "daily" + +[[services.api.volume]] +name = "data" +path = "/var/lib/data" + +[services.api.backup] +postgres = ["orders"] +volume = ["data"] +``` + +A declared database or volume that the service's backup declaration does not +reference is not a backup target. + +Administrative bind mounts are never backup targets. `[services..backup]` accepts +only declared databases and volumes, and Gordon never archives an +operator-owned host path exposed through a bind. + +## gordon backups Database backups are logical PostgreSQL backups made with `pg_dump`. ```bash -gordon backups databases +gordon backups ``` Subcommands: -- `list [domain]` - List database backups -- `run [--db ]` - Trigger an immediate database backup -- `detect ` - Detect supported databases for a domain -- `status` - Show database backup status - -Compatibility aliases remain available: `gordon backups list`, `run`, `detect`, and `status` map to database backups. +- `list [app]` - List stored backups, for every app or one app +- `run --service --database ` - Run one declared + database backup now +- `status` - Show stored backups plus declared targets that have no completed + backup yet -## gordon backups volumes +## gordon backups volume -Volume backups are best-effort filesystem archives of Gordon-managed named volumes, uploaded to S3. +Volume backups are best-effort filesystem archives of the app's declared +volumes, uploaded to S3. ```bash -gordon backups volumes +gordon backups volume ``` Subcommands: -- `list [domain]` - List completed volume backup archives -- `run [domain] [--volume ]` - Trigger volume backups now +- `list [app]` - List completed volume backup archives +- `run --service --volume ` - Run one declared volume + backup now - `status` - Show completed archives plus current/recent in-memory job state -Volume backups exclude bind mounts, tmpfs mounts, anonymous volumes, and non-Gordon volumes. Live archives are not application-consistent unless the application is quiesced or stopped. +Volume archives are not application-consistent unless the application is +quiesced or stopped. + +## Selectors + +`--service` and the resource flag (`--database`, `--volume`) select the target: + +- both given: the target must exist, otherwise the command reports not found; +- one omitted: it is only allowed when exactly one compatible target remains; +- several candidates: the command reports the ambiguity and lists the safe + `service/resource` names. + +Nothing is ever chosen by guessing. ## Examples ```bash # Database backups -gordon backups databases list -gordon backups databases run app.example.com --db postgres -gordon backups databases detect app.example.com -gordon backups databases status +gordon backups list +gordon backups list shop +gordon backups run shop --service api --database orders +gordon backups status # Volume backups -gordon backups volumes list -gordon backups volumes list app.example.com -gordon backups volumes run -gordon backups volumes run app.example.com --volume gordon-app-example-com-data -gordon backups volumes status +gordon backups volume list +gordon backups volume list shop +gordon backups volume run shop --service api --volume data +gordon backups volume status ``` +## Scheduling + +The installation backup schedule runs every declared database whose own +`schedule` matches the firing tier, and applies retention under the app name. +Volume declarations carry no schedule of their own: the installation volume +backup interval runs the declared volume targets. + ## JSON Output -Database and volume list commands support `--json`. +Every list, run, and status command supports `--json`. ```bash -gordon backups volumes list --json +gordon backups volume list --json ``` ## Required Permissions -- Read operations (`list`, `status`, `detect`) require `admin:status:read`. +- Read operations (`list`, `status`) require `admin:status:read`. - `run` requires `admin:config:write`. ## Related - [CLI Commands](./index.md) - [Backups Configuration](../config/backups.md) +- [Apps Configuration](../config/apps.md) diff --git a/docs/cli/bootstrap.md b/docs/cli/bootstrap.md deleted file mode 100644 index 01f14fcfa..000000000 --- a/docs/cli/bootstrap.md +++ /dev/null @@ -1,80 +0,0 @@ -# Bootstrap Command - -Create or update route configuration, attachments, and secrets in one command. - -## gordon bootstrap - -### Synopsis - -```bash -gordon bootstrap [options] -``` - -### Arguments - -| Argument | Description | -|----------|-------------| -| `` | The route domain to create or update | -| `` | The image to assign to the route | - -### Options - -| Option | Description | -|--------|-------------| -| `--attachment` | Add an attachment to the route (repeatable) | -| `--env` | Set an environment variable for the route (repeatable, `KEY=VALUE`) | -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | - -### Description - -`gordon bootstrap` is the recommended first-step setup workflow. - -- Creates the route when it does not exist. -- Updates the route when it already exists. -- Can attach services and set environment variables as part of the same command. -- Does not push or deploy the image. - -Unlike `gordon push`, `gordon bootstrap` does not require the route to exist first. Run `gordon push` separately after bootstrap to upload and deploy an image. - -### Examples - -```bash -# First-time route setup -gordon bootstrap app.example.com myapp:latest - -# Then push and deploy the image -gordon push myapp:latest --domain app.example.com --build --no-confirm - -# First-time route setup with a database attachment and environment variable -gordon bootstrap app.example.com myapp:latest --attachment postgres:18 --env APP_ENV=production - -# Multiple attachments and environment variables -gordon bootstrap app.example.com myapp:latest \ - --attachment postgres:18 \ - --attachment redis:7-alpine \ - --env APP_ENV=production \ - --env LOG_LEVEL=info - -# Remote target -gordon bootstrap app.example.com myapp:latest --remote https://gordon.mydomain.com --token $TOKEN - -# Push custom attachment image first -gordon attachments push pitlane-pgsql --build - -# Then push and deploy the route image -gordon push myapp --build --no-confirm -``` - -### Notes - -- `gordon bootstrap` is idempotent for route configuration: rerunning it re-applies config instead of failing. -- Run `gordon push` after bootstrap to upload and deploy the image. -- If attachments use custom images that are not available from a public registry, push those attachment images with `gordon attachments push` before deploying the route image. - -## Related - -- [CLI Overview](./index.md) -- [Attachments Commands](./attachments.md) -- [Routes Commands](./routes.md) -- [Push Command](./push.md) diff --git a/docs/cli/config.md b/docs/cli/config.md deleted file mode 100644 index 25a190404..000000000 --- a/docs/cli/config.md +++ /dev/null @@ -1,72 +0,0 @@ -# Config Commands - -Inspect Gordon server configuration. - -Remote targeting uses client config or an active remote by default. -Use `--remote` and `--token` to override. See [CLI Overview](./index.md). - -## gordon config - -### Subcommands - -| Subcommand | Description | -|------------|-------------| -| `show` | Show server configuration | - ---- - -## gordon config show - -Display the Gordon server configuration including server settings, -auto-route, network isolation, routes, and external route domains. Sensitive filesystem paths and upstream external route targets are redacted by default. - -```bash -gordon config show -gordon config show --json -gordon config show --remote https://gordon.mydomain.com --token $TOKEN -``` - -### Flags - -| Flag | Description | -|------|-------------| -| `--json` | Output as JSON | - -### JSON Output - -```json -{ - "server": { - "port": 1111, - "registry_port": 5000, - "registry_domain": "reg.example.com" - }, - "auto_route": { - "enabled": true, - "allowed_domains": ["example.com", "*.staging.example.com"] - }, - "network_isolation": { - "enabled": true, - "prefix": "gordon_" - }, - "routes": [ - {"domain": "app.example.com", "image": "myapp:latest"} - ], - "external_routes": [ - {"domain": "reg.example.com"} - ] -} -``` - -External route targets and `server.data_dir` are intentionally omitted from the default admin config response because they reveal internal network and filesystem layout. - -### Auto-Route Allowed Domains - -The `auto_route.allowed_domains` field lists domain patterns that auto-route may assign to containers. Manage this list with [`gordon autoroute allow`](./autoroute.md). - -## Related - -- [CLI Overview](./index.md) -- [Auto-Route Commands](./autoroute.md) -- [Status Command](./status.md) -- [Routes Command](./routes.md) diff --git a/docs/cli/daemon.md b/docs/cli/daemon.md new file mode 100644 index 000000000..4ef024268 --- /dev/null +++ b/docs/cli/daemon.md @@ -0,0 +1,518 @@ +# Daemon Commands + +Inspect and operate the Gordon daemon: status, process logs, reload, configuration, TLS, traffic, and networks. + +These commands target the local daemon through its owner-only socket, or the daemon selected with `--remote`, `GORDON_REMOTE`, or the active remote. See [CLI Overview](./index.md). + +## gordon daemon + +| Subcommand | Description | +|------------|-------------| +| `status` | Show daemon status and the app fleet summary | +| `logs` | Show daemon process logs | +| `reload` | Reload installation configuration | +| `config show` | Show installation configuration | +| `config validate` | Statically validate a local configuration file | +| `tls` | Show public TLS certificate status | +| `traffic` | Show traffic entrypoint, router, and counter status | +| `networks` | List Gordon-managed networks | + +--- + +## gordon daemon status + +Display installation identity plus one status line per app (from desired/active state, no container inspection). Per-service detail lives under `gordon apps show APP` and `gordon apps status APP`. + +```bash +gordon daemon status +gordon daemon status --remote prod +``` + +`gordon daemon status` works in local mode and remote mode. + +- Local mode reads status from in-process services. +- Remote mode reads status from the target admin API. + +### Output + +``` +Gordon Status + +Gordon Domain: gordon.example.com +Registry Port: 5000 +Server Port: 8088 +Apps: 3 +Network Isolation: false + +Container Status: + blog: active + shop: deploying + old-site: stopped +``` + +### Information Displayed + +| Field | Description | +|-------|-------------| +| Gordon Domain | Public Gordon domain from configuration | +| Registry Port | Docker registry port | +| Server Port | Gordon admin port | +| Apps | Total apps in desired/active state | +| Network Isolation | Whether installation network policy is enabled | +| Container Status | Fleet status per app (see states below) | + +### App States + +| State | Description | +|-------|-------------| +| active | App deployed and converged on desired state | +| deploying | Desired state diverges from effective state | +| pending | App applied but never deployed | +| stopped | Durable stopped intent (stays stopped across reboot) | + +### Flags + +Uses the global remote flags: + +| Flag | Description | +|------|-------------| +| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | +| `--token-file` | Read the remote token from a mode 0600 file | + +### Environment Variables + +| Variable | Description | +|----------|-------------| +| `GORDON_REMOTE` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | +| `GORDON_TOKEN` | Authentication token | + +### Examples + +### Check Local or Remote Status + +```bash +# Local +gordon daemon status + +# Using a saved remote +gordon daemon status --remote prod + +# Using environment variables +export GORDON_REMOTE=https://gordon.mydomain.com +export GORDON_TOKEN=your-token +gordon daemon status +``` + +### Quick Fleet Check + +```bash +# Check for non-converged apps +gordon daemon status --remote prod | grep -E "(deploying|pending|stopped)" +``` + +### Required Permissions (Remote Only) + +Remote status calls require `admin:status:read` scope in the authentication token. + +```bash +# Generate token with required scope +gordon auth token generate --subject admin --scopes admin:status:read +``` + +--- + +## gordon daemon logs + +Display Gordon daemon process logs. Application workload output is read with +`gordon apps logs APP --service SVC`, which resolves the app's active container +through the daemon. + +### Synopsis + +```bash +gordon daemon logs [options] +``` + +### Options + +| Option | Short | Default | Description | +|--------|-------|---------|-------------| +| `--config` | `-c` | Auto | Path to config file | +| `--follow` | `-f` | false | Follow log output (like `tail -f`) | +| `--lines` | `-n` | 50 | Number of lines to show | +| `--remote, -r` | | | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | +| `--token-file` | | | Read the remote token from a mode 0600 file | + +Remote targeting uses client config or an active remote by default. Use +`--remote` and `--token-file` to override. See [CLI Overview](./index.md). + +Remote log access requires an admin token with `admin:logs:read` (or `admin:*:*`). `admin:status:read` is not sufficient for logs. + +### Examples + +```bash +# Gordon process logs +gordon daemon logs # Last 50 lines +gordon daemon logs -f # Follow logs +gordon daemon logs -n 100 # Last 100 lines +gordon daemon logs -f -n 200 # Follow, starting from last 200 lines + +# App service logs +gordon apps logs blog --service web +gordon apps logs blog --service web --follow --remote prod + +# Remote process logs (override) +gordon daemon logs --remote prod +``` + +### Log Locations + +```bash +# Using gordon daemon logs +gordon daemon logs -f + +# Direct file access +tail -f ~/.gordon/logs/gordon.log + +# With systemd +journalctl --user -u gordon -f + +# App service logs through Gordon +gordon apps logs blog --service web --tail 50 +gordon apps logs blog --service web --follow +``` + +--- + +## gordon daemon reload + +Reload installation configuration. + +### Synopsis + +```bash +gordon daemon reload +``` + +### Description + +Sends `SIGUSR1` to the running Gordon process, triggering: + +- Installation settings reload (live keys apply, restart-required keys are reported) +- Traffic/proxy state refresh from ACTIVE app projection + +Reload never activates pending desired app state, re-resolves images, or flips intent. Obsolete application keys (`routes`, `attachments`, `services`, `auto`, previews, …) are rejected with a `config-retired` diagnostic before any mutation. + +### Example + +```bash +# After editing gordon.toml, apply changes without restart +vim ~/.config/gordon/gordon.toml +gordon daemon reload +``` + +--- + +## gordon daemon config show + +Display the Gordon installation configuration including server settings, +network isolation, volumes, and external route domains. App routes live +under `gordon apps show`, never here. Sensitive filesystem paths and upstream external route targets are redacted by default. + +```bash +gordon daemon config show +gordon daemon config show --json +gordon daemon config show --remote prod +``` + +### Flags + +| Flag | Description | +|------|-------------| +| `--json` | Output as JSON | + +### JSON Output + +```json +{ + "server": { + "port": 1111, + "registry_port": 5000, + "registry_domain": "reg.example.com" + }, + "network_isolation": { + "enabled": true, + "prefix": "gordon" + }, + "volumes": { + "auto_create": true, + "prefix": "gordon", + "preserve": true + }, + "external_routes": [ + {"domain": "reg.example.com"} + ] +} +``` + +External route targets and `server.data_dir` are intentionally omitted from the default admin config response because they reveal internal network and filesystem layout. + +--- + +## gordon daemon config validate + +Statically validates a candidate configuration file before it is installed. +This command is local-only: `--remote` is rejected. Validation is static — +runtime, ACTIVE-state, secret, pull, and listener checks are not performed. +A file that fails validation exits non-zero; `--json` is still written first. + +```bash +gordon daemon config validate --file ./gordon.toml +gordon daemon config validate --file ./gordon.toml --json +``` + +### Flags + +| Flag | Description | +|------|-------------| +| `--file` | Local candidate configuration file (required) | +| `--json` | Output as JSON | + +### JSON Output + +On success: + +```json +{ + "valid": true, + "diagnostics": [], + "scope": "static" +} +``` + +On failure, `valid` is `false` and `diagnostics` carries the failure: + +```json +{ + "valid": false, + "diagnostics": [ + {"code": "config-invalid", "key": "", "message": "configuration failed static validation"} + ], + "scope": "static" +} +``` + +--- + +## gordon daemon tls + +Inspect public TLS/ACME certificate status. + +Gordon serves normal HTTPS fallback on TLS-capable entrypoints such as `entrypoints.edge` with `protocol = "smart_tcp"`. Certificate priority is static certificates first, then public ACME certificates, then Gordon's internal CA. + +ACME challenge notes: + +- DNS-01 (`cloudflare-dns-01`) does not require a special external port 80 edge. +- HTTP-01 requires an HTTP-capable smart TCP entrypoint reachable on external port 80 for every hostname being validated. +- TLS-ALPN-01 is not supported. + +Display the current public TLS/ACME certificate status, including ACME mode, +certificate details, route coverage, and any errors. + +```bash +gordon daemon tls +gordon daemon tls --json +gordon daemon tls --remote prod +``` + +### Flags + +| Flag | Description | +|------|-------------| +| `--json` | Output as JSON | + +### Human Output + +```text +Public TLS / ACME Status + +ACME: enabled +Configured Mode: auto +Effective Mode: http-01 +Reason: configured +Token Source: env + +Certificates + ID: cert-abc123 + Names: example.com, www.example.com + Status: valid + Not After: 2026-05-29 12:00:00 + +Route Coverage + example.com covered=yes covered_by=cert-abc123 + internal.local covered=no error=self-signed cert + +Errors + route internal.local has no ACME cert +``` + +### JSON Output + +```json +{ + "acme_enabled": true, + "configured_mode": "auto", + "effective_mode": "http-01", + "selection_reason": "configured", + "token_source": "env", + "certificates": [ + { + "id": "cert-abc123", + "names": ["example.com", "www.example.com"], + "challenge": "http-01", + "status": "valid", + "not_after": "2026-05-29T12:00:00Z", + "renewal_pending": false + } + ], + "routes": [ + { + "domain": "example.com", + "covered": true, + "covered_by": "cert-abc123", + "required_acme": true + }, + { + "domain": "internal.local", + "covered": false, + "required_acme": false, + "error": "self-signed cert" + } + ], + "errors": ["route internal.local has no ACME cert"] +} +``` + +### Token Source + +The `token_source` field indicates where the ACME token was sourced from +(e.g., `env`, `file`, `config`). The token value is never displayed. + +--- + +## gordon daemon traffic + +```bash +gordon daemon traffic --remote prod +gordon daemon traffic --remote https://gordon.example.com --json +``` + +Remote mode queries the running Gordon admin API. Local mode does not synthesize runtime traffic state from a fresh config load; use a configured remote target for authoritative counters and reload status. + +### Flags + +| Flag | Description | +|------|-------------| +| `--json` | Output machine-readable JSON | +| `--remote`, `-r` | Remote Gordon instance or saved remote name | + +### JSON Output + +```json +{ + "last_reload_status": "ok", + "entrypoints": [ + { + "name": "postgres", + "address": "0.0.0.0:5432", + "protocol": "tcp", + "active": true, + "active_tcp_connections": 1, + "active_udp_sessions": 0, + "total_accepted": 12, + "total_refused": 0, + "total_errors": 0, + "bytes_in": 4096, + "bytes_out": 8192, + "smart_tcp": { + "http_accepted": 0, + "h2c_accepted": 0, + "https_fallback_accepted": 0, + "tls_passthrough_accepted": 0, + "raw_fallback_accepted": 0, + "entrypoint_cidr_refused": 0, + "raw_fallback_cidr_refused": 0, + "proxy_refused": 0, + "unknown_no_fallback_refused": 0, + "malformed_rejected": 0, + "sniff_timeout": 0, + "client_hello_too_large": 0 + } + } + ], + "routers": [], + "services": [], + "counters": { + "active_tcp_connections": 1, + "active_udp_sessions": 0, + "total_accepted": 12, + "total_refused": 0, + "total_errors": 0, + "bytes_in": 4096, + "bytes_out": 8192, + "smart_tcp": { + "http_accepted": 0, + "h2c_accepted": 0, + "https_fallback_accepted": 0, + "tls_passthrough_accepted": 0, + "raw_fallback_accepted": 0, + "entrypoint_cidr_refused": 0, + "raw_fallback_cidr_refused": 0, + "proxy_refused": 0, + "unknown_no_fallback_refused": 0, + "malformed_rejected": 0, + "sniff_timeout": 0, + "client_hello_too_large": 0 + } + } +} +``` + +--- + +## gordon daemon networks + +Display Docker networks managed by Gordon, including which containers +are connected to each network. + +```bash +gordon daemon networks +gordon daemon networks --json +gordon daemon networks --remote prod +``` + +### Flags + +| Flag | Description | +|------|-------------| +| `--json` | Output as JSON | + +### JSON Output + +```json +[ + { + "name": "gordon_myapp", + "driver": "bridge", + "containers": ["container1", "container2"] + } +] +``` + +## Related + +- [CLI Overview](./index.md) +- [Apps Commands](./apps.md) +- [Serve Command](./serve.md) +- [Traffic configuration](../config/traffic.md) +- [Remote CLI Management](/wiki/guides/remote-cli.md) diff --git a/docs/cli/images.md b/docs/cli/images.md index dfc15a855..c81f6928c 100644 --- a/docs/cli/images.md +++ b/docs/cli/images.md @@ -1,6 +1,6 @@ # Images Command -List and prune runtime/registry images. +Push, list, and prune runtime/registry images. ## gordon images @@ -10,12 +10,126 @@ gordon images Subcommands: +- `push` - Tag and push an image to the Gordon registry (never deploys). - `list` - List runtime images and registry tags. - `prune` - Prune dangling runtime images and old registry tags. - `tags` - List registry tags for a specific repository. > **Note:** Images commands require remote mode (`--remote` + `--token`, or configured remotes). +## gordon images push + +Tag and push an image to the Gordon registry. Push transfers OCI content +only: it never deploys. Deploy separately with `gordon apps deploy` after +applying the manifest that references the pushed tag. + +### Synopsis + +```bash +gordon images push [image] [options] +``` + +### Arguments + +| Argument | Description | +|----------|-------------| +| `[image]` | Image name to push (optional). If omitted, auto-detected from Dockerfile labels or current directory name | + +### Options + +| Option | Description | +|--------|-------------| +| `--build` | Build the image first using `docker buildx` | +| `-f, --file` | Path to Dockerfile (default: `./Dockerfile`, used with `--build`) | +| `--platform` | Target platform for buildx (default: `linux/amd64`) | +| `--build-arg` | Additional build args (repeatable, `KEY=VALUE`) | +| `--tag` | Override pushed version tag (default: tag ref from CI, then `git describe --tags --dirty`) | +| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | +| `--token` | Authentication token for remote | + +### Description + +`gordon images push` tags the selected image for the Gordon registry and pushes it. + +If you do not pass `--remote` and no active remote is configured, Gordon tries to +infer the correct saved remote by probing your saved remotes for the image. +It auto-selects only when exactly one remote matches. If multiple remotes match, +or a saved remote cannot be probed safely, the command stops and asks you to use +`--remote` explicitly. + +Domain-style push targets are retired: push an image name, then +`gordon apps apply` the manifest that references the pushed tag. + +- The version tag defaults to a CI tag ref (like `refs/tags/v1.2.3`) when available, + then falls back to `git describe --tags --dirty` (for example + `v1.2.3-4-gabc1234` or `v1.2.3-dirty`). If no tag is found, `latest` is used. +- When `--build` is set, the command builds with `docker buildx build --load` + and injects `VERSION`, `GIT_TAG`, `GIT_SHA`, and `BUILD_TIME` into the build + environment plus any `--build-arg` values. To use these in your Dockerfile, + declare them with `ARG` (e.g., `ARG VERSION`) then reference via `ENV` or + in build steps. +- Use `-f/--file` to build from a Dockerfile outside the current directory root. +- The version tag and `latest` are both pushed (unless the version is `latest`). + +### Authentication + +When used with `--remote`, gordon images push authenticates in two ways: + +- **Admin API** (tag listing for inference): uses `--token` or `$GORDON_TOKEN` as Bearer token +- **Registry push**: automatically exchanges the token for a short-lived (5 min) registry access token via `/auth/token` -- no `docker login` required + +This means CI/CD pipelines only need a single secret (`GORDON_TOKEN`). + +### Version Auto-Detection + +Gordon reads version tags from CI environment variables (in priority order): + +| CI System | Variable | Example | +|-----------|----------|---------| +| GitHub Actions | `$GITHUB_REF` | `refs/tags/v1.2.0` | +| GitHub Actions | `$GITHUB_REF_TYPE` + `$GITHUB_REF_NAME` | `tag` + `v1.2.0` | +| GitLab CI | `$CI_COMMIT_TAG` | `v1.2.0` | +| Azure DevOps | `$BUILD_SOURCEBRANCH` | `refs/tags/v1.2.0` | +| Any | `git describe --tags` | `v1.2.3-4-gabc1234` | +| Fallback | - | `latest` | + +### Examples + +```bash +# Build, push (auto-detect image name) +gordon images push --build --remote https://gordon.example.com + +# Push an image +gordon images push myapp --remote https://gordon.example.com + +# Push a fully qualified ref +gordon images push registry.example.com/myapp:v1.2.3 --build --remote https://gordon.example.com + +# Push existing local image with an explicit tag +gordon images push myapp --tag v1.2.0 --remote https://gordon.example.com + +# Build for ARM and pass build args +gordon images push myapp --build --platform linux/arm64 --build-arg CGO_ENABLED=0 --remote https://gordon.example.com + +# Build from a custom Dockerfile path +gordon images push myapp --build -f docker/app/Dockerfile --remote https://gordon.example.com + +# CI/CD usage (single env var, no docker login needed) +export GORDON_TOKEN="your-token" +gordon images push myapp --build --remote https://gordon.example.com +``` + +### Notes + +- Remote mode required. See [CLI Overview](./index.md) for targeting options. +- `--build` requires Docker with Buildx. Docker Desktop includes it; on Linux, + install the `docker-buildx-plugin` package. +- Gordon uses native registry uploads instead of shelling out to `docker push`. + Image layers are sent in 50MB chunks, which stays under Cloudflare's 100MB + per-request limit so proxied pushes keep working. Keep the server's + `max_blob_chunk_size` larger than the client chunk size; the default `95MB` + works out of the box. + ## gordon images list ```bash @@ -59,11 +173,13 @@ gordon images prune [--dry-run] [--keep-releases ] [--dangling] [--registry] By default, prune removes dangling runtime images **and** applies registry tag retention (keeping `latest` + 3 previous non-`latest` tags per repository). A confirmation prompt is shown before destructive operations. +A prune only ever deletes resources it can positively prove safe. Every other candidate is reported as `protected` or `unknown` and left in place; the command still succeeds. See [Prune safety](#prune-safety). + Flags: | Flag | Default | Description | |------|---------|-------------| -| `--dry-run` | `false` | Show prune behavior without applying changes | +| `--dry-run` | `false` | Run the full inventory and planning, then report; nothing is deleted | | `--keep-releases` | `3` | Number of previous non-`latest` tags to keep per repository (`latest` is always preserved) | | `--dangling` | `false` | Restrict scope to dangling runtime images only | | `--registry` | `false` | Restrict scope to registry tag retention only | @@ -80,7 +196,22 @@ Flags: - `latest` is always preserved when present. - `--keep-releases` counts non-`latest` tags, ordered by most recent first. -- `--keep-releases=0` with registry scope enabled still runs registry cleanup but keeps no non-`latest` tags. +- A tag named by durable app state survives beyond the retention window. +- `--keep-releases=0` skips registry tag and blob cleanup entirely; dangling runtime prune still runs. + +## Prune safety + +Gordon prunes by ownership, not by name or age. Each candidate gets one verdict: + +| Verdict | Meaning | +|---------|---------| +| `eligible` | Every fact needed to prove the resource safe was read completely, and no durable record claims it. It is deleted. | +| `protected` | A durable fact claims it: a desired or active app service (including stopped apps), a recovery inhibition, a staged or committed apply intent, an unfinished operation journal entry, container use, a pending upload, an ownership record, or the OCI closure of any of those. It survives. | +| `unknown` | A fact needed to prove it safe was missing, unreadable, or unsupported. It survives. | + +Runtime images used by any container, running or stopped, always survive. Registry blobs reachable from a retained or protected manifest survive, including blobs shared with another tag. When a retained manifest cannot be read, the blobs whose safety depended on it become `unknown` rather than eligible. + +Both `--dry-run` and the executed prune return the same report: the verdict counts, the deleted identities, any deletion failures, every inventory gap, and the skipped candidates with their reason codes. A prune that deletes nothing is a success. ## gordon images tags @@ -128,7 +259,8 @@ gordon images prune --dangling # Prune registry tags only, keeping latest + 5 previous gordon images prune --registry --keep-releases 5 -# Preview cleanup without applying +# Inspect what prune would remove without applying +# (also lists every skipped candidate with its reason) gordon images prune --dry-run # Skip confirmation prompt diff --git a/docs/cli/index.md b/docs/cli/index.md index 75648344a..b1cfafdd1 100644 --- a/docs/cli/index.md +++ b/docs/cli/index.md @@ -1,8 +1,8 @@ # CLI Commands -Gordon provides a command-line interface for server management, deployment, and authentication. +Gordon provides a command-line interface for server management, app deployment, and authentication. -Most `list` commands also support `--json` for machine-readable output. +Most `list`/`show` commands also support `--json` for machine-readable output. Commands are organized by where they run: @@ -20,30 +20,15 @@ Commands are organized by where they run: ## Management Commands (local or remote) -Management commands run locally through in-process services by default. Add `--remote` to target another Gordon instance. +Management commands use the authenticated daemon API. By default they connect through the owner-only local Unix socket; add `--remote` to use authenticated HTTP/TLS against another Gordon instance. | Command | Description | Documentation | |---------|-------------|---------------| -| `gordon attachments` | Manage container attachments | [attachments](./attachments.md) | -| `gordon autoroute` | Manage auto-route domain allowlist | [autoroute](./autoroute.md) | -| `gordon backups` | Manage database backups | [backup](./backup.md) | -| `gordon bootstrap` | Configure a route, attachments, and secrets for an app | [bootstrap](./bootstrap.md) | -| `gordon config show` | Show server configuration | [config](./config.md) | -| `gordon deploy` | Manually deploy or redeploy a route | [serve](./serve.md#gordon-deploy) | -| `gordon images` | List and prune images | [images](./images.md) | -| `gordon logs` | Display Gordon process or container logs | [serve](./serve.md#gordon-logs) | -| `gordon networks list` | List Gordon-managed Docker networks | [networks](./networks.md) | -| `gordon preview` | Create or manage preview environments | [preview](../config/preview.md) | -| `gordon push` | Tag, push, and optionally deploy an image | [push](./push.md) | -| `gordon attachments push` | Build/push attachment images to registry | [attachments](./attachments.md) | -| `gordon reload` | Reload configuration and sync containers | [serve](./serve.md#gordon-reload) | -| `gordon restart` | Restart a running container | [restart](./restart.md) | -| `gordon pin` | Pin a route to a specific image tag | [pin](./pin.md) | -| `gordon routes` | Manage routes | [routes](./routes.md) | -| `gordon secrets` | Manage secrets | [secrets](./secrets.md) | -| `gordon status` | Show Gordon server status | [status](./status.md) | -| `gordon traffic` | Inspect traffic plane status | [traffic](./traffic.md) | -| `gordon volumes` | Manage volumes | - | +| `gordon apps` | Manage applications, operations, and app secrets | [apps](./apps.md) | +| `gordon backups` | Manage declared app database and volume backups | [backup](./backup.md) | +| `gordon daemon` | Daemon status, logs, reload, config, TLS, traffic, and networks | [daemon](./daemon.md) | +| `gordon images` | Push, list, and prune images | [images](./images.md) | +| `gordon volumes` | Manage volumes | [volumes](./volumes.md) | ## Client Commands @@ -60,54 +45,59 @@ Management commands run locally through in-process services by default. Add `--r gordon serve gordon serve --config /path/to/config.toml -# Reload configuration -gordon reload - -# Deploy a specific route -gordon deploy myapp.example.com - -# Restart a running container -gordon restart myapp.example.com - -# First-time route setup -gordon bootstrap app.example.com myapp:latest --attachment postgres:18 --env APP_ENV=production - -# Then push and deploy -gordon push myapp:latest --domain app.example.com --build --no-confirm - -# Push an image and deploy -gordon push myapp --build - -# Push and deploy without confirmation -gordon push myapp --no-confirm - -# Pin to a specific tag -gordon pin myapp.example.com +# Reload installation configuration (never activates app state) +gordon daemon reload + +# Applications (daemon-owned; fail without a reachable daemon) +gordon apps apply --file blog.toml +gordon apps apply --file blog.toml --deploy +gordon apps list +gordon apps show blog +gordon apps diff blog +gordon apps deploy blog +gordon apps restart blog +gordon apps stop blog +gordon apps start blog +gordon apps remove blog +gordon apps operations show blog --key +gordon apps operations watch blog --key +gordon apps secrets list blog + +# Daemon +gordon daemon status +gordon daemon config show +gordon daemon config validate --file /path/to/gordon.toml +gordon daemon tls +gordon daemon networks + +# Push an image (OCI transfer only; deploy separately) +gordon images push myapp --build --remote prod # View logs -gordon logs # Gordon process logs -gordon logs -f # Follow process logs -gordon logs -n 100 # Last 100 lines -gordon logs myapp.example.com # Container logs for myapp.example.com -gordon logs myapp.example.com -f # Follow container logs +gordon daemon logs # Gordon process logs +gordon daemon logs -f # Follow process logs +gordon daemon logs -n 100 # Last 100 process-log lines +gordon apps logs blog --service web # App service logs +gordon apps logs blog --service web --follow # Follow app service logs # Check version gordon version # Traffic plane -gordon traffic status --remote prod -gordon traffic status --remote prod --json +gordon daemon traffic --remote prod +gordon daemon traffic --remote prod --json # Backups gordon backups list -gordon backups run app.example.com -gordon backups detect app.example.com +gordon backups run shop --service api --database orders +gordon backups volume run shop --service api --volume data gordon backups status # Images +gordon images push myapp --build --remote prod gordon images list -gordon images prune --runtime-only -gordon images prune --keep 3 +gordon images prune --dry-run +gordon images prune --keep-releases 3 # Authentication gordon auth login --remote https://gordon.example.com --token $TOKEN @@ -119,43 +109,14 @@ gordon auth token list gordon auth token revoke gordon auth internal -# Routes -gordon routes list -gordon routes status -gordon routes add myapp.example.com myapp:latest -gordon routes remove myapp.example.com -gordon routes deploy myapp.example.com - -# Attachments -gordon attachments list -gordon attachments add app.example.com postgres:18 -gordon attachments remove app.example.com postgres:18 - -# Secrets -gordon secrets list myapp.example.com -gordon secrets set myapp.example.com DATABASE_URL "postgres://..." -gordon secrets remove myapp.example.com DATABASE_URL - # Remotes gordon remotes add prod https://gordon.mydomain.com --token $TOKEN gordon remotes list gordon remotes use prod -# Preview environments -gordon preview create app.example.com --branch feature-x -gordon preview list -gordon preview extend app.example.com --branch feature-x -gordon preview delete app.example.com --branch feature-x - # Volumes gordon volumes list gordon volumes prune - -# Auto-route allowlist -gordon autoroute allow list -gordon autoroute allow add example.com -gordon autoroute allow add "*.staging.example.com" -gordon autoroute allow remove example.com ``` ## Global Options @@ -173,64 +134,35 @@ The CLI can target remote Gordon instances using client config, an active remote or `GORDON_REMOTE` environment variable. Use `--remote` and `--token` as global overrides when you want to bypass your saved configuration. -When no explicit remote is selected and no active remote is configured, Gordon can now -**auto-infer a saved remote** for commands that already have a concrete target. It probes your -saved remotes and uses the remote automatically when exactly one matches. If multiple remotes -match, Gordon stops with an ambiguity error and asks you to use `--remote`. If any remote probe -fails, Gordon also stops rather than guessing. - -Auto-inference currently applies to target-based commands such as: -- `gordon push` -- `gordon attachments push` -- `gordon deploy ` -- `gordon restart ` -- `gordon pin ` / `gordon pin list ` -- `gordon routes show ` / `gordon routes remove ` -- `gordon secrets list|set|remove ` -- `gordon attachments list ` / `gordon attachments add|remove ...` -- `gordon backups list ` / `gordon backups run ` / `gordon backups detect ` -- `gordon images tags ` -- `gordon logs ` - -`gordon routes list` and `gordon routes status` are the exceptions: when neither `--remote` -nor `GORDON_REMOTE` is set, they show local routes first, then every saved remote. Set either -one to force a single target. +When no explicit remote is selected and no active remote is configured, Gordon can +**auto-infer a saved remote** for `gordon images push`. It probes your saved remotes and uses the +remote automatically when exactly one matches. If multiple remotes match, Gordon stops with +an ambiguity error and asks you to use `--remote`. If any remote probe fails, Gordon also +stops rather than guessing. ```bash -# Aggregate routes views -gordon routes list -gordon routes status - -# Single-target override -gordon routes list --remote prod -GORDON_REMOTE=prod gordon routes status - # Auto-inferred single match from saved remotes -gordon push myapp --build -gordon deploy app.example.com +gordon images push myapp --build ``` **Important:** The remote URL must be the `gordon_domain` configured on the remote Gordon instance. This is the domain that serves both the container registry and the Admin API. -If remote CLI gets a `404` during `/auth/token` exchange, the server likely still sets only `server.registry_domain` and needs `server.gordon_domain` configured. Use `--insecure` when the remote endpoint uses a self-signed or otherwise untrusted TLS certificate. You can make this persistent with `insecure_tls = true` in `[client]` of `~/.config/gordon/gordon.toml` or in a specific entry in `~/.config/gordon/remotes.toml`. -For Tailscale setups, you can also avoid `--insecure` by using the machine `*.ts.net` name with Tailscale-issued TLS certs in Gordon server config. -Your public app domains can still use wildcard DNS and normal reverse-proxy routing. ```bash # Using flags (use the gordon_domain from remote Gordon config) -gordon routes list --remote https://gordon.example.com --token $TOKEN +gordon daemon status --remote https://gordon.example.com --token $TOKEN # Against self-signed/private CA endpoint -gordon --remote https://gordon.example.com --token $TOKEN --insecure status +gordon --remote https://gordon.example.com --token $TOKEN --insecure daemon status # Using environment variables export GORDON_REMOTE=https://gordon.example.com export GORDON_TOKEN=$TOKEN export GORDON_INSECURE=true -gordon routes list +gordon daemon status ``` ## Exit Codes diff --git a/docs/cli/networks.md b/docs/cli/networks.md deleted file mode 100644 index cda021a3d..000000000 --- a/docs/cli/networks.md +++ /dev/null @@ -1,51 +0,0 @@ -# Networks Commands - -Inspect Gordon-managed Docker networks. - -Remote targeting uses client config or an active remote by default. -Use `--remote` and `--token` to override. See [CLI Overview](./index.md). - -## gordon networks - -### Subcommands - -| Subcommand | Description | -|------------|-------------| -| `list` | List Gordon-managed Docker networks | - ---- - -## gordon networks list - -Display Docker networks managed by Gordon, including which containers -are connected to each network. - -```bash -gordon networks list -gordon networks list --json -gordon networks list --remote https://gordon.mydomain.com --token $TOKEN -``` - -### Flags - -| Flag | Description | -|------|-------------| -| `--json` | Output as JSON | - -### JSON Output - -```json -[ - { - "name": "gordon_myapp", - "driver": "bridge", - "containers": ["container1", "container2"] - } -] -``` - -## Related - -- [CLI Overview](./index.md) -- [Routes Command](./routes.md) -- [Attachments Command](./attachments.md) diff --git a/docs/cli/pin.md b/docs/cli/pin.md deleted file mode 100644 index 259e34628..000000000 --- a/docs/cli/pin.md +++ /dev/null @@ -1,72 +0,0 @@ -# Pin Command - -Pin a route to a specific image tag. - -## gordon pin - -### Synopsis - -```bash -gordon pin [options] -gordon pin list -``` - -### Arguments - -| Argument | Description | -|----------|-------------| -| `` | The route domain to pin | - -### Options - -| Option | Description | -|--------|-------------| -| `--tag` | Target tag (skips interactive selection) | -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | -| `--json` | Output as JSON (for `pin list`) | - -### Description - -`gordon pin` deploys a selected image tag for the given route. Tags are -listed from the Gordon registry, with semver tags sorted in descending -order first, followed by non-semver tags. - -Use cases: -- **Rollback** — revert to a previous stable version -- **Forward** — advance to a newer version -- **Lock** — pin to a specific version instead of following `latest` -- **Test** — switch to an experimental tag - -`gordon pin list ` lists available tags without deploying and -marks the current tag. - -If the selected tag matches the current running tag, no action is taken. - -### Examples - -```bash -# Interactive selection -gordon pin myapp.example.com - -# Pin to a specific tag -gordon pin myapp.example.com --tag v1.2.0 - -# List available tags for a domain -gordon pin list myapp.example.com - -# List available tags as JSON -gordon pin list myapp.example.com --json -``` - -### Notes - -- Local by default; remote mode optional. See [CLI Overview](./index.md) for targeting options. -- Tags are read from the Gordon registry configured on the server. - -## Related - -- [CLI Overview](./index.md) -- [Push Command](./push.md) -- [Routes Command](./routes.md) -- [Deployment Rollback Strategies](../deployment/rollback.md) diff --git a/docs/cli/preview.md b/docs/cli/preview.md deleted file mode 100644 index 8360b6f7c..000000000 --- a/docs/cli/preview.md +++ /dev/null @@ -1,61 +0,0 @@ -# gordon preview - -Create and manage ephemeral preview environments. Each preview gets its own subdomain, container, and optionally cloned volumes from the base route. - -## Commands - -### `gordon preview create` - -Build, push, and deploy a preview environment. - -```bash -gordon preview create myapp -gordon preview create myapp --ttl 72h -gordon preview create myapp --no-build --no-data -``` - -| Flag | Description | -|------|-------------| -| `--ttl` | Override TTL duration (e.g., `72h`) | -| `--no-build` | Skip image build (use existing image) | -| `--no-data` | Skip volume cloning from base route | -| `--platform` | Target platform for build (default `linux/amd64`) | - -### `gordon preview list` - -List active preview environments with status and remaining TTL. - -```bash -gordon preview list -gordon preview list --json -``` - -| Flag | Description | -|------|-------------| -| `--json` | Output as JSON | - -### `gordon preview delete` - -Tear down a preview environment and clean up its resources. - -```bash -gordon preview delete myapp-feature-x -``` - -### `gordon preview extend` - -Extend the TTL of an active preview. - -```bash -gordon preview extend myapp-feature-x -gordon preview extend myapp-feature-x --ttl 48h -``` - -| Flag | Description | -|------|-------------| -| `--ttl` | Additional TTL to add (default `24h`) | - -## Related - -- [Preview Configuration](../config/preview.md) -- [Deploy](./serve.md#gordon-deploy) diff --git a/docs/cli/push.md b/docs/cli/push.md deleted file mode 100644 index 0a4f02b4c..000000000 --- a/docs/cli/push.md +++ /dev/null @@ -1,151 +0,0 @@ -# Push Command - -Tag, push, and optionally deploy an image to the Gordon registry. - -## gordon push - -### Synopsis - -```bash -gordon push [image] [options] -``` - -### Arguments - -| Argument | Description | -|----------|-------------| -| `[image]` | Image name to push (optional). If omitted, auto-detected from Dockerfile labels or current directory name | - -### Options - -| Option | Description | -|--------|-------------| -| `--build` | Build the image first using `docker buildx` | -| `-f, --file` | Path to Dockerfile (default: `./Dockerfile`, used with `--build`) | -| `--platform` | Target platform for buildx (default: `linux/amd64`) | -| `--build-arg` | Additional build args (repeatable, `KEY=VALUE`) | -| `--tag` | Override pushed version tag (default: tag ref from CI, then `git describe --tags --dirty`) | -| `--no-confirm` | Skip deploy confirmation prompt | -| `--no-deploy` | Push only; skip deployment prompt | -| `--domain` | Explicit deploy target override | -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | - -### Description - -`gordon push` tags the selected image for the Gordon registry, pushes it, and -optionally deploys matching routes. - -If you do not pass `--remote` and no active remote is configured, Gordon tries to -infer the correct saved remote by probing your saved remotes for a matching route. -It auto-selects only when exactly one remote matches. If multiple remotes match, -or a saved remote cannot be probed safely, the command stops and asks you to use -`--remote` explicitly. - -Image resolution order: -1. `--domain` is the explicit deploy target override for legacy workflows -2. Positional refs resolve routes by image name; tagged refs still use the image name for lookup -3. Dotted positional refs probe image routes first, then fall back to legacy domain lookup -4. No-arg mode auto-detects from the Dockerfile label or current directory - -The pushed version still comes from `--tag`, CI tag refs, or `git describe --tags --dirty`. - -To push attachment images (databases, caches, etc.), use `gordon attachments push`. - -- For first deploys, run `gordon bootstrap` first to create or update the route, attachments, and secrets, then run `gordon push` to upload and deploy the image. -- The registry and repository are derived from the route image on the server. -- The version tag defaults to a CI tag ref (like `refs/tags/v1.2.3`) when available, - then falls back to `git describe --tags --dirty` (for example - `v1.2.3-4-gabc1234` or `v1.2.3-dirty`). If no tag is found, `latest` is used. -- When `--build` is set, the command builds with `docker buildx build --load` - and injects `VERSION`, `GIT_TAG`, `GIT_SHA`, and `BUILD_TIME` into the build - environment plus any `--build-arg` values. To use these in your Dockerfile, - declare them with `ARG` (e.g., `ARG VERSION`) then reference via `ENV` or - in build steps. -- Use `-f/--file` to build from a Dockerfile outside the current directory root. -- The version tag and `latest` are both pushed (unless the version is `latest`). - -### Authentication - -When used with `--remote`, gordon push authenticates in two ways: - -- **Admin API** (route resolution, deploy): uses `--token` or `$GORDON_TOKEN` as Bearer token -- **Registry push**: automatically exchanges the token for a short-lived (5 min) registry access token via `/auth/token` -- no `docker login` required - -This means CI/CD pipelines only need a single secret (`GORDON_TOKEN`). - -### Deploy Modes - -When the token has `admin:config:write` scope, the CLI manages deployment explicitly -(DeployIntent → push → Deploy). This gives the CLI control over deploy timing. - -When the token lacks `admin:config:write`, the CLI pushes the image and the server -auto-deploys when it receives it via its event listener. The CLI logs: -`info: deploy intent skipped (insufficient scope), server will auto-deploy on image receive` - -### Version Auto-Detection - -Gordon reads version tags from CI environment variables (in priority order): - -| CI System | Variable | Example | -|-----------|----------|---------| -| GitHub Actions | `$GITHUB_REF` | `refs/tags/v1.2.0` | -| GitHub Actions | `$GITHUB_REF_TYPE` + `$GITHUB_REF_NAME` | `tag` + `v1.2.0` | -| GitLab CI | `$CI_COMMIT_TAG` | `v1.2.0` | -| Azure DevOps | `$BUILD_SOURCEBRANCH` | `refs/tags/v1.2.0` | -| Any | `git describe --tags` | `v1.2.3-4-gabc1234` | -| Fallback | - | `latest` | - -### Examples - -```bash -# Build, push, and deploy (auto-detect image name) -gordon push --build --remote https://gordon.example.com --no-confirm - -# Push an image and deploy it -gordon push myapp --remote https://gordon.example.com --no-confirm - -# Push and deploy to an explicit domain -gordon push myapp:latest --domain app.example.com --remote https://gordon.example.com --no-confirm - -# Tagged refs still resolve routes by image name -gordon push myapp:v1.2.3 --tag v1.2.3 --no-deploy - -# Push existing local image, skip deploy -gordon push myapp --tag v1.2.0 --no-deploy - -# Build for ARM and pass build args -gordon push myapp --build --platform linux/arm64 --build-arg CGO_ENABLED=0 - -# Build from a custom Dockerfile path -gordon push myapp --build -f docker/app/Dockerfile - -# Legacy compatibility: domain-looking positional target -gordon push app.example.com --no-confirm - -# CI/CD usage (single env var, no docker login needed) -export GORDON_TOKEN="your-token" -gordon push myapp --build --remote https://gordon.example.com --no-confirm -``` - -### Notes - -- Remote mode required. See [CLI Overview](./index.md) for targeting options. -- `gordon push` requires the target route to already exist so it can resolve the deploy target. - Use `gordon bootstrap` for first deploys. -- `--build` requires Docker with Buildx. Docker Desktop includes it; on Linux, - install the `docker-buildx-plugin` package. -- Gordon uses native registry uploads instead of shelling out to `docker push`. - Image layers are sent in 50MB chunks, which stays under Cloudflare's 100MB - per-request limit so proxied pushes keep working. Keep the server's - `max_blob_chunk_size` larger than the client chunk size; the default `95MB` - works out of the box. - -## Related - -- [CLI Overview](./index.md) -- [Deployment Overview](../deployment/index.md) -- [GitHub Actions](../deployment/github-actions.md) -- [Attachments Commands](./attachments.md) -- [Routes Command](./routes.md) -- [Authentication](../config/auth.md) diff --git a/docs/cli/remotes.md b/docs/cli/remotes.md index df4a45ae3..6bd6edd2c 100644 --- a/docs/cli/remotes.md +++ b/docs/cli/remotes.md @@ -163,15 +163,13 @@ gordon remotes use ### Description When a remote is active, it's used automatically for most remote-capable commands without needing to specify `--remote` and `--token`. -`gordon routes list` and `gordon routes status` are the exception: they aggregate local + saved remotes unless you set `--remote` or `GORDON_REMOTE`. -If no remote is active and you do not pass `--remote`, Gordon can also auto-infer a saved remote for commands with a concrete target (for example `gordon push myapp`, `gordon deploy app.example.com`, or `gordon images tags myapp`). It only auto-selects when exactly one saved remote matches. Ambiguous matches and probe failures require an explicit `--remote`. +If no remote is active and you do not pass `--remote`, Gordon can also auto-infer a saved remote for `gordon images push` and `gordon images tags `. It only auto-selects when exactly one saved remote matches. Ambiguous matches and probe failures require an explicit `--remote`. ```bash gordon remotes use prod -gordon secrets list app.com # Uses prod remote automatically -gordon routes list # Aggregates local + saved remotes -gordon routes status # Aggregates local + saved remotes +gordon apps list # Uses prod remote automatically +gordon daemon status # Uses prod remote automatically ``` ### Example @@ -246,7 +244,7 @@ token = "eyJ..." When multiple sources specify remote or token, the CLI uses this priority: -For `routes list` and `routes status`, `--remote` and `GORDON_REMOTE` are the explicit single-target selectors. Without either one, those commands aggregate local + saved remotes even when an active remote exists. +For aggregate views, `--remote` and `GORDON_REMOTE` are the explicit single-target selectors. Without either one, those commands aggregate local + saved remotes even when an active remote exists. Auto-inference is not used for those aggregate views. **Remote target selection:** @@ -271,13 +269,8 @@ Auto-inference is not used for those aggregate views. This allows overriding specific values while keeping defaults: ```bash -# Aggregate routes views -gordon routes list -gordon routes status - # Force one target -GORDON_REMOTE=prod gordon routes list -GORDON_REMOTE=prod gordon routes status +GORDON_REMOTE=prod gordon daemon status ``` --- @@ -294,12 +287,12 @@ gordon remotes add dev https://gordon.dev.example.com --token-env DEV_TOKEN # Work with prod gordon remotes use prod -gordon secrets list myapp.example.com -gordon routes list --remote prod +gordon apps list +gordon daemon status # Switch to staging gordon remotes use staging -GORDON_REMOTE=staging gordon routes status +GORDON_REMOTE=staging gordon daemon status ``` ### CI/CD Pipeline @@ -311,25 +304,21 @@ env: GORDON_TOKEN: ${{ secrets.GORDON_TOKEN }} steps: - - name: Deploy to Gordon + - name: Push to Gordon run: | - gordon routes deploy myapp.example.com + gordon images push myapp --build --remote ${{ secrets.GORDON_URL }} ``` ### Compare Environments ```bash -# Aggregate route inventory/status across local + saved remotes -gordon routes list -gordon routes status - # Compare one target at a time -gordon routes list --remote https://gordon.example.com --token $PROD_TOKEN -GORDON_REMOTE=staging gordon routes status +gordon daemon status --remote https://gordon.example.com --token $PROD_TOKEN +GORDON_REMOTE=staging gordon daemon status # Or switch between active remotes for other commands -gordon remotes use prod && gordon secrets list myapp.example.com -gordon remotes use staging && gordon secrets list myapp.example.com +gordon remotes use prod && gordon apps list +gordon remotes use staging && gordon apps list ``` ### Private Admin + Wildcard App Domains @@ -354,7 +343,7 @@ A common Tailscale setup is to point your domain's DNS at the machine's Tailscal If you have your own certificate for the domain (e.g. from a corporate CA), you can provide it via `tls_cert_file`/`tls_key_file` — see [Server Configuration](../config/server.md#custom-certificates). The static cert is served for SNI-matching domains; everything else falls through to the internal CA. -`insecure_tls` only affects CLI -> Gordon admin HTTPS verification. It does not change runtime routing: Gordon reverse proxy and container routes can still serve wildcard app domains like `*.example.com`. +`insecure_tls` only affects CLI -> Gordon admin HTTPS verification. It does not change runtime routing: the reverse proxy can still serve wildcard app domains like `*.example.com`. ## Migration from [client] Config diff --git a/docs/cli/restart.md b/docs/cli/restart.md deleted file mode 100644 index 984f4d50f..000000000 --- a/docs/cli/restart.md +++ /dev/null @@ -1,54 +0,0 @@ -# Restart Command - -Restart a running container for a route. - -## gordon restart - -### Synopsis - -```bash -gordon restart [options] -``` - -### Arguments - -| Argument | Description | -|----------|-------------| -| `` | The route domain to restart | - -### Options - -| Option | Description | -|--------|-------------| -| `--with-attachments` | Also restart attached services (databases, caches) | -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | - -### Description - -Restarts the container for the specified route. This is useful after updating -environment variables or secrets without performing a full redeploy. - -When `--with-attachments` is set, Gordon also restarts any attachment containers -for the route. - -### Examples - -```bash -# Restart main container only -gordon restart myapp.example.com - -# Restart with attachments -gordon restart myapp.example.com --with-attachments -``` - -### Notes - -- Local mode works when run on the Gordon host. It uses the same deploy-signal path as `gordon deploy `. -- `--with-attachments` is only supported in remote mode. -- In remote mode, target your Gordon admin endpoint (for example `https://gordon.example.com`), not the registry host (for example `https://reg.example.com`). - -## Related - -- [CLI Overview](./index.md) -- [Secrets Command](./secrets.md) diff --git a/docs/cli/routes.md b/docs/cli/routes.md deleted file mode 100644 index 117c7b2c4..000000000 --- a/docs/cli/routes.md +++ /dev/null @@ -1,442 +0,0 @@ -# Routes Commands - -Manage routes on local or remote Gordon instances. - -Most remote-capable commands use client config or an active remote by default. -`gordon routes list` and `gordon routes status` are different: when neither -`--remote` nor `GORDON_REMOTE` is set, they show local routes first, then each -saved remote under its own heading. See [CLI Overview](./index.md). - -## gordon routes - -### Subcommands - -| Subcommand | Description | -|------------|-------------| -| `list` | List routes by domain and image | -| `status` | Show detailed route status | -| `show` | Show details for a single route | -| `add` | Create or update a route | -| `remove` | Remove a route | -| `deploy` | Deploy a specific route | - ---- - -## gordon routes list - -List routes by domain and image only. - -When no target is selected, Gordon shows local routes first, then each saved -remote under its own heading. Use `--remote` or `GORDON_REMOTE` to show one -target only. - -```bash -gordon routes list -gordon routes list --json -gordon routes list --remote https://gordon.mydomain.com --token $TOKEN -GORDON_REMOTE=prod gordon routes list -``` - -### Output - -```text -Routes - -Local - app.example.com myapp:latest - api.example.com myapi:v2.1.0 - -Remote: hetzner-vps - gordon.example.com gordon-webapp:latest - -Remote: igor - grafana.supri.xyz grafana - test.supri.xyz hello-test -``` - -### JSON Output - -```bash -gordon routes list --json -``` - -```json -[ - { - "kind": "local", - "name": "local", - "routes": [ - { - "domain": "app.example.com", - "image": "myapp:latest" - } - ] - }, - { - "kind": "remote", - "name": "igor", - "url": "https://gordon.supri.xyz", - "routes": [ - { - "domain": "grafana.supri.xyz", - "image": "grafana" - } - ] - } -] -``` - -Single-target mode still returns a one-element array. Sections can also include -an `error` field when a target is unavailable. - -### Options - -| Option | Description | -|--------|-------------| -| `--json` | Output routes as JSON | -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | - ---- - -## gordon routes status - -Show detailed route status for each target. - -`routes status` uses the same target selection rules as `routes list`. When no -target is selected, Gordon shows local status first, then each saved remote -under its own heading. - -```bash -gordon routes status -gordon routes status --json -gordon routes status --remote https://gordon.mydomain.com --token $TOKEN -GORDON_REMOTE=prod gordon routes status -``` - -### Output - -```text -Route Status - -Local - - -Remote: hetzner-vps - - -Remote: igor - -``` - -The rich view keeps network grouping, container status, HTTP probe status, and -attachments within each target. - -### JSON Output - -```json -[ - { - "kind": "local", - "name": "local", - "routes": [ - { - "domain": "app.example.com", - "image": "myapp:latest", - "container_id": "abcd1234", - "container_status": "running", - "http_status": 200, - "network": "gordon-shared", - "attachments": [ - { - "name": "postgres", - "image": "postgres:18", - "status": "running" - } - ] - } - ] - } -] -``` - -Single-target mode still returns a one-element array. Sections can also include -an `error` field when a target is unavailable. - -### Options - -| Option | Description | -|--------|-------------| -| `--json` | Output routes as JSON | -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | - ---- - -## gordon routes add - -Create a new route or update an existing route. - -```bash -gordon routes add -gordon routes add myapp.example.com myapp:latest -``` - -If the route already exists, Gordon updates it to the new image instead of failing. -The image does not need to be pushed to the Gordon registry before you add the route. - -### Arguments - -| Argument | Description | -|----------|-------------| -| `` | The domain name for the route | -| `` | The container image to deploy | - -### Options - -| Option | Description | -|--------|-------------| -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | - -### Examples - -```bash -# Local -gordon routes add myapp.example.com myapp:latest -gordon routes add api.example.com myapi:v2.1.0 - -# Update an existing route -gordon routes add myapp.example.com myapp:v2 - -# Remote (override) -gordon routes add myapp.example.com myapp:latest --remote https://gordon.mydomain.com --token $TOKEN -``` - -### Notes - -- `gordon routes add` is idempotent: it creates the route when missing and updates it when present. -- You can add the route before the image is pushed. Deploy happens when the image is later available or when you deploy an available image. - ---- - -## gordon routes show - -Show detailed information about a single route. - -```bash -gordon routes show -gordon routes show myapp.example.com -gordon routes show myapp.example.com --json -``` - -### Arguments - -| Argument | Description | -|----------|-------------| -| `` | The domain name of the route to inspect | - -### Options - -| Option | Description | -|--------|-------------| -| `--json` | Output route details as JSON | -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | - -### Description - -Displays the configured image for the route plus any available container and HTTP health information. -In local-only mode, health data may be unavailable. - -### Examples - -```bash -# Local -gordon routes show myapp.example.com - -# JSON -gordon routes show myapp.example.com --json - -# Remote (override) -gordon routes show myapp.example.com --remote https://gordon.mydomain.com --token $TOKEN -``` - -### JSON Output - -```json -{ - "domain": "myapp.example.com", - "image": "myapp:latest", - "container_status": "running", - "http_status": 200 -} -``` - ---- - -## gordon routes diagnose - -Diagnose route configuration, runtime state, preserved volumes, and orphaned attachment containers. - -```bash -gordon routes diagnose -gordon routes diagnose myapp.example.com --json -``` - -Use this after route deletion or failed deployment to see whether runtime state still exists and which safe cleanup commands to run next. - -### Output - -```text -Route diagnosis: myapp.example.com -Image: myapp:latest -Container: running abc123def456 -Preserved volume: gordon-myapp-example-com-data -Orphaned attachment: postgres -Hint: persistent volumes are preserved by default -Hint: run 'gordon attachments prune --stop' to stop orphaned attachment containers while preserving volumes -``` - -### JSON Output - -```json -{ - "domain": "myapp.example.com", - "configured": true, - "route": {"domain": "myapp.example.com", "image": "myapp:latest", "https": true}, - "runtime": {"domain": "myapp.example.com", "container_id": "abc123def456", "container_status": "running"}, - "volumes": [{"name": "gordon-myapp-example-com-data", "in_use": false}], - "orphaned_attachments": [{"container_id": "pg123", "name": "postgres", "image": "postgres:16", "status": "running", "owner": "myapp.example.com"}], - "hints": ["persistent volumes are preserved by default"] -} -``` - -### Next Steps - -```bash -gordon routes purge myapp.example.com # dry-run retained resources -gordon routes purge myapp.example.com --attachments --force -``` - ---- - -## gordon routes purge - -Review retained resources for a route and execute supported explicit cleanup actions only when forced. - -```bash -gordon routes purge myapp.example.com # dry-run -gordon routes purge myapp.example.com --attachments --force -gordon routes purge myapp.example.com --volumes # report preserved volumes for manual review -``` - -### Dry-run Output - -```text -Purge dry-run: myapp.example.com -Preserved attachments: - postgres (running) -Purge candidate: route_container abc123def456 -Preserved volume: gordon-myapp-example-com-data -Hint: dry-run only; add --force with explicit category flags to execute supported purge actions -``` - -### Forced Attachment Cleanup Output - -```text -Purge completed: myapp.example.com -Removed containers: - postgres -Hint: attachment volumes and data were preserved -``` - -Purge is intentionally conservative. Volumes are reported and preserved unless a dedicated volume deletion flow supports safe targeted deletion. - ---- - -## gordon routes remove - -Remove a route. - -Route removal is safe by default. Gordon removes the route from configuration and reconciles the active route container so it is no longer served or restarted by Gordon's monitor. Stateful data is preserved. Volumes and attachment data are not deleted unless a separate explicit purge flow requests destructive cleanup. - -```bash -gordon routes remove -gordon routes remove myapp.example.com -``` - -### Arguments - -| Argument | Description | -|----------|-------------| -| `` | The domain name of the route to remove | - -### Options - -| Option | Description | -|--------|-------------| -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | - -### Examples - -```bash -# Local -gordon routes remove myapp.example.com - -# Remote (override) -gordon routes remove myapp.example.com --remote https://gordon.mydomain.com --token $TOKEN - -# Typical output -Route removed: myapp.example.com - -Removed containers: - gordon-myapp.example.com (abc123def456) - -Preserved attachments: - postgres (running) -``` - ---- - -## gordon routes deploy - -Deploy or redeploy a specific route. - -```bash -gordon routes deploy -gordon routes deploy myapp.example.com -``` - -### Arguments - -| Argument | Description | -|----------|-------------| -| `` | The domain name of the route to deploy | - -### Options - -| Option | Description | -|--------|-------------| -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | - -### Description - -Triggers a fresh image pull and container redeployment for the specified route. - -### Examples - -```bash -# Local -gordon routes deploy myapp.example.com - -# Remote (override) -gordon routes deploy myapp.example.com --remote https://gordon.mydomain.com --token $TOKEN -``` - -## Related - -- [CLI Overview](./index.md) -- [Routes Configuration](../config/routes.md) diff --git a/docs/cli/secrets.md b/docs/cli/secrets.md deleted file mode 100644 index 97f70e806..000000000 --- a/docs/cli/secrets.md +++ /dev/null @@ -1,251 +0,0 @@ -# Secrets Commands - -Manage secrets on local or remote Gordon instances. - -Remote targeting uses client config or an active remote by default. -When you provide a concrete domain and no remote is selected, Gordon can also -auto-infer a saved remote when exactly one match is found. -Use `--remote` and `--token` to override. See [CLI Overview](./index.md). - -Storage depends on the secrets backend: -- `pass`: secrets are stored in pass under `gordon/env//` -- `sops`: secrets live in domain `.env` files and can reference SOPS-encrypted values -- `unsafe`: secrets are stored in plain-text domain `.env` files - -## gordon secrets - -### Subcommands - -| Subcommand | Description | -|------------|-------------| -| `list` | List all secrets for a domain | -| `set` | Set a secret value | -| `remove` | Remove a secret | - ---- - -## gordon secrets list - -List all secrets for a specific domain. When attachment secrets are present, they are displayed in a tree view below the domain secrets. - -```bash -gordon secrets list -``` - -### Arguments - -| Argument | Description | -|----------|-------------| -| `` | The domain name to list secrets for | - -### Options - -| Option | Description | -|--------|-------------| -| `--json` | Output secrets as JSON | -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | - -### Examples - -```bash -# Local -gordon secrets list myapp.example.com - -# Remote (override) -gordon secrets list myapp.example.com --remote https://gordon.mydomain.com --token $TOKEN -``` - -### Output - -``` -Secrets for app.mydomain.com - -Key Value -DATABASE_URL **** -API_KEY **** -├─ [postgres] -│ ├─ POSTGRES_USER **** -│ └─ POSTGRES_PASSWORD **** -└─ [redis] - └─ REDIS_PASSWORD **** -``` - -### JSON Output - -```bash -gordon secrets list app.mydomain.com --json -``` - -```json -{ - "domain": "app.mydomain.com", - "secrets": { - "DATABASE_URL": "****", - "API_KEY": "****" - }, - "attachments": { - "postgres": { - "POSTGRES_USER": "****", - "POSTGRES_PASSWORD": "****" - }, - "redis": { - "REDIS_PASSWORD": "****" - } - } -} -``` - ---- - -## gordon secrets set - -Set a secret value for a domain. - -```bash -gordon secrets set ... -gordon secrets set myapp.example.com DATABASE_URL="postgres://..." -``` - -### Arguments - -| Argument | Description | -|----------|-------------| -| `` | The domain name to set the secret for | -| `...` | One or more secret key/value pairs | - -### Options - -| Option | Description | -|--------|-------------| -| `--attachment` / `-a` | Target an attachment service (e.g., postgres, redis) | -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | - -### Examples - -```bash -# Local -gordon secrets set myapp.example.com DATABASE_URL="postgres://user:pass@postgres:5432/db" -gordon secrets set myapp.example.com API_KEY="your-api-key" - -# Remote (override) -gordon secrets set myapp.example.com DATABASE_URL="$DATABASE_URL" --remote https://gordon.mydomain.com --token $TOKEN - -# Set attachment secrets -gordon secrets set app.mydomain.com --attachment postgres POSTGRES_PASSWORD=secret -gordon secrets set app.mydomain.com -a redis REDIS_PASSWORD=mysecret - -# Multiple secrets at once -gordon secrets set app.mydomain.com -a postgres POSTGRES_USER=admin POSTGRES_PASSWORD=secret -``` - ---- - -## gordon secrets remove - -Remove a secret from a domain. - -```bash -gordon secrets remove -``` - -### Arguments - -| Argument | Description | -|----------|-------------| -| `` | The domain name | -| `` | The secret key to remove | - -### Options - -| Option | Description | -|--------|-------------| -| `--attachment` / `-a` | Target an attachment service (e.g., postgres, redis) | -| `--force` | Remove without confirmation | -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | - -### Examples - -```bash -# Local -gordon secrets remove myapp.example.com DATABASE_URL - -# Remote (override) -gordon secrets remove myapp.example.com DATABASE_URL --remote https://gordon.mydomain.com --token $TOKEN - -# Remove attachment secret -gordon secrets remove app.mydomain.com --attachment postgres POSTGRES_PASSWORD -``` - ---- - -## Workflow Examples - -### Setting Up Application Secrets - -```bash -# Database connection -gordon secrets set myapp.example.com DATABASE_URL="postgres://user:pass@postgres:5432/mydb" - -# API keys -gordon secrets set myapp.example.com STRIPE_KEY="sk_live_..." -gordon secrets set myapp.example.com SENDGRID_KEY="SG..." - -# JWT secret -gordon secrets set myapp.example.com JWT_SECRET="your-jwt-secret-here" - -# Verify -gordon secrets list myapp.example.com -``` - -### CI/CD Secret Management - -```bash -# In your CI/CD pipeline -export GORDON_REMOTE=https://gordon.mydomain.com -export GORDON_TOKEN=$GORDON_TOKEN - -# Update secrets before deploy -gordon secrets set myapp.example.com DATABASE_URL="$DATABASE_URL" -gordon secrets set myapp.example.com API_KEY="$API_KEY" - -# Deploy -gordon deploy myapp.example.com -``` - -### Rotating Secrets - -```bash -# Generate new secret -NEW_JWT_SECRET=$(openssl rand -base64 32) - -# Update the secret -gordon secrets set myapp.example.com JWT_SECRET="$NEW_JWT_SECRET" - -# Redeploy to pick up new secret -gordon deploy myapp.example.com -``` - -### Attachment Secrets - -```bash -# Configure database credentials -gordon secrets set app.mydomain.com -a postgres POSTGRES_USER=admin POSTGRES_PASSWORD=secret - -# Configure cache credentials -gordon secrets set app.mydomain.com -a redis REDIS_PASSWORD=cache-secret - -# Verify -gordon secrets list app.mydomain.com - -# Redeploy to pick up new secrets -gordon deploy app.mydomain.com -``` - -## Related - -- [CLI Overview](./index.md) -- [Secrets Configuration](../config/secrets.md) -- [Environment Variables](../config/env.md) diff --git a/docs/cli/serve.md b/docs/cli/serve.md index 00800e70e..edeb98a5a 100644 --- a/docs/cli/serve.md +++ b/docs/cli/serve.md @@ -23,8 +23,8 @@ gordon serve [options] Starts the Gordon server, which includes: - **Container Registry** - Receives image pushes on the registry port -- **HTTP Proxy** - Routes traffic to containers on the proxy port -- **Event Bus** - Coordinates deployments and updates +- **Reverse Proxy** - Routes app hosts to recorded loopback backends +- **Event Bus** - Coordinates runtime notifications - **Config Watcher** - Monitors configuration file for changes ### Configuration File Detection @@ -68,8 +68,8 @@ Gordon responds to these signals: |--------|--------| | `SIGTERM` | Graceful shutdown | | `SIGINT` | Graceful shutdown (Ctrl+C) | -| `SIGUSR1` | Reload configuration | -| `SIGUSR2` | Manual deploy request (used by local `gordon deploy`) | +| `SIGUSR1` | Reload installation configuration | +| `SIGUSR2` | Reserved; no app deployment action | ### Running with systemd @@ -106,21 +106,16 @@ sudo loginctl enable-linger $USER 7. Register event handlers 8. Start config file watcher 9. Start HTTP servers -10. Run best-effort startup recovery (sync existing containers, then recover configured routes) +10. Run best-effort startup recovery (reconcile apps intended to run from active state) ### Startup Recovery After Gordon starts, including after a host reboot, it runs a best-effort recovery pass once the listeners are bound. Errors are logged, but Gordon keeps starting. -1. **Sync existing containers** - Gordon reconciles runtime state with configured routes. -2. **Recover configured routes** - Gordon runs `AutoStart` for any configured route that has no running container. -3. **Start the background monitor** - Ongoing crash recovery resumes after the startup pass. +1. **Reconcile app boot state** - Gordon ensures apps intended to run are running from active state (stopped-intent apps stay stopped, pending desired revisions are never activated). +4. **Start the background monitor** - Ongoing crash recovery resumes after the startup pass. -This recovery runs even when `[auto_route].enabled = false`. `auto_route` only controls whether new image pushes create routes automatically; it does not disable restart recovery for routes already in the config. - -Recovery stays inside Gordon's own control flow instead of relying on Docker or Podman restart policies. That keeps the behavior consistent across both runtimes. - -Startup recovery is intentionally narrower than a manual deploy. It only starts routes that are missing a running container, skips readiness checks during boot, and does not perform drain/replacement logic for routes that are already running. +Reload and startup recovery never activate pending desired app state, re-resolve images, or flip intent. They touch installation settings and runtime reconciliation only. ### Shutdown Sequence @@ -133,164 +128,6 @@ Startup recovery is intentionally narrower than a manual deploy. It only starts --- -## gordon reload - -Reload configuration and sync containers to match. - -### Synopsis - -```bash -gordon reload -``` - -### Description - -Sends `SIGUSR1` to the running Gordon process, triggering: - -- Configuration file reload -- Route synchronization -- Deployment of containers for routes missing containers -- Attachment deployment - -### Example - -```bash -# After editing gordon.toml, apply changes without restart -vim ~/.config/gordon/gordon.toml -gordon reload -``` - ---- - -## gordon deploy - -Manually deploy or redeploy a specific route. - -### Synopsis - -```bash -gordon deploy [options] -``` - -### Arguments - -| Argument | Description | -|----------|-------------| -| `` | The domain name of the route to deploy (required) | - -### Options - -| Option | Description | -|--------|-------------| -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token for remote | - -Remote targeting uses client config or an active remote by default. -Use `--remote` and `--token` to override. See [CLI Overview](./index.md). - -### Description - -**Local mode:** On the Gordon host, Gordon uses the explicit deploy path for the selected route. When the CLI cannot execute that path directly, it falls back to queueing the request with `SIGUSR2` for the running server. This is different from startup recovery: startup recovery uses `AutoStart`, while a manual deploy performs an explicit redeploy for the selected route. - -**Remote mode:** Calls the remote Gordon Admin API to trigger deployment. The remote Gordon instance still performs the actual deploy internally; the CLI only submits the request. - -Both local and remote manual deploys use Gordon's explicit deploy path: - -- Fresh image content is pulled for the route -- The specified route is redeployed -- Configured readiness checks and drain behavior apply when needed - -### Examples - -```bash -# Local deployment -gordon deploy myapp.example.com -gordon deploy api.example.com - -# Remote deployment (override) -gordon deploy myapp.example.com --remote https://gordon.mydomain.com --token $TOKEN -``` - -### Use Cases - -- Recover from a failed deployment -- Force redeploy without pushing a new image -- Manual deployment when automatic deploy didn't trigger -- Trigger deployments on remote Gordon instances from CI/CD - ---- - -## gordon logs - -Display Gordon process logs or container logs. - -### Synopsis - -```bash -gordon logs [domain] [options] -``` - -### Arguments - -| Argument | Description | -|----------|-------------| -| `[domain]` | Optional. Container domain to view logs for. Without this, shows Gordon process logs. | - -### Options - -| Option | Short | Default | Description | -|--------|-------|---------|-------------| -| `--config` | `-c` | Auto | Path to config file | -| `--follow` | `-f` | false | Follow log output (like `tail -f`) | -| `--lines` | `-n` | 50 | Number of lines to show | -| `--remote, -r` | | | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | | | Authentication token for remote | - -Remote targeting uses client config or an active remote by default. -When you provide a concrete domain and no remote is selected, Gordon can also -auto-infer a saved remote when exactly one match is found. -Use `--remote` and `--token` to override. See [CLI Overview](./index.md). - -Remote log access requires an admin token with `admin:logs:read` (or `admin:*:*`). `admin:status:read` is not sufficient for logs. - -### Examples - -```bash -# Gordon process logs -gordon logs # Last 50 lines -gordon logs -f # Follow logs -gordon logs -n 100 # Last 100 lines -gordon logs -f -n 200 # Follow, starting from last 200 lines - -# Container logs -gordon logs myapp.example.com # Last 50 lines from container -gordon logs myapp.example.com -f # Follow container logs -gordon logs myapp.example.com -n 100 # Last 100 lines from container - -# Remote mode (override) -gordon logs --remote https://gordon.mydomain.com --token $TOKEN -gordon logs myapp.example.com --remote https://gordon.mydomain.com --token $TOKEN -``` - -### Log Locations - -```bash -# Using gordon logs -gordon logs -f - -# Direct file access -tail -f ~/.gordon/logs/gordon.log - -# With systemd -journalctl --user -u gordon -f - -# Container logs via docker (local alternative) -docker logs --tail 50 myapp.example.com -docker logs -f myapp.example.com -``` - ---- - ## gordon version Print version information. diff --git a/docs/cli/status.md b/docs/cli/status.md deleted file mode 100644 index d9d403bea..000000000 --- a/docs/cli/status.md +++ /dev/null @@ -1,113 +0,0 @@ -# Status Command - -Show Gordon server status and container health. - -## gordon status - -Display server configuration and container status for all routes. - -```bash -gordon status -gordon status --remote https://gordon.mydomain.com --token $TOKEN -``` - -`gordon status` works in local mode and remote mode. - -- Local mode reads status from in-process services. -- Remote mode reads status from the target admin API. - -### Output - -``` -Gordon Status - -Gordon Domain: gordon.example.com -Registry Port: 5000 -Server Port: 8088 -Routes: 3 -Auto-Route: true -Network Isolation: false - -Container Status: - app.example.com: running - api.example.com: running - worker.example.com: stopped -``` - -### Information Displayed - -| Field | Description | -|-------|-------------| -| Gordon Domain | Public Gordon domain from configuration | -| Registry Port | Docker registry port | -| Server Port | Gordon HTTP proxy port | -| Routes | Total configured routes | -| Auto-Route | Whether auto-routing is enabled | -| Network Isolation | Whether network isolation is enabled | -| Container Status | Status of each route's container | - -### Container States - -| State | Description | -|-------|-------------| -| running | Container is running and healthy | -| restarting | Container is in runtime restart backoff or restart cycle | -| stopped | Container was stopped | -| exited | Container exited (check logs for errors) | -| paused | Container is paused | -| unknown | Unable to determine container state | - -## Flags - -The status command uses global flags for remote access: - -| Flag | Description | -|------|-------------| -| `--remote, -r` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `--token` | Authentication token | - -## Environment Variables - -| Variable | Description | -|----------|-------------| -| `GORDON_REMOTE` | Remote name or URL (e.g., prod, https://gordon.mydomain.com) | -| `GORDON_TOKEN` | Authentication token | - -## Examples - -### Check Local or Remote Status - -```bash -# Local -gordon status - -# Using flags -gordon status --remote https://gordon.mydomain.com --token $TOKEN - -# Using environment variables -export GORDON_REMOTE=https://gordon.mydomain.com -export GORDON_TOKEN=your-token -gordon status -``` - -### Quick Health Check - -```bash -# Check if all containers are running -gordon status --remote https://gordon.mydomain.com --token $TOKEN | grep -E "(running|stopped|exited)" -``` - -## Required Permissions (Remote Only) - -Remote status calls require `admin:status:read` scope in the authentication token. - -```bash -# Generate token with required scope -gordon auth token generate --subject admin --scopes admin:status:read -``` - -## Related - -- [Serve Command](./serve.md) -- [Routes Command](./routes.md) -- [Remote CLI Management](/wiki/guides/remote-cli.md) diff --git a/docs/cli/tls.md b/docs/cli/tls.md deleted file mode 100644 index 5ab14b6d8..000000000 --- a/docs/cli/tls.md +++ /dev/null @@ -1,114 +0,0 @@ -# TLS Commands - -Inspect public TLS/ACME certificate status. - -Gordon serves normal HTTPS fallback on TLS-capable entrypoints such as `entrypoints.edge` with `protocol = "smart_tcp"`. Certificate priority is static certificates first, then public ACME certificates, then Gordon's internal CA. - -ACME challenge notes: - -- DNS-01 (`cloudflare-dns-01`) does not require a special external port 80 edge. -- HTTP-01 requires an HTTP-capable smart TCP entrypoint reachable on external port 80 for every hostname being validated. -- TLS-ALPN-01 is not supported. - -Remote targeting uses client config or an active remote by default. -Use `--remote` and `--token` to override. See [CLI Overview](./index.md). - -## gordon tls - -### Subcommands - -| Subcommand | Description | -|------------|-------------| -| `status` | Show public TLS certificate status | - ---- - -## gordon tls status - -Display the current public TLS/ACME certificate status, including ACME mode, -certificate details, route coverage, and any errors. - -```bash -gordon tls status -gordon tls status --json -gordon tls status --remote https://gordon.mydomain.com --token $TOKEN -``` - -### Flags - -| Flag | Description | -|------|-------------| -| `--json` | Output as JSON | - -### Human Output - -```text -Public TLS / ACME Status - -ACME: enabled -Configured Mode: auto -Effective Mode: http-01 -Reason: configured -Token Source: env - -Certificates - ID: cert-abc123 - Names: example.com, www.example.com - Status: valid - Not After: 2026-05-29 12:00:00 - -Route Coverage - example.com covered=yes covered_by=cert-abc123 - internal.local covered=no error=self-signed cert - -Errors - route internal.local has no ACME cert -``` - -### JSON Output - -```json -{ - "acme_enabled": true, - "configured_mode": "auto", - "effective_mode": "http-01", - "selection_reason": "configured", - "token_source": "env", - "certificates": [ - { - "id": "cert-abc123", - "names": ["example.com", "www.example.com"], - "challenge": "http-01", - "status": "valid", - "not_after": "2026-05-29T12:00:00Z", - "renewal_pending": false - } - ], - "routes": [ - { - "domain": "example.com", - "covered": true, - "covered_by": "cert-abc123", - "required_acme": true - }, - { - "domain": "internal.local", - "covered": false, - "required_acme": false, - "error": "self-signed cert" - } - ], - "errors": ["route internal.local has no ACME cert"] -} -``` - -### Token Source - -The `token_source` field indicates where the ACME token was sourced from -(e.g., `env`, `file`, `config`). The token value is never displayed. - -## Related - -- [CLI Overview](./index.md) -- [Status Command](./status.md) -- [Routes Command](./routes.md) diff --git a/docs/cli/traffic.md b/docs/cli/traffic.md deleted file mode 100644 index 3b07f6a76..000000000 --- a/docs/cli/traffic.md +++ /dev/null @@ -1,93 +0,0 @@ -# gordon traffic - -Inspect Gordon traffic plane status. - -## Subcommands - -| Command | Description | -|---------|-------------| -| `gordon traffic status` | Show traffic entrypoints, counters, and reload status | - -## Usage - -```bash -gordon traffic status --remote prod -gordon traffic status --remote https://gordon.example.com --json -``` - -Remote mode queries the running Gordon admin API. Local mode does not synthesize runtime traffic state from a fresh config load; use a configured remote target for authoritative counters and reload status. - -## Flags - -| Flag | Description | -|------|-------------| -| `--json` | Output machine-readable JSON | -| `--remote`, `-r` | Remote Gordon instance or saved remote name | - -## JSON Output - -```json -{ - "last_reload_status": "ok", - "entrypoints": [ - { - "name": "postgres", - "address": "0.0.0.0:5432", - "protocol": "tcp", - "active": true, - "active_tcp_connections": 1, - "active_udp_sessions": 0, - "total_accepted": 12, - "total_refused": 0, - "total_errors": 0, - "bytes_in": 4096, - "bytes_out": 8192, - "smart_tcp": { - "http_accepted": 0, - "h2c_accepted": 0, - "https_fallback_accepted": 0, - "tls_passthrough_accepted": 0, - "raw_fallback_accepted": 0, - "entrypoint_cidr_refused": 0, - "raw_fallback_cidr_refused": 0, - "proxy_refused": 0, - "unknown_no_fallback_refused": 0, - "malformed_rejected": 0, - "sniff_timeout": 0, - "client_hello_too_large": 0 - } - } - ], - "routers": [], - "services": [], - "counters": { - "active_tcp_connections": 1, - "active_udp_sessions": 0, - "total_accepted": 12, - "total_refused": 0, - "total_errors": 0, - "bytes_in": 4096, - "bytes_out": 8192, - "smart_tcp": { - "http_accepted": 0, - "h2c_accepted": 0, - "https_fallback_accepted": 0, - "tls_passthrough_accepted": 0, - "raw_fallback_accepted": 0, - "entrypoint_cidr_refused": 0, - "raw_fallback_cidr_refused": 0, - "proxy_refused": 0, - "unknown_no_fallback_refused": 0, - "malformed_rejected": 0, - "sniff_timeout": 0, - "client_hello_too_large": 0 - } - } -} -``` - -## Related - -- [Traffic configuration](../config/traffic.md) -- [Standalone services](../config/services.md) -- [Server status](./status.md) diff --git a/docs/cli/volumes.md b/docs/cli/volumes.md index 3867d456d..5f11ad145 100644 --- a/docs/cli/volumes.md +++ b/docs/cli/volumes.md @@ -19,7 +19,7 @@ gordon volumes list --json ### `gordon volumes prune` -Remove orphaned volumes no longer associated with a running container. +Remove app-owned volumes that were explicitly released and are no longer used by any container. ```bash gordon volumes prune @@ -29,10 +29,22 @@ gordon volumes prune --no-confirm | Flag | Description | |------|-------------| -| `--dry-run` | Show what would be removed without deleting | +| `--dry-run` | Report the plan without deleting | | `--no-confirm` | Skip confirmation prompt | | `--json` | Output as JSON | +#### What survives + +A volume is removed only when all of the following hold: + +- A durable ownership record explicitly marks it `released`. +- The runtime labels agree with that record: the app, the app incarnation UUID, and the service. +- No container mounts it. + +Everything else survives. Attached and retained volumes, volumes carrying only `gordon.managed`, volumes with no Gordon provenance, and volumes whose ownership cannot be established all survive. Ownership is never inferred from the volume name, and labels alone never make a volume deletable. + +With current metadata no lifecycle marks a volume `released`, so a volume prune normally reports a zero-deletion success. The command still lists every candidate with its verdict and reason, and `--json` returns the same plan shape as `gordon images prune`. + ## Related - [Volumes Configuration](../config/volumes.md) diff --git a/docs/concepts.md b/docs/concepts.md index 28c22161d..c2ed6961d 100644 --- a/docs/concepts.md +++ b/docs/concepts.md @@ -9,276 +9,193 @@ Your development machine likely has 8-16 cores and 16-32GB RAM. Your VPS has 1-2 Gordon flips the typical deployment model: 1. **Build locally** where you have computing power -2. **Push the finished image** to your VPS -3. **Gordon deploys automatically** +2. **Push the finished image** to your VPS registry +3. **Apply the app manifest** declaring the image +4. **Deploy** to activate it This means faster builds, less VPS resource usage, and a simpler deployment workflow. -## Push-to-Deploy +## Push, Apply, Deploy -Gordon combines a Docker registry with automatic deployment: +Gordon combines a Docker registry with a declarative app runtime: ``` ┌──────────────┐ push ┌──────────────┐ │ docker build │ ──────────────>│ Gordon │ -│ docker push │ │ Registry │ -└──────────────┘ └──────┬───────┘ +│ docker push │ (OCI only, │ Registry │ +└──────────────┘ never deploys)└──────┬───────┘ │ - │ event: image.pushed + ┌───────────────────┴───────────────────┐ + │ gordon apps apply --file blog.toml │ desired state + │ gordon apps deploy blog │ activation + └───────────────────┬───────────────────┘ v ┌──────────────┐ - │ Deploy │ - │ Container │ + │ App │ + │ Containers │ └──────────────┘ ``` -When you push an image, Gordon: +Pushing an image only stores it. Nothing runs, no route is created, no manifest is modified. Activation is always an explicit `gordon apps deploy`. -1. Stores the image in its registry -2. Fires an `image.pushed` event -3. Looks up the route for that image -4. Deploys a new container -5. Updates the proxy routing -6. Stops the old container +## Declarative Apps -## Zero-Downtime Updates +One standalone TOML file defines one globally named app with one or more explicitly named image-backed services. The app owns its entrypoints and routes; containers are replaceable runtime instances, not public identities. -Gordon ensures your app stays available during updates: - -1. **New container starts** while old container is still running -2. **Health check** waits for new container to be ready -3. **Traffic switches** to the new container -4. **Old container stops** after traffic has moved - -``` -Time ─────────────────────────────────────────────> - -Old Container: [═══════════════════] - ↓ stop -New Container: [═════════════════════════> - ↑ start ↑ traffic routed -``` - -## Deletion and Cleanup Lifecycle +```toml +name = "blog" -Gordon separates configuration removal from destructive data cleanup: +[env] +APP_ENV = "production" # app-wide public env, injected into all services -- **configured** — an entity exists in Gordon configuration, such as a route in `gordon.toml`. -- **active** — an entity has runtime state, such as a running Gordon-managed container. -- **preserved** — state intentionally kept after configuration removal, such as volumes or attachment data. -- **orphaned** — runtime state no longer referenced by configuration and requiring follow-up cleanup or diagnosis. -- **purged** — state explicitly deleted by a destructive cleanup command. -- **retained** — preserved state that Gordon reports so operators can decide whether to keep or purge it later. +[services.web] +image = "gordon.mydomain.com/blog:1.4.2" -Safe deletion is the default. Removing a route reconciles active route containers so the app is no longer served or restarted by Gordon's monitor. Stateful data such as volumes and attachment data is preserved unless a later purge command explicitly requests deletion. +[[services.web.http]] +host = "blog.mydomain.com" +port = 3000 -Cleanup reports use additive-only JSON fields so humans and automation can rely on stable keys while Gordon adds more details over time. +[services.web.secrets] # ENV name -> secret name (values stay in pass) +DATABASE_URL = "database-url" +``` -Current runtime cleanup capabilities are intentionally conservative: +- `apps apply --file FILE` validates and persists desired configuration only. +- `deploy APP` activates it. `apply --deploy` chains both using exactly the revision accepted by apply. +- `--dry-run` validates and previews without persistence or runtime effects. +- Version tags are recommended, not constrained to SemVer. `latest` remains valid; explicit deploy re-resolves mutable tags while restart uses the active pinned content. +- The file is intended for Git: it must never contain secret values. Service-specific values use `secrets` even when non-confidential. +- Staging is an ordinary app in another TOML file. There is no pin, no preview environments, and no historical rollback command. -- Gordon can identify managed route and attachment containers by labels such as `gordon.managed`, `gordon.route`, and `gordon.attachment`. -- Gordon can stop and remove containers. -- Existing volume attribution is limited for older volumes; Gordon may need naming heuristics when labels do not identify owner, mount path, or category. -- Updating runtime restart policy and relabeling existing volumes are runtime-dependent capabilities and should be treated as best-effort when added. +## Updates -## Routes +Gordon deploys an app's services one at a time in sorted order. A deploy may cause a short service interruption; there is no zero-downtime promise for app services. Gordon never runs two Gordon-managed generations of the same service at the same time. -Routes map domains to container images: +Every replaced service follows the same flow: preflight completes before anything is disrupted (image resolution and pull, secrets, volumes, networks, bind policy, reservations); the service is withdrawn from traffic; the old container is stopped and removed; the new container is created and started; the declared readiness probe runs; the new `ACTIVE` state is published; and traffic is rebuilt and published for the new container. -```toml -[routes] -"app.example.com" = { image = "myapp:latest" } -"api.example.com" = { image = "myapi:v2.1.0" } +``` +Preflight ─► withdraw traffic ─► stop + remove old ─► start new ─► readiness ─► publish ACTIVE ─► route traffic ``` -Route domains must be plain hostnames such as `app.example.com`. Gordon rejects `http://` and `https://` prefixes, `.local` and `.internal` suffixes, localhost names, and IP literals. Legacy `http://...` keys are still read for backward compatibility and rewritten on the next save. - -When a request comes in for `app.example.com`, Gordon: +If preflight fails, the running service is left untouched. If replacement fails after the old container was removed, the failure is explicit: Gordon does not recreate the old container during that operation and does not promise automatic data rollback. A stateful service's recovery inhibition stays in place until its never-published candidate is confirmed removed; a later boot/start/restart then clears it and rebuilds the generation `ACTIVE` records from its pinned digest, while a candidate that cannot be removed fails recovery closed. -1. Looks up the route configuration -2. Finds the running container for `myapp:latest` -3. Proxies the request to that container +Deployment stops at the first service failure: services already deployed in that run are kept and are not rolled back. This flow covers every service, including TCP, UDP, mixed, and volume-owning services. An open UDP socket is not application readiness, and Gordon never restarts an old volume-owning image automatically after a replacement may have written data. -### Route Domains +A deploy interrupted after the replacement container was created but before it was published leaves the failure explicit: the candidate is recorded in the deployment journal, and Gordon removes it before it creates or rebuilds any generation of that app — at boot and before every mutation — so two generations of one service never run together. An interrupted operation is finalized instead of staying in flight: as a failure, or as the success it was when every step had already completed. If the recorded candidate cannot be removed, reconciliation fails closed: the operation is not finalized, stays the latest journal, and every later mutation is refused so a newer operation can never mask the orphan. The next recovery pass republishes routing within its 15-second cadence if the final traffic publication was the step that failed. -Routes use plain hostnames. HTTPS is enabled by default for routes; see the [routes example](./config/routes.md#development-setup) with `dev-app.example.com`. Cloudflare is one deployment option, but Gordon can also terminate TLS with its built-in listener and certificates issued by Gordon's internal CA, with static certificates, or with public ACME certificates when `[tls.acme]` is enabled. ACME supports HTTP-01 and Cloudflare DNS-01 challenge modes. +Deploy is the only command that applies changes: it recreates a service whose image digest, spec, app environment, or secret values changed, and keeps an unchanged one, reported `unchanged`. -```toml -[routes] -"dev-app.example.com" = { image = "internal-app:latest", https = false } -``` +`gordon apps restart` restarts the same container and applies nothing: it withdraws traffic, restarts the pinned container, verifies readiness, and republishes traffic. No second container is created. When the recorded container is gone, restart rebuilds the service from its pinned `ACTIVE` digest and publishes it. -## Network Isolation - -Each app runs in its own isolated Docker network: +## Deletion and Cleanup Lifecycle -``` -┌────────────────────────────────────────────────┐ -│ gordon-app-mydomain-com │ -│ │ -│ ┌──────────┐ ┌──────────┐ ┌──────────┐ │ -│ │ App │───>│ Postgres │ │ Redis │ │ -│ │ :3000 │ │ :5432 │ │ :6379 │ │ -│ └──────────┘ └──────────┘ └──────────┘ │ -│ │ -└────────────────────────────────────────────────┘ -``` +Gordon separates workload removal from destructive data cleanup: -Benefits: +- **desired** — the persisted manifest revision waiting to be activated. +- **active** — the pinned, running definition (never inferred from containers). +- **stopped** — durable stopped intent: workloads are down, data preserved, reboot keeps them stopped. +- **retained** — volumes and secrets kept after app removal under the old internal UUID, visible but never implicitly adopted by a new app reusing the name. -- Containers can't access each other's services -- Services are only accessible by name within their network -- No port conflicts between apps +Safe removal is the default. `gordon apps remove` withdraws workloads and frees the name; volumes and secrets are retained as owned orphans. There is deliberately no `purge`: destructive volume deletion requires a separately accepted destructive-action contract. Ordinary apply/deploy/restart/stop/remove never delete user volumes. -## Attachments +## App HTTP Hosts -Attachments are service dependencies for your apps: +An app's HTTP interfaces declare the hosts Gordon serves: ```toml -[attachments] -"app.mydomain.com" = ["postgres:latest", "redis:latest"] +[[services.web.http]] +host = "app.example.com" +port = 3000 ``` -Gordon deploys attachments to the same network as your app. Services are accessible by their image name: +When a request comes in for `app.example.com`, Gordon: -```javascript -// In your app -const db = await connect("postgresql://postgres:5432/mydb"); -const cache = await connect("redis://redis:6379"); -``` +1. Looks up the host in the ACTIVE projection (merged with installation external routes) +2. Finds the recorded loopback backend for that host (never a container IP — rootless-first) +3. Proxies the request to that backend -## Network Groups +Hostnames must be plain hostnames. `https` behavior per host follows the `tls` mode (`auto`, `always`, `never`). -Network groups allow multiple apps to share services: +## Networks -```toml -[network_groups] -"backend" = ["app.mydomain.com", "api.mydomain.com"] +Each app gets a private network automatically. Services can additionally join named shared networks, created and reused only within verified Gordon ownership: -[attachments] -"backend" = ["shared-postgres:latest", "shared-redis:latest"] +```toml +[[network.shared]] +network = "backend" +services = ["web", "worker"] ``` -Both `app.mydomain.com` and `api.mydomain.com` can access the shared services. +Deploy adds AND removes memberships without disconnecting unrelated services. Short DNS names resolve privately; app-qualified aliases apply on shared networks. ## Volumes -Gordon automatically creates persistent storage from Dockerfile `VOLUME` directives: +Services declare persistent storage in the app manifest. Every Dockerfile `VOLUME` path must be declared explicitly; deployment rejects unmanaged image volumes: -```dockerfile -FROM postgres:18 -VOLUME ["/var/lib/postgresql/data"] +```toml +[[services.web.volume]] +name = "web-data" +path = "/data" ``` -Volume behavior: +Docker/Podman own the named volumes; Gordon records app UUID, service, and logical volume ownership. App manifests allow neither bind mounts nor service-shared volumes. Replacement and restart reuse volumes. Removing a service or app retains its volumes under the original app UUID, and a new app reusing the public name never adopts them. -- **auto_create**: Volumes are created automatically (default: true) -- **prefix**: Volume names are prefixed with `gordon-` (configurable) -- **preserve**: Volumes persist across container updates (default: true) +Use `gordon volumes prune --dry-run` to inspect Gordon's ownership-aware plan. Do not use `docker volume prune` or an equivalent runtime command for Gordon data: it bypasses Gordon's retention checks and can delete unmounted retained volumes. -## Environment Variables +## Environment and Secrets -Gordon loads environment variables from files based on the domain: +App-wide public env is declared in the manifest: +```toml +[env] +APP_ENV = "production" ``` -~/.gordon/env/ -├── app_mydomain_com.env -├── api_mydomain_com.env -└── admin_mydomain_com.env -``` - -Domain dots become underscores: `app.mydomain.com` → `app_mydomain_com.env` - -Variables are merged in order: - -1. Dockerfile `ENV` directives (lowest priority) -2. `.env` file values (highest priority) -### Secret Providers +Service-specific values use `secrets` even when non-confidential: -Environment files support secret provider syntax: - -```bash -# From Unix password manager (pass) -DATABASE_PASSWORD=${pass:myapp/db-password} - -# From SOPS encrypted files -API_SECRET=${sops:secrets.yaml:api.secret} +```toml +[services.web.secrets] +DATABASE_URL = "database-url" ``` -## Configuration Hot-Reload - -Gordon watches its config file and reloads automatically: +Secret values stay in pass under `gordon/apps///`, keyed by the stable internal UUID so a removed app's secrets are never adopted by a new app reusing the name. Running containers keep the values they were created with; `gordon apps deploy` applies new values. `restart` does not. Write values with `gordon apps secrets set`; only key names are ever echoed back, never values. -1. Edit `~/.config/gordon/gordon.toml` -2. Save the file -3. Gordon reloads hot-reloaded settings such as attachments, network groups, and routes -4. Containers and proxy config sync to match the new configuration +## Installation Reload -Route file edits now reload automatically again. - -You can also trigger a reload: +Gordon watches `gordon.toml` and reloads installation-only settings (entrypoints, TLS, limits, external routes). Reload never activates pending app desired state, never re-resolves image tags, and never starts app workloads. ```bash -gordon reload +gordon daemon reload ``` -`gordon reload` sends `SIGUSR1` to the running Gordon process. +`gordon daemon reload` sends `SIGUSR1` to the running Gordon process. `gordon.toml` holds installation settings only — app workloads live in app files. ## Event System -Gordon uses an internal event system for coordination: - -| Event | Trigger | Action | -|-------|---------|--------| -| `image.pushed` | Image pushed to registry | Deploy container | -| `config.reload` | Config file changed or `gordon reload` sends `SIGUSR1` | Reload config, sync containers, and refresh proxy state | -| `manual.deploy` | `gordon deploy ` command | Deploy specific route | -| `container.deployed` | Container started | Update proxy cache | +Gordon uses an internal event system for coordination. Registry storage events no longer deploy anything: push events never deploy, create routes, or modify manifests. ## Backups and Recovery -Gordon can run logical PostgreSQL backups for attachment containers. +Gordon runs PostgreSQL logical backups and volume archives to S3 for explicitly declared app targets. Declarations live in the app manifest (`[[services..database]]`, `[services..backup]`); storage infrastructure (destinations, schedules, retention) stays global. -Current design: +Stored backups are never deleted when declarations change — only schedules update on deploy. -1. Detect PostgreSQL attachments attached to a route -2. Execute `pg_dump -Fc` through Gordon runtime operations -3. Store backup artifacts on local filesystem storage -4. Expose backup actions through admin API and CLI - -This is intentionally scoped for operational safety and predictable behavior. -Future extensions can add physical backups, PITR, and remote object storage adapters. For configuration details and usage examples, see the [Backups Configuration guide](./config/backups.md), [Backup CLI reference](./cli/backup.md), and [Configuration Reference](./config/reference.md). -## Container Labels +## Container Identity -Gordon uses labels to track managed containers: +Gordon stamps app ownership labels on every container and volume it creates: | Label | Purpose | |-------|---------| -| `gordon.managed=true` | Identifies Gordon-managed containers | -| `gordon.domain` | Domain this container serves | -| `gordon.image` | Image name and tag | -| `gordon.route` | Route this container handles | -| `gordon.attachment=true` | Container is an attachment service | -| `gordon.attached-to` | Which route this attachment serves | - -## Proxy Port Selection - -When a container exposes multiple ports, Gordon needs to know which one serves HTTP: - -```dockerfile -FROM gitea/gitea:latest -LABEL gordon.proxy.port=3000 # Route HTTP to port 3000 -EXPOSE 22 # SSH -EXPOSE 3000 # HTTP -``` +| `gordon.managed=true` | Identifies Gordon-managed resources | +| `gordon.app` | App public name | +| `gordon.app.service` | Service name | +| `gordon.app.revision` | Active revision that created it | -Without the label, Gordon uses the first exposed port. +Queries by logical identity use labels, never name parsing. Unknown resources (no labels, old labels) are preserved, never adopted or deleted. ## Related - [Configuration Reference](./config/index.md) -- [Docker Labels Reference](./reference/docker-labels.md) -- [Environment Variables](./config/env.md) +- [Apps CLI](./cli/apps.md) +- [Getting Started](./getting-started.md) diff --git a/docs/config/apps.md b/docs/config/apps.md new file mode 100644 index 000000000..7b3862ec5 --- /dev/null +++ b/docs/config/apps.md @@ -0,0 +1,191 @@ +# App Manifest + +Application workloads are declared in standalone TOML files — one file per app — not in `gordon.toml`. Apply with `gordon apps apply --file .toml`, activate with `gordon apps deploy `. + +The file is intended for Git: it must never contain secret values. Service-specific values use `secrets` even when non-confidential; values stay in pass. + +## Minimal Example + +```toml +name = "blog" + +[services.web] +image = "gordon.mydomain.com/blog:1.4.2" + +[[services.web.http]] +host = "blog.mydomain.com" +port = 3000 +``` + +Each `[services.]` table declares one service, and the table key is the service name. A name may contain dots, so quote the key when it does: `[services."web.api"]`, `[[services."web.api".http]]`. + +## Full Example + +```toml +name = "blog" + +[env] # optional, app-wide public env for all services +APP_ENV = "production" + +[services.web] # one table per service; the key is the name +image = "gordon.mydomain.com/blog:1.4.2" +command = ["node", "server.js"] # optional override +stop_grace = "30s" # optional, default 30s +devices = ["transcode-gpu"] # 0..n logical names from [app_devices.] + +[services.web.readiness] # optional explicit readiness +type = "http" +path = "/healthz" +port = 3000 +timeout = "30s" + +[[services.web.http]] # 0..n HTTP interfaces +host = "blog.mydomain.com" +port = 3000 +tls = "auto" # auto | always | never + +[[services.web.http]] # optional private interface +visibility = "internal" # public (default) | internal +port = 8080 # required; no host or tls + +[services.web.secrets] # ENV name -> secret name (values in pass) +DATABASE_URL = "database-url" + +[[services.web.volume]] # 0..n named volumes only +name = "web-data" +path = "/data" +readonly = false + +[[services.web.bind]] # 0..n; name references [app_mounts.] +name = "app-logs" # policy name, never a host path +path = "/var/log/app" # absolute container destination +readonly = true + +[[services.web.database]] # explicit database declarations +name = "main" +type = "postgres" +schedule = "daily" + +[services.web.backup] # backup targets by reference +postgres = ["main"] +volume = ["web-data"] + +[[network.shared]] # optional shared-network memberships +network = "backend" +services = ["web"] +``` + +## Identity Rules + +- App name: DNS label (lowercase alphanumerics and hyphens, max 63), must not contain `--`, reserved: `gordon`, `registry`, `admin`, `localhost`. Case-insensitive uniqueness. +- Service name: the `[services.]` table key — `[a-z0-9_.-]`, max 63, quoted when it contains a dot. TOML keys are unique, so one table exists per name; names that share a runtime identifier after normalization (`web.api` and `web-api`) are rejected. +- Bind name: `[a-z0-9_.-]`, max 63, must not contain `--`, unique within its service. +- Device name: `[a-z0-9_.-]`, max 63, must not contain `--`, unique within its service. +- Removing an app ends its incarnation: the name is freed, volumes and secrets are archived as retained under the old internal UUID, desired/active/intent state is cleared, and the next apply allocates a new UUID. A new app reusing the name never adopts the old secrets or volumes. + +## Services + +One or more explicitly named image-backed services per app, each in its own `[services.]` table. Exactly one container per service — no replicas field. A service may expose HTTP, TCP, UDP interfaces, or none. + +```toml +[[services.web.tcp]] +entrypoint = "game" # names installation entrypoints. +port = 25565 +publish = "0.0.0.0:25565" + +[[services.web.udp]] +entrypoint = "game" +port = 28015 +publish = "0.0.0.0:28015" +``` + +RCON is ordinary TCP: use TCP interfaces, no special RCON kind. + +`publish` must match the entrypoint listener it attaches to exactly: the same host, port, and transport. The traffic manager binds the entrypoint address, so a declaration that differs (for example a loopback bind on a wildcard entrypoint, or a different port) is rejected at apply time instead of being silently widened. `0.0.0.0` and an omitted host are the same wildcard. + +Image registry names and digest syntax are validated during manifest apply, resolution, deployment preflight, and immediately before every pull, including boot, restart, and recovery. Docker Hub (`docker.io` and canonical pull host `registry-1.docker.io`), `ghcr.io`, `quay.io`, and Gordon's configured registry are allowed by default. Add other exact registry hostname+port entries with `images.allowed_registries`; this does not configure credentials, and external resolution and pulls use anonymous access. When `images.require_digest` is enabled, every image reference, including Gordon registry images, must use `@sha256:<64 hex chars>`. See [Images](./images.md). This is a hostname allowlist, not DNS/IP validation or runtime egress enforcement. + +## Environment and Secrets + +- `[env]` is app-wide public env injected into all services. Each key must be disjoint from every `[services..secrets]` key in the app. +- `[services..secrets]` maps ENV var name to service-local secret name. Values are written with `gordon apps secrets set` and stay in pass under `gordon/apps///`. Running containers keep the values they were created with; `gordon apps deploy` applies new values. `restart` does not. +- There is no `[services..env]` key — service-specific values must use secrets. + +## Volumes and Databases + +- Named volumes only, and no service-shared volumes. Replacement reuses volumes; removed services leave volumes retained and visible. +- Manifests never carry host paths. A `[[services..bind]]` references an `[app_mounts.]` policy the operator declares in `gordon.toml`; a bind whose name has no matching policy is rejected at apply time. See [Volumes](./volumes.md) and [Security Hardening](./security-hardening.md). +- Named volumes are Gordon-owned app data: created, labeled, retained, backed up, and pruned by Gordon. Administrative binds are operator-owned host locations: Gordon mounts them and never creates, deletes, owns, backs up, or prunes them. +- `[[services..bind]]` requires an absolute, normalized container `path` and an optional `readonly`. The policy's `read_only` and the bind's `readonly` force read-only together: either side wins and a manifest can never weaken its policy. Reserved destinations (`/`, `/proc`, `/sys`, `/dev`, `/boot` and their children) and paths colliding with a declared volume or another bind are rejected at apply time. +- `devices` lists logical device names granted by `[app_devices.]` policy the operator declares in `gordon.toml`; a device whose name has no matching policy, or whose policy does not allow the app+service pair, is rejected at apply time. Gordon resolves each name to explicit CDI IDs at activation time and encodes them as one native CDI `DeviceRequest`. Revisions persist the logical names, never the host resolution. A device add/remove shows as `service//devices` in diffs; reorder-only input is a no-op. Changing a mapping never recreates a running container: the next deploy serves the new resolution. Revoking a grant fails subsequent deploys closed while the running service is untouched. Device-bearing creates require Podman 5.4+ or Docker 28.3+ with native CDI configured; older or unrecognized engines return a structured `runtime-unsupported` error and never run without devices. Gordon installs no drivers, manages no quotas, and injects no `NVIDIA_*` environment: images carry their own runtime expectations. +- A service with any bind runs as a single writer, like a volume-backed service: replacements never serve two generations at once. +- A volume declared `readonly = true` is mounted read-only in the container; the service cannot modify protected data. +- `[[services..database]]` declares databases explicitly (no image inference). Only PostgreSQL is supported, and each database declares its own backup `schedule` (`hourly`, `daily`, `weekly`, or `monthly`). +- `[services..backup]` lists the declared databases (`postgres`) and volumes (`volume`) that are backup targets. A declared database or volume that is not referenced here is never backed up. Schedules follow the declaration through deploys; stored backups are never deleted when declarations change. + +## HTTP Interfaces + +`[[services..http]]` declares 0..n HTTP interfaces per service. `visibility` is optional and defaults to `public` when the key is absent or empty. + +```toml +[[services.web.http]] # public: proxied by host +host = "blog.mydomain.com" +port = 3000 +tls = "auto" # auto | always | never + +[[services.web.http]] # internal: not published (app private network only) +visibility = "internal" +port = 8080 +``` + +- `public`: `host` is required and must be a valid public hostname, and `tls` is `auto` (default when absent), `always`, or `never`. A public interface gets a proxy route, a global host reservation, a certificate target when TLS applies, and a `127.0.0.1` loopback backend publication. +- `internal`: creates no proxy route, no host reservation, no certificate target, and no host port publication, so it is never reachable through the host or proxy plane. `port` is required, `host` must be absent, and `tls` must be absent — any declared TLS value is rejected. It is still a declared TCP-capable container port for readiness metadata. Reachability follows network membership, not visibility: any container attached to a network this service joins can reach the port — sibling services on the app's own private network, and peers on any `[[network.shared]]` network the service is enrolled in. + +There is no `.internal` pseudo-domain: internal interfaces carry no hostname at all. + +A container port declared by both an internal HTTP interface and an externally backed interface (public HTTP or TCP) is rejected at apply time, because publication is socket-level. Duplicate internal HTTP ports within a service are rejected the same way. + +## Readiness + +`type = "http"` requires an origin-form `path` beginning with a single `/`. Absolute URLs, authority forms such as `@host:port`, scheme-relative paths, and control characters are rejected at apply time. The probe always dials the declared loopback backend, never follows redirects, ignores environment proxy settings, and is bounded per request and for the whole operation. A public or otherwise published interface keeps this loopback probe. An internal HTTP port is instead probed over the app private network by one bounded, short-lived helper, using HTTP or TCP according to the interface's `[services..readiness]` type; the internal port is never temporarily published to the host to probe it. + +## TLS + +TLS applies to public interfaces only; an internal interface never declares `tls`. + +- `auto` keeps the host eligible for HTTP and HTTPS; plaintext is redirected when an HTTPS endpoint exists and redirects are enabled. +- `always` is enforced: a plaintext request to an `always` host is redirected whenever an HTTPS endpoint exists, and refused with `421 Misdirected Request` when none does. The backend is never reached over plaintext. +- `never` stays plaintext and is never issued an app certificate. + +## Networks + +Each app gets a private network automatically. `[[network.shared]]` adds services to named shared networks, created/reused only within verified Gordon ownership. Deploy adds AND removes memberships without disconnecting unrelated services. + +Services of the same app communicate over that private network and resolve each other by service alias. Different apps are isolated by default; cross-app traffic requires both services to declare the same `[[network.shared]]` membership. See [Network Isolation](./network-isolation.md). + +## Telemetry + +When [telemetry log export](./telemetry.md#logs) is enabled, the stdout and stderr of every service are exported by default under `service.name = "."`. `logs = false` opts out; a service setting overrides the app setting: + +```toml +[telemetry] +logs = false # no service of this app exports... + +[services.web.telemetry] +logs = true # ...except web +``` + +Changing either setting is a service change: it applies on the next deploy. + +## Strictness + +Unknown fields, duplicate service names, unresolved service/entrypoint/backup refs, secret/env collisions, unsafe identities, and forbidden mounts are hard errors at apply time. Canonical host conflicts (including system domains and external routes) fail before persistence. + +The keyed `[services.]` form is the only accepted app schema: manifests written for an earlier v3 alpha using `[[service]]` with a `name` key and `[service.*]` tables are rejected with a keyed-schema diagnostic, not converted. See [Migrating a v3 alpha app manifest](../upgrading.md#migrating-a-v3-alpha-app-manifest). + +## Related + +- [Apps CLI](../cli/apps.md) +- [Secrets](./secrets.md) +- [Volumes](./volumes.md) +- [External Routes](./external-routes.md) diff --git a/docs/config/attachments.md b/docs/config/attachments.md deleted file mode 100644 index dd2cfdec9..000000000 --- a/docs/config/attachments.md +++ /dev/null @@ -1,315 +0,0 @@ -# Attachments Configuration - -Attach service dependencies (databases, caches, queues) to your applications. - -## Requirements - -**Network isolation must be enabled** for attachments to work properly: - -```toml -[network_isolation] -enabled = true -``` - -Without network isolation, containers run on Docker's default bridge network which **does not provide DNS resolution**. Your application won't be able to reach attachments by hostname (e.g., `postgres:5432`). - -## Configuration - -```toml -[attachments] -"app.mydomain.com" = ["postgres:latest", "redis:latest"] -"api.mydomain.com" = ["postgres:latest"] -``` - -## Syntax - -```toml -[attachments] -"" = [":", ":"] -``` - -| Component | Description | -|-----------|-------------| -| `domain-or-group` | Route domain or network group name | -| `image:tag` | Service images to deploy alongside the app | - -## How Attachments Work - -When you define attachments: - -1. Gordon deploys attachment containers to the same network as your app -2. Services are accessible by their image name (before the colon) -3. Attachments start before your main application -4. Attachments persist across app updates - -``` -┌───────────────────────────────────────────────────┐ -│ Network: gordon-app-mydomain-com │ -│ │ -│ ┌──────────┐ ┌──────────┐ ┌──────────┐ │ -│ │ App │───>│ postgres │ │ redis │ │ -│ │ :3000 │ │ :5432 │ │ :6379 │ │ -│ └──────────┘ └──────────┘ └──────────┘ │ -│ │ -└───────────────────────────────────────────────────┘ -``` - -## Service Discovery - -Attachments are accessible by their image name within the network: - -```toml -[attachments] -"app.mydomain.com" = ["postgres:latest", "redis:latest"] -``` - -In your application: -```javascript -// Connect using simple hostnames -const db = await pg.connect("postgresql://postgres:5432/mydb"); -const cache = await redis.connect("redis://redis:6379"); -``` - -## Persistent Storage - -Use Dockerfile VOLUME directives for persistent data: - -```dockerfile -# postgres.Dockerfile -FROM postgres:18 -VOLUME ["/var/lib/postgresql/data"] -ENV POSTGRES_DB=myapp -ENV POSTGRES_USER=app -# Password is injected via attachment secrets — do NOT hardcode here -``` - -Configure credentials via attachment secrets instead of hardcoding them: - -```bash -# Set database password securely -gordon secrets set app.mydomain.com --attachment postgres POSTGRES_PASSWORD=secret - -# Verify -gordon secrets list app.mydomain.com -``` - -This works with all secrets backends (pass, sops, unsafe). See [Secrets Configuration](./secrets.md) for backend details. - -Build and push to Gordon: -```bash -docker build -f postgres.Dockerfile -t my-postgres:latest . -docker push registry.mydomain.com/my-postgres:latest -``` - -Use in attachments: -```toml -[attachments] -"app.mydomain.com" = ["my-postgres:latest"] -``` - -## Attachment Secrets - -Inject environment variables into attachment containers using the `--attachment` flag on secrets commands: - -```bash -# Set secrets for the postgres attachment -gordon secrets set app.mydomain.com --attachment postgres POSTGRES_USER=admin POSTGRES_PASSWORD=secret - -# Set secrets for the redis attachment -gordon secrets set app.mydomain.com --attachment redis REDIS_PASSWORD=cache-secret - -# View all secrets (domain + attachments) in a tree view -gordon secrets list app.mydomain.com - -# Remove an attachment secret -gordon secrets remove app.mydomain.com --attachment postgres POSTGRES_PASSWORD -``` - -The `--attachment` flag takes the service name (the image name before the colon in your attachments config). For example, if your config has `"app.mydomain.com" = ["postgres:18", "redis:7-alpine"]`, the service names are `postgres` and `redis`. - -Storage depends on your secrets backend: -- **pass**: `gordon/env/attachments//` -- **sops/unsafe**: `gordon-.env` files - -See [Secrets Configuration](./secrets.md) and [Secrets Commands](../cli/secrets.md) for details. - -## Shared Attachments with Network Groups - -Share services between multiple apps using network groups: - -```toml -[network_groups] -"backend" = ["app.mydomain.com", "api.mydomain.com"] - -[attachments] -"backend" = ["shared-postgres:latest", "shared-redis:latest"] -``` - -Both `app.mydomain.com` and `api.mydomain.com` can access the shared services. - -## Per-App vs Shared Attachments - -### Per-App Attachments - -Each app gets its own isolated service instances: - -```toml -[attachments] -"app.mydomain.com" = ["postgres:latest"] # App's own postgres -"api.mydomain.com" = ["postgres:latest"] # API's own postgres -``` - -### Shared Attachments - -Multiple apps share the same service instances: - -```toml -[network_groups] -"backend" = ["app.mydomain.com", "api.mydomain.com"] - -[attachments] -"backend" = ["postgres:latest", "redis:latest"] # Shared by both -``` - -### Mixed Approach - -```toml -[network_groups] -"backend" = ["app.mydomain.com", "api.mydomain.com"] - -[attachments] -"backend" = ["redis:latest"] # Shared cache -"app.mydomain.com" = ["postgres:latest"] # App's own database -"api.mydomain.com" = ["postgres:latest"] # API's own database -``` - -## Examples - -### Web App with Database - -```toml -[routes] -"app.mydomain.com" = "myapp:latest" - -[attachments] -"app.mydomain.com" = ["postgres:18", "redis:7-alpine"] -``` - -### Microservices with Shared Queue - -```toml -[routes] -"orders.mydomain.com" = "orders-service:latest" -"inventory.mydomain.com" = "inventory-service:latest" -"notifications.mydomain.com" = "notifications-service:latest" - -[network_groups] -"services" = ["orders.mydomain.com", "inventory.mydomain.com", "notifications.mydomain.com"] - -[attachments] -"services" = ["rabbitmq:3-management"] -"orders.mydomain.com" = ["orders-db:latest"] -"inventory.mydomain.com" = ["inventory-db:latest"] -``` - -### Custom Service Images - -Build custom service images with your configuration: - -```dockerfile -# my-postgres.Dockerfile -FROM postgres:18 -VOLUME ["/var/lib/postgresql/data"] -ENV POSTGRES_DB=production -ENV POSTGRES_USER=appuser -# Password should come from env file -``` - -```dockerfile -# my-redis.Dockerfile -FROM redis:7-alpine -VOLUME ["/data"] -CMD ["redis-server", "--appendonly", "yes"] -``` - -```toml -[attachments] -"app.mydomain.com" = ["my-postgres:latest", "my-redis:latest"] -``` - -## Container Naming - -Attachment containers are named: -- `gordon--` -- Example: `gordon-app-mydomain-com-postgres` - -## Labels - -Gordon adds labels to attachment containers: - -| Label | Value | -|-------|-------| -| `gordon.managed` | `true` | -| `gordon.attachment` | `true` | -| `gordon.attached-to` | Domain or group name | - -## CLI Management - -Manage attachments via the CLI without editing configuration files. - -### List Attachments - -```bash -# List all attachments -gordon attachments list - -# List attachments for a specific domain or network group -gordon attachments list app.mydomain.com -gordon attachments list backend - -# Remote mode -gordon attachments list --remote https://gordon.mydomain.com --token $TOKEN -``` - -### Add Attachments - -```bash -# Add attachment to a domain -gordon attachments add app.mydomain.com postgres:18 - -# Add attachment to a network group -gordon attachments add backend redis:7-alpine - -# Remote mode -gordon attachments add app.mydomain.com postgres:18 --remote https://gordon.mydomain.com --token $TOKEN -``` - -### Remove Attachments - -```bash -# Remove attachment from a domain -gordon attachments remove app.mydomain.com postgres:18 - -# Remove from network group -gordon attachments remove backend redis:7-alpine - -# Remote mode -gordon attachments remove app.mydomain.com postgres:18 --remote https://gordon.mydomain.com --token $TOKEN -``` - -### Alias - -The `gordon attach` command is an alias for `gordon attachments`: - -```bash -gordon attach list -gordon attach add app.mydomain.com postgres:18 -gordon attach remove app.mydomain.com postgres:18 -``` - -## Related - -- [Network Isolation](./network-isolation.md) -- [Network Groups](./network-groups.md) -- [Routes](./routes.md) -- [CLI Commands](/docs/cli/index.md) diff --git a/docs/config/auth.md b/docs/config/auth.md index d0d6a9986..30e9ef753 100644 --- a/docs/config/auth.md +++ b/docs/config/auth.md @@ -87,16 +87,14 @@ Admin scopes (for remote CLI): | Scope | Permission | |-------|------------| | `admin:*:*` | Full admin access | -| `admin:routes:read` | Read-only routes access | -| `admin:routes:write` | Routes write access | +| `admin:apps:read` | Read-only apps access (list, show, diff, status) | +| `admin:apps:write` | App mutations (apply, deploy, lifecycle, secrets) | | `admin:config:read` | Read-only config access | | `admin:config:write` | Config write access | | `admin:status:read` | Read-only status/health | | `admin:logs:read` | Read-only logs access | | `admin:volumes:read` | List Docker volumes | | `admin:volumes:write` | Prune eligible Gordon-managed volumes | -| `admin:secrets:read` | List secret keys | -| `admin:secrets:write` | Set/delete secrets | Examples: @@ -143,10 +141,19 @@ gordon auth token generate --subject temp --expiry 30d Authentication is enabled by default. If you set `auth.enabled=false`, Gordon switches to local-only mode: -- `/admin/*` endpoints are not registered. +- `/admin/*` endpoints are not registered on the TCP listener. - `/v2/*` registry endpoints are restricted to loopback (`127.0.0.1` / `::1`). - Remote registry and remote admin access are disabled. +Local `gordon apps` commands keep working through an owner-only administration socket: + +- When `XDG_RUNTIME_DIR` is set, the daemon uses `$XDG_RUNTIME_DIR/gordon/admin.sock`; an invalid or unsafe preferred path fails closed and never falls back. When `XDG_RUNTIME_DIR` is unset, the daemon uses `~/.gordon/run/admin.sock`. The CLI probes `$XDG_RUNTIME_DIR/gordon/admin.sock` when set, then `/run/user//gordon/admin.sock`, then `~/.gordon/run/admin.sock`, accepting only a safe owner-owned socket. +- The runtime directory is `0700` and the socket `0600`, both owned by the effective user. A foreign owner, group/other permission bits, or a symlink makes the daemon refuse to start and the CLI refuse to connect; an existing live listener is never replaced. +- The socket exposes only app administration and app log reads. The implicit `local-owner` principal receives exactly `admin:apps:read`, `admin:apps:write`, and `admin:logs:read`; configuration, auth, prune, route, backup, and volume routes are denied. +- Unix platforms only. Never share, mount, or forward the socket: filesystem access to it is the credential. + +With neither a running daemon nor an explicit remote, app commands fail with `daemon-unavailable`. An explicit `--remote`/`GORDON_REMOTE` is authoritative and never falls back to the local socket. + Use this only when Gordon is intended for local machine usage. ## Internal Registry Auth diff --git a/docs/config/auto-route.md b/docs/config/auto-route.md deleted file mode 100644 index a28ca06ed..000000000 --- a/docs/config/auto-route.md +++ /dev/null @@ -1,281 +0,0 @@ -# Auto Route Configuration - -Gordon supports two methods for automatic route creation: - -1. **Image Name Detection** - Routes from domain-like image names -2. **Image Labels** - Routes from Dockerfile labels (recommended) - -## Configuration - -```toml -[auto_route] -enabled = true -``` - -## Options - -| Option | Type | Default | Description | -|--------|------|---------|-------------| -| `enabled` | bool | `false` | Enable automatic route creation | - -## How It Works - -When auto-route is enabled, Gordon automatically creates routes for images with domain-like names: - -```bash -# Push an image with a domain name -docker push registry.mydomain.com/app.example.com:latest -``` - -Gordon automatically: -1. Detects `app.example.com` looks like a domain -2. Creates a route: `app.example.com` → `app.example.com:latest` -3. Deploys the container -4. Routes traffic to it - -## Domain Detection - -Gordon recognizes image names as domains when they: - -- Contain at least one dot (`.`) -- Don't start or end with a dot -- Look like valid hostnames - -| Image Name | Detected as Domain? | -|------------|-------------------| -| `app.example.com` | Yes | -| `myapp.dev` | Yes | -| `staging.api.io` | Yes | -| `myapp` | No (no dots) | -| `.invalid` | No (starts with dot) | -| `myapp.` | No (ends with dot) | - -## Use Cases - -### Simplified Deployment - -Without auto-route: -```toml -# Must define each route in config -[routes] -"app.example.com" = "app.example.com:latest" -``` - -With auto-route: -```toml -[auto_route] -enabled = true -# No routes needed - just push! -``` - -```bash -docker push registry.mydomain.com/app.example.com:latest -# Route automatically created -``` - -### Development Workflow - -```bash -# Create a new app instantly -docker build -t myapp . -docker tag myapp registry.mydomain.com/newapp.mydomain.com:latest -docker push registry.mydomain.com/newapp.mydomain.com:latest -# newapp.mydomain.com is now live! -``` - -### Multi-Tenant SaaS - -```bash -# Quickly provision customer subdomains -docker push registry.mydomain.com/acme.saas.com:latest -docker push registry.mydomain.com/beta.saas.com:latest -docker push registry.mydomain.com/gamma.saas.com:latest -``` - -## Combining with Manual Routes - -Auto-route works alongside manual routes: - -```toml -[auto_route] -enabled = true - -[routes] -# Pinned versions override auto-route -"api.mydomain.com" = "myapi:v2.1.0" -``` - -- `api.mydomain.com` uses the manually configured `myapi:v2.1.0` -- Other domain-named images create automatic routes - -## Priority Order - -1. **Manual routes** in `[routes]` take precedence -2. **Auto-routes** are created for domain-named images without manual routes - -## Examples - -### Enable Auto-Route - -```toml -[auto_route] -enabled = true -``` - -### Development with Auto-Route - -```toml -[server] -gordon_domain = "gordon.local" - -[auto_route] -enabled = true - -# No routes defined - all come from image names -``` - -Usage: -```bash -docker push gordon.local/myapp.local:latest -docker push gordon.local/api.local:latest -# Both routes created automatically -``` - -### Production with Auto-Route - -```toml -[auto_route] -enabled = true - -[routes] -# Critical services with pinned versions -"api.company.com" = "company-api:v2.1.0" -"app.company.com" = "company-app:v1.5.0" - -# Other services can use auto-route -``` - -## Limitations (Image Name Detection) - -- Auto-routes always use the pushed image tag -- No support for HTTP-only routes (all HTTPS) -- No automatic attachment configuration -- Cannot pin versions (always uses pushed tag) - -For more control, define routes manually in `[routes]` or use image labels. - ---- - -## Image Labels (Recommended) - -The preferred method for automatic routing uses Dockerfile labels. This gives you full control over routing behavior directly in your application. - -### Supported Labels - -| Label | Description | Example | -|-------|-------------|---------| -| `gordon.domains` | Comma-separated list of domains | `app.example.com,www.app.example.com` | -| `gordon.proxy.port` | Container port to proxy | `3000` | -| `gordon.health` | Health check endpoint path | `/health` | -| `gordon.env-file` | Path to .env file in image | `/app/.env.example` | - -### Dockerfile Example - -```dockerfile -FROM node:20-alpine -WORKDIR /app -COPY . . -RUN npm install - -# Gordon routing labels -LABEL gordon.domains="myapp.example.com,www.myapp.example.com" -LABEL gordon.proxy.port="3000" -LABEL gordon.health="/api/health" -LABEL gordon.env-file="/app/.env.example" - -EXPOSE 3000 -CMD ["npm", "start"] -``` - -### How It Works - -When you push an image with Gordon labels: - -1. Gordon reads the `gordon.domains` label -2. Creates routes for each domain automatically -3. Uses `gordon.proxy.port` for the proxy target, otherwise auto-detects an exposed port preferring common HTTP ports -4. Configures health checks if `gordon.health` is set -5. Extracts environment variables from `gordon.env-file` if specified - -### Label Priority - -Labels take precedence over image name detection: - -1. **Labels present** → Use `gordon.domains` for routing -2. **No labels** → Fall back to image name detection (if enabled) -3. **Manual routes** → Always take highest priority - -### Multi-Domain Example - -```dockerfile -# Single image serving multiple domains -LABEL gordon.domains="api.example.com,api.staging.example.com" -``` - -Both domains will route to the same container. - -### Environment File Extraction - -The `gordon.env-file` label tells Gordon where to find a template `.env` file: - -```dockerfile -# Include a .env.example in your image -COPY .env.example /app/.env.example -LABEL gordon.env-file="/app/.env.example" -``` - -Gordon will: -1. Extract the file from the image -2. Store it in the env directory -3. Use the variables for container deployment - -### Best Practices - -1. **Always set `gordon.proxy.port`** - Don't rely on EXPOSE detection -2. **Use `gordon.health`** - Enables reliable deployment verification -3. **Include `.env.example`** - Self-documenting environment requirements -4. **Separate domains with commas** - No spaces around commas - -### Example: Complete Setup - -```dockerfile -FROM golang:1.25-alpine AS builder -WORKDIR /app -COPY . . -RUN go build -o server - -FROM alpine:3.21 -WORKDIR /app -COPY --from=builder /app/server . -COPY .env.example . - -# Gordon labels -LABEL gordon.domains="api.mycompany.com" -LABEL gordon.proxy.port="8080" -LABEL gordon.health="/health" -LABEL gordon.env-file="/app/.env.example" - -EXPOSE 8080 -CMD ["./server"] -``` - -```bash -# Push and Gordon handles everything -docker push registry.mycompany.com/api:latest -# Route automatically created: api.mycompany.com → container:8080 -``` - -## Related - -- [Routes](./routes.md) -- [Configuration Overview](./index.md) diff --git a/docs/config/backups.md b/docs/config/backups.md index d477513dd..921da7b46 100644 --- a/docs/config/backups.md +++ b/docs/config/backups.md @@ -3,7 +3,11 @@ Gordon has two backup flows: - **Database backups**: PostgreSQL logical dumps via `pg_dump`, stored on the local filesystem. -- **Volume backups**: best-effort filesystem archives of Gordon-managed named volumes, uploaded to S3. +- **Volume backups**: best-effort filesystem archives of Gordon-managed named volumes, uploaded to S3. Administrative bind mounts are operator-owned host paths and are never archived. + +Backups are identified by app, service, and the declared database or volume +(see [Apps](../config/apps.md)). Domains are routing addresses and are never a +backup identity. ## Database backups @@ -64,19 +68,24 @@ keep = 14 | `backups.volumes.s3.prefix` | `""` | Object key prefix | | `backups.volumes.s3.endpoint` | `""` | Optional S3-compatible endpoint | | `backups.volumes.s3.path_style` | `false` | Use path-style addressing for S3-compatible storage | -| `backups.volumes.retention.keep` | `14` | Completed archives to keep per domain + volume | +| `backups.volumes.retention.keep` | `14` | Completed archives to keep per app + volume | -Volume backup objects are stored under: +Volume backup objects are stored under the canonical app name: ```text -/domains//volumes//-.tar.gz # gzip -/domains//volumes//-.tar.zst # zstd +/apps//volumes//-.tar.gz # gzip +/apps//volumes//-.tar.zst # zstd ``` +Objects written by earlier versions under a domain prefix are not read or +migrated: back them up and re-run the app's volume backups if you need them in +the current layout. + ## Notes -- Volume backups include named volumes mounted by Gordon-managed route or attachment containers. -- Bind mounts, tmpfs mounts, anonymous volumes, and non-Gordon volumes are excluded. +- Volume backups archive the app's declared volumes that its backup declaration references. +- Bind mounts, tmpfs mounts, anonymous volumes, and undeclared volumes are never backed up. +- Database backups archive the app's declared PostgreSQL databases that its backup declaration references. - Live volume archives are best-effort; consistency requires application quiesce, pause, or stop. - Automated volume restore is not part of the MVP. - Snapshots are not generic across Docker/Podman volumes and require backend-specific support such as ZFS, Btrfs, LVM, EBS, or a snapshot-capable volume driver. @@ -84,6 +93,6 @@ Volume backup objects are stored under: ## Related -- [Attachments Configuration](./attachments.md) +- [App Manifest](./apps.md) - [CLI Backup Command](../cli/backup.md) - [Configuration Reference](./reference.md) diff --git a/docs/config/deploy.md b/docs/config/deploy.md deleted file mode 100644 index 6b7aca510..000000000 --- a/docs/config/deploy.md +++ /dev/null @@ -1,121 +0,0 @@ -# Deploy Configuration - -Controls how Gordon pulls images during deployment. - -## Example - -```toml -[deploy] -pull_policy = "if-tag-changed" -readiness_mode = "auto" -health_timeout = "90s" -readiness_delay = "5s" -drain_mode = "auto" -drain_timeout = "30s" -drain_delay = "2s" -``` - -## Settings - -### `deploy.pull_policy` - -How to decide when to pull an image: - -- `always`: always pull, even if the tag exists locally. -- `if-not-present`: only pull when the tag is missing locally. -- `if-tag-changed`: pull for tag references to check for updates; skip pulls for digest references (`@sha256:...`). - -Default: `if-tag-changed` - -### `deploy.readiness_delay` - -How long Gordon waits after a container first reports `running` before it is -considered ready for traffic. - -- Uses Go duration format (examples: `"5s"`, `"15s"`, `"1m"`). -- If the container briefly exits right after this delay, Gordon now waits up to - 30 additional seconds for it to recover before failing the deploy. - -Default: `"5s"` - -### `deploy.readiness_mode` - -How Gordon determines that a newly started container is ready for traffic. - -- `auto`: use Docker health status if the container has a healthcheck; otherwise - fall back to `deploy.readiness_delay`. -- `docker-health`: require a Docker healthcheck and wait for `healthy`. - Deploy fails with `no healthcheck detected` if absent. -- `delay`: always use `deploy.readiness_delay`. - -Default: `"auto"` - -### `deploy.health_timeout` - -Maximum time Gordon waits for health-based readiness in `auto`/`docker-health` -mode before failing the deploy. - -- Uses Go duration format (examples: `"30s"`, `"90s"`, `"2m"`). -- Default `"90s"` balances typical app startup time with fast failure feedback. -- Increase for slow cold-starts, migrations, or heavy warmup workloads. - -Default: `"90s"` - -### `deploy.drain_delay` - -How long Gordon waits after synchronous proxy cache invalidation before stopping -the previous container during zero-downtime replacement. - -- Uses Go duration format (examples: `"2s"`, `"10s"`, `"1m"`). -- Applied only when a previous container exists and cache invalidation was - triggered for the deployed domain. -- If explicitly set to `"0s"` (or a negative duration), delay is disabled and - Gordon switches immediately after invalidation. - -Default: `"2s"` - -### `deploy.drain_mode` - -How Gordon decides when it is safe to stop the previous container after routing -traffic to the new one. - -- `auto`: wait for in-flight proxy requests to drain when available; otherwise - fall back to `deploy.drain_delay`. -- `inflight`: wait for in-flight proxy requests on the previous container to - reach zero (bounded by `deploy.drain_timeout`). -- `delay`: always use `deploy.drain_delay`. - -Default: `"auto"` - -### `deploy.drain_timeout` - -Maximum time Gordon waits for in-flight request drain before continuing with old -container shutdown. - -- Uses Go duration format (examples: `"10s"`, `"30s"`, `"1m"`). -- Default `"30s"` is a practical upper bound for most HTTP workloads. -- Increase for long-lived requests (large uploads, SSE, long polling). - -Default: `"30s"` - -## Container security profile - -Runtime hardening is configured under `[containers]`: - -```toml -[containers] -security_profile = "compat" # "compat" or "strict" -``` - -- `compat` is the default and preserves existing image compatibility while keeping `no-new-privileges` and capability drop/add defaults. -- `strict` enables a read-only root filesystem, drops all capabilities, and only adds `NET_BIND_SERVICE`. Images that write outside mounted volumes or need extra Linux capabilities may require changes before using this profile. - -## Notes - -- Changes to `deploy.pull_policy` require a restart to take effect. -- Changes to `deploy.readiness_mode` require a restart to take effect. -- Changes to `deploy.health_timeout` require a restart to take effect. -- Changes to `deploy.readiness_delay` require a restart to take effect. -- Changes to `deploy.drain_mode` require a restart to take effect. -- Changes to `deploy.drain_timeout` require a restart to take effect. -- Changes to `deploy.drain_delay` require a restart to take effect. diff --git a/docs/config/env.md b/docs/config/env.md deleted file mode 100644 index 69cdcddb7..000000000 --- a/docs/config/env.md +++ /dev/null @@ -1,207 +0,0 @@ -# Environment Variables - -Configure per-application environment variables using domain-based files. - -## Configuration - -```toml -[env] -dir = "~/.gordon/env" # Default location -``` - -## Options - -| Option | Type | Default | Description | -|--------|------|---------|-------------| -| `dir` | string | `~/.gordon/env` | Directory containing .env files | - -## How It Works - -Gordon loads environment variables from files named after the domain: - -``` -~/.gordon/env/ -├── app_mydomain_com.env # For app.mydomain.com -├── api_mydomain_com.env # For api.mydomain.com -└── admin_mydomain_com.env # For admin.mydomain.com -``` - -**Naming convention:** Replace dots with underscores, add `.env` extension. - -| Domain | File Name | -|--------|-----------| -| `app.mydomain.com` | `app_mydomain_com.env` | -| `api.company.io` | `api_company_io.env` | -| `staging.app.dev` | `staging_app_dev.env` | - -## Backend Behavior - -The `auth.secrets_backend` setting changes where route secrets live: - -### Pass Backend - -- Env files are not used for route secrets. -- Use `gordon secrets set` to store per-domain secrets in pass. -- Existing `.env` files are migrated on startup and renamed to `.env.migrated`. - -### SOPS or Unsafe Backend - -- Env files remain the source of truth. -- Use `${sops:...}` syntax for encrypted values when `secrets_backend = "sops"`. - -## File Format - -Standard `.env` file format: - -```bash -# app_mydomain_com.env -NODE_ENV=production -PORT=3000 -DATABASE_URL=postgresql://db:5432/myapp -API_KEY=sk-1234567890 -DEBUG=false -``` - -## Variable Merging - -Variables are merged from multiple sources (lowest to highest priority): - -1. **Dockerfile ENV** - Base defaults from image -2. **.env file** - Overrides Dockerfile values - -```dockerfile -# Dockerfile -ENV NODE_ENV=development -ENV PORT=3000 -``` - -```bash -# app_mydomain_com.env -NODE_ENV=production # Overrides Dockerfile -# PORT not set, uses 3000 from Dockerfile -``` - -Result: `NODE_ENV=production`, `PORT=3000` - -## Secret Provider Syntax - -Reference secrets from configured backends in env files: - -### Pass (Unix Password Manager) - -```bash -DATABASE_PASSWORD=${pass:myapp/database/password} -API_SECRET=${pass:company/api-secret} -JWT_KEY=${pass:production/jwt-signing-key} -``` - -### SOPS (Encrypted Files) - -```bash -DATABASE_PASSWORD=${sops:secrets.yaml:database.password} -API_SECRET=${sops:production.yaml:api.secret} -STRIPE_KEY=${sops:secrets.yaml:stripe.api_key} -``` - -### Syntax Reference - -| Provider | Syntax | Example | -|----------|--------|---------| -| pass | `${pass:}` | `${pass:myapp/db-password}` | -| sops | `${sops::}` | `${sops:secrets.yaml:db.password}` | - -## Examples - -### Basic Application - -```bash -# app_mydomain_com.env -NODE_ENV=production -PORT=3000 -LOG_LEVEL=info -``` - -### Database Connection - -```bash -# ~/.gordon/env/api_mydomain_com.env -DATABASE_HOST=postgres -DATABASE_PORT=5432 -DATABASE_NAME=myapi -DATABASE_USER=apiuser -DATABASE_PASSWORD=${pass:myapi/db-password} -DATABASE_URL=postgresql://${DATABASE_USER}:${DATABASE_PASSWORD}@${DATABASE_HOST}:${DATABASE_PORT}/${DATABASE_NAME} -``` - -### Full Production Setup - -```bash -# ~/.gordon/env/app_company_com.env -# Application settings -NODE_ENV=production -PORT=3000 -LOG_LEVEL=warn - -# Database -DATABASE_URL=postgresql://postgres:5432/production -DATABASE_PASSWORD=${pass:company/db-password} - -# Redis -REDIS_URL=redis://redis:6379 - -# External APIs -STRIPE_SECRET_KEY=${pass:company/stripe-secret} -SENDGRID_API_KEY=${pass:company/sendgrid-key} - -# Auth -JWT_SECRET=${pass:company/jwt-secret} -SESSION_SECRET=${pass:company/session-secret} - -# Feature flags -ENABLE_ANALYTICS=true -MAINTENANCE_MODE=false -``` - -### Development Environment - -```bash -# ~/.gordon/env/app_local.env -NODE_ENV=development -PORT=3000 -LOG_LEVEL=debug -DATABASE_URL=postgresql://postgres:5432/dev -DEBUG=* -``` - -## File Permissions - -Gordon creates env files with secure permissions: -- Directory: `0700` (owner only) -- Files: `0600` (owner only) - -Create files with proper permissions: - -```bash -mkdir -p ~/.gordon/env -chmod 700 ~/.gordon/env -touch ~/.gordon/env/app_mydomain_com.env -chmod 600 ~/.gordon/env/app_mydomain_com.env -``` - -## Auto-Creation - -Gordon creates empty env files automatically when deploying a new route. You can then edit the file to add your variables. - -## Viewing Effective Environment - -To see what environment a container receives: - -```bash -docker inspect gordon-app-mydomain-com | jq '.[0].Config.Env' -``` - -## Related - -- [Secrets Configuration](./secrets.md) -- [Routes](./routes.md) -- [Attachments](./attachments.md) diff --git a/docs/config/external-routes.md b/docs/config/external-routes.md index f64dc4616..7abf5bfc9 100644 --- a/docs/config/external-routes.md +++ b/docs/config/external-routes.md @@ -79,5 +79,5 @@ vim ~/.config/gordon/gordon.toml ## Related -- [Routes Configuration](./routes.md) +- [App Manifest](./apps.md) - [Configuration Overview](./index.md) diff --git a/docs/config/images.md b/docs/config/images.md index f9c0d6ed7..6d13a190a 100644 --- a/docs/config/images.md +++ b/docs/config/images.md @@ -6,18 +6,20 @@ Configure automatic image cleanup for Docker runtime images and local registry s When enabled, Gordon runs a scheduled image prune job that: -- Prunes dangling runtime images. +- Prunes dangling runtime images that carry positive Gordon provenance. - Applies tag retention to registry repositories. - Preserves the `latest` tag. - Removes unreferenced blobs after tag cleanup. +- Reports every protected and unknown candidate with its reason. -The CLI `gordon images prune` uses the same defaults as the scheduled job (keep `latest` + 3 previous tags, both scopes enabled: dangling runtime images and registry tag retention). +The CLI `gordon images prune` runs the same use case and planner as the scheduled job, with the same defaults (keep `latest` + 3 previous tags, both scopes enabled: dangling runtime images and registry tag retention). App state existing on the server is normal and never disables pruning; the plan decides per candidate. ## Configuration ```toml [images] -allowed_registries = [] +# Docker Hub, ghcr.io, quay.io, and Gordon's registry are already allowed. +allowed_registries = ["registry.internal:5000"] require_digest = false [images.prune] @@ -30,19 +32,32 @@ keep_last = 3 | Key | Type | Default | Description | |-----|------|---------|-------------| -| `images.allowed_registries` | array | `[]` | Allowlist for external image registries. Empty means no external registries are accepted. Unqualified image names such as `nginx:latest` are treated as Docker Hub (`docker.io`) for policy checks. Gordon always allows its configured registry and rejects localhost/private/link-local registries. Include ports when needed, e.g. `"registry.example.com:5000"`. | -| `images.require_digest` | bool | `false` | Require allowlisted external image references to use a valid `@sha256:<64 hex chars>` digest. Gordon registry images are exempt. | +| `images.allowed_registries` | array | `[]` | Additional registry hostname+port entries. Docker Hub (`docker.io`, canonical pull host `registry-1.docker.io`), `ghcr.io`, `quay.io`, and Gordon's configured registry are always allowed. Include non-default ports, e.g. `"registry.example.com:5000"`. Hostnames are case-insensitive; one trailing dot and port `443` are canonicalized. This setting does not configure registry credentials. | +| `images.require_digest` | bool | `false` | Require every image reference, including Gordon registry images, to use a valid `@sha256:<64 hex chars>` digest. | | `images.prune.enabled` | bool | `false` | Enables scheduled image cleanup | | `images.prune.schedule` | string | `"daily"` | Schedule preset: `hourly`, `daily`, `weekly`, `monthly` | | `images.prune.keep_last` | int | `3` | Number of newest non-`latest` tags kept per repository during registry cleanup (`latest` is always kept when present) | +The policy validates registry names and strict SHA-256 digest syntax at manifest apply, digest resolution, deployment preflight, and immediately before each pull. It rejects malformed, userinfo-bearing, and ambiguous authorities. External app images currently require digest-pinned references, independently of `require_digest`; external tag resolution is not available. External pulls are anonymous, so adding a host to `allowed_registries` does not enable authenticated private-registry access. This hostname allowlist does **not** prove that DNS resolves to a public address and does not constrain runtime egress; enforce destination-level restrictions in the host firewall or runtime network policy. + ## Retention Behavior - `latest` is always preserved. - `keep_last` applies per repository and counts non-`latest` tags. -- `keep_last = 0` skips registry tag/blob cleanup (runtime dangling prune still runs). +- `keep_last` sets a minimum retained set, not a maximum: a tag named by durable app state, or referenced by the OCI closure of a protected manifest, survives beyond the window. +- `keep_last = 0` skips registry tag/blob cleanup entirely (runtime dangling prune still runs). - Negative `keep_last` values are invalid. +## Prune Safety + +A prune deletes only resources it can positively prove safe. Each candidate gets one verdict: `eligible` (deleted), `protected` (a durable fact claims it), or `unknown` (a required fact could not be read). Protected and unknown candidates are reported and left in place; an operation that deletes nothing is a success. + +Durable protection covers the DESIRED and ACTIVE state of every app, including stopped and partially converged services; recovery inhibitions; staged and committed apply intents; unfinished operation journal entries; runtime container use for running and stopped containers; recent uploads; and the transitive OCI closure of every retained manifest. Durable digest roots are repository-qualified, so moving a tag away from a deployed image keeps that manifest's config, layers, child manifests, and subject content protected. An unreadable, unsupported, or unresolvable manifest makes the blobs whose safety depended on it `unknown` instead of eligible. + +Registry content is repository-scoped: a blob is served only to a repository that completed an upload of that digest, and a manifest may only reference config, layers, and child manifests that repository owns. Image labels never authorize deletion of a runtime image; eligibility requires a matching released ownership claim from the app that pinned it, no container use, and a complete inventory. + +An exclusive GC lease serializes prune against app apply, deploy, start, restart, remove, recovery, and restore. Those paths hold a shared lease from validation/resource acquisition through the durable publication of that resource's protection, and an apply verifies the expected desired revision so a stale apply cannot overwrite a newer one. + ## Related - [CLI Images Command](../cli/images.md) diff --git a/docs/config/index.md b/docs/config/index.md index 269de208c..46e9b8c1b 100644 --- a/docs/config/index.md +++ b/docs/config/index.md @@ -19,12 +19,11 @@ gordon_domain = "gordon.mydomain.com" [entrypoints.edge] address = ":443" protocol = "smart_tcp" - -[routes] -"app.mydomain.com" = "myapp:latest" ``` -> **Note:** `gordon_domain` is the canonical key. Migrate older `registry_domain` values before restarting. +Application workloads live in standalone app files, not in `gordon.toml` — see [App Manifest](./apps.md). Retired workload keys such as `[routes]`, `[attachments]`, `[network_groups]`, `[service_routes]`, `[auto_route]`, and `[previews]` are rejected at startup. Installation-level `[[services]]` for standalone L4 workloads remains valid. + +> **Note:** `gordon_domain` is the canonical registry and Admin API host. > > For a staged registry host rename, set the new `server.gordon_domain` and keep old Gordon registry hosts in `server.legacy_registry_domains` until clients move. See [Server](./server.md#gordon-domain) and [Upgrading](../upgrading.md#staged-registry-host-rename). @@ -61,16 +60,6 @@ per_ip_rps = 50 # Max requests/second per IP burst = 100 # Burst size trusted_proxies = [] # IPs/CIDRs trusted for X-Forwarded-For -# Deploy behavior -[deploy] -pull_policy = "if-tag-changed" # always, if-not-present, if-tag-changed -readiness_mode = "auto" # auto, docker-health, delay -health_timeout = "90s" # Max wait for health-based readiness -readiness_delay = "5s" # Wait after running before ready -drain_mode = "auto" # auto, inflight, delay -drain_timeout = "30s" # Max wait for in-flight drain -drain_delay = "2s" # Wait after proxy invalidation before stopping the old container - # Container runtime profile [containers] security_profile = "compat" # compat or strict @@ -87,12 +76,8 @@ max_size = 100 # MB before rotation max_backups = 3 # Old files to keep max_age = 28 # Days to keep -[logging.container_logs] -enabled = true -dir = "~/.gordon/logs/containers" # Default location -max_size = 100 -max_backups = 3 -max_age = 28 +# Workload logs are streamed directly from the container runtime with +# `gordon apps logs APP --service SERVICE`. # Telemetry (OpenTelemetry) [telemetry] @@ -101,42 +86,24 @@ endpoint = "http://localhost:5080/api/default" # OTLP HTTP endpoint auth_token = "" # Base64 user:password for Basic auth traces = true # Export traces metrics = true # Export metrics -logs = true # Bridge zerolog to OTLP logs +logs = true # Export Gordon, access, and app logs trace_sample_rate = 1.0 # 0.0 = none, 1.0 = all -# Environment variables -[env] -dir = "~/.gordon/env" # Default location - # Volume settings [volumes] auto_create = true # Auto-create from Dockerfile VOLUME prefix = "gordon" # Volume name prefix preserve = true # Keep volumes on container removal -# Network isolation +# Installation network policy (prefix filter for `gordon daemon networks`) [network_isolation] -enabled = true # Per-app isolated networks +enabled = true # Gordon-managed network policy network_prefix = "gordon" # Network name prefix internal = false # Set true to block direct egress from isolated networks -# Auto-route -[auto_route] -enabled = false # Auto-create routes from image names - -# Routes (required) -[routes] -"app.mydomain.com" = "myapp:latest" -"api.mydomain.com" = "myapi:v2.1.0" - -# Network groups -[network_groups] -"backend" = ["app.mydomain.com", "api.mydomain.com"] - -# Attachments -[attachments] -"app.mydomain.com" = ["postgres:latest", "redis:latest"] -"backend" = ["rabbitmq:latest"] +# REMOVED in v3 (declare apps in standalone files, see ./apps.md): +# [routes], [attachments], [network_groups], +# [service_routes], [auto_route], [previews] # Backups [backups] @@ -152,8 +119,8 @@ monthly = 12 # Images [images] -allowed_registries = [] # Explicit external registries allowed for deploy/attachments -require_digest = false # Require digests for allowlisted external registries +allowed_registries = [] # Additional exact registry hostname+port entries +require_digest = false # Require SHA-256 digests for every image registry [images.prune] enabled = false @@ -161,6 +128,8 @@ schedule = "daily" keep_last = 3 ``` +Docker Hub (`docker.io` and `registry-1.docker.io`), `ghcr.io`, `quay.io`, and Gordon's registry are allowed by default. Add private or other registries with exact hostname+port entries. This hostname policy does not enforce resolved IP destinations or runtime egress; see [Images](./images.md). + ## Configuration Sections | Section | Description | Documentation | @@ -168,19 +137,14 @@ keep_last = 3 | `[server]` | Core server settings | [Server](./server.md) | | `[auth]` | Authentication and secrets backend | [Auth](./auth.md) | | `[api.rate_limit]` | Rate limiting configuration | [Rate Limiting](./rate-limiting.md) | -| `[deploy]` | Deployment behavior | [Deploy](./deploy.md) | | `[logging]` | Logging configuration | [Logging](./logging.md) | | `[telemetry]` | OpenTelemetry observability export | [Telemetry](./telemetry.md) | -| `[env]` | Environment variable settings | [Environment](./env.md) | | `[volumes]` | Volume management | [Volumes](./volumes.md) | -| `[network_isolation]` | Network isolation settings | [Network Isolation](./network-isolation.md) | -| `[auto_route]` | Automatic route creation | [Auto Route](./auto-route.md) | -| `[routes]` | Domain to image mapping | [Routes](./routes.md) | +| `[network_isolation]` | Installation network policy | [Network Isolation](./network-isolation.md) | | `[external_routes]` | Non-containerized service proxying | [External Routes](./external-routes.md) | | `[entrypoints]`, `[traffic]`, `[[network_services]]`, `[[services]]` | L4 and TLS passthrough traffic plane | [Traffic](./traffic.md) | -| `[network_groups]` | Shared service networks | [Network Groups](./network-groups.md) | -| `[attachments]` | Service dependencies | [Attachments](./attachments.md) | -| `[backups]` | Database backups | [Backups](./backups.md) | +| App files (`.toml`) | Declarative apps: services, hosts, secrets, volumes, backup targets | [App Manifest](./apps.md) | +| `[backups]` | Database backup storage, scheduling, and retention | [Backups](./backups.md) | | `[images.prune]` | Scheduled image cleanup | [Images](./images.md) | | Security hardening | Security controls and recommended knobs | [Security Hardening](./security-hardening.md) | @@ -200,13 +164,6 @@ keep_last = 3 | `api.rate_limit.per_ip_rps` | `50` | | `api.rate_limit.burst` | `100` | | `api.rate_limit.trusted_proxies` | `[]` | -| `deploy.pull_policy` | `"if-tag-changed"` | -| `deploy.readiness_mode` | `"auto"` | -| `deploy.health_timeout` | `"90s"` | -| `deploy.readiness_delay` | `"5s"` | -| `deploy.drain_mode` | `"auto"` | -| `deploy.drain_timeout` | `"30s"` | -| `deploy.drain_delay` | `"2s"` | | `containers.security_profile` | `"compat"` | | `logging.level` | `"info"` | | `logging.format` | `"console"` | @@ -214,13 +171,11 @@ keep_last = 3 | `logging.file.max_size` | `100` | | `logging.file.max_backups` | `3` | | `logging.file.max_age` | `28` | -| `logging.container_logs.enabled` | `true` | | `volumes.auto_create` | `true` | | `volumes.prefix` | `"gordon"` | | `volumes.preserve` | `true` | | `network_isolation.enabled` | `true` | | `network_isolation.internal` | `false` | -| `auto_route.enabled` | `false` | | `backups.enabled` | `false` | | `backups.schedule` | `"daily"` (`"hourly"`, `"daily"`, `"weekly"`, `"monthly"`) | | `images.allowed_registries` | `[]` | @@ -236,14 +191,14 @@ keep_last = 3 | `telemetry.logs` | `true` | | `telemetry.trace_sample_rate` | `1.0` | -When `auth.enabled=false`, Gordon runs in local-only mode: `/admin/*` is disabled and `/v2/*` is loopback-only. +When `auth.enabled=false`, Gordon runs in local-only mode: `/admin/*` is not registered on the TCP listener and `/v2/*` is loopback-only. Local `gordon apps` commands discover the daemon's owner-only admin socket in `$XDG_RUNTIME_DIR/gordon`, `/run/user//gordon`, or `~/.gordon/run`; the daemon itself uses the XDG location when configured and otherwise the home fallback. See [Authentication](./auth.md#local-only-mode). ## Hot Reload Gordon watches the configuration file and reloads automatically when changes are detected. You can also trigger a manual reload: ```bash -gordon reload +gordon daemon reload ``` ### Hot-reloaded (no restart needed) @@ -256,7 +211,6 @@ gordon reload | `server.max_proxy_response_size` | | `server.max_concurrent_conns` | -> **Note:** Routes are hot-reloaded from the config file. You can still use the API or CLI (`gordon routes add/update/remove`) for live route changes. ### Requires restart @@ -267,11 +221,6 @@ gordon reload | `server.max_blob_chunk_size` | | `server.max_blob_size` | | `auth.*` | -| `deploy.readiness_mode` | -| `deploy.readiness_delay` | -| `deploy.health_timeout` | -| `deploy.drain_mode` | -| `deploy.drain_timeout` | ## Environment Variable Override @@ -286,7 +235,7 @@ Pattern: `GORDON_SECTION_KEY` (uppercase, underscores instead of dots) ## Related - [Server Configuration](./server.md) -- [Routes Configuration](./routes.md) +- [App Manifest](./apps.md) - [External Routes](./external-routes.md) - [Standalone Services](./services.md) - [Traffic Plane](./traffic.md) diff --git a/docs/config/logging.md b/docs/config/logging.md index 9b44a3880..5cd3b7002 100644 --- a/docs/config/logging.md +++ b/docs/config/logging.md @@ -1,6 +1,6 @@ # Logging Configuration -Configure application and container log collection. +Configure Gordon process and HTTP access logging. Workload logs are read directly from the container runtime. ## Configuration @@ -16,13 +16,6 @@ max_size = 100 max_backups = 3 max_age = 28 -[logging.container_logs] -enabled = true -dir = "~/.gordon/logs/containers" -max_size = 100 -max_backups = 3 -max_age = 28 - [logging.access_log] enabled = false format = "json" @@ -44,24 +37,19 @@ syslog_identifier = "gordon-access" | Option | Type | Default | Description | |--------|------|---------|-------------| -| `file.enabled` | bool | `false` | Enable file-based logging | -| `file.path` | string | - | Path to main log file | +| `file.enabled` | bool | `false` | Enable process file logging | +| `file.path` | string | `{data_dir}/logs/gordon.log` | Process log path | | `file.max_size` | int | `100` | Max file size in MB before rotation | | `file.max_backups` | int | `3` | Number of old files to keep | | `file.max_age` | int | `28` | Days to keep old files | -The Admin API and `gordon logs` read from the process log file. Keep -`logging.file.enabled` set to `true` if you need process log streaming. +The Admin API and `gordon daemon logs` read from the process log file. Keep `logging.file.enabled` set to `true` if you need process log streaming. -### Container Logs +### Workload Logs -| Option | Type | Default | Description | -|--------|------|---------|-------------| -| `container_logs.enabled` | bool | `true` | Collect container stdout/stderr | -| `container_logs.dir` | string | `{data_dir}/logs/containers` | Directory for container logs | -| `container_logs.max_size` | int | `100` | Max file size in MB | -| `container_logs.max_backups` | int | `3` | Old files to keep | -| `container_logs.max_age` | int | `28` | Days to keep files | +`gordon apps logs APP --service SERVICE` reads stdout and stderr directly from the container runtime. Gordon does not persist workload logs to files; configure retention in the container runtime's logging driver. + +The accepted `logging.container_logs` configuration fields are currently not connected to a production log sink and do not create files. ### Access Log @@ -89,31 +77,9 @@ Use the access log for reverse-proxy traffic analysis, CrowdSec/fail2ban ingesti | `warn` | Warnings | | `error` | Errors only | -## Log Directory Structure - -``` -~/.gordon/logs/ -├── gordon.log # Main application logs -├── gordon.log.1 # Rotated log -├── gordon.log.2.gz # Compressed old log -└── containers/ - ├── app_mydomain_com.log # Container logs by domain - └── api_mydomain_com.log -``` - ## Log Rotation -Gordon uses automatic log rotation: - -1. **Size-based**: Rotates when file exceeds `max_size` MB -2. **Age-based**: Removes files older than `max_age` days -3. **Count-based**: Keeps only `max_backups` old files -4. **Compression**: Old files are gzip compressed - -Example with defaults: -- Log grows to 100MB → rotated to `gordon.log.1` -- After 3 rotations → oldest is compressed -- Files older than 28 days → deleted +Process file logs and file-based access logs rotate by size and retain files by count and age. Old files are compressed. ## Examples @@ -131,13 +97,6 @@ max_size = 10 max_backups = 2 max_age = 7 -[logging.container_logs] -enabled = true -dir = "./logs/containers" -max_size = 10 -max_backups = 2 -max_age = 7 - [logging.access_log] enabled = true format = "json" @@ -160,12 +119,11 @@ max_size = 100 max_backups = 10 max_age = 90 -[logging.container_logs] +[logging.access_log] enabled = true -dir = "~/.gordon/logs/containers" -max_size = 100 -max_backups = 10 -max_age = 90 +format = "json" +output = "journald" +syslog_identifier = "gordon-access" ``` ### Minimal (Console Only) @@ -177,60 +135,31 @@ format = "console" [logging.file] enabled = false - -[logging.container_logs] -enabled = true ``` ## Viewing Logs -### Gordon Logs +### Gordon Process Logs ```bash -# Using gordon logs command -gordon logs -f # Follow logs -gordon logs -n 100 # Last 100 lines - -# Direct file access +gordon daemon logs -f +gordon daemon logs -n 100 tail -f ~/.gordon/logs/gordon.log - -# With journalctl (if using systemd) journalctl --user -u gordon -f ``` -### Container Logs +### Workload Logs ```bash -# View container logs -tail -f ~/.gordon/logs/containers/app_mydomain_com.log - -# All container logs -ls ~/.gordon/logs/containers/ +gordon apps logs blog --service web +gordon apps logs blog --service web --follow ``` -## Log Format - -### Console Format - -``` -2024-01-15T10:30:00Z INF container deployed domain=app.mydomain.com image=myapp:latest -2024-01-15T10:30:01Z INF proxy routing updated route=app.mydomain.com -``` - -### JSON Format - -```json -{"level":"info","time":"2024-01-15T10:30:00Z","message":"container deployed","domain":"app.mydomain.com","image":"myapp:latest"} -{"level":"info","time":"2024-01-15T10:30:01Z","message":"proxy routing updated","route":"app.mydomain.com"} -``` +These commands stream from the container runtime; they do not read Gordon-managed workload log files. ## Security -Log files are created with secure permissions: -- Directories: `0700` (owner only) -- Files: `0600` (owner only) - -Sensitive values (passwords, tokens) are automatically redacted. +Process and access log files are created with owner-only permissions. Gordon redacts common credential patterns when serving logs, but application output may still contain sensitive data. Restrict process-log, journal, runtime, and `admin:logs:read` access. ## Related diff --git a/docs/config/network-groups.md b/docs/config/network-groups.md deleted file mode 100644 index f0c00e566..000000000 --- a/docs/config/network-groups.md +++ /dev/null @@ -1,188 +0,0 @@ -# Network Groups - -Group multiple applications into shared networks for inter-service communication. - -## Configuration - -```toml -[network_groups] -"backend" = ["app.mydomain.com", "api.mydomain.com"] -"monitoring" = ["grafana.mydomain.com", "prometheus.mydomain.com"] -``` - -## Syntax - -```toml -[network_groups] -"" = ["", "", ...] -``` - -| Component | Description | -|-----------|-------------| -| `group-name` | Name for the shared network | -| `domains` | List of routes that share this network | - -## How It Works - -Network groups create shared Docker networks for multiple applications: - -```toml -[network_groups] -"backend" = ["app.mydomain.com", "api.mydomain.com"] -``` - -Creates a shared network: `gordon-backend` - -``` -┌────────────────────────────────────────────────┐ -│ gordon-backend │ -│ │ -│ ┌──────────┐ ┌──────────┐ ┌──────────┐ │ -│ │ App │←──→│ API │←──→│ Postgres │ │ -│ │ :3000 │ │ :8080 │ │ :5432 │ │ -│ └──────────┘ └──────────┘ └──────────┘ │ -│ │ -└────────────────────────────────────────────────┘ -``` - -## Service Discovery - -Applications in the same group can communicate using container names: - -```toml -[routes] -"app.mydomain.com" = "frontend:latest" -"api.mydomain.com" = "backend:latest" - -[network_groups] -"services" = ["app.mydomain.com", "api.mydomain.com"] -``` - -Server-side code (SSR, API routes) can reach other containers by name: -```javascript -// In server-side code (SSR / API route) - NOT browser JavaScript -const response = await fetch("http://backend:8080/api/data"); -``` - -Browser clients should use the public route instead: -```javascript -// In browser JavaScript -const response = await fetch("https://api.mydomain.com/api/data"); -``` - -## Shared Attachments - -Combine network groups with attachments for shared services: - -```toml -[network_groups] -"backend" = ["app.mydomain.com", "api.mydomain.com"] - -[attachments] -"backend" = ["postgres:latest", "redis:latest"] -``` - -Both `app.mydomain.com` and `api.mydomain.com` can access: -- `postgres:5432` -- `redis:6379` - -## Multiple Groups - -An application can only be in one network group: - -```toml -# Each app belongs to one group -[network_groups] -"customer-facing" = ["app.mydomain.com", "api.mydomain.com"] -"internal" = ["admin.mydomain.com", "metrics.mydomain.com"] -``` - -## Mixed Isolation - -Combine network groups with per-app attachments: - -```toml -[network_groups] -"backend" = ["app.mydomain.com", "api.mydomain.com"] - -[attachments] -# Shared by both apps -"backend" = ["redis:latest", "rabbitmq:latest"] - -# App-specific databases -"app.mydomain.com" = ["app-postgres:latest"] -"api.mydomain.com" = ["api-postgres:latest"] -``` - -Result: -- Shared: Redis and RabbitMQ -- Isolated: Each app has its own Postgres - -## Examples - -### Microservices Architecture - -```toml -[routes] -"orders.mydomain.com" = "orders-service:latest" -"inventory.mydomain.com" = "inventory-service:latest" -"shipping.mydomain.com" = "shipping-service:latest" -"payments.mydomain.com" = "payments-service:latest" - -[network_groups] -"order-processing" = ["orders.mydomain.com", "inventory.mydomain.com", "shipping.mydomain.com"] -"payment-processing" = ["payments.mydomain.com"] - -[attachments] -"order-processing" = ["rabbitmq:latest"] -"payments.mydomain.com" = ["payments-db:latest"] -``` - -### Monitoring Stack - -```toml -[routes] -"grafana.mydomain.com" = "grafana/grafana:latest" -"prometheus.mydomain.com" = "prom/prometheus:latest" -"alertmanager.mydomain.com" = "prom/alertmanager:latest" - -[network_groups] -"monitoring" = ["grafana.mydomain.com", "prometheus.mydomain.com", "alertmanager.mydomain.com"] - -[attachments] -"monitoring" = ["prometheus-data:latest"] -``` - -### Multi-Tenant SaaS - -```toml -[routes] -"app.saas.com" = "saas-app:latest" -"api.saas.com" = "saas-api:latest" -"admin.saas.com" = "saas-admin:latest" - -# API and App share backend services -[network_groups] -"platform" = ["app.saas.com", "api.saas.com"] - -# Admin is isolated -[attachments] -"platform" = ["postgres:latest", "redis:latest", "elasticsearch:latest"] -"admin.saas.com" = ["admin-db:latest"] -``` - -## Network Naming - -Group networks are named: `{network_prefix}-{group-name}` - -| Group | Network Name | -|-------|--------------| -| `backend` | `gordon-backend` | -| `monitoring` | `gordon-monitoring` | -| `order-processing` | `gordon-order-processing` | - -## Related - -- [Network Isolation](./network-isolation.md) -- [Attachments](./attachments.md) -- [Routes](./routes.md) diff --git a/docs/config/network-isolation.md b/docs/config/network-isolation.md index c63266b94..82b21961f 100644 --- a/docs/config/network-isolation.md +++ b/docs/config/network-isolation.md @@ -1,6 +1,6 @@ # Network Isolation -Isolate applications in separate Docker networks for enhanced security. +Installation network policy for Gordon-managed networks. ## Configuration @@ -11,177 +11,40 @@ network_prefix = "gordon" internal = false ``` -## Migration Note - -As of this release, `network_isolation.enabled` defaults to `true` (previously `false`). -Existing installs that rely on a shared network must explicitly opt out: - -```toml -[network_isolation] -enabled = false -``` - ## Options | Option | Type | Default | Description | |--------|------|---------|-------------| -| `enabled` | bool | `true` | Enable per-app network isolation (changed from `false`) | -| `network_prefix` | string | `"gordon"` | Prefix for created networks | -| `internal` | bool | `false` | Create isolated Docker networks with Docker's `Internal` flag, blocking direct external egress from containers on those networks. Default remains `false` for compatibility. | - -## How It Works - -When network isolation is enabled, each application gets its own Docker network: - -``` -[network_isolation] -enabled = true -network_prefix = "gordon" - -[routes] -"app.mydomain.com" = "myapp:latest" -"api.mydomain.com" = "myapi:latest" -``` - -Creates two isolated networks: -- `gordon-app-mydomain-com` -- `gordon-api-mydomain-com` - -## Network Naming - -Networks are named: `{prefix}-{domain-with-dashes}` - -| Domain | Network Name | -|--------|--------------| -| `app.mydomain.com` | `gordon-app-mydomain-com` | -| `api.company.io` | `gordon-api-company-io` | -| `staging.app.dev` | `gordon-staging-app-dev` | - -## Security Benefits - -### Without Network Isolation - -All containers can potentially communicate: - -``` -┌───────────────────────────────────────┐ -│ Default Bridge Network │ -│ │ -│ App A ←──────→ App B ←──────→ App C │ -│ ↕ ↕ ↕ │ -│ DB A ←──────→ DB B ←──────→ DB C │ -│ │ -└───────────────────────────────────────┘ -``` - -### With Network Isolation - -Each app is isolated with its dependencies: - -``` -┌─────────────────┐ ┌─────────────────┐ ┌─────────────────┐ -│ gordon-app-a │ │ gordon-app-b │ │ gordon-app-c │ -│ │ │ │ │ │ -│ App A ←→ DB A │ │ App B ←→ DB B │ │ App C ←→ DB C │ -│ │ │ │ │ │ -└─────────────────┘ └─────────────────┘ └─────────────────┘ - ↑ ↑ ↑ - └───────── No direct communication ─────┘ -``` - -## Service Discovery - -Within an isolated network, services are discoverable by name: +| `enabled` | bool | `true` | Enable Gordon-managed network policy | +| `network_prefix` | string | `"gordon"` | Prefix filter for `gordon daemon networks` | +| `internal` | bool | `false` | Create isolated Docker networks with Docker's `Internal` flag, blocking direct external egress from containers on those networks. | -```toml -[network_isolation] -enabled = true - -[attachments] -"app.mydomain.com" = ["postgres:latest", "redis:latest"] -``` - -Your application connects using simple hostnames: - -```python -# These work within the isolated network -db = connect("postgresql://postgres:5432/mydb") -cache = connect("redis://redis:6379") -``` - -## Examples - -### Basic Isolation - -```toml -[network_isolation] -enabled = true - -[routes] -"app.mydomain.com" = "myapp:latest" -"api.mydomain.com" = "myapi:latest" - -[attachments] -"app.mydomain.com" = ["app-postgres:latest"] -"api.mydomain.com" = ["api-postgres:latest"] -``` - -Each app gets its own network with its own database. - -### Production Configuration - -```toml -[network_isolation] -enabled = true -network_prefix = "prod" - -[routes] -"app.company.com" = "company-app:v2.1.0" -"api.company.com" = "company-api:v1.5.0" -"admin.company.com" = "admin-panel:v1.0.0" -``` - -Creates networks: -- `prod-app-company-com` -- `prod-api-company-com` -- `prod-admin-company-com` - -### Shared Services with Network Groups - -When apps need to communicate, use network groups: +Per-app isolation is declared in app files, not here: each app gets a private network automatically, and services can join named shared networks with `[[network.shared]]` (see [App Manifest](./apps.md)). Deploy adds AND removes memberships without disconnecting unrelated services. Shared networks are created/reused only within verified Gordon ownership. -```toml -[network_isolation] -enabled = true +## Service-to-Service Networking -[network_groups] -"backend" = ["app.mydomain.com", "api.mydomain.com"] +Every app container joins an incarnation-owned private network. Services of the same app communicate over that network and can resolve each other by service alias. -[attachments] -"backend" = ["shared-postgres:latest", "shared-redis:latest"] -``` +Different apps are isolated by default: a container on one app network cannot reach another app's network. Cross-app communication happens only when both services declare the same `[[network.shared]]` membership, which attaches the explicitly enrolled containers to a shared network. -Both apps share the `gordon-backend` network. +Readiness helpers never join shared networks and exist only on the target app's private network. ## Inspecting Networks -View created networks: +View Gordon-managed networks: ```bash +gordon daemon networks docker network ls | grep gordon -# gordon-app-mydomain-com -# gordon-api-mydomain-com -# gordon-backend ``` Inspect a network: ```bash -docker network inspect gordon-app-mydomain-com +docker network inspect ``` ## Related -- [Network Groups](./network-groups.md) -- [Attachments](./attachments.md) +- [App Manifest](./apps.md) - [Configuration Overview](./index.md) diff --git a/docs/config/preview.md b/docs/config/preview.md deleted file mode 100644 index ef7925543..000000000 --- a/docs/config/preview.md +++ /dev/null @@ -1,273 +0,0 @@ -# Preview Environments - -Push a branch, get a URL. Preview environments give every branch or pull request its own isolated deployment — automatically torn down when it expires. - -Gordon provisions a preview environment when a matching image tag is pushed to the registry. Each preview gets its own subdomain derived from the base route, lives for a configurable TTL, and is cleaned up automatically when the TTL expires. - -## Configuration - -```toml -[auto.preview] -enabled = true -ttl = "48h" -separator = "--" -tag_patterns = ["preview-*", "pr-*"] -data_copy = true -env_copy = true -``` - -## Options - -| Option | Type | Default | Description | -|--------|------|---------|-------------| -| `enabled` | bool | `false` | Enable automatic preview environment creation | -| `ttl` | duration string | `"48h"` | How long a preview environment lives before automatic teardown | -| `separator` | string | `"--"` | String inserted between the base domain and the branch slug | -| `tag_patterns` | string array | `[]` | Glob patterns matched against image tags to trigger preview creation | -| `data_copy` | bool | `true` | Clone the production route's volumes into the preview environment on creation | -| `env_copy` | bool | `true` | Inherit environment variables from the base route's secret store into the preview container | - -### `ttl` Format - -The `ttl` field accepts Go duration strings: - -| Value | Meaning | -|-------|---------| -| `"24h"` | 24 hours | -| `"48h"` | 48 hours (default) | -| `"168h"` | 7 days | -| `"0"` | Never expire (manual deletion only) | - -### `tag_patterns` Matching - -Patterns use standard glob syntax. A tag must match at least one pattern in the list for a preview environment to be created. - -| Pattern | Matches | -|---------|---------| -| `"preview-*"` | `preview-my-feature`, `preview-fix-123` | -| `"pr-*"` | `pr-42`, `pr-123` | -| `"feat/*"` | `feat/new-api`, `feat/dark-mode` | - -## Naming Scheme - -Preview domains are derived from the base route for the deployed image. Gordon combines the base domain, the separator, and a URL-safe slug of the image tag. - -```text - -``` - -For example, with `separator = "--"` and base route `myapp.example.com`: - -| Image Tag | Preview Domain | -|-----------|----------------| -| `preview-my-feature` | `myapp--my-feature.example.com` | -| `pr-42` | `myapp--42.example.com` | -| `preview-fix-login` | `myapp--fix-login.example.com` | - -The tag prefix matched by `tag_patterns` is stripped to keep domains short. Slashes in tags are replaced with hyphens. - -## CLI Usage - -### List Preview Environments - -```bash -gordon preview list -``` - -Shows all active preview environments, their domains, TTL remaining, and source route. - -### Create or Refresh a Preview - -Previews are created automatically on push. To manually create or reset the TTL of an existing preview: - -```bash -gordon preview create app.example.com --tag preview-my-feature -``` - -### Extend a Preview - -Reset the TTL on an existing preview without redeploying: - -```bash -gordon preview extend myapp--my-feature.example.com -gordon preview extend myapp--my-feature.example.com --ttl 24h -``` - -Without `--ttl`, the configured default TTL is applied from the current time. - -### Delete a Preview - -```bash -gordon preview delete myapp--my-feature.example.com -``` - -Stops the container, removes the route, and (unless `volumes.preserve = true`) removes any cloned volumes. - -## CI Usage - -### GitHub Actions - -Use tag-based deploys to trigger preview environments automatically. - -```yaml -name: Preview Environment - -on: - pull_request: - types: [opened, synchronize] - -jobs: - deploy-preview: - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v4 - - - name: Log in to Gordon registry - run: echo "${{ secrets.GORDON_TOKEN }}" | docker login ${{ vars.GORDON_REGISTRY }} -u deploy --password-stdin - - - name: Build and push preview image - env: - TAG: pr-${{ github.event.pull_request.number }} - run: | - docker build -t ${{ vars.GORDON_REGISTRY }}/myapp:$TAG . - docker push ${{ vars.GORDON_REGISTRY }}/myapp:$TAG - - - name: Comment preview URL - uses: actions/github-script@v7 - with: - script: | - github.rest.issues.createComment({ - issue_number: context.issue.number, - owner: context.repo.owner, - repo: context.repo.repo, - body: `Preview deployed: https://app--${{ github.event.pull_request.number }}.example.com` - }) -``` - -Configure Gordon to match the `pr-*` tag convention: - -```toml -[auto.preview] -enabled = true -ttl = "72h" -tag_patterns = ["pr-*"] -``` - -### Cleanup on PR Close - -```yaml -name: Teardown Preview - -on: - pull_request: - types: [closed] - -jobs: - teardown: - runs-on: ubuntu-latest - steps: - - name: Delete preview environment - env: - PR_NUMBER: ${{ github.event.pull_request.number }} - run: | - gordon --server ${{ vars.GORDON_SERVER }} preview delete app--${PR_NUMBER}.example.com -``` - -## Lifecycle - -### Creation - -When a push matches a `tag_patterns` entry: - -1. Gordon creates a new route: `` → pushed image -2. If `data_copy = true`, volumes from the base route are cloned into the preview -3. If `env_copy = true`, environment variables from the base route's secret store are loaded and passed to the preview container (without persisting them) -4. The preview TTL timer starts -5. The container is deployed with zero-downtime rules disabled (previews are always cold-starts) - -### TTL and Automatic Teardown - -Gordon checks preview TTLs on a background schedule. When a preview expires: - -1. In-flight connections are drained -2. The container is stopped and removed -3. The route is removed from the proxy -4. Cloned volumes are removed (unless `volumes.preserve = true`) - -Use `gordon preview extend` to reset the timer without redeploying. - -### Volume Cloning - -When `data_copy = true`, Gordon copies the named volumes attached to the base route into fresh volumes for the preview. This gives the preview a realistic dataset without sharing state with production. - -- Cloned volumes are prefixed with the preview slug -- Cloned volumes are removed on preview teardown unless `volumes.preserve = true` -- Set `data_copy = false` to start previews with empty volumes (faster creation, no production data) - -### Environment Variable Inheritance - -Preview containers receive environment variables from three sources, merged in this order (last wins): - -1. **Dockerfile `ENV`** — baked into the image at build time -2. **Base route secrets** — inherited from the production domain's secret store on first deploy (when `env_copy = true`) -3. **Preview domain secrets** — set directly on the preview domain, picked up on redeploy - -On initial creation, Gordon loads the base route's secrets and passes them to the container. The variables are **not** persisted under the preview domain — they are resolved at deploy time. - -#### Overriding variables for a preview - -To change a variable for a specific preview, set it on the preview domain: - -```bash -gordon secrets set gordon--pr42.example.com API_URL=https://staging-api.example.com -``` - -This writes the secret to the preview domain's store and triggers an automatic redeploy (~60s debounce). On redeploy, the container picks up its own domain secrets instead of the inherited base route ones. - -#### Disabling env inheritance - -To start previews with a clean environment (Dockerfile `ENV` only): - -```toml -[auto.preview] -env_copy = false -``` - -## Example Config - -```toml -[server] -gordon_domain = "registry.example.com" - -[auto_route] -enabled = true - -[auto.preview] -enabled = true -ttl = "48h" -separator = "--" -tag_patterns = ["preview-*", "pr-*"] -data_copy = true -env_copy = true - -[routes] -"app.example.com" = "myapp:latest" -"api.example.com" = "myapi:latest" - -[volumes] -auto_create = true -preserve = false -``` - -With this config: -- Pushing `myapp:pr-99` creates `myapp--99.example.com` with cloned volumes and inherited env vars -- Pushing `myapi:preview-auth-refactor` creates `myapi--auth-refactor.example.com` -- Both previews expire after 48 hours and volumes are removed on teardown -- Env vars are loaded at deploy time from the base route's secret store and are not persisted for the preview domain - -## Related - -- [Routes](./routes.md) -- [Auto Route](./auto-route.md) -- [Volumes](./volumes.md) -- [Configuration Overview](./index.md) diff --git a/docs/config/reference.md b/docs/config/reference.md index 9b85743c8..a5c39cc6b 100644 --- a/docs/config/reference.md +++ b/docs/config/reference.md @@ -83,12 +83,9 @@ max_size = 100 # Max size in MB before rotation max_backups = 3 # Number of old files to keep max_age = 28 # Days to keep old files -[logging.container_logs] -enabled = true # Enable container log collection -dir = "" # Log directory (default: {data_dir}/container-logs) -max_size = 100 # Max size in MB before rotation -max_backups = 3 # Number of old files to keep -max_age = 28 # Days to keep old files +# Workload logs are streamed from the container runtime with +# `gordon apps logs APP --service SERVICE`. The accepted +# logging.container_logs fields are not connected to a production file sink. [logging.access_log] enabled = false # Dedicated HTTP access log for reverse-proxy traffic @@ -110,45 +107,21 @@ endpoint = "" # OTLP HTTP endpoint URL auth_token = "" # Base64-encoded user:password for Basic auth traces = true # Export distributed traces metrics = true # Export metrics -logs = true # Bridge zerolog output to OTLP logs +logs = true # Export Gordon, access, and app logs trace_sample_rate = 1.0 # Fraction of traces to sample (0.0–1.0) -# ============================================================================= -# ENVIRONMENT -# ============================================================================= -[env] -dir = "" # Env files directory (default: {data_dir}/env) - -# ============================================================================= -# DEPLOYMENT -# ============================================================================= -[deploy] -pull_policy = "if-tag-changed" # "always", "if-tag-changed", "never" -readiness_mode = "auto" # "auto", "docker-health", "delay" -health_timeout = "90s" # Max wait for health-based readiness -readiness_delay = "5s" # Wait after running before considered ready -drain_mode = "auto" # "auto", "inflight", "delay" -drain_timeout = "30s" # Max wait for in-flight request drain -drain_delay = "2s" # Wait after cache invalidation before old stop - # ============================================================================= # CONTAINERS # ============================================================================= [containers] security_profile = "compat" # "compat" or "strict" -# ============================================================================= -# AUTO-ROUTE -# ============================================================================= -[auto_route] -enabled = false # Create routes from image labels automatically - # ============================================================================= # NETWORK ISOLATION # ============================================================================= [network_isolation] -enabled = true # Enable per-app Docker networks -network_prefix = "gordon" # Prefix for created networks +enabled = true # Installation network policy for Gordon-managed networks +network_prefix = "gordon" # Prefix filter for `gordon daemon networks` internal = false # Create Docker internal networks (blocks direct egress) # ============================================================================= @@ -160,12 +133,38 @@ prefix = "gordon" # Volume name prefix preserve = true # Keep volumes when containers are removed # ============================================================================= -# ROUTES +# ADMINISTRATIVE APP BIND MOUNTS +# ============================================================================= +# A named policy is the only way an app manifest may reference a host path. +# App manifests declare [[services..bind]] with name = ""; direct host +# paths are always rejected, and the policy's allowlists are exact and +# non-empty. read_only on the policy and readonly on the bind both force +# read-only: either side wins and a bind never weakens its policy. +# [app_mounts.app-logs] +# source = "/srv/gordon/host-logs" # Required; absolute, normalized +# read_only = true # Force read-only for this mount +# allowed_apps = ["metrics-agent"] # Required; exact, non-empty +# allowed_services = ["web"] # Required; exact, non-empty +# root = "/srv/gordon" # Optional; default: parent of source + +# ============================================================================= +# ADMINISTRATIVE APP DEVICES (CDI) # ============================================================================= -[routes] -# "domain.com" = { image = "image:tag" } -# "insecure.domain.com" = { image = "image:tag", https = false } -# Legacy "http://domain.com" keys are read for compatibility and rewritten on save. +# A named device grant is the only way an app manifest may request host +# devices. Manifests declare devices = [""]; raw /dev paths are +# always rejected, and the policy's allowlists are exact and non-empty. +# Aggregate "=all" CDI selectors are rejected: name every device +# explicitly. Device-bearing creates require Podman 5.4+ or Docker 28.3+ +# with native CDI configured. +# [app_devices.transcode-gpu] +# cdi = ["example.com/gpu=GPU-device-uuid"] # Required; explicit, non-empty +# allowed_apps = ["video"] # Required; exact, non-empty +# allowed_services = ["transcoder"] # Required; exact, non-empty + +# REMOVED in v3: [routes], [attachments], [network_groups], +# app-like [[services]], [service_routes], [auto_route], [previews]. +# Declare apps in standalone files (see ./apps.md). The standalone +# [[services]] table below stays valid for installation-level L4 workloads. # ============================================================================= # EXTERNAL ROUTES @@ -195,18 +194,6 @@ preserve = true # Keep volumes when containers are # publish = "127.0.0.1:38016" # trusted_cidrs = ["100.64.0.0/10"] -# ============================================================================= -# NETWORK GROUPS -# ============================================================================= -[network_groups] -# "group-name" = ["domain1.com", "domain2.com"] - -# ============================================================================= -# ATTACHMENTS -# ============================================================================= -[attachments] -# "domain-or-group" = ["image1:tag", "image2:tag"] - # ============================================================================= # BACKUPS # ============================================================================= @@ -225,6 +212,8 @@ monthly = 0 # Keep N monthly backups per DB # IMAGES # ============================================================================= [images] +# Defaults: docker.io/registry-1.docker.io, ghcr.io, quay.io, Gordon registry. +# Add private or other registries as exact hostname+port entries. allowed_registries = [] require_digest = false @@ -265,7 +254,7 @@ keep_last = 3 # Keep N newest tags per repository | `tls.acme.email` | `""` | ACME account email when enabled | | `tls.acme.challenge` | `"auto"` | ACME challenge mode: `auto`, `http-01`, or `cloudflare-dns-01` | | `tls.acme.obtain_batch_size` | `1` | Maximum new ACME certificate orders per reconcile run | -| `auth.enabled` | `true` | Enable authentication; when `false`, run local-only mode (loopback-only `/v2/*`, `/admin/*` disabled) | +| `auth.enabled` | `true` | Enable authentication; when `false`, run local-only mode (loopback-only `/v2/*`, TCP `/admin/*` not registered, owner-only admin socket for local `gordon apps`) | | `auth.secrets_backend` | `"unsafe"` | Secrets storage | | `auth.token_expiry` | `"30d"` | 30 days | | `auth.access_token_ttl` | `"15m"` | Ephemeral access token lifetime | @@ -280,10 +269,6 @@ keep_last = 3 # Keep N newest tags per repository | `logging.file.max_size` | `100` | 100 MB | | `logging.file.max_backups` | `3` | Keep 3 old files | | `logging.file.max_age` | `28` | 28 days | -| `logging.container_logs.enabled` | `true` | Container logs enabled | -| `logging.container_logs.max_size` | `100` | 100 MB | -| `logging.container_logs.max_backups` | `3` | Keep 3 old files | -| `logging.container_logs.max_age` | `28` | 28 days | | `logging.access_log.enabled` | `false` | Dedicated HTTP access log disabled | | `logging.access_log.format` | `"json"` | Access log format (`json`, `clf`, `combined`) | | `logging.access_log.output` | `"stdout"` | Access log sink (`stdout`, `file`, `journald`) | @@ -297,23 +282,23 @@ keep_last = 3 # Keep N newest tags per repository | `telemetry.auth_token` | `""` | Base64 `user:password` for Basic auth | | `telemetry.traces` | `true` | Export distributed traces | | `telemetry.metrics` | `true` | Export metrics | -| `telemetry.logs` | `true` | Bridge zerolog to OTLP logs | +| `telemetry.logs` | `true` | Export Gordon, proxy access, and app container logs to OTLP | | `telemetry.trace_sample_rate` | `1.0` | Fraction of traces to sample (0.0–1.0) | -| `deploy.pull_policy` | `"if-tag-changed"` | Pull on tag change | -| `deploy.readiness_mode` | `"auto"` | Readiness strategy (`auto`, `docker-health`, `delay`) | -| `deploy.health_timeout` | `"90s"` | Max wait for health-based readiness before deploy fails | -| `deploy.readiness_delay` | `"5s"` | Delay before container is considered ready | -| `deploy.drain_mode` | `"auto"` | Drain strategy (`auto`, `inflight`, `delay`) | -| `deploy.drain_timeout` | `"30s"` | Max wait for in-flight request drain before old stop | -| `deploy.drain_delay` | `"2s"` | Delay before stopping previous container after cache invalidation | | `containers.security_profile` | `"compat"` | Runtime hardening profile: `compat` preserves existing behavior, `strict` enables read-only rootfs and narrower capabilities | -| `auto_route.enabled` | `false` | Auto-route disabled | | `network_isolation.enabled` | `true` | Network isolation enabled | | `network_isolation.network_prefix` | `"gordon"` | Network prefix | | `network_isolation.internal` | `false` | Create Docker internal networks without direct external egress | | `volumes.auto_create` | `true` | Auto-create volumes | | `volumes.prefix` | `"gordon"` | Volume prefix | | `volumes.preserve` | `true` | Keep volumes | +| `app_mounts..source` | none | Required host source of a named administrative bind; absolute and normalized. The only host path that mount may come from | +| `app_mounts..allowed_apps` | none | Required exact, non-empty app allowlist for this mount (no wildcard or empty-means-all form) | +| `app_mounts..allowed_services` | none | Required exact, non-empty service allowlist for this mount | +| `app_mounts..read_only` | `false` | Force every bind resolved under this policy read-only; a manifest bind can never weaken it | +| `app_mounts..root` | parent of `source` | Optional administrative boundary the resolved source must stay under | +| `app_devices..cdi` | none | Required explicit, non-empty CDI device IDs granted under this logical name (no `=all` aggregate form) | +| `app_devices..allowed_apps` | none | Required exact, non-empty app allowlist for this grant (no wildcard or empty-means-all form) | +| `app_devices..allowed_services` | none | Required exact, non-empty service allowlist for this grant | | `services[].name` | none | Standalone service name used by `service::` traffic refs | | `services[].image` | none | Container image for enabled standalone services | | `services[].enabled` | `false` | Whether Gordon creates, starts, and reconciles the service container | @@ -342,14 +327,27 @@ keep_last = 3 # Keep N newest tags per repository | `backups.retention.daily` | `0` | Keep no daily backups by default (recommend `7`) | | `backups.retention.weekly` | `0` | Keep no weekly backups by default | | `backups.retention.monthly` | `0` | Keep no monthly backups by default | -| `images.allowed_registries` | `[]` | Explicit external registry allowlist; empty rejects explicit external registries; dangerous local/private registries are always rejected | -| `images.require_digest` | `false` | Require digest-pinned references for allowlisted external registries | +| `images.allowed_registries` | `[]` | Additional exact hostname+port entries. Defaults allow Docker Hub (`docker.io` and `registry-1.docker.io`), `ghcr.io`, `quay.io`, and Gordon's registry. This allowlist does not configure registry credentials and is not DNS/IP or runtime egress enforcement. | +| `images.require_digest` | `false` | Require valid SHA-256 digest-pinned references for every registry, including Gordon | | `images.prune.enabled` | `false` | Scheduled image cleanup disabled | | `images.prune.schedule` | `"daily"` | Cleanup schedule preset | | `images.prune.keep_last` | `3` | Number of recent tags kept per repository | Note: for all `backups.retention.*` keys, `0` means keep no backups for that retention tier. +## App Manifest HTTP Interfaces + +`[[services..http]]` declares HTTP interfaces in a standalone app manifest (see [App Manifest](./apps.md)). + +| Key | Values | Default | Description | +|-----|--------|---------|-------------| +| `services..http[].visibility` | `"public"`, `"internal"` | `"public"` | `public` is proxied by host and may terminate TLS; `internal` is reachable only from the app's private network | +| `services..http[].host` | hostname | none | Required for public interfaces; must be absent for `visibility = "internal"` | +| `services..http[].port` | integer | none | Required container port for every interface | +| `services..http[].tls` | `"auto"`, `"always"`, `"never"` | `"auto"` | Public interfaces only; `visibility = "internal"` rejects any declared value | + +`visibility = "internal"` creates no proxy route, host reservation, certificate target, or host port publication. A container port declared by both an internal HTTP interface and an externally backed interface (public HTTP or TCP) is rejected, as are duplicate internal HTTP ports. + ## Environment Variables All configuration options can be set via environment variables using the pattern: @@ -419,5 +417,6 @@ gordon serve - [Telemetry](./telemetry.md) - [Network Isolation](./network-isolation.md) - [Volumes](./volumes.md) +- [App Manifest](./apps.md) - [Standalone Services](./services.md) - [Images](./images.md) diff --git a/docs/config/routes.md b/docs/config/routes.md deleted file mode 100644 index 10f344a63..000000000 --- a/docs/config/routes.md +++ /dev/null @@ -1,206 +0,0 @@ -# Routes Configuration - -Routes map hostnames to container images. - -## Configuration - -```toml -[routes] -"app.mydomain.com" = { image = "myapp:latest" } -"api.mydomain.com" = { image = "myapi:v2.1.0" } -"admin.mydomain.com" = { image = "admin-panel:v1.0.0" } -``` - -## Syntax - -```toml -[routes] -"" = { image = ":" } -``` - -| Component | Description | -|-----------|-------------| -| `domain` | Public, fully qualified domain name | -| `image` | Full container image reference, including tag | -| `https` | Optional; add `false` for HTTP-only routes | - -Legacy `http://...` route keys are still read for backward compatibility and rewritten on the next save. - -## Route Types - -### HTTPS Routes (Default) - -Standard routes expect HTTPS traffic (terminated by Cloudflare or similar): - -```toml -[routes] -"app.mydomain.com" = { image = "myapp:latest" } -``` - -Route domains must be plain hostnames. Gordon rejects `http://` and `https://` prefixes, `.local` and `.internal` suffixes, localhost names, and IP literals. - -### Development Routes - -For local testing, use a hostname you can resolve yourself: - -```toml -[routes] -"dev-app.example.com" = { image = "internal-app:latest", https = false } -``` - -## Version Strategies - -### Latest Tag - -Always deploy the most recent push: - -```toml -[routes] -"app.mydomain.com" = { image = "myapp:latest" } -``` - -### Pinned Version - -Deploy a specific version: - -```toml -[routes] -"app.mydomain.com" = { image = "myapp:v2.1.0" } -``` - -Update the config to deploy a new version: - -```toml -[routes] -"app.mydomain.com" = { image = "myapp:v2.2.0" } # Changed -``` - -### Semantic Versioning - -Use different routes for different versions: - -```toml -[routes] -"app.mydomain.com" = { image = "myapp:v2.1.0" } # Production -"staging.mydomain.com" = { image = "myapp:staging" } # Staging -"canary.mydomain.com" = { image = "myapp:canary" } # Canary -``` - -## Multiple Routes - -### Same Image, Different Domains - -```toml -[routes] -"app.mydomain.com" = { image = "myapp:latest" } -"www.mydomain.com" = { image = "myapp:latest" } -"mydomain.com" = { image = "myapp:latest" } -``` - -### Multiple Services - -```toml -[routes] -"app.mydomain.com" = { image = "frontend:latest" } -"api.mydomain.com" = { image = "backend:v2.1.0" } -"docs.mydomain.com" = { image = "documentation:latest" } -"status.mydomain.com" = { image = "status-page:v1.0.0" } -``` - -## How Routing Works - -1. Request arrives for `app.mydomain.com` -2. Gordon looks up the route in configuration -3. Finds running container for `myapp:latest` -4. Proxies request to container's exposed port -5. Returns response to client - -``` -Client ─> Gordon Proxy ─> Container - (port 80) (exposed port) -``` - -## Deployment Flow - -When you push an image: - -```bash -docker push registry.mydomain.com/myapp:latest -``` - -1. Gordon receives the image -2. Checks routes for any that use `myapp:latest` -3. Deploys new container for each matching route -4. Updates proxy to route traffic to new container -5. Stops old container - -## Route Changes - -Route changes in the config file are hot-reloaded automatically: - -1. Edit `[routes]` in `gordon.toml` -2. Save the file -3. Gordon reloads the updated routes and proxy config - -Legacy `http://...` route keys are still read here for backward compatibility and rewritten the next time Gordon saves the config. - -You can still manage routes with the API or CLI if you prefer live mutations: - -```bash -gordon routes add newapp.mydomain.com newapp:latest -gordon routes add app.mydomain.com myapp:v2.2.0 -gordon routes remove oldapp.mydomain.com -``` - -## Examples - -### Development Setup - -```toml -[routes] -"dev-app.example.com" = { image = "myapp:latest" } -"dev-api.example.com" = { image = "myapi:latest" } -``` - -Add to `/etc/hosts`: - -```text -127.0.0.1 dev-app.example.com dev-api.example.com registry.example.com -``` - -### Production Setup - -```toml -[routes] -"app.company.com" = { image = "company-app:v2.1.0" } -"api.company.com" = { image = "company-api:v1.5.2" } -"admin.company.com" = { image = "admin-panel:v1.0.1" } -"docs.company.com" = { image = "company-docs:latest" } -``` - -### Multi-Tenant SaaS - -```toml -[routes] -# Platform services -"app.saas-platform.com" = { image = "saas-frontend:v2.1.0" } -"api.saas-platform.com" = { image = "saas-api:v3.2.1" } - -# Customer subdomains -"acme.saas-platform.com" = { image = "saas-app:v2.1.0" } -"beta.saas-platform.com" = { image = "saas-app:v2.1.0" } - -# Customer custom domains -"portal.acme-corp.com" = { image = "saas-app:v2.1.0" } -``` - -## External Services - -For non-containerized services (databases, legacy apps, etc.), see [External Routes](./external-routes.md). - -## Related - -- [Configuration Overview](./index.md) -- [Auto Route](./auto-route.md) -- [Attachments](./attachments.md) -- [External Routes](./external-routes.md) diff --git a/docs/config/secrets.md b/docs/config/secrets.md index 2ada0c338..840c6cfa9 100644 --- a/docs/config/secrets.md +++ b/docs/config/secrets.md @@ -1,6 +1,6 @@ # Secrets Configuration -Configure how Gordon stores and retrieves sensitive data. +Configure how Gordon stores and retrieves sensitive data: auth secrets such as `token_secret` (this page) and app secret values (managed with `gordon apps secrets`, values in pass under `gordon/apps///`). ## Configuration @@ -52,12 +52,6 @@ token_secret = "gordon/auth/token_secret" # Path in pass store - Standard Unix tooling - Works with team GPG keys -**Route secrets storage:** -- `gordon secrets set` stores per-domain secrets in pass under `gordon/env//` (dots/colons/slashes → underscores) -- Existing `.env` files are auto-migrated on startup and renamed to `.env.migrated` -- Attachment secrets are stored under `gordon/env/attachments//` with a `.keys` manifest -- Use `gordon secrets set --attachment KEY=value` to manage them - ### SOPS Uses Mozilla SOPS for encrypted file-based secrets: @@ -77,13 +71,6 @@ brew install sops # macOS sops secrets.yaml ``` -**Usage in env files:** -```bash -# ~/.gordon/env/app_mydomain_com.env -API_SECRET=${sops:secrets.yaml:api.secret} -DB_PASSWORD=${sops:secrets.yaml:database.password} -``` - **Benefits:** - Multiple encryption backends (AWS KMS, GCP KMS, Azure Key Vault, PGP) - YAML/JSON file encryption @@ -94,15 +81,6 @@ DB_PASSWORD=${sops:secrets.yaml:database.password} - Path traversal (`..`) is blocked - Only relative paths from your config directory are allowed -**Route secrets storage:** -- Domain secrets stay in `.env` files -- Use `${sops:...}` syntax to resolve encrypted values - -**Attachment secrets storage:** -- Attachment secrets are stored in `gordon-.env` files alongside domain env files -- Use `${sops:...}` syntax inside attachment env files to resolve encrypted values -- Use `gordon secrets set --attachment KEY=value` to manage them - ### Unsafe (Development Only) Stores secrets as plain text files: @@ -120,11 +98,6 @@ secrets_backend = "unsafe" │ └── token_secret ``` -**Attachment secrets:** -- Stored as `gordon-.env` files in the env directory -- Example: `gordon-app__mydomain__com-postgres.env` -- Use `gordon secrets set --attachment KEY=value` to manage them - **Usage:** ```bash # Create secret @@ -134,25 +107,15 @@ echo "your-token-secret" > ~/.gordon/secrets/gordon/auth/token_secret > **Warning:** Only use for local development. Secrets are stored in plain text. -## Secret Provider Syntax - -In environment files, reference secrets using provider syntax: +## App Secret Values -### Pass Provider +App secret names are registered in the app manifest (`[services..secrets]` maps ENV name to secret name); values are written separately and stay in pass: ```bash -# ${pass:} -DATABASE_PASSWORD=${pass:myapp/database/password} -API_KEY=${pass:myapp/api-key} +gordon apps secrets set blog --service web DATABASE_URL=... ``` -### SOPS Provider - -```bash -# ${sops::} -DATABASE_PASSWORD=${sops:secrets.yaml:database.password} -API_SECRET=${sops:production.yaml:api.secret.key} -``` +Values affect the next deploy/restart, never running containers. Only key names are ever echoed back, never values. See [App Manifest](./apps.md) and [Apps CLI](../cli/apps.md). ## Examples @@ -182,29 +145,10 @@ enabled = false secrets_backend = "unsafe" ``` -### Enterprise with SOPS - -```toml -[auth] -enabled = true -secrets_backend = "sops" -token_secret = "gordon/auth/token_secret" -``` - -Environment file: -```bash -# ~/.gordon/env/app_company_com.env -NODE_ENV=production -DATABASE_URL=postgresql://db:5432/app -DATABASE_PASSWORD=${sops:secrets.yaml:database.password} -API_KEY=${sops:secrets.yaml:api.key} -JWT_SECRET=${sops:secrets.yaml:jwt.secret} -``` - ## Security Recommendations 1. **Production**: Always use `pass` or `sops` backend -2. **Never commit**: Don't commit unencrypted secrets to git +2. **Never commit**: Don't commit unencrypted secrets to git (app files must never contain secret values) 3. **Rotate regularly**: Regenerate tokens and passwords periodically 4. **Least privilege**: Use separate secrets per environment 5. **Path validation**: SOPS provider rejects absolute paths and path traversal attempts for security @@ -212,5 +156,5 @@ JWT_SECRET=${sops:secrets.yaml:jwt.secret} ## Related - [Authentication](./auth.md) -- [Environment Variables](./env.md) +- [App Manifest](./apps.md) - [Configuration Overview](./index.md) diff --git a/docs/config/security-hardening.md b/docs/config/security-hardening.md index beb796834..20a5b498e 100644 --- a/docs/config/security-hardening.md +++ b/docs/config/security-hardening.md @@ -24,37 +24,79 @@ Container and deploy logs can include environment-derived output. Gordon gates l gordon auth token generate --subject ops --scopes "admin:logs:read" --expiry 30d ``` -- `/admin/logs` and deploy failure logs require `admin:logs:read`. +- `/admin/logs`, app failure diagnostics, and deploy failure logs require `admin:logs:read`. - `admin:status:read` does not grant log access. -- Common secret patterns are redacted before logs are returned. +- Operation errors returned to `admin:apps:read` callers are stable and log-free. Application diagnostics are a separate field, redacted of the affected service's resolved secret values before storage, and returned only to callers holding `admin:logs:read`. If a declared secret cannot be resolved for redaction, Gordon drops the diagnostics instead of storing unredacted output. Mutation responses never include application output. +- Process and container log sources remain operator-owned data. Gordon redacts common credential patterns when serving logs through its API, but this is defense in depth rather than proof that arbitrary application output is secret-free. Restrict filesystem, journal, runtime, backup, and `admin:logs:read` access accordingly. ## Volume pruning scope -Volume pruning only removes unused Docker volumes explicitly managed by Gordon (`gordon.managed=true`). It ignores unrelated Docker volumes even if they are unused. +Volume pruning removes a volume only when Gordon has a durable `released` ownership record, the runtime app/incarnation/service labels agree with that record, and no container mounts it. A `gordon.managed=true` label by itself is not deletion authority; retained, unknown, contradictory, and unrelated volumes survive. + +Use `gordon volumes prune --dry-run` before deletion. Do not substitute `docker volume prune` or an equivalent runtime command: runtime-native pruning bypasses Gordon's ownership and retention checks. Use dedicated admin scopes: - `admin:volumes:read` for listing volumes. - `admin:volumes:write` for prune operations. -## Pass migration plaintext handling +## Administrative app bind mounts + +App manifests cannot name host paths. A host bind exists only when the operator declares a named policy in `gordon.toml`: + +```toml +[app_mounts.app-logs] +source = "/srv/gordon/host-logs" +read_only = true +allowed_apps = ["metrics-agent"] # exact, non-empty +allowed_services = ["web"] # exact, non-empty +root = "/srv/gordon" # optional boundary +``` + +The manifest references only the policy name: `[[services..bind]] name = "app-logs"` with an absolute container `path`. + +- `allowed_apps` and `allowed_services` are exact, non-empty allowlists, so least privilege is enforced by construction. There is no wildcard, prefix, or empty-means-all form. +- `source` is resolved through symlinks and must be a regular file or directory under `root`. When `root` is omitted, the source's parent directory is the boundary. Devices, sockets, FIFOs, and escaping symlinks are refused. +- Read-only precedence: `read_only` on the policy or `readonly` on the manifest bind forces the mount read-only; a manifest never weakens its policy. +- Destinations must be absolute, normalized, and outside reserved container paths (`/`, `/proc`, `/sys`, `/dev`, `/boot`, and their children). +- The policy is re-resolved immediately before every container create, restart, and recovery. Removing a policy or an allowlist entry blocks future deploys; running containers keep their current mounts until redeployed. +- `gordon serve` reload republishes validated policies atomically. An edit that fails validation is rejected and the previous policies stay live. +- Gordon never creates, deletes, chowns, backs up, or prunes the host path; ownership, permissions, and backup remain the operator's responsibility. + +## Administrative app devices (CDI) -When Gordon migrates legacy plaintext `.env` files into `pass`, it removes the plaintext source after a successful migration and does not leave `.env.migrated` copies by default. If pass entries already exist, migration fails closed and leaves the plaintext file in place for manual operator review rather than deleting potentially unique values. +App manifests cannot name host devices. A device grant exists only when the operator declares a named policy in `gordon.toml`: + +```toml +[app_devices.transcode-gpu] +cdi = ["example.com/gpu=GPU-device-uuid"] +allowed_apps = ["video"] # exact, non-empty +allowed_services = ["transcoder"] # exact, non-empty +``` + +The manifest references only the logical name: `devices = ["transcode-gpu"]`. + +- `cdi` holds explicit CDI device IDs. Raw `/dev` paths, unqualified names, and aggregate `=all` selectors are rejected. +- `allowed_apps` and `allowed_services` are exact, non-empty allowlists, so least privilege is enforced by construction. There is no wildcard, prefix, or empty-means-all form. +- Gordon resolves logical names to CDI IDs at activation time and encodes them as one native CDI `DeviceRequest`. Revisions persist the logical names, never the host resolution. +- The grant is re-resolved immediately before every preflight, container create, restart, and recovery. Revoking a grant blocks future deploys while the running service is untouched; changing a mapping never recreates a running container. +- Device-bearing creates require Podman 5.4+ or Docker 28.3+ with native CDI configured. Older or unrecognized engines fail with a structured `runtime-unsupported` error and never run without devices. +- Gordon installs no drivers, manages no quotas, and injects no device environment: images carry their own runtime expectations. ## External image registries -Gordon's configured registry is always allowed. Explicit external registries must be allowlisted: +Docker Hub, `ghcr.io`, `quay.io`, and Gordon's configured registry are always allowed. Add every other registry hostname and non-default port explicitly: ```toml [images] -allowed_registries = ["docker.io", "ghcr.io", "registry.example.com:5000"] +allowed_registries = ["registry.internal:5000"] require_digest = true ``` -- Empty `allowed_registries` rejects explicit external registries. -- `localhost`, loopback, private, link-local, unspecified, and metadata-style registries are rejected. -- `require_digest = true` requires allowlisted external images to use `@sha256:<64 hex chars>`. -- Include ports in allowlist entries when the registry uses a non-default port. +- `docker.io` and Docker Hub's canonical pull host `registry-1.docker.io` are equivalent. +- Other registries are accepted only when their exact hostname+port is configured. Allowlisting does not configure credentials; external resolution and pulls are anonymous unless the runtime already has suitable access. +- `require_digest = true` requires every image, including Gordon registry images, to use `@sha256:<64 hex chars>`. +- The allowlist restricts hostnames, not resolved IPs. It cannot prove a DNS hostname is non-private and does not replace firewall or runtime egress controls. ## Smart TCP Raw Fallback @@ -95,7 +137,28 @@ enabled = true internal = true ``` -`internal = false` remains the compatibility default because some applications and attachments need direct egress during startup. +`internal = false` remains the compatibility default because some applications need direct egress during startup. + +Every app container joins an incarnation-owned private network derived from the app's internal UUID; shared memberships come only from explicit `[[network.shared]]` declarations, and memory/CPU/PID limits apply on every create and recovery path. + +## Readiness helper containers + +Probing an internal HTTP port uses one bounded helper container per readiness attempt, removed immediately after. Gordon force-removes the helper under an independent cleanup context, including on failure and timeout. There is no operator configuration for the helper; its image and limits are fixed. + +- The helper image is digest-pinned and multi-arch: `alpine@sha256:d9e853e87e55526f6b2917df91a2115c36dd7c696a35be12163d44e6e2a4b6bc`. The image must be pre-provisioned and available offline on the target host, because the probe never pulls. +- The helper attaches only to the target app's private network. It never joins a shared network or a host network. +- It never publishes a host port and has no mounts, volumes, secrets, environment, or runtime socket. +- It runs non-root (uid/gid `65534`) with a read-only root filesystem, all capabilities dropped, `no-new-privileges`, and bounded CPU, memory, and PIDs. +- It targets the exact inspected container IP on that network, never a service alias, and revalidates the container's execution start before trusting the result; a restarted generation is discarded. +- A target that listens only on its own loopback fails readiness. +- Helpers left behind by an abrupt daemon or host stop are reclaimed by the next probe once they are older than ten minutes; a live session is never touched. + +## Registry exposure + +- With `auth.enabled = true`, registry requests are authenticated and repository-scoped; the public proxy forwards registry domains to the internal registry. +- With `auth.enabled = false`, the registry is local-only: the public proxy refuses registry-domain requests, and the registry handler accepts only direct loopback connections carrying the instance credentials. +- Registry requests are parsed once into a validated operation used by both authorization and dispatch, so the repository a token is checked against is exactly the repository served. +- Blob and upload access is repository-scoped: an upload UUID is usable only by the repository that started it, and a blob is served only to a repository that completed an upload of that digest. ## Container runtime profile @@ -114,6 +177,5 @@ Use `strict` for images designed to write only to mounted volumes and run withou - [Auth](./auth.md) - [Images](./images.md) - [Network Isolation](./network-isolation.md) -- [Deploy](./deploy.md) - [Volumes](./volumes.md) - [Reference](./reference.md) diff --git a/docs/config/server.md b/docs/config/server.md index d47a0529a..c4bc9ac32 100644 --- a/docs/config/server.md +++ b/docs/config/server.md @@ -32,7 +32,7 @@ protocol = "smart_tcp" | `gordon_domain` | string | **required** | Domain for Gordon (registry + admin API) | | `registry_domain` | string | - | Deprecated migration key. Set `gordon_domain` instead. | | `legacy_registry_domains` | []string | `[]` | Additional Gordon registry hosts treated as aliases during staged migration. See [Upgrading: Staged Registry Host Rename](../upgrading.md#staged-registry-host-rename). | -| `data_dir` | string | `~/.gordon` | Directory for registry data, logs, and env files | +| `data_dir` | string | `~/.gordon` | Directory for registry data, logs, and state | | `max_proxy_body_size` | string | `"512MB"` | Maximum request body size for proxied requests | | `max_blob_chunk_size` | string | `"95MB"` | Maximum request body size for a single registry blob upload chunk | | `max_blob_size` | string | `"1GB"` | Maximum cumulative size for one registry blob/layer upload | @@ -102,7 +102,7 @@ Challenge behavior: HTTP-01 requires public access to external port 80 for each hostname being validated. If you use Cloudflare in DNS-only/gray-cloud mode, your firewall/NAT must allow direct public traffic to the HTTP-capable smart TCP entrypoint. A firewall rule that only allows Cloudflare source IPs on port 80 is compatible with orange-cloud proxying, but it blocks gray-cloud HTTP-01 validation; use DNS-01 or temporarily open port 80 for direct validation. -Gordon automatically includes `server.gordon_domain` in public ACME coverage in addition to configured HTTPS routes. The management hostname therefore needs the same challenge reachability: public port 80 for HTTP-01, or zone-read and DNS-edit permissions for its zone with Cloudflare DNS-01. +Gordon automatically includes `server.gordon_domain` in public ACME coverage in addition to app HTTPS hosts. The management hostname therefore needs the same challenge reachability: public port 80 for HTTP-01, or zone-read and DNS-edit permissions for its zone with Cloudflare DNS-01. Gordon limits new ACME certificate orders to `obtain_batch_size` per reconcile run (default `1`) so enabling ACME on an existing multi-route server does not burst through every route and hit Let's Encrypt rate limits. The management hostname consumes a place in the same batch; later reloads, restarts, or other explicit reconcile runs continue issuing remaining certificates. @@ -116,7 +116,7 @@ DNS-01 uses the Cloudflare API to create TXT records, then checks public DNS vis The defaults avoid relying on host-local DNS. This matters on hosts using Tailscale MagicDNS, split-horizon corporate DNS, or Pi-hole, where `/etc/resolv.conf` may not reflect public DNS visibility as Let's Encrypt sees it. -Because Gordon's ACME challenge mode is global, `cloudflare-dns-01` requires a Cloudflare token that can read zones and edit DNS records for every zone used by configured HTTPS routes. If that is not desirable, use `http-01` until Gordon supports per-route or per-zone challenge policy. +Because Gordon's ACME challenge mode is global, `cloudflare-dns-01` requires a Cloudflare token that can read zones and edit DNS records for every zone used by app HTTPS hosts. If that is not desirable, use `http-01` until Gordon supports per-route or per-zone challenge policy. #### Direct HTTP CA onboarding paths (`/.well-known/gordon/ca`) @@ -128,7 +128,7 @@ When Gordon is serving TLS-capable edge traffic, direct cleartext HTTP clients ( This lets new clients discover and trust the internal CA over plain HTTP without exposing the full application. Trusted proxy traffic (e.g. from Cloudflare) continues through the normal HTTP proxy path unaffected. -On HTTPS, onboarding paths are served only on `server.gordon_domain`; app hosts can use their own routes without Gordon intercepting them. If `gordon_domain` is empty, HTTPS onboarding paths are not served at all to avoid intercepting app hosts. +On HTTPS, onboarding paths are served only on `server.gordon_domain`; app hosts are served without Gordon intercepting them. If `gordon_domain` is empty, HTTPS onboarding paths are not served at all to avoid intercepting app hosts. #### HTTP to HTTPS redirect @@ -187,7 +187,7 @@ gordon_domain = "gordon.mydomain.com" This domain is used for: - Docker login and image push/pull operations - Admin API access (`/admin/*` endpoints) -- CLI remote targeting (`gordon routes --remote https://gordon.mydomain.com`) +- CLI remote targeting (`gordon apps --remote https://gordon.mydomain.com`) - Authentication endpoints (`/auth/*`) > **Warning:** If you are upgrading an older config, copy `server.registry_domain` to `server.gordon_domain` before restarting. @@ -220,13 +220,9 @@ data_dir = "~/.gordon" # Default for user installations ├── registry/ # Container images and manifests │ ├── blobs/ │ └── manifests/ -├── env/ # Environment files per domain -│ ├── app_mydomain_com.env -│ └── api_mydomain_com.env -├── logs/ # Application and container logs +├── logs/ # Gordon process and file-based access logs │ ├── gordon.log -│ ├── proxy.log -│ └── containers/ +│ └── access.log └── secrets/ # Secrets (unsafe backend only) ``` @@ -430,6 +426,6 @@ sudo firewall-cmd --reload - [Configuration Overview](./index.md) - [Installation Guide](../installation.md) -- [Routes Configuration](./routes.md) +- [App Manifest](./apps.md) - [Standalone Services](./services.md) - [Traffic Plane Configuration](./traffic.md) diff --git a/docs/config/services.md b/docs/config/services.md index 6cfc61a1e..bdb282807 100644 --- a/docs/config/services.md +++ b/docs/config/services.md @@ -91,4 +91,4 @@ By default Gordon removes old or disabled service containers while preserving vo - [Traffic Plane Configuration](./traffic.md) - [Configuration Reference](./reference.md) -- [CLI traffic status](../cli/traffic.md) +- [CLI traffic status](../cli/daemon.md#gordon-daemon-traffic) diff --git a/docs/config/telemetry.md b/docs/config/telemetry.md index 06bccc3db..643264327 100644 --- a/docs/config/telemetry.md +++ b/docs/config/telemetry.md @@ -26,7 +26,7 @@ trace_sample_rate = 1.0 | `auth_token` | string | `""` | Base64-encoded `user:password` for Basic auth | | `traces` | bool | `true` | Export distributed traces | | `metrics` | bool | `true` | Export metrics (deploy counters, container lifecycle, registry, events) | -| `logs` | bool | `true` | Bridge zerolog output to OTLP logs | +| `logs` | bool | `true` | Export Gordon, proxy access, and app container logs to OTLP | | `trace_sample_rate` | float | `1.0` | Fraction of traces to sample (`0.0` = none, `1.0` = all) | ## How It Works @@ -35,10 +35,59 @@ Gordon initializes an OTel provider at startup. When telemetry is enabled: 1. **Traces** -- Spans wrap critical operations: container deploy, image pull, registry manifest push, and proxy target resolution. The `otelhttp` middleware adds a span to every HTTP request on both the proxy and registry servers. 2. **Metrics** -- Gordon records custom counters and histograms for deploys, container restarts, crash loops, managed container count, registry pushes, and event bus throughput. -3. **Logs** -- A zerowrap/otel hook bridges all structured log output to the OTLP log pipeline. Every log line automatically carries `trace_id` and `span_id` when emitted inside a traced request. +3. **Logs** -- Gordon exports three log families (see [Logs](#logs)). When telemetry is disabled (the default), all OTel instruments are noop -- zero overhead. +## Logs + +With `logs = true`, Gordon exports: + +| Family | Source | `service.name` | +|--------|--------|----------------| +| Gordon process logs | zerowrap output, with `trace_id`/`span_id` inside traced requests | `gordon` | +| Proxy access logs | Every proxied request, attributed to the app service owning the host | `.` (`gordon` for registry and unknown hosts) | +| App container logs | stdout and stderr of every running app service container | `.` | + +Every app record carries these resource attributes: + +| Attribute | Example | Use | +|-----------|---------|-----| +| `service.name` | `blog.web` | One exact service; unique across apps | +| `service.namespace` | `blog` | All services of one app | +| `gordon.app` | `blog` | Filter by app | +| `gordon.service` | `web` | Filter by service name across apps | + +Record attributes: + +- `log.type`: `access` or `container` +- `log.iostream`: `stdout` or `stderr` (container logs) +- Access logs: `http.request.method`, `http.response.status_code`, `server.address`, `url.path`, `client.address`, `request.id`, and more. Severity follows the status: `5xx` is `ERROR`, `4xx` is `WARN`, otherwise `INFO`. + +Loki indexes `service.name` and `service.namespace` as stream labels by default; the other attributes are structured metadata, filtered after the stream selector. Example queries in Grafana: + +```text +{service_namespace="blog"} # every log of the blog app +{service_name="blog.web"} | log_type="access" # access logs of one service +{service_namespace=~".+"} | gordon_service="web" # every "web" service, all apps +``` + +Container output is exported from the moment Gordon starts, including the first lines of every container deployed afterwards. Output emitted while Gordon is stopped is not exported. Lines longer than 64 KiB are truncated. Export is buffered and never slows the proxy or the app: on overload, the oldest records are dropped. + +### Per-app opt-out + +App container logs are exported by default. Opt out in the [app manifest](./apps.md#telemetry), per app or per service; a service setting overrides the app setting: + +```toml +[telemetry] +logs = true # app default + +[services.db.telemetry] +logs = false # only "db" stops exporting +``` + +The opt-out covers container output only; access logs of the app's hosts are still exported. + ## Endpoint URL The `endpoint` field accepts a full URL. Gordon parses it to extract: @@ -61,7 +110,7 @@ Set `auth_token` to the Base64-encoded `user:password` string. Gordon sends it a For OpenObserve, copy the token from **Ingestion > OTLP** in the web UI. -Since Gordon itself is the platform (not a managed container), it does not use `gordon secrets set`. Store the token with one of these methods: +Gordon itself is the platform, not an app, so the token does not go through `gordon apps secrets`. Store it with one of these methods: | Method | How | |--------|-----| @@ -98,9 +147,9 @@ Attributes: `domain`, `image` |--------|------|------|-------------| | `gordon.container.restarts` | Counter | - | Container restart count | | `gordon.container.crash_loops` | Counter | - | Crash loop detections | -| `gordon.container.managed` | UpDownCounter | - | Currently tracked containers | +| `gordon.container.managed` | Gauge | - | Containers of running app services, read from ACTIVE app state at each export | -Attributes: `source` (restarts only: `monitor` or `api`); `gordon.container.managed` is a global gauge with no attributes +Attributes: `source` (restarts only: `monitor` or `api`). `gordon.container.managed` has no attributes; tell instances apart with the `host.name` resource attribute. ### Registry diff --git a/docs/config/traffic.md b/docs/config/traffic.md index be53c2c37..b132384c2 100644 --- a/docs/config/traffic.md +++ b/docs/config/traffic.md @@ -4,7 +4,7 @@ Gordon models network listeners as a traffic graph: entrypoints receive packets ## Smart TCP Edge Entrypoint -Public application traffic is normally exposed with a `smart_tcp` entrypoint. The conventional route-capable entrypoint name is `edge`, but Gordon does not require that name or assign a built-in public port. When exactly one route-capable `smart_tcp` or `tls_mux` entrypoint exists, normal Gordon routes use it even if it has a custom name. Choose the address that matches your deployment, firewall, and container mapping: +Public application traffic is normally exposed with a `smart_tcp` entrypoint. The conventional route-capable entrypoint name is `edge`, but Gordon does not require that name or assign a built-in public port. When exactly one route-capable `smart_tcp` or `tls_mux` entrypoint exists, app HTTP hosts use it even if it has a custom name. Choose the address that matches your deployment, firewall, and container mapping: ```toml [entrypoints.edge] @@ -15,7 +15,7 @@ trusted_cidrs = [] Do not treat `entrypoints.edge.address` as an HTTP port or an HTTPS port. It is one TCP socket that sniffs each new connection and dispatches the original byte stream. -Supported entrypoint protocols are `smart_tcp`, `tls_mux`, `tcp`, and `udp`. `smart_tcp` is the primary public edge model; `tls_mux` can also serve normal Gordon routes through TLS fallback, `tcp` and `udp` are for explicit L4 services, and UDP remains separate from the TCP entrypoint. +Supported entrypoint protocols are `smart_tcp`, `tls_mux`, `tcp`, and `udp`. `smart_tcp` is the primary public edge model; `tls_mux` can also serve app HTTP hosts through TLS fallback, `tcp` and `udp` are for explicit L4 services, and UDP remains separate from the TCP entrypoint. ## Smart TCP Dispatch Order @@ -73,7 +73,7 @@ protocol = "tcp" Service references use: -- `route:` for configured HTTP routes +- `route:` for app HTTP hosts from ACTIVE state (merged with installation external routes) - `external_route:` for configured external HTTP routes - `network_service::` for manually managed TCP, UDP, and TLS passthrough backends - `service::` for Gordon-managed standalone service backends @@ -96,7 +96,7 @@ sni = "raw.example.com" service = "network_service:raw:tls" ``` -Exact SNI matches win over wildcard matches. Ambiguous wildcard overlaps and HTTP-host/TLS-passthrough conflicts on the same smart TCP entrypoint are rejected at validation time. HTTPS application routes that do not match a passthrough SNI use Gordon's normal HTTPS fallback and certificate selection. +Exact SNI matches win over wildcard matches. Ambiguous wildcard overlaps and HTTP-host/TLS-passthrough conflicts on the same smart TCP entrypoint are rejected at validation time. App HTTPS hosts that do not match a passthrough SNI use Gordon's normal HTTPS fallback and certificate selection. ## Raw TCP Fallback @@ -176,6 +176,6 @@ UDP sessions are keyed by client address and expire after `idle_timeout`. If `ma ## Related - [Server Settings](./server.md) -- [Routes](./routes.md) +- [App Manifest](./apps.md) - [Standalone Services](./services.md) -- [CLI traffic status](../cli/traffic.md) +- [CLI traffic status](../cli/daemon.md#gordon-daemon-traffic) diff --git a/docs/config/volumes.md b/docs/config/volumes.md index 053fed2f2..5781829c2 100644 --- a/docs/config/volumes.md +++ b/docs/config/volumes.md @@ -1,165 +1,126 @@ # Volumes Configuration -Configure automatic persistent storage for containers. +Persistent app storage is declared in each app manifest. The installation-level `[volumes]` settings apply to non-app volume management and do not control declarative app storage. -## Configuration +## Declarative app volumes + +Services declare persistent mounts in the app manifest: ```toml -[volumes] -auto_create = true -prefix = "gordon" -preserve = true +[[services.web.volume]] +name = "database-data" +path = "/var/lib/postgresql/data" ``` -## Options - -| Option | Type | Default | Description | -|--------|------|---------|-------------| -| `auto_create` | bool | `true` | Automatically create volumes from Dockerfile VOLUME | -| `prefix` | string | `"gordon"` | Prefix for volume names | -| `preserve` | bool | `true` | Keep volumes when containers are removed | +A volume name is unique within its service. Gordon creates an incarnation-owned runtime volume, records the app, app UUID, service, and logical volume ownership, and reuses it across replacement and restart. Manifests never carry host paths, and sharing one volume between services is not supported. When an app must read an operator-approved host location, use a named administrative bind instead. -## How It Works +## Administrative bind mounts -Gordon automatically creates Docker volumes from Dockerfile `VOLUME` directives: +Manifests cannot name host paths directly. The operator declares a named policy in `gordon.toml`; the manifest only references the policy name: -```dockerfile -FROM postgres:18 -VOLUME ["/var/lib/postgresql/data"] +```toml +# gordon.toml +[app_mounts.app-logs] +source = "/srv/gordon/host-logs" # required; the only host path this mount may come from +read_only = true +allowed_apps = ["metrics-agent"] # required; exact, non-empty +allowed_services = ["web"] # required; exact, non-empty +# root = "/srv/gordon" # optional administrative boundary ``` -When Gordon deploys this container: - -1. Reads `VOLUME` directives from image metadata -2. Creates named volumes with `prefix-domain-path` naming -3. Mounts volumes to the container -4. Preserves data across container updates - -## Volume Naming - -Volumes are named: `{prefix}-{domain}-{path}` - -| Domain | Volume Path | Volume Name | -|--------|-------------|-------------| -| `app.mydomain.com` | `/data` | `gordon-app-mydomain-com-data` | -| `db.mydomain.com` | `/var/lib/postgresql/data` | `gordon-db-mydomain-com-var-lib-postgresql-data` | - -## Persistence - -### Default: Preserve Volumes - ```toml -[volumes] -preserve = true +# app manifest +[[services.web.bind]] +name = "app-logs" # must match [app_mounts.] +path = "/var/lib/collector/host-logs" # container destination +readonly = true ``` -With `preserve = true`: -- Volumes persist when containers are updated -- Data survives container restarts -- Volumes remain even if container is removed +- `source` must be absolute, normalized, and resolve through symlinks to a regular file or directory that stays under `root`. When `root` is omitted, the source's parent directory is the boundary, so a symlinked source may resolve within it but never escape it. +- `allowed_apps` and `allowed_services` are exact, non-empty allowlists; there is no wildcard, prefix, or empty-means-all form. +- Read-only precedence: `read_only` on the policy or `readonly` on the bind forces the resolved mount read-only. A bind never weakens its policy. +- Gordon re-resolves the policy immediately before every container create, restart, and recovery. Removing a policy or an allowlist entry blocks future deploys; running containers keep their current mounts until redeployed. `gordon serve` reload republishes validated policies atomically, and a failing edit keeps the previous policies live. +- Gordon never creates, copies, deletes, chowns, backs up, or prunes the host path; ownership and permissions stay with the operator. -### Remove with Container +A read-only log collector is the typical use. Its own state is a named volume; the host logs are exposed through a bind: ```toml -[volumes] -preserve = false -``` - -With `preserve = false`: -- Volumes are removed when containers are removed -- Useful for stateless containers -- Frees up disk space automatically +name = "metrics-agent" -## Examples +[services.collector] +image = "registry.example.com/metrics/collector:2.1.0" -### Database Container +[[services.collector.volume]] +name = "collector-state" +path = "/var/lib/collector" -```dockerfile -# my-postgres.Dockerfile -FROM postgres:18 -VOLUME ["/var/lib/postgresql/data"] -ENV POSTGRES_DB=myapp -ENV POSTGRES_USER=app +[[services.collector.bind]] +name = "app-logs" +path = "/var/lib/collector/host-logs" +readonly = true ``` -Gordon automatically: -- Creates `gordon-db-mydomain-com-var-lib-postgresql-data` volume -- Mounts it to `/var/lib/postgresql/data` -- Preserves data across postgres container updates +Collection is one-way: the collector reads operator-owned logs and cannot write back through the bind. -### Application with Uploads +Every Dockerfile `VOLUME` path must be mapped explicitly. A matching `[[services..volume]]` declaration maps it to a Gordon-owned volume; a `[[services..bind]]` whose container `path` equals the `VOLUME` path also satisfies the mapping, because the bind mounts over that image-declared volume. Any `VOLUME` path with neither mapping fails closed at deploy with an unmanaged-volume error. A bind destination may not collide with a declared volume path. -```dockerfile -FROM node:18 -WORKDIR /app -VOLUME ["/app/uploads", "/app/data"] -COPY . . -CMD ["npm", "start"] -``` +Runtime volume names are implementation details. Use `gordon volumes list` and ownership labels to inspect them; do not derive ownership from a name, rename volumes, or edit Gordon's ownership records. -Creates two volumes: -- `gordon-app-mydomain-com-app-uploads` -- `gordon-app-mydomain-com-app-data` +## Retention -### Custom Prefix +Ordinary app operations retain data: -```toml -[volumes] -prefix = "prod" -``` +- deploy and restart reuse the service's volumes; +- removing a service retains its volumes; +- `gordon apps remove` retains volumes under the removed app's internal UUID; +- a new app that reuses the public name does not adopt retained volumes; +- administrative binds are unaffected by app operations: their host paths are never created, adopted, or retained by Gordon. -Volume names become: -- `prod-app-mydomain-com-data` -- `prod-db-mydomain-com-var-lib-postgresql-data` +Gordon does not automatically delete retained app volumes. Back up persistent data before any manual deletion, and use database-native backup and restore procedures for databases. -## Managing Volumes +## Manual migration -### List Volumes +Gordon does not copy data between volumes. Stop every container that can write to the source, verify a backup, create an empty target volume, and perform the copy through the runtime so rootless UID/GID mappings are preserved. For Podman, a disposable helper container is the preferred generic method: ```bash -docker volume ls | grep gordon +podman run --rm --pull=never --network=none \ + --volume ":/source:ro" \ + --volume ":/target" \ + docker.io/library/alpine@sha256:d9e853e87e55526f6b2917df91a2115c36dd7c696a35be12163d44e6e2a4b6bc \ + sh -c 'cp -a /source/. /target/' ``` -### Inspect Volume +Replace the literal `<...>` placeholders before running; they are quoted above so a shell cannot mistake them for redirection. The approved helper image is digest-pinned, must already be available locally, and runs without network access. If direct access to storage paths is unavoidable with rootless Podman, enter its user namespace: ```bash -docker volume inspect gordon-app-mydomain-com-data +podman unshare cp -a ""/. ""/ ``` -### Backup Volume +Do not use a plain root `cp` as the default procedure: host ownership IDs may not match the container's user namespace. Applications that rely on ACLs, extended attributes, sparse files, or database consistency need an application-specific export/restore or copy tool. Validate ownership and application behavior against the target, and retain the source volume until the migration is accepted. -```bash -docker run --rm \ - -v gordon-db-mydomain-com-var-lib-postgresql-data:/data \ - -v $(pwd):/backup \ - alpine tar -czf /backup/db-backup.tar.gz -C /data . -``` +## Pruning -### Restore Volume +Use Gordon's ownership-aware command: ```bash -docker run --rm \ - -v gordon-db-mydomain-com-var-lib-postgresql-data:/data \ - -v $(pwd):/backup \ - alpine tar -xzf /backup/db-backup.tar.gz -C /data +gordon volumes prune --dry-run +gordon volumes prune ``` -## Volume Cleanup +A volume is eligible only when all of these requirements hold: -If you have orphaned volumes: +1. Gordon has a durable ownership record marking it `released`. +2. Runtime labels agree with the record's app, app incarnation UUID, and service. +3. No container mounts the volume. -```bash -# List all gordon volumes -docker volume ls -f name=gordon +Retained, attached, unknown, contradictory, and unrelated volumes survive. A `gordon.managed=true` label or a matching name is not sufficient deletion authority. Administrative binds are out of scope entirely: they are operator-owned host paths, no ownership record exists for them, and they are never candidates. With current lifecycle metadata, no operation marks app volumes `released`, so prune normally succeeds with no deletions. -# Remove specific volume (warning: deletes data!) -docker volume rm gordon-old-app-data +> **Warning:** Do not use `docker volume prune`, `podman volume prune`, or equivalent runtime cleanup for Gordon data. Those commands bypass Gordon's ownership and retention checks and can delete an unmounted retained volume. -# Prune unused volumes (be careful!) -docker volume prune -``` +See the [Volumes CLI reference](../cli/volumes.md) for flags and plan output. ## Related -- [Attachments](./attachments.md) +- [App Manifest](./apps.md) +- [Volumes CLI](../cli/volumes.md) - [Configuration Overview](./index.md) diff --git a/docs/deployment/generic-ci.md b/docs/deployment/generic-ci.md index 0ffbe92cd..55d741421 100644 --- a/docs/deployment/generic-ci.md +++ b/docs/deployment/generic-ci.md @@ -5,10 +5,10 @@ Deploy with Gordon from any CI/CD system. ## Requirements - Docker available on CI runner (for building images) -- Network access to your Gordon server (HTTPS) +- Network access to the public Gordon HTTPS domain; do not expose or target the loopback `server.registry_port` - Gordon binary (optional but recommended) -## Recommended: gordon push +## Recommended: gordon images push ### 1. Generate Deployment Token @@ -17,7 +17,7 @@ On your Gordon server: ```bash gordon auth token generate \ --subject ci-deploy \ - --scopes "push,pull,admin:routes:read,admin:config:write" \ + --scopes "push,pull,admin:apps:read,admin:apps:write" \ --expiry 0 ``` @@ -40,9 +40,9 @@ chmod +x /usr/local/bin/gordon # 2. Build, push, and deploy export GORDON_TOKEN="$GORDON_TOKEN" -gordon push --build \ +gordon images push --build \ --remote "$GORDON_REMOTE" \ - --no-confirm + ``` Gordon auto-detects the version from CI environment variables: @@ -52,7 +52,7 @@ Gordon auto-detects the version from CI environment variables: | GitHub Actions | `$GITHUB_REF` | `refs/tags/v1.2.0` | | GitLab CI | `$CI_COMMIT_TAG` | `v1.2.0` | | Azure DevOps | `$BUILD_SOURCEBRANCH` | `refs/tags/v1.2.0` | -| Other | `--tag` flag | `gordon push --tag v1.2.0` | +| Other | `--tag` flag | `gordon images push --tag v1.2.0` | ## Alternative: docker push @@ -69,7 +69,7 @@ docker push gordon.example.com/myapp:v1.0.0 ``` This requires the token subject (`ci-deploy`) as the username. -Gordon auto-deploys when it receives the image. +Pushing only stores the image; deploy explicitly with `gordon apps deploy` afterwards. ## CI System Examples @@ -89,7 +89,7 @@ pipeline { curl -fsSL https://github.com/bnema/gordon/releases/latest/download/gordon_linux_amd64 \ -o /usr/local/bin/gordon chmod +x /usr/local/bin/gordon - gordon push --build --remote "$GORDON_REMOTE" --no-confirm + gordon images push --build --remote "$GORDON_REMOTE" ''' } } @@ -119,9 +119,8 @@ jobs: - run: name: Deploy command: | - gordon push --build \ - --remote "$GORDON_REMOTE" \ - --no-confirm + gordon images push --build \ + --remote "$GORDON_REMOTE" workflows: deploy: @@ -153,7 +152,7 @@ steps: - apk add --no-cache curl - curl -fsSL https://github.com/bnema/gordon/releases/latest/download/gordon_linux_amd64 -o /usr/local/bin/gordon - chmod +x /usr/local/bin/gordon - - gordon push --build --remote "$GORDON_REMOTE" --no-confirm + - gordon images push --build --remote "$GORDON_REMOTE" trigger: event: @@ -162,11 +161,13 @@ trigger: ## Token Scopes Reference +Use the minimum scopes for the steps the pipeline performs. A repository restriction limits registry access only; it does not narrow `admin:*` permissions. + | Workflow | Required Scopes | |----------|----------------| -| Build + push + deploy | `push,pull,admin:routes:read,admin:config:write` | -| Push only (no deploy) | `push,pull,admin:routes:read` | -| docker push (auto-deploy) | `push,pull` | +| Build + push, then apply + deploy | `push,pull,admin:apps:read,admin:apps:write` | +| Push only (no deploy) | `push,pull` | +| docker push, then apply + deploy | `push,pull` + `admin:apps:read,admin:apps:write` for the CLI step | ## Troubleshooting @@ -174,21 +175,20 @@ trigger: The token is invalid or has been revoked. Generate a new one. -### "no route configured for image" +### "deploy target not found" / app has no desired state -The route must exist before pushing. Create it with: +Push only stores the image. Declare the app first: ```bash -gordon routes add myapp.example.com myapp -# or for first deploy: -gordon bootstrap myapp.example.com myapp +gordon apps apply --file myapp.toml --remote "$GORDON_REMOTE" +gordon apps deploy myapp --remote "$GORDON_REMOTE" ``` ### Version shows "latest" Gordon could not detect a version tag. Either: - Tag your git repo: `git tag v1.0.0 && git push --tags` -- Pass explicitly: `gordon push --tag v1.0.0` +- Pass explicitly: `gordon images push --tag v1.0.0` ### Large images time out @@ -200,4 +200,4 @@ Gordon uploads in 50MB chunks. For very large images (> 1GB), the push may take - [GitLab CI](./gitlab-ci.md) - [Deployment Overview](./index.md) - [Authentication](../config/auth.md) -- [Push Command](../cli/push.md) +- [Images Commands](../cli/images.md#gordon-images-push) diff --git a/docs/deployment/github-actions.md b/docs/deployment/github-actions.md index 2874e0f6f..61348f56f 100644 --- a/docs/deployment/github-actions.md +++ b/docs/deployment/github-actions.md @@ -1,16 +1,16 @@ # GitHub Actions Deployment -Two approaches for deploying with GitHub Actions: the `gordon push` CLI (recommended) or a Docker-based workflow using Gordon's official action. +Two approaches for deploying with GitHub Actions: the `gordon images push` CLI (recommended) or a Docker-based workflow using Gordon's official action. ## Prerequisites -1. Gordon server running with registry authentication enabled -2. Deployment token generated with the required scopes +1. Gordon server running with registry authentication enabled; CI reaches the public Gordon HTTPS domain, not the loopback `server.registry_port` +2. Deployment token generated with the minimum required scopes: `push,pull` for image transfer, plus `admin:apps:read,admin:apps:write` only when the workflow applies or deploys apps 3. GitHub repository secrets configured -## Recommended: gordon push +## Recommended: gordon images push -Use the `gordon push` CLI for a lightweight, single-step build and deploy. +Use the `gordon images push` CLI for a lightweight, single-step build and deploy. ### 1. Generate Token @@ -19,7 +19,7 @@ On your Gordon server: ```bash gordon auth token generate \ --subject github-actions \ - --scopes "push,pull,admin:routes:read,admin:config:write" \ + --scopes "push,pull,admin:apps:read,admin:apps:write" \ --expiry 0 ``` @@ -71,9 +71,8 @@ jobs: env: GORDON_TOKEN: ${{ secrets.GORDON_TOKEN }} run: | - gordon push --build \ - --remote ${{ secrets.GORDON_REMOTE }} \ - --no-confirm + gordon images push --build \ + --remote ${{ secrets.GORDON_REMOTE }} ``` #### Continuous Deploy on Main @@ -103,10 +102,9 @@ jobs: env: GORDON_TOKEN: ${{ secrets.GORDON_TOKEN }} run: | - gordon push --build \ + gordon images push --build \ --remote ${{ secrets.GORDON_REMOTE }} \ - --tag latest \ - --no-confirm + --tag latest ``` #### Manual Dispatch @@ -141,7 +139,7 @@ jobs: GORDON_REMOTE: ${{ secrets.GORDON_REMOTE }} DEPLOY_TAG: ${{ inputs.tag }} run: | - args=(push --build --remote "$GORDON_REMOTE" --no-confirm) + args=(push --build --remote "$GORDON_REMOTE") if [[ -n "$DEPLOY_TAG" ]]; then [[ "$DEPLOY_TAG" =~ ^[A-Za-z0-9][A-Za-z0-9._-]{0,127}$ ]] || exit 1 args+=(--tag "$DEPLOY_TAG") @@ -151,7 +149,7 @@ jobs: #### Monorepo -Deploy multiple services with separate `gordon push` calls: +Deploy multiple services with separate `gordon images push` calls: ```yaml name: Deploy Services @@ -173,21 +171,19 @@ jobs: env: GORDON_TOKEN: ${{ secrets.GORDON_TOKEN }} run: | - gordon push --build \ + gordon images push --build \ --remote ${{ secrets.GORDON_REMOTE }} \ --file ./services/api/Dockerfile \ - myapp-api \ - --no-confirm + myapp-api - name: Deploy Web env: GORDON_TOKEN: ${{ secrets.GORDON_TOKEN }} run: | - gordon push --build \ + gordon images push --build \ --remote ${{ secrets.GORDON_REMOTE }} \ --file ./services/web/Dockerfile \ - myapp-web \ - --no-confirm + myapp-web ``` #### With Build Args @@ -199,12 +195,12 @@ Pass build arguments to the Docker build: env: GORDON_TOKEN: ${{ secrets.GORDON_TOKEN }} run: | - gordon push --build \ + gordon images push --build \ --remote ${{ secrets.GORDON_REMOTE }} \ --build-arg NODE_ENV=production \ --build-arg API_URL=https://api.example.com \ --build-arg BUILD_DATE=${{ github.event.head_commit.timestamp }} \ - --no-confirm + ``` ### setup-gordon Action Reference @@ -220,7 +216,7 @@ Pass build arguments to the Docker build: ## Alternative: Docker-based Workflow -For environments where installing the Gordon binary is not desired, use Docker directly to build and push to the Gordon registry. Gordon auto-deploys when it receives the image — no explicit deploy step is needed. +For environments where installing the Gordon binary is not desired, use Docker directly to build and push to the Gordon registry. Pushing only stores the image — deploy explicitly with `gordon apps deploy` afterwards. ### Setup @@ -482,5 +478,5 @@ Error: unauthorized: authentication required - [Generic CI](./generic-ci.md) - [Deployment Overview](./index.md) - [Authentication](../config/auth.md) -- [Push Command](../cli/push.md) +- [Images Commands](../cli/images.md#gordon-images-push) - [Rollback](./rollback.md) diff --git a/docs/deployment/gitlab-ci.md b/docs/deployment/gitlab-ci.md index 44ae83a5e..d63b5c61d 100644 --- a/docs/deployment/gitlab-ci.md +++ b/docs/deployment/gitlab-ci.md @@ -4,8 +4,8 @@ Automated deployment with GitLab CI/CD using Gordon's push command. ## Prerequisites -1. Gordon server running with registry authentication enabled -2. Generated deployment token +1. Gordon server running with registry authentication enabled; CI reaches the public Gordon HTTPS domain, not the loopback `server.registry_port` +2. A least-privilege token: `push,pull` for image transfer, plus `admin:apps:read,admin:apps:write` only when the pipeline applies or deploys apps 3. GitLab CI/CD variables configured ## Quick Setup @@ -17,7 +17,7 @@ On your Gordon server: ```bash gordon auth token generate \ --subject gitlab-ci \ - --scopes "push,pull,admin:routes:read,admin:config:write" \ + --scopes "push,pull,admin:apps:read,admin:apps:write" \ --expiry 0 ``` @@ -45,10 +45,10 @@ deploy: - curl -fsSL https://github.com/bnema/gordon/releases/latest/download/gordon_linux_amd64 -o /usr/local/bin/gordon - chmod +x /usr/local/bin/gordon script: - - gordon push --build + - gordon images push --build --remote "$GORDON_REMOTE" --token "$GORDON_TOKEN" - --no-confirm + only: - tags ``` @@ -73,10 +73,10 @@ deploy: - curl -fsSL https://github.com/bnema/gordon/releases/latest/download/gordon_linux_amd64 -o /usr/local/bin/gordon - chmod +x /usr/local/bin/gordon script: - - gordon push --build + - gordon images push --build --remote "$GORDON_REMOTE" --token "$GORDON_TOKEN" - --no-confirm + rules: - if: $CI_COMMIT_TAG ``` @@ -96,10 +96,10 @@ deploy: - curl -fsSL https://github.com/bnema/gordon/releases/latest/download/gordon_linux_amd64 -o /usr/local/bin/gordon - chmod +x /usr/local/bin/gordon script: - - gordon push --build + - gordon images push --build --remote "$GORDON_REMOTE" --token "$GORDON_TOKEN" - --no-confirm + rules: - if: $CI_COMMIT_BRANCH == "main" ``` @@ -117,10 +117,10 @@ deploy: - curl -fsSL https://github.com/bnema/gordon/releases/latest/download/gordon_linux_amd64 -o /usr/local/bin/gordon - chmod +x /usr/local/bin/gordon script: - - gordon push --build + - gordon images push --build --remote "$GORDON_REMOTE" --token "$GORDON_TOKEN" - --no-confirm + when: manual ``` @@ -137,12 +137,12 @@ deploy: - curl -fsSL https://github.com/bnema/gordon/releases/latest/download/gordon_linux_amd64 -o /usr/local/bin/gordon - chmod +x /usr/local/bin/gordon script: - - gordon push --build + - gordon images push --build --remote "$GORDON_REMOTE" --token "$GORDON_TOKEN" --build-arg NODE_ENV=production --build-arg API_URL=https://api.example.com - --no-confirm + rules: - if: $CI_COMMIT_TAG ``` @@ -166,7 +166,7 @@ deploy: ``` This requires additional CI/CD variables: `GORDON_REGISTRY` and `GORDON_USERNAME`. -Gordon auto-deploys when it receives the image. +Pushing only stores the image; deploy explicitly with `gordon apps deploy` afterwards. ## Version Detection @@ -208,4 +208,4 @@ GitLab CI clones your repository automatically. The build context defaults to th - [Generic CI](./generic-ci.md) - [Deployment Overview](./index.md) - [Authentication](../config/auth.md) -- [Push Command](../cli/push.md) +- [Images Commands](../cli/images.md#gordon-images-push) diff --git a/docs/deployment/index.md b/docs/deployment/index.md index e8644e0c6..ebb5ac6d4 100644 --- a/docs/deployment/index.md +++ b/docs/deployment/index.md @@ -1,78 +1,68 @@ # Deployment Overview -Gordon deploys containers when you push images to its built-in registry. Three deployment methods are available depending on your workflow and infrastructure. +Gordon deploys apps in two explicit steps: push the image to its built-in registry, then apply the app manifest and deploy. Push transfers OCI content only — it never deploys, creates routes, or modifies manifests. -## Recommended: gordon push +## Recommended: gordon images push + apps deploy -`gordon push --build --remote` is the simplest way to build, push, and deploy from CI/CD pipelines. +`gordon images push --build --remote` builds, pushes, and stores the image from CI/CD pipelines. Activation is a separate explicit step with the Gordon CLI. -- Single command: build + push + deploy - Single secret (`GORDON_TOKEN`): auto-exchanges for a short-lived registry token - Auto-detects version from CI environment (`$GITHUB_REF`, `$CI_COMMIT_TAG`, `$BUILD_SOURCEBRANCH`, or `git describe`) - Chunked uploads (50MB chunks) — works behind Cloudflare and restrictive proxies ```bash -gordon push --build --remote https://gordon.example.com --no-confirm +gordon images push --build --remote https://gordon.example.com +``` + +Then activate (from CI with the Gordon binary, or from your machine): + +```bash +gordon apps apply --file blog.toml --remote https://gordon.example.com +gordon apps deploy blog --remote https://gordon.example.com ``` ## All Deployment Methods | Method | Best For | Secrets Needed | Registry Access | Deploy Control | |--------|----------|----------------|-----------------|----------------| -| `gordon push` (Recommended) | CI/CD pipelines | 1 (`GORDON_TOKEN`) | Via gordon domain (HTTPS) | Explicit (CLI-triggered) | -| `docker push` | Simple setups, existing Docker workflows | 2 (`username` + `token`) | Via gordon domain (HTTPS) | Automatic (event-based) | -| Docker labels + auto-route | GitOps, zero-config deploys | 2 (`username` + `token`) | Via gordon domain (HTTPS) | Automatic (label-driven) | +| `gordon images push` + `apps deploy` (Recommended) | CI/CD pipelines | 1 (`GORDON_TOKEN`) | Via gordon domain (HTTPS) | Explicit (CLI-triggered) | +| `docker push` + `apps deploy` | Simple setups, existing Docker workflows | 2 (`username` + `token`) | Via gordon domain (HTTPS) | Explicit (CLI-triggered) | -### Method 1: gordon push (Recommended) +### Method 1: gordon images push + apps deploy (Recommended) -The Gordon CLI handles authentication, image building, and deployment in a single step. +The Gordon CLI handles authentication, image building, and registry upload in a single step. Deploy stays explicit. - Single token handles everything: admin API access + registry auth via automatic token exchange - Version tag auto-detected from `$GITHUB_REF`, `$CI_COMMIT_TAG`, `$BUILD_SOURCEBRANCH`, or `git describe` -- Use `--no-deploy` for push-only workflows (useful for staging images without triggering deployment) - Requires the Gordon binary on the CI runner ```bash -# Build, push, and deploy -gordon push --build --remote https://gordon.example.com --no-confirm +# Build and push (OCI transfer only) +gordon images push --build --remote https://gordon.example.com -# Push only, no deploy -gordon push --build --remote https://gordon.example.com --no-deploy --no-confirm +# Apply the manifest that references the pushed tag, then deploy +gordon apps apply --file blog.toml --remote https://gordon.example.com +gordon apps deploy blog --remote https://gordon.example.com ``` -### Method 2: docker push +### Method 2: docker push + apps deploy -Standard Docker workflow — no Gordon binary required on the runner. +Standard Docker workflow — no Gordon binary required on the runner for the push itself. - Use `docker login`, `docker build`, and `docker push` as usual -- Gordon auto-deploys when it receives the image (event-based, no explicit trigger needed) -- No Gordon binary needed on the CI runner +- Pushing only stores the image; deploy explicitly with the Gordon CLI afterwards - Registry endpoint is `gordon.example.com` (not a separate registry host) ```bash echo "$GORDON_TOKEN" | docker login -u ci-bot --password-stdin gordon.example.com docker build -t gordon.example.com/myapp:v1.2.0 . docker push gordon.example.com/myapp:v1.2.0 +# Then: gordon apps apply --file blog.toml + gordon apps deploy blog ``` -### Method 3: Docker labels + auto-route - -Add a `gordon.domain` label to your image and Gordon creates the route automatically on push. - -- Add `gordon.domain=app.example.com` label to your Dockerfile -- Push image — Gordon creates or updates the route without manual config -- Gated by the `auto_route_allowed_domains` allowlist in `gordon.toml` -- Best for GitOps workflows where routes are defined alongside the application - -```dockerfile -LABEL gordon.domain="app.example.com" -``` - -See [Auto-Route](../config/auto-route.md) for configuration details. - ## Registry Access -Gordon's registry is served through the main gordon domain over HTTPS on port 443. The internal registry port (5000) is never exposed externally. All push methods use `https://gordon.example.com/v2/...`. +Clients use the main Gordon domain over HTTPS, normally on port 443. Gordon binds `server.registry_port` to `127.0.0.1`; the public edge proxies authenticated registry requests to that loopback listener. Do not publish or forward the loopback registry port. With `auth.enabled=false`, registry access is loopback-only and the public edge refuses registry-domain requests. ### Network Topologies @@ -87,34 +77,27 @@ Gordon's registry is served through the main gordon domain over HTTPS on port 44 See the examples below for the right scopes for each workflow. ```bash -# Minimum scope for gordon push — route lookup + registry push -# Server auto-deploys when it receives the image +# Push only — registry scopes gordon auth token generate \ --subject ci-bot \ - --scopes "push,pull,admin:routes:read" \ + --scopes "push,pull" \ --expiry 90d -# With explicit CLI-managed deploy (adds config:write for deploy control) +# Push + apply/deploy — adds app mutation scopes gordon auth token generate \ --subject ci-bot \ - --scopes "push,pull,admin:routes:read,admin:config:write" \ + --scopes "push,pull,admin:apps:read,admin:apps:write" \ --expiry 90d -# Scoped to a specific repository +# Repository-scoped registry token for push only gordon auth token generate \ --subject ci-bot \ --repo myapp \ - --scopes "push,pull,admin:routes:read" \ - --expiry 90d - -# For docker push — registry scopes only -gordon auth token generate \ - --subject ci-bot \ --scopes "push,pull" \ --expiry 90d ``` -Set the generated token as `GORDON_TOKEN` in your CI environment. +Set the generated token as `GORDON_TOKEN` in your CI environment. `--repo` limits registry push/pull to that repository; it does not constrain `admin:*` scopes. Use a separate least-privilege token for app administration when repository and deployment duties must be isolated. ## Version Strategies @@ -127,26 +110,23 @@ docker tag myapp gordon.example.com/myapp:latest docker push gordon.example.com/myapp:latest ``` +Reference it from the app manifest: + ```toml -[routes] -"app.example.com" = "myapp:latest" +[services.web] +image = "gordon.example.com/myapp:latest" ``` ### Semantic Versioning -Pin routes to specific versions and update config to roll forward: +Pin services to specific versions and apply a new manifest to roll forward: ```bash docker tag myapp gordon.example.com/myapp:v2.1.0 docker push gordon.example.com/myapp:v2.1.0 ``` -```toml -[routes] -"app.example.com" = "myapp:v2.1.0" -``` - -To deploy a new version, update the tag in `gordon.toml` and push the new image. +To deploy a new version, update the tag in the app file, apply, and deploy. ### Git SHA Tags @@ -158,30 +138,22 @@ docker tag myapp gordon.example.com/myapp:$VERSION docker push gordon.example.com/myapp:$VERSION ``` -## Zero-Downtime Deployment - -Gordon performs zero-downtime deployments by default: +## Updates -1. **New container starts** while the old one is still running -2. **Health check** waits for the new container to become ready -3. **Traffic switches** to the new container -4. **Old container stops** after traffic moves +Gordon deploys an app's services one at a time in sorted order. A deploy may cause a short service interruption; there is no zero-downtime promise for app services, and Gordon never runs two Gordon-managed generations of the same service at the same time. ``` -Timeline ─────────────────────────────────────────> - -Old Container: [═══════════════════] - ↓ stop -New Container: [═════════════════════════> - ↑ start ↑ traffic routed +Preflight ─► withdraw traffic ─► stop + remove old ─► start new ─► readiness ─► publish ACTIVE ─► route traffic ``` +Preflight (image resolution and pull, secrets, volumes, networks, bind policy, reservations) completes before anything is disrupted, so a preflight failure leaves the running service untouched. If replacement fails after the old container was removed, the failure is explicit: Gordon does not recreate the old container during that operation and does not promise automatic data rollback. A stateful service's recovery inhibition stays in place until its never-published candidate is confirmed removed; a later boot/start/restart then clears it and rebuilds the generation `ACTIVE` records from its pinned digest, while a candidate that cannot be removed fails recovery closed. Deployment stops at the first service failure: services already deployed in that run are kept and are not rolled back. + +If a deploy is interrupted between creating the replacement container and publishing it, that candidate is recorded durably and Gordon removes it before creating or rebuilding any generation of that app — at boot and before every mutation — so two generations of one service never run together. An interrupted operation is finalized instead of staying in flight: as a failure, or as the success it was when every step had already completed. If the recorded candidate cannot be removed, reconciliation fails closed: the operation is not finalized, stays the latest journal, and every later mutation is refused so a newer operation can never mask the orphan. A failed final traffic publication is republished by the next recovery pass, within 15 seconds by default. + ## Related - [GitHub Actions](./github-actions.md) - [GitLab CI](./gitlab-ci.md) - [Generic CI](./generic-ci.md) -- [Rollback](./rollback.md) -- [Routes Configuration](../config/routes.md) +- [Apps CLI](../cli/apps.md) - [Authentication](../config/auth.md) -- [Auto-Route](../config/auto-route.md) diff --git a/docs/deployment/rollback.md b/docs/deployment/rollback.md index 8ab50636a..dad39bb04 100644 --- a/docs/deployment/rollback.md +++ b/docs/deployment/rollback.md @@ -1,62 +1,28 @@ # Rollback -Roll back to previous versions when deployments fail. +Roll forward to a previous version when a deploy misbehaves. There is no historical rollback command in v3: rolling back means applying and deploying a previous manifest or image reference. Redeploying a mutable tag (for example `latest`) does not guarantee the previous image bytes; keep immutable versioned tags. -## Recommended: Use `gordon pin` +## Roll Forward to a Previous Tag -`gordon pin` is the built-in rollback and roll-forward workflow for route images: +The image tags are still in the registry. Point the app file at the last good tag, apply, deploy: ```bash -gordon pin app.example.com -gordon pin app.example.com --tag v2.30.1 -gordon pin list app.example.com +# 1. Edit blog.toml: image = "gordon.mydomain.com/myapp:v2.0.0" +gordon apps apply --file blog.toml --remote prod +gordon apps deploy blog --remote prod ``` -Use it when the target image already exists in the Gordon registry and you want Gordon to update the route and redeploy it for you. +A rollback deploy replaces services one at a time like any other deploy: a short service interruption is possible, and Gordon never runs two generations of the same service at the same time. If the replacement fails after the old container was removed, Gordon does not recreate it and does not promise automatic data rollback. -## Rollback Strategies +## Re-push the Previous Image as latest -### 1. Config-Based Rollback - -Update the route to point to the previous version: - -```toml -# Current (broken) -[routes] -"app.mydomain.com" = "myapp:v2.1.0" - -# Rollback to previous -[routes] -"app.mydomain.com" = "myapp:v2.0.0" -``` - -Then reload: - -```bash -gordon reload -``` - -### 2. Push Previous Version - -Re-push the previous image with the `latest` tag: +When the app file tracks `latest`, re-tag and re-push, then deploy (deploy re-resolves mutable tags): ```bash -docker pull registry.mydomain.com/myapp:v2.0.0 -docker tag registry.mydomain.com/myapp:v2.0.0 registry.mydomain.com/myapp:latest -docker push registry.mydomain.com/myapp:latest -``` - -### 3. Manifest-Based Rollback - -Use OCI manifest annotations for version control: - -```bash -# Rollback to v2.0.0 -export VERSION=v2.0.0 -podman manifest create myapp:latest --amend -podman manifest add myapp:latest registry.mydomain.com/myapp:$VERSION -podman manifest annotate myapp:latest --annotation version=$VERSION registry.mydomain.com/myapp:$VERSION -podman manifest push myapp:latest registry.mydomain.com/myapp:latest +docker pull gordon.mydomain.com/myapp:v2.0.0 +docker tag gordon.mydomain.com/myapp:v2.0.0 gordon.mydomain.com/myapp:latest +docker push gordon.mydomain.com/myapp:latest +gordon apps deploy blog --remote prod ``` ## Version Management @@ -67,169 +33,56 @@ Always push versioned tags alongside `latest`: ```bash VERSION=$(git describe --tags) -docker tag myapp registry.mydomain.com/myapp:$VERSION -docker tag myapp registry.mydomain.com/myapp:latest -docker push registry.mydomain.com/myapp:$VERSION -docker push registry.mydomain.com/myapp:latest +docker tag myapp gordon.mydomain.com/myapp:$VERSION +docker tag myapp gordon.mydomain.com/myapp:latest +docker push gordon.mydomain.com/myapp:$VERSION +docker push gordon.mydomain.com/myapp:latest ``` ### Semantic Versioning -Use semantic versions for clear rollback targets: +Use semantic versions for clear recovery targets: ``` v2.1.0 ← Current (broken) -v2.0.0 ← Rollback target +v2.0.0 ← Recovery target v1.9.0 ← Older stable ``` ### Git SHA Tags -Tag with commit SHA for precise rollbacks: +Tag with commit SHA for precise recovery: ```bash # Deploy SHA=$(git rev-parse --short HEAD) -docker push registry.mydomain.com/myapp:$SHA - -# Rollback to specific commit -docker pull registry.mydomain.com/myapp:abc1234 -docker tag registry.mydomain.com/myapp:abc1234 registry.mydomain.com/myapp:latest -docker push registry.mydomain.com/myapp:latest -``` - -## Rollback Workflow - -### Quick Rollback - -```bash -# 1. Identify last working version -docker image ls registry.mydomain.com/myapp - -# 2. Tag as latest -docker tag registry.mydomain.com/myapp:v2.0.0 registry.mydomain.com/myapp:latest - -# 3. Push -docker push registry.mydomain.com/myapp:latest -``` - -### Config Rollback - -```bash -# 1. Edit config -vim ~/.config/gordon/gordon.toml - -# 2. Change version -# "app.mydomain.com" = "myapp:v2.0.0" - -# 3. Reload -gordon reload -``` - -## Automated Rollback - -### GitHub Actions - -Add rollback capability to your workflow: - -```yaml -name: Rollback - -on: - workflow_dispatch: - inputs: - version: - description: 'Version to rollback to (e.g., v2.0.0)' - required: true - -jobs: - rollback: - runs-on: ubuntu-latest - steps: - - name: Login to Registry - env: - GORDON_TOKEN: ${{ secrets.GORDON_TOKEN }} - GORDON_USERNAME: ${{ secrets.GORDON_USERNAME }} - GORDON_REGISTRY: ${{ secrets.GORDON_REGISTRY }} - run: printf '%s' "$GORDON_TOKEN" | docker login -u "$GORDON_USERNAME" --password-stdin "$GORDON_REGISTRY" - - - name: Rollback - env: - GORDON_REGISTRY: ${{ secrets.GORDON_REGISTRY }} - VERSION: ${{ inputs.version }} - run: | - [[ "$VERSION" =~ ^[A-Za-z0-9][A-Za-z0-9._-]{0,127}$ ]] || exit 1 - REPOSITORY="$GORDON_REGISTRY/myapp" - SOURCE_IMAGE="$REPOSITORY:$VERSION" - LATEST_IMAGE="$REPOSITORY:latest" - SOURCE_DIGEST="$(docker buildx imagetools inspect "$SOURCE_IMAGE" --format '{{json .Manifest}}' | jq -er '.digest')" - IMMUTABLE_IMAGE="$REPOSITORY@$SOURCE_DIGEST" - docker buildx imagetools create --tag "$LATEST_IMAGE" "$IMMUTABLE_IMAGE" - LATEST_DIGEST="$(docker buildx imagetools inspect "$LATEST_IMAGE" --format '{{json .Manifest}}' | jq -er '.digest')" - [[ "$LATEST_DIGEST" == "$SOURCE_DIGEST" ]] || { - echo "rollback verification failed: latest resolved to $LATEST_DIGEST, expected $SOURCE_DIGEST" >&2 - exit 1 - } - - - name: Summary - env: - VERSION: ${{ inputs.version }} - run: | - echo "## Rollback Complete" >> "$GITHUB_STEP_SUMMARY" - printf 'Rolled back to version: %s\n' "$VERSION" >> "$GITHUB_STEP_SUMMARY" -``` - -### Rollback Script - -Create a local rollback script: - -```bash -#!/bin/bash -# rollback.sh - -REGISTRY="registry.mydomain.com" -IMAGE="myapp" -VERSION=$1 - -if [ -z "$VERSION" ]; then - echo "Usage: ./rollback.sh " - echo "Available versions:" - docker image ls "$REGISTRY/$IMAGE" --format "{{.Tag}}" - exit 1 -fi - -echo "Rolling back to $IMAGE:$VERSION..." -docker pull "$REGISTRY/$IMAGE:$VERSION" -docker tag "$REGISTRY/$IMAGE:$VERSION" "$REGISTRY/$IMAGE:latest" -docker push "$REGISTRY/$IMAGE:latest" -echo "Rollback complete!" +docker push gordon.mydomain.com/myapp:$SHA ``` -## Verifying Rollback +## Verifying Recovery -After rollback: +After re-deploying the previous version: ```bash -# Check container is running correct version -docker inspect gordon-app-mydomain-com | grep Image +# Check the app's effective vs observed state +gordon apps status blog --remote prod # Check application responds curl -I https://app.mydomain.com # Check logs -gordon logs -f +gordon apps logs blog --remote prod ``` ## Best Practices 1. **Always tag versions** - Don't rely solely on `latest` -2. **Keep N previous versions** - Maintain rollback options +2. **Keep N previous versions** - Maintain recovery options 3. **Test before deploy** - Reduce need for rollbacks 4. **Document known-good versions** - Track stable releases -5. **Automate rollback** - Reduce time to recovery ## Related - [Deployment Overview](./index.md) - [GitHub Actions](./github-actions.md) -- [Routes Configuration](../config/routes.md) +- [Apps CLI](../cli/apps.md) diff --git a/docs/getting-started.md b/docs/getting-started.md index ef667701c..8b382f683 100644 --- a/docs/getting-started.md +++ b/docs/getting-started.md @@ -12,7 +12,7 @@ Deploy your first app with Gordon in under 5 minutes. ## 1. Install Gordon ```bash -curl -fsSL https://gordon.bnema.dev/install | bash +curl -fsSL https://bnema.dev/gordon/install | bash ``` This script automatically detects your OS and architecture, downloads the appropriate binary, and installs it to `/usr/local/bin`. @@ -79,11 +79,10 @@ gordon_domain = "gordon.mydomain.com" # Registry + Admin API domain [entrypoints.edge] address = ":443" # Public smart TCP edge (choose your bind/mapping) protocol = "smart_tcp" - -[routes] -"app.mydomain.com" = "myapp:latest" # Domain → Image mapping ``` +Application workloads are NOT declared in `gordon.toml`. Each app lives in its own standalone TOML file (see step 8). The old `[routes]`, `[attachments]`, `[network_groups]`, `[service_routes]`, `[auto_route]`, and `[previews]` keys were removed in v3 — Gordon refuses to start when any of them is present. Installation-level `[[services]]` for standalone L4 workloads remains valid. + ## 5. Set Up DNS (Including Wildcard) `gordon_domain` is the single domain used by both the registry and admin API. @@ -96,8 +95,8 @@ In Cloudflare (or your DNS provider), create: | A/CNAME | `*` | `YOUR_SERVER_IP` or `gordon.mydomain.com` | Yes | Why wildcard (`*`)? -- It automatically covers app routes like `app.mydomain.com`, `api.mydomain.com`, `demo.mydomain.com`, etc. -- You can add new domains in `[routes]` without creating DNS records one by one. +- It automatically covers app hosts like `app.mydomain.com`, `api.mydomain.com`, `demo.mydomain.com`, etc. +- You can add new app HTTP hosts without creating DNS records one by one. If your DNS provider supports wildcard CNAME flattening (Cloudflare does), `* -> gordon.mydomain.com` is usually the cleanest option. @@ -129,10 +128,10 @@ sudo loginctl enable-linger $USER ## 7. Generate a Deploy Token -Create a token for remote CLI deploys (skip if auth is disabled): +Create a token for remote CLI use (skip if auth is disabled): ```bash -gordon auth token generate --subject deploy --scopes push,pull --expiry 0 +gordon auth token generate --subject deploy --scopes "push,pull,admin:apps:read,admin:apps:write" --expiry 90d ``` `--expiry 0` creates a non-expiring token. Prefer a finite expiry and a rotation policy unless you explicitly need a long-lived deploy token. @@ -148,53 +147,65 @@ On your local machine: gordon remotes add prod https://gordon.mydomain.com --token gordon remotes use prod -# Recommended first-time setup -gordon bootstrap app.example.com myapp:latest --attachment postgres:18 --env APP_ENV=production - -# Then build, push, and deploy -gordon push myapp:latest --domain app.example.com --build --no-confirm +# Build and push the image (OCI transfer only, never deploys) +gordon images push myapp --build --remote prod ``` -What this command does: +Write the app file (`blog.toml`). The file is intended for Git: it must never contain secret values. -- `gordon bootstrap` creates or updates the route, applies requested attachments, - and stores environment variables. -- This is the recommended path for first deploys because it does not require the - route to exist ahead of time. -- Run `gordon push` separately after bootstrap to build, upload, and deploy the image. +```toml +name = "blog" + +[services.web] +image = "gordon.mydomain.com/myapp:latest" + +[[services.web.http]] +host = "app.mydomain.com" +port = 3000 +``` -If the route already exists and you only need to push a new image version, use: +Apply the manifest, then deploy: ```bash -gordon push myapp --build --no-confirm +gordon apps apply --file blog.toml --remote prod +gordon apps deploy blog --remote prod ``` -`gordon push` still requires the route to already exist so it can resolve the -deploy target. +What these commands do: + +- `gordon images push` builds, uploads, and stores the image. It never deploys. +- `gordon apps apply` validates the manifest and persists it as desired state. +- `gordon apps deploy` activates the accepted revision: runs preflight, withdraws the service from traffic, stops and removes the old container, starts the new container, waits for readiness, publishes the new `ACTIVE` state, and routes traffic to it. + +If the app needs secrets, register their names in the manifest (`[services..secrets]` maps ENV name to secret name), then write values — values stay in pass, never in the file: + +```bash +gordon apps secrets set blog --service web APP_ENV=production --remote prod +``` Your app is now live at `https://app.mydomain.com`! ## 9. Update Your App -Push a new image to deploy with zero downtime: +Push a new image, update the manifest tag if needed, apply, deploy: ```bash -# Make changes, then build + push + deploy -gordon push myapp --build --no-confirm +# Make changes, then build + push +gordon images push myapp --build --remote prod + +# Deploy the new tag (re-resolves mutable tags) +gordon apps deploy blog --remote prod ``` -Gordon automatically: -1. Starts the new container -2. Waits for it to be ready -3. Routes traffic to the new container -4. Stops the old container +Gordon deploys services one at a time in sorted order. A deploy may cause a short service interruption, and Gordon never runs two Gordon-managed generations of the same service at the same time. If preflight fails, the running service is left untouched. This applies to every service, including TCP, UDP, and volume-owning services. ## Next Steps - [Installation Guide](./installation.md) - Production setup with firewall and rootless containers -- [Configuration Reference](./config/index.md) - All configuration options +- [Configuration Reference](./config/index.md) - All installation configuration options +- [Apps CLI](./cli/apps.md) - Apply, deploy, and lifecycle commands - [Authentication](./config/auth.md) - Secure your registry -- [Environment Variables](./config/env.md) - Configure per-app settings +- [App secrets](./cli/apps.md#gordon-apps-secrets) - Service-scoped secret values in pass ## Related diff --git a/docs/index.md b/docs/index.md index 24c49edc6..b5743d19b 100644 --- a/docs/index.md +++ b/docs/index.md @@ -1,15 +1,16 @@ # Gordon Documentation -Gordon is a self-hosted container deployment platform that combines a private Docker registry with automatic container deployment. +Gordon is a self-hosted container deployment platform: a private container registry, a declarative app runtime, and a reverse proxy. ## What is Gordon? Gordon runs on your VPS and provides: - **Private Container Registry** - Push images from your local machine or CI -- **HTTP Reverse Proxy** - Routes domains to containers automatically -- **Push-to-Deploy** - Containers deploy when you push new images -- **Zero-Downtime Updates** - New containers start before old ones stop +- **Declarative Apps** - One TOML file per app, explicit apply then deploy +- **HTTP Reverse Proxy** - Routes app hosts to containers +- **Push, Apply, Deploy** - Push stores images; deploy is always explicit +- **Sequential Service Replacement** - One service generation at a time: withdraw traffic, replace, verify readiness, republish - **Single Binary** - ~15MB RAM footprint ## How It Works @@ -31,7 +32,7 @@ Gordon runs on your VPS and provides: 1. Build your container locally where you have computing power 2. Push to your Gordon registry -3. Gordon automatically deploys and routes traffic to your container +3. Apply the app manifest and deploy to activate it ## Quick Navigation @@ -46,15 +47,13 @@ Gordon runs on your VPS and provides: - [Configuration Overview](./config/index.md) - All configuration options - [Server Settings](./config/server.md) - Ports, domains, and runtime -- [Routes](./config/routes.md) - Domain to container mapping +- [App Manifest](./config/apps.md) - Declarative app files (services, hosts, secrets, volumes) +- [Migrate to Gordon v3](./migrate-to-v3.md) - Breaking upgrade and explicit secret migration - [Traffic Plane](./config/traffic.md) - TCP, UDP, and TLS passthrough entrypoints - [Authentication](./config/auth.md) - Registry auth plus remote CLI login/token workflows -- [Secrets](./config/secrets.md) - Secure credential storage -- [Network Isolation](./config/network-isolation.md) - Per-app network isolation -- [Preview Environments](./config/preview.md) - Ephemeral per-branch deployments -- [Attachments](./config/attachments.md) - Service dependencies +- [Secrets](./config/secrets.md) - Installation secrets and app secret values +- [Network Isolation](./config/network-isolation.md) - Installation network policy - [Logging](./config/logging.md) - Log collection and rotation -- [Environment Variables](./config/env.md) - Per-route environment configuration ### CLI Reference @@ -71,7 +70,6 @@ Gordon runs on your VPS and provides: ### Reference - [Docker Labels](./reference/docker-labels.md) - Container and image labels -- [Environment Variables](./reference/env-variables.md) - Environment variable syntax - [Troubleshooting](./reference/troubleshooting.md) - Common issues and solutions ## Requirements diff --git a/docs/installation.md b/docs/installation.md index be11a38dc..3f56fea57 100644 --- a/docs/installation.md +++ b/docs/installation.md @@ -19,7 +19,7 @@ Detailed installation guide for production environments. set -euo pipefail installer="$(mktemp)" trap 'rm -f "$installer"' EXIT - curl -fsSL --output "$installer" https://gordon.bnema.dev/install + curl -fsSL --output "$installer" https://bnema.dev/gordon/install if command -v less >/dev/null 2>&1; then less "$installer" else @@ -29,7 +29,25 @@ Detailed installation guide for production environments. ) ``` -Download and inspect the installer before executing it. The installer verifies the release archive checksum before installation and automatically detects your OS (Linux/macOS) and architecture (amd64/arm64), downloads the appropriate binary from GitHub releases, and installs it to `/usr/local/bin`. +Download and inspect the installer before executing it. The installer verifies the release archive checksum, detects Linux/macOS and amd64/arm64, and installs Gordon to `~/.local/bin` without `sudo`. If that directory is absent from `PATH`, an interactive install can offer to add an idempotent marker block to Fish (`~/.config/fish/config.fish`), Bash (`~/.bashrc`), or Zsh (`~/.zshrc`). Restart the shell after accepting, or follow the printed command to update the current shell. + +For unattended installs, control PATH configuration explicitly: + +```bash +# Update supported shell configuration without prompting +curl -fsSL https://bnema.dev/gordon/install | GORDON_UPDATE_PATH=1 sh + +# Never modify shell configuration +curl -fsSL https://bnema.dev/gordon/install | GORDON_UPDATE_PATH=0 sh + +# Use another user-local directory +curl -fsSL https://bnema.dev/gordon/install | GORDON_INSTALL_DIR="$HOME/bin" GORDON_UPDATE_PATH=1 sh + +# Explicit global installation (may request sudo) +curl -fsSL https://bnema.dev/gordon/install | GORDON_INSTALL_DIR=/usr/local/bin GORDON_UPDATE_PATH=0 sh +``` + +`GORDON_INSTALL_DIR` accepts arbitrary safe absolute destinations, including `"$HOME/bin"`, `"$HOME/.local/bin"`, and `/usr/local/bin`. PATH detection and shell configuration use that effective directory. Relative paths, PATH separators, and control characters are rejected. `GORDON_UPDATE_PATH` accepts only `0` or `1`. Do not run the default installer through `sudo`: it refuses to infer a user home and install silently under `/root`. ### Manual Installation @@ -136,6 +154,16 @@ insecure = true EOF ``` +Podman assigns loopback backend ports from the host's ephemeral port range when Gordon publishes managed service ports. If an nftables output policy restricts loopback access, allow the Gordon daemon's primary UID—not its subordinate container UIDs—to connect to that range for the protocols used by managed services: + +```nftables +# Confirm the range with: sysctl net.ipv4.ip_local_port_range +meta skuid ip daddr 127.0.0.1 tcp dport 32768-60999 accept +meta skuid ip daddr 127.0.0.1 udp dport 32768-60999 accept +``` + +Place these rules before any broader loopback rejection. Keep public entrypoint and administrative-port policy separate; these rules only let Gordon reach rootless Podman backends on the local host. + ## Firewall Configuration Gordon needs ports accessible for the registry and the public edge entrypoint. @@ -195,11 +223,10 @@ gordon_domain = "gordon.yourdomain.com" [entrypoints.edge] address = ":443" protocol = "smart_tcp" - -[routes] -"app.yourdomain.com" = "myapp:latest" ``` +Application workloads live in standalone app files, not in `gordon.toml` (see [Getting Started](./getting-started.md#8-deploy-your-first-app)). + See [Configuration Reference](./config/index.md) for all options. ## Systemd Service @@ -310,17 +337,20 @@ Without this setting, Cloudflare traffic receives `403 Forbidden: Only certifica > **Note:** This is separate from `[api.rate_limit] trusted_proxies`, which controls IP extraction from `X-Forwarded-For`. Both should list your proxy IPs. See [Proxy Origin IP Allowlist](./config/server.md#proxy-origin-ip-allowlist) for details. -## Installing a Specific Version or Pre-Release +## Choosing an Install Channel ```bash -# Install an exact version -curl -fsSL https://gordon.bnema.dev/install | GORDON_VERSION=v2.30.1 bash +# Install an exact release +curl -fsSL https://bnema.dev/gordon/install | GORDON_VERSION=v2.30.1 sh + +# Install the latest pre-release +curl -fsSL https://bnema.dev/gordon/install | GORDON_PRERELEASE=1 sh -# Install the latest pre-release instead of the latest stable release -curl -fsSL https://gordon.bnema.dev/install | GORDON_PRERELEASE=1 bash +# Build the current next branch commit from source +curl -fsSL https://bnema.dev/gordon/install | GORDON_CHANNEL=next sh ``` -By default, the installer resolves `latest` to the newest stable release and verifies the downloaded checksum before installing. +Stable, exact-version, and pre-release installs download release binaries and verify their published checksums. The `next` channel is an **unverified development source build**: it resolves the branch through the GitHub API, pins the resulting commit SHA, downloads that exact source snapshot, and builds it locally for the detected platform. It requires the Go version declared by that commit's `go.mod`, is not covered by release checksums, and may be unstable. Do not combine `GORDON_CHANNEL=next` with `GORDON_VERSION` or `GORDON_PRERELEASE`. ## Verify Installation @@ -347,7 +377,6 @@ Gordon stores data in the following locations: | `~/.config/gordon/gordon.toml` | Configuration file | | `~/.gordon/` | Default data directory | | `~/.gordon/registry/` | Container images | -| `~/.gordon/env/` | Environment files | | `~/.gordon/logs/` | Application logs | | `~/.gordon/secrets/` | Secrets (unsafe backend only) | diff --git a/docs/migrate-to-v3.md b/docs/migrate-to-v3.md new file mode 100644 index 000000000..51d6faf3a --- /dev/null +++ b/docs/migrate-to-v3.md @@ -0,0 +1,146 @@ +# Migrate to Gordon v3 + +Gordon v3 replaces domain-based route workloads with declarative apps. The upgrade is intentionally explicit: Gordon does not infer app ownership from domains, does not adopt existing volumes, and does not copy domain secrets into app secrets automatically. + +Use this guide before starting the v3 daemon with production traffic. + +## Before the upgrade + +1. Stop deployments and automatic image pruning. +2. Back up Gordon's configuration, state, registry, databases, volumes, and secret store using your existing procedures. +3. Record every current domain, image tag or digest, attachment, network relationship, volume, and secret key. +4. Keep the previous signed Gordon binary and its matching configuration and state backup available for rollback. + +Do not delete old containers, volumes, pass entries, or registry tags during the migration. Gordon's ownership-aware reconciliation and prune paths preserve resources whose ownership cannot be proven. Runtime-native cleanup commands and manual deletion do not provide that guarantee. + +## 1. Update the installation configuration + +Remove workload declarations that v3 no longer accepts from `gordon.toml`: + +- `[routes]`; +- `[attachments]`; +- `[network_groups]`; +- app-workload entries formerly declared as `[[services]]`, and `[service_routes]` (installation-level `[[services]]` used for standalone L4 workloads stays valid); +- `[auto_route]` and `[auto_route_allowed_domains]`; +- `[previews]`. + +Keep installation-level settings such as entrypoints, TLS, limits, registry policy, external routes, backup destinations, logging, and authentication. + +Review every app image registry before applying manifests. v3 allows Docker Hub (`docker.io`, including its canonical pull host `registry-1.docker.io`), `ghcr.io`, `quay.io`, and Gordon's configured registry by default. Add each other registry, including private registries, as an exact hostname and optional non-default port: + +```toml +[images] +allowed_registries = ["registry.internal:5000"] +``` + +Hostname matching is case-insensitive, ignores one trailing dot, and treats port `443` as the default. The allowlist is checked during manifest apply, deployment preflight, and immediately before pull. It controls registry names only: it does not prove that DNS resolves to a public IP or enforce runtime egress. Use firewall or runtime network policy when destination-level restrictions are required. + +Create one app manifest per workload. See [App Manifest](./config/apps.md). + +```toml +name = "example-app" + +[services.web] +image = "registry.example.com/example-app:1.2.3" + +[services.web.secrets] +DATABASE_URL = "database-url" + +[[services.web.http]] +host = "app.example.com" +port = 8080 +``` + +The manifest contains secret **names**, never secret values. + +Manifests written for an early v3 alpha used `[[service]]` with a `name` field; that shape is rejected, not converted. Rewrite each service as a keyed `[services.]` table with `[services..*]` children — see [Migrating a v3 alpha app manifest](./upgrading.md#migrating-a-v3-alpha-app-manifest). + +## 2. Migrate domain secrets + +Gordon does not convert domain-scoped secrets into app secrets, and v3 has no command to list them. With the `pass` backend, inventory the old key names with `pass ls gordon/env`, declare them under `[services..secrets]`, then apply the manifest before setting values: + +```bash +gordon apps apply --file ./example-app.toml --remote production +printf 'DATABASE_URL=%s\n' "$(pass show gordon/env/app_example_com/DATABASE_URL)" \ + | gordon apps secrets set example-app --service web --stdin --remote production +``` + +Pass values through standard input. Do not put them in manifests, command arguments, logs, shell history, or temporary files. Repeat explicitly when several services need the same value. + +Deploy and restart the app to verify that it uses the new app-scoped entries. Keep the old `gordon/env/...` entries throughout the rollback window; Gordon never uses them as a fallback. Refer to the `pass` documentation for password-store and GPG backup or restore procedures. + +## 3. Transfer existing volume data + +Gordon's reconciliation and prune paths leave existing unowned volumes untouched, but Gordon does not adopt them. After applying the manifest, let Gordon create the new app-owned volumes, then transfer the data with your container runtime's tools while the workload is stopped. Do not run runtime-wide volume prune, edit `state.db`, rename volumes, or alter ownership labels. + +Use a database-native backup and restore for databases instead of copying live database files. Keep the old volumes unchanged until the new app has passed data, deploy, restart, and rollback-readiness checks. Refer to the Docker or Podman documentation for runtime-specific copy and inspection commands. + +## 4. Apply and deploy apps + +For every manifest: + +```bash +gordon apps apply --file ./example-app.toml --remote production +gordon apps diff example-app --remote production +gordon apps deploy example-app --remote production +gordon apps status example-app --json --remote production +``` + +Push only uploads OCI content in v3; it does not deploy. Apply and deploy explicitly. + +## 5. Replace tokens and automation + +Replace removed route scopes with app scopes: + +- `admin:apps:read` for list, show, diff, and status; +- `admin:apps:write` for apply, deploy, lifecycle, and app secrets. + +Generate replacement CI and operator tokens with the minimum required scopes. Update automation that expected push-to-deploy, route mutation commands, previews, attachments, bootstrap, autoroute, pin, or rollback commands. + +To roll back an application release in v3, apply a manifest containing the previous image tag and deploy it. + +## 6. Validate before opening traffic + +Verify all of the following: + +- each app is converged on the expected pinned digest; +- HTTP, TCP, and UDP entrypoints work as applicable; +- app secrets survive a restart and no value appears in output or logs; +- named volumes contain the expected data after deploy and restart; +- stopped intent survives a daemon restart; +- retained and unknown resources remain untouched; +- `gordon images prune --dry-run` reports protected app content correctly before any real prune; +- the previous binary and matching data backup remain available. + +Run destructive prune commands only after their dry-run candidate list matches the expected ownership and retention policy. + +## Security guarantees to rely on after the upgrade + +- **Containment first.** Keep `auth.enabled = true` and automatic registry prune disabled (`images.prune.enabled = false`) until the upgraded build is running and validated. A disabled-auth registry is local-only: public ingress cannot reach it through the proxy. +- **Network isolation.** Every app container joins an app-owned private network; shared memberships come only from explicit manifest declarations, and container memory/CPU/PID limits are applied on create and recovery. +- **Registry content is repository-scoped.** Pulls require the repository to have completed the upload of a blob, and manifests may only reference content that repository owns. Content uploaded by an older build has no durable ownership marker, so a repository whose blobs predate this build must re-push its images before they can be pulled again. +- **Runtime image prune needs released ownership.** An app's pinned images are recorded as attached while deployed and released when the app is removed. Dangling images with no durable ownership record are never deleted, and image labels alone never authorize deletion. +- **Durable roots keep their closure.** A digest pinned by active/desired state, an apply intent, an operation, or ownership retains its manifest, config, and layers even if the tag moves away. +- **Plaintext TLS policy.** Hosts declared `tls = always` never receive plaintext traffic: they redirect when an HTTPS endpoint exists and are refused with `421` when none does. +- **Failure output is log-free.** Mutation responses and operation lookup for `admin:apps:read` carry stable errors only; application diagnostics require `admin:logs:read` and are redacted before storage. + +## Rollback + +A binary downgrade against v3 app state is unsupported. To roll back: + +1. stop the v3 daemon; +2. restore the previous signed binary; +3. restore its matching configuration and state backup as one set; +4. restore secret-store entries if any were removed; +5. restart with automatic pruning disabled; +6. verify workloads and data before reopening traffic. + +Do not attempt rollback by adopting unknown containers or volumes into v3 state. + +## Related + +- [Upgrading Gordon](./upgrading.md) +- [App Manifest](./config/apps.md) +- [Apps CLI](./cli/apps.md) +- [Secrets](./config/secrets.md) +- [Authentication](./config/auth.md) diff --git a/docs/reference/docker-labels.md b/docs/reference/docker-labels.md index e66db83d3..54feba251 100644 --- a/docs/reference/docker-labels.md +++ b/docs/reference/docker-labels.md @@ -4,122 +4,57 @@ Labels used by Gordon for container and image metadata. ## Container Labels -Gordon adds these labels to managed containers: +Gordon stamps these ownership labels on every container it creates: | Label | Value | Description | |-------|-------|-------------| | `gordon.managed` | `"true"` | Identifies Gordon-managed containers | -| `gordon.domain` | Domain name | Domain this container serves | -| `gordon.image` | Image:tag | Original image from configuration | -| `gordon.route` | Domain name | Route this container handles | +| `gordon.app` | App name | App this container serves | +| `gordon.app.service` | Service name | Service this container runs | +| `gordon.app.revision` | Revision | Active revision that created it | | `gordon.created` | Timestamp | When Gordon created the container | -### Propagated Image Labels - -When an image defines these labels, Gordon copies them onto the container for runtime use (readiness probing, proxy routing): - -| Label | Example | Description | -|-------|---------|-------------| -| `gordon.proxy.port` | `"3000"` | Port the proxy routes HTTP traffic to | -| `gordon.health` | `"/healthz"` | Health check endpoint for readiness probing | - -These labels only appear on containers whose source image defined them. +Queries by logical identity use labels, never name parsing. Resources without these labels (old or foreign containers) are preserved, never adopted or deleted. -### Attachment Labels +### Legacy Labels -Additional labels for attachment containers: +Containers created before v3 may carry these labels. They are read-only provenance hints for prune guards — Gordon never infers app state from them: | Label | Value | Description | |-------|-------|-------------| -| `gordon.attachment` | `"true"` | Container is an attachment service | -| `gordon.attached-to` | Domain/group | Route or network group this serves | - -### Backup Labels +| `gordon.domain` | Domain name | Pre-v3 domain this container served | +| `gordon.image` | Image:tag | Original image from configuration | +| `gordon.route` | Domain name | Pre-v3 route this container handled | -Labels used by the backup subsystem: +### Backup Metadata -| Label | Value | Description | -|-------|-------|-------------| -| `gordon.backup` | `"true"` / `"false"` | Enables or disables backup behavior for a container | -| `gordon.backup.type` | e.g. `"postgresql"` | Explicit database type override | -| `gordon.backup.version` | e.g. `"17"` | Explicit database version override | -| `gordon.backup.schedule` | e.g. `"hourly,daily"` | Schedule override hint | -| `gordon.backup.sidecar` | `"true"` | Identifies backup sidecar containers | +Backups are declarative rather than label-driven. App manifests declare database and volume targets in `[services..backup]`; Gordon does not define or write Docker labels to enable backups, select database types or schedules, or identify backup sidecars. ## Image Labels -Labels you can set in your Dockerfile: - -| Label | Example | Description | -|-------|---------|-------------| -| `gordon.domains` | `"app.example.com,www.app.example.com"` | Comma-separated domains for auto-route | -| `gordon.proxy.port` | `"3000"` | Port to proxy HTTP traffic to | -| `gordon.health` | `"/healthz"` | HTTP health check endpoint path for readiness probing | -| `gordon.env-file` | `"/app/.env.example"` | Path to env template file inside the image | - -### Health Check Label - -When `gordon.health` is set, Gordon performs HTTP GET requests to the specified -path during deployment and waits for a 2xx or 3xx response before routing traffic -to the new container: - -```dockerfile -FROM node:20-alpine -LABEL gordon.health="/api/health" -EXPOSE 3000 -CMD ["node", "server.js"] -``` - -Gordon probes `http://:3000/api/health` until it gets a successful -response or the `deploy.http_probe_timeout` is reached. - -### Proxy Port Label - -When an image exposes multiple ports, Gordon needs to know which one serves HTTP: - -```dockerfile -# Gitea exposes SSH (22) and HTTP (3000) -FROM gitea/gitea:latest -LABEL gordon.proxy.port=3000 -EXPOSE 22 -EXPOSE 3000 -``` - -Without this label, Gordon uses the first exposed port. - -**When to use:** -- Image exposes multiple ports -- First exposed port isn't the HTTP service -- You want explicit control over routing +No image-label inference exists in v3: Gordon never creates routes, deploys, or resolves image names from Dockerfile labels. Push with a domain-like name or a `gordon.domain` label is treated as an ordinary image name. Readiness and proxy ports come from the app manifest (`[services..readiness]`, `[[services..http]]`), not from image labels. ## Container Naming -Gordon names containers: - -| Pattern | Example | -|---------|---------| -| `gordon-{domain}` | `gordon-app-mydomain-com` | -| `gordon-{domain}-new` | `gordon-app-mydomain-com-new` (during updates) | -| `gordon-{domain}-{service}` | `gordon-app-mydomain-com-postgres` (attachments) | +Gordon names containers `gordon-----` where `` is the creating operation ID. Retiring containers keep their instance names until removal. ## Inspecting Labels View labels on a container: ```bash -docker inspect gordon-app-mydomain-com --format '{{json .Config.Labels}}' | jq +docker inspect --format '{{json .Config.Labels}}' | jq ``` Example output: ```json { - "gordon.created": "2024-01-15T10:30:00Z", - "gordon.domain": "app.mydomain.com", - "gordon.image": "myapp:latest", "gordon.managed": "true", - "gordon.proxy.port": "3000", - "gordon.route": "app.mydomain.com" + "gordon.app": "blog", + "gordon.app.service": "web", + "gordon.app.revision": "rev-3", + "gordon.created": "2024-01-15T10:30:00Z" } ``` @@ -131,75 +66,12 @@ Find Gordon-managed containers: # All Gordon containers docker ps -f "label=gordon.managed=true" -# Containers for specific domain -docker ps -f "label=gordon.domain=app.mydomain.com" - -# Attachment containers -docker ps -f "label=gordon.attachment=true" - -# Attachments for specific route -docker ps -f "label=gordon.attached-to=app.mydomain.com" - -# Backup sidecars -docker ps -f "label=gordon.backup.sidecar=true" -``` - -## Examples - -### Standard Application Container - -```bash -docker inspect gordon-app-mydomain-com --format '{{json .Config.Labels}}' -``` - -```json -{ - "gordon.managed": "true", - "gordon.domain": "app.mydomain.com", - "gordon.image": "myapp:v2.1.0", - "gordon.route": "app.mydomain.com", - "gordon.created": "2024-01-15T10:30:00Z" -} -``` - -### Attachment Container - -```bash -docker inspect gordon-app-mydomain-com-postgres --format '{{json .Config.Labels}}' +# Containers for a specific app +docker ps -f "label=gordon.app=blog" ``` -```json -{ - "gordon.managed": "true", - "gordon.attachment": "true", - "gordon.attached-to": "app.mydomain.com", - "gordon.created": "2024-01-15T10:29:00Z" -} -``` - -### Multi-Port Dockerfile - -```dockerfile -FROM node:18 -LABEL gordon.proxy.port=3000 - -WORKDIR /app -COPY . . - -# HTTP server on 3000 -EXPOSE 3000 -# Metrics on 9090 -EXPOSE 9090 -# WebSocket on 8080 -EXPOSE 8080 - -CMD ["npm", "start"] -``` - -Gordon routes HTTP to port 3000. - ## Related - [Configuration Overview](../config/index.md) -- [Attachments](../config/attachments.md) +- [App Manifest](../config/apps.md) - [Concepts](../concepts.md) diff --git a/docs/reference/env-variables.md b/docs/reference/env-variables.md deleted file mode 100644 index 8f1ae370d..000000000 --- a/docs/reference/env-variables.md +++ /dev/null @@ -1,222 +0,0 @@ -# Environment Variables Reference - -How Gordon handles environment variables for containers. - -## Variable Sources - -Environment variables are loaded from multiple sources and merged: - -| Source | Priority | Description | -|--------|----------|-------------| -| Dockerfile `ENV` | Lowest | Default values from image | -| `.env` file | Highest | Per-route overrides | - -Higher priority sources override lower priority values. - -## Environment File Location - -Files are stored in the env directory (default: `~/.gordon/env/`): - -``` -~/.gordon/env/ -├── app_mydomain_com.env -├── api_mydomain_com.env -└── admin_mydomain_com.env -``` - -**Naming:** Replace dots with underscores, add `.env` suffix. - -| Domain | File Name | -|--------|-----------| -| `app.mydomain.com` | `app_mydomain_com.env` | -| `api.company.io` | `api_company_io.env` | -| `staging.app.dev` | `staging_app_dev.env` | - -## File Format - -Standard `.env` format: - -```bash -# Comments start with # -KEY=value -ANOTHER_KEY=another value - -# Quotes are optional but recommended for values with spaces -MESSAGE="Hello World" - -# No spaces around = -DATABASE_URL=postgresql://localhost:5432/mydb -``` - -## Secret Provider Syntax - -Reference secrets from configured backends: - -### Pass Provider - -```bash -# Syntax: ${pass:} -DATABASE_PASSWORD=${pass:myapp/database/password} -API_KEY=${pass:company/api-key} -JWT_SECRET=${pass:production/jwt-secret} -``` - -### SOPS Provider - -```bash -# Syntax: ${sops::} -DATABASE_PASSWORD=${sops:secrets.yaml:database.password} -API_SECRET=${sops:production.yaml:api.secret.key} -STRIPE_KEY=${sops:secrets.yaml:stripe.api_key} -``` - -## Variable Expansion - -### From Secrets - -```bash -# Password from pass -DB_PASSWORD=${pass:myapp/db-password} - -# Then use in connection string -DATABASE_URL=postgresql://user:${DB_PASSWORD}@db:5432/mydb -``` - -### Static Values - -```bash -# Simple values -NODE_ENV=production -PORT=3000 - -# Values with special characters (use quotes) -MESSAGE="Hello, World!" -JSON_CONFIG='{"key": "value"}' -``` - -## Dockerfile ENV - -Variables from Dockerfile serve as defaults: - -```dockerfile -FROM node:18 -ENV NODE_ENV=development -ENV PORT=3000 -ENV LOG_LEVEL=info -``` - -Override in `.env` file: - -```bash -# Override NODE_ENV, keep PORT and LOG_LEVEL defaults -NODE_ENV=production -``` - -Result: -- `NODE_ENV=production` (from .env) -- `PORT=3000` (from Dockerfile) -- `LOG_LEVEL=info` (from Dockerfile) - -## Common Variables - -### Node.js - -```bash -NODE_ENV=production -PORT=3000 -LOG_LEVEL=warn -``` - -### Python - -```bash -FLASK_ENV=production -DJANGO_SETTINGS_MODULE=myapp.settings.production -PYTHONUNBUFFERED=1 -``` - -### Database Connections - -```bash -# PostgreSQL -DATABASE_URL=postgresql://user:pass@postgres:5432/mydb - -# MySQL -DATABASE_URL=mysql://user:pass@mysql:3306/mydb - -# Redis -REDIS_URL=redis://redis:6379/0 -``` - -### External Services - -```bash -# AWS -AWS_ACCESS_KEY_ID=${pass:aws/access-key} -AWS_SECRET_ACCESS_KEY=${pass:aws/secret-key} -AWS_REGION=us-east-1 - -# Stripe -STRIPE_SECRET_KEY=${pass:stripe/secret-key} -STRIPE_PUBLISHABLE_KEY=pk_live_... - -# SendGrid -SENDGRID_API_KEY=${pass:sendgrid/api-key} -``` - -## Examples - -### Minimal - -```bash -NODE_ENV=production -PORT=3000 -``` - -### With Database - -```bash -NODE_ENV=production -PORT=3000 -DATABASE_URL=postgresql://postgres:5432/myapp -DATABASE_PASSWORD=${pass:myapp/db-password} -``` - -### Full Production - -```bash -# Application -NODE_ENV=production -PORT=3000 -LOG_LEVEL=warn - -# Database -DATABASE_URL=postgresql://postgres:5432/production -DATABASE_PASSWORD=${pass:company/db-password} - -# Cache -REDIS_URL=redis://redis:6379/0 - -# External APIs -STRIPE_SECRET_KEY=${pass:company/stripe-secret} -SENDGRID_API_KEY=${pass:company/sendgrid-key} - -# Auth -JWT_SECRET=${pass:company/jwt-secret} -SESSION_SECRET=${pass:company/session-secret} - -# Feature flags -ENABLE_ANALYTICS=true -MAINTENANCE_MODE=false -``` - -## Viewing Container Environment - -```bash -docker inspect gordon-app-mydomain-com --format '{{json .Config.Env}}' | jq -``` - -## Related - -- [Environment Configuration](../config/env.md) -- [Secrets Configuration](../config/secrets.md) diff --git a/docs/reference/index.md b/docs/reference/index.md index 136b14d3f..4f33471dd 100644 --- a/docs/reference/index.md +++ b/docs/reference/index.md @@ -5,7 +5,6 @@ Technical reference documentation for Gordon. ## Contents - [Docker Labels](./docker-labels.md) - Container and image labels used by Gordon -- [Environment Variables](./env-variables.md) - Environment variable syntax and interpolation - [Telemetry & Metrics](../config/telemetry.md) - OpenTelemetry configuration, custom metrics, and trace spans - [Troubleshooting](./troubleshooting.md) - Common issues and solutions diff --git a/docs/reference/troubleshooting.md b/docs/reference/troubleshooting.md index 483887ebc..f9643c7c6 100644 --- a/docs/reference/troubleshooting.md +++ b/docs/reference/troubleshooting.md @@ -91,7 +91,7 @@ No restart needed — Podman reads this on each pull/push. 1. Per-command flag: ```bash - gordon push myapp:latest --insecure + gordon images push myapp:latest --insecure ``` 2. Environment variable: @@ -118,17 +118,13 @@ No restart needed — Podman reads this on each pull/push. ### "unknown: image not found" -**Cause:** Image was pushed but no route configured. +**Cause:** The app manifest references an image that is unavailable to the runtime. -**Solution:** Add route to config: -```toml -[routes] -"app.mydomain.com" = "myapp:latest" -``` +**Solution:** Push the image to the configured registry, verify the service's `image` reference in the app manifest, then apply and deploy the accepted revision: -Then reload: ```bash -gordon reload +gordon apps apply --file app.toml +gordon apps deploy app ``` ## Deployment Issues @@ -161,14 +157,14 @@ gordon reload **Solutions:** -1. Check container logs: +1. Check workload logs: ```bash - docker logs gordon-app-mydomain-com + gordon apps logs blog --service web ``` 2. Check Gordon logs: ```bash - gordon logs -f + gordon daemon logs -f ``` 3. Run container manually to debug: @@ -178,28 +174,23 @@ gordon reload ### Environment variables not loaded -**Cause:** Env file not found or wrong format. +**Cause:** The variable is not declared in the app file, or its secret value was never set. **Solutions:** -1. Check file exists with correct name: +1. Declare public values under `[env]` and service values under `[services..secrets]` in the app file, then apply it: ```bash - ls ~/.gordon/env/ - # Should show: app_mydomain_com.env (dots → underscores) + gordon apps apply --file ./blog.toml ``` -2. Check file permissions: +2. Check which secret values are set (names only): ```bash - chmod 600 ~/.gordon/env/app_mydomain_com.env + gordon apps secrets list blog ``` -3. Check file format (no spaces around `=`): +3. Deploy or restart: values apply on the next deploy/restart, never to running containers. ```bash - # Correct - KEY=value - - # Wrong - KEY = value + gordon apps restart blog ``` ### Secrets not resolved @@ -234,18 +225,14 @@ gordon reload **Solutions:** -1. Check attachments are configured: - ```toml - [attachments] - "app.mydomain.com" = ["postgres:latest"] - ``` +1. Check that both services declare the same shared network in the app manifest. -2. Check containers are in same network: +2. Check containers are attached to that network: ```bash - docker network inspect gordon-app-mydomain-com + docker network inspect NETWORK_NAME ``` -3. Use correct hostname (image name before colon): +3. Use the service name as the internal hostname: ```javascript // Correct connect("postgresql://postgres:5432/mydb") @@ -326,7 +313,7 @@ gordon reload **Solution:** Manual reload: ```bash -gordon reload +gordon daemon reload ``` ### Stale `targets.toml` in config directory @@ -369,15 +356,9 @@ enabled = true path = "~/.gordon/logs/gordon.log" ``` -### Container logs missing - -**Cause:** Container log collection disabled. +### Workload logs unavailable -**Solution:** -```toml -[logging.container_logs] -enabled = true # default: true -``` +`gordon apps logs APP --service SERVICE` reads directly from the container runtime. Confirm that the app has an active deployment, use the exact service name, and check the runtime's logging driver and retention settings. Gordon does not write workload logs to `logging.container_logs` files. ## Diagnostic Commands @@ -386,7 +367,7 @@ enabled = true # default: true systemctl --user status gordon # View Gordon logs -gordon logs -f +gordon daemon logs -f journalctl --user -u gordon -f # List containers @@ -398,11 +379,11 @@ docker network ls | grep gordon # List volumes docker volume ls | grep gordon -# Check container logs -docker logs gordon-app-mydomain-com +# Check workload logs through Gordon +gordon apps logs blog --service web -# Inspect container -docker inspect gordon-app-mydomain-com +# Inspect runtime containers when diagnosing locally +docker ps -f "label=gordon.app=blog" # Check connectivity curl -v http://localhost:5000/v2/ diff --git a/docs/upgrading.md b/docs/upgrading.md index 36e4925d4..2f9085493 100644 --- a/docs/upgrading.md +++ b/docs/upgrading.md @@ -2,11 +2,69 @@ This guide covers breaking changes and migration steps between major versions. +## v3.0.0: Declarative Apps (breaking) + +Gordon v3 replaces route-container management with declarative apps. One standalone TOML file defines one app; `gordon apps apply` persists desired state and `gordon apps deploy` activates it. Push transfers OCI content only and never deploys. + +Follow [Migrate to Gordon v3](./migrate-to-v3.md) for the complete cutover procedure, including explicit migration from domain-scoped secrets to app- and service-scoped secrets. + +### Removed + +- App-workload keys in `gordon.toml`: `[routes]`, `[attachments]`, `[network_groups]`, `[service_routes]`, `[auto_route]` (+ `_allowed_domains`), and `[previews]`. Gordon fails boot/reload closed with a `config-retired` diagnostic naming the fix when any of them is present — never a silent migration. +- Installation-level `[[services]]` (standalone L4 workloads) and `[[network_services]]` (L4 traffic plane) stay valid. Only their old app-workload semantics were removed; declare application workloads in standalone files instead. See [Standalone Services](./config/services.md). +- CLI: `pin`, `preview`, `attachments`, `bootstrap`, `autoroute allow`, `routes add/remove/purge`, push deploy/route inference, implicit deploys on push/reload, label/env-file inference. Removed HTTP mutation endpoints answer `410 Gone`. +- Domain secrets: `gordon secrets`, `/admin/secrets` (now `410 Gone`), the `admin:secrets:*` scopes, the `[env]` config section, and startup `.env` import into pass. Use `gordon apps secrets`. Existing `gordon/env/...` pass entries are left untouched. +- CLI layout: `gordon push` is now `gordon images push`. `status`, `logs`, `reload`, `config`, `tls status`, `traffic status`, and `networks list` moved under `gordon daemon` (`gordon daemon status`, `gordon daemon tls`, `gordon daemon traffic`, `gordon daemon networks`, …). There are no aliases. +- Scopes `admin:routes:*` and `admin:secrets:*` are replaced by `admin:apps:read` (list, show, diff, status) and `admin:apps:write` (apply, deploy, lifecycle, secrets). Regenerate CI tokens, e.g. `--scopes "push,pull,admin:apps:read,admin:apps:write"`. +- No historical rollback command: roll back by applying a manifest that references the previous tag and deploying again. + +### Migrating a v3 alpha app manifest + +Early v3 alpha manifests used the array-of-tables form `[[service]]` with a `name` field and `[service.*]` children. That shape is rejected with a keyed-schema diagnostic, not converted. Rewrite each service as a `[services.]` table. + +Before (alpha, rejected): + +```toml +name = "blog" + +[[service]] +name = "web" +image = "registry.example.com/blog/web:1.2.3" + +[[service.http]] +host = "blog.example.com" +port = 8080 +``` + +After (current keyed schema): + +```toml +name = "blog" + +[services.web] +image = "registry.example.com/blog/web:1.2.3" + +[[services.web.http]] +host = "blog.example.com" +port = 8080 +``` + +Names containing dots must be quoted: `[services."web.api"]` and `[[services."web.api".http]]`. See [App Manifest](./config/apps.md). + +### Manual migration + +1. Back up databases/volumes and pass entries with the existing procedures. Record original ownership and image versions. No update hook deletes volumes: unknown resources are preserved, never adopted. +2. Delete the removed keys, including `[env]`, from `gordon.toml` (installation settings only: entrypoints, TLS, limits, external routes, images policy, backups destinations stay). +3. Write one `.toml` per app (see [App Manifest](./config/apps.md)): services, `[[services..http]]` hosts, `[services..secrets]` names, volumes, `[[network.shared]]`, backup declarations. +4. Migrate secret values explicitly with `gordon apps secrets set` after applying each manifest. Gordon does not copy `gordon/env//...` entries into `gordon/apps///...`; follow the [v3 secrets migration procedure](./migrate-to-v3.md#2-migrate-domain-secrets). +5. `gordon apps apply --file .toml`, then `gordon apps deploy `. +6. Staging is an ordinary app in another file. A binary downgrade against the new app-state format is unsupported: restore the old installation/config/state and backups through an operator-approved procedure. + ## Route-Domain Validation Route keys must be plain hostnames. Use inline tables like `"app.example.com" = { image = "myapp:latest" }`. Gordon still reads legacy `http://...` route entries for backward compatibility and rewrites them on the next save. Update `[routes]`, CLI commands, and automation that reference the old values. -## Next major: Unified smart TCP entrypoints +## v2.31.0: Unified smart TCP entrypoints ### Breaking: `server.port` / `server.tls_port` no longer define public listeners @@ -59,26 +117,6 @@ Keep `server.registry_port` for Docker/Podman push and pull traffic; it is separ - TLS-ALPN-01 is unsupported. - Normal HTTPS fallback certificate priority is static certificates, then public ACME certificates, then Gordon's internal CA. -## v2.30.0 to v2.31.0 - -### Breaking: Public listener migration - -`server.port` and `server.tls_port` no longer create public listeners. Gordon refuses to start when either legacy key is present without an `[entrypoints]` entry, preventing routes from silently becoming unavailable. - -Replace the legacy listener settings with a route-capable entrypoint: - -```toml -[server] -registry_port = 5000 -gordon_domain = "gordon.example.com" - -[entrypoints.edge] -address = ":443" -protocol = "smart_tcp" -``` - -Keep `server.registry_port` for Docker and Podman registry traffic. For public TLS, configure `[tls.acme]` or static certificates; do not use `server.tls_port`. - ### Required for Cloudflare/Proxy Setups: `proxy_allowed_ips` The internal CA's HTTP onboarding gate rejects non-localhost HTTP requests by default. If Gordon sits behind Cloudflare or another reverse proxy, add the proxy's edge IPs to `proxy_allowed_ips`: @@ -114,7 +152,7 @@ registry_domain = "gordon.example.com" gordon_domain = "gordon.example.com" ``` -If you do not migrate, `gordon status --remote ...` and `gordon routes list --remote ...` can fail with `/auth/token` `404`, and `reg-domain/v2/` or `/admin/status` can return `404`. +If you do not migrate, `gordon daemon status --remote ...` and `gordon apps list --remote ...` can fail with `/auth/token` `404`, and `reg-domain/v2/` or `/admin/status` can return `404`. ### Staged Registry Host Rename @@ -214,7 +252,7 @@ Gordon v2.30.0 removes password-based authentication entirely. Only token-based - `gordon auth show-token` prints the stored token for a remote - `gordon auth logout` removes the stored token locally - Automatic token exchange: the CLI transparently exchanges long-lived tokens for ephemeral ones before API calls -- Admin scopes (`admin:*:*`, `admin:routes:read`, etc.) allow fine-grained access control for remote CLI operations +- Admin scopes (`admin:*:*`, `admin:apps:read`, `admin:apps:write`, etc.) allow fine-grained access control for remote CLI operations **Migration steps:** @@ -262,7 +300,7 @@ Gordon v2.30.0 removes password-based authentication entirely. Only token-based 5. **Upgrade the binary** and restart: ```bash - curl -fsSL https://gordon.bnema.dev/install | bash + curl -fsSL https://bnema.dev/gordon/install | bash systemctl --user restart gordon ``` @@ -294,8 +332,8 @@ gordon auth token generate --subject admin --scopes "push,pull,admin:*:*" --expi # Read-only monitoring gordon auth token generate --subject monitor --scopes "admin:status:read" --expiry 30d -# CI deploy with route read + config write -gordon auth token generate --subject ci --scopes "push,pull,admin:routes:read,admin:config:write" --expiry 0 +# CI deploy with app read + write +gordon auth token generate --subject ci --scopes "push,pull,admin:apps:read,admin:apps:write" --expiry 0 ``` See [Token Scopes](./config/auth.md#token-scopes) for the full list. @@ -433,7 +471,7 @@ If using pass or sops, update your secret paths: 3. **Test in staging** if possible 4. **Upgrade the binary**: ```bash - curl -fsSL https://gordon.bnema.dev/install | bash + curl -fsSL https://bnema.dev/gordon/install | bash ``` 5. **Restart Gordon**: ```bash @@ -441,10 +479,10 @@ If using pass or sops, update your secret paths: ``` 6. **Check logs** for any errors: ```bash - gordon logs + gordon daemon logs ``` ## Getting Help - [GitHub Issues](https://github.com/bnema/gordon/issues) -- [Documentation](https://gordon.bnema.dev/docs) +- [Documentation](https://bnema.dev/gordon/docs) diff --git a/go.mod b/go.mod index 22433041f..11259e805 100644 --- a/go.mod +++ b/go.mod @@ -3,10 +3,10 @@ module github.com/bnema/gordon go 1.27 require ( - github.com/aws/aws-sdk-go-v2 v1.45.1 - github.com/aws/aws-sdk-go-v2/config v1.33.1 - github.com/aws/aws-sdk-go-v2/feature/s3/transfermanager v0.4.1 - github.com/aws/aws-sdk-go-v2/service/s3 v1.109.1 + github.com/aws/aws-sdk-go-v2 v1.47.1 + github.com/aws/aws-sdk-go-v2/config v1.33.6 + github.com/aws/aws-sdk-go-v2/feature/s3/transfermanager v0.4.10 + github.com/aws/aws-sdk-go-v2/service/s3 v1.113.4 github.com/bnema/zerowrap v1.4.1 github.com/charmbracelet/bubbles v1.0.0 github.com/charmbracelet/bubbletea v1.3.10 @@ -17,12 +17,12 @@ require ( github.com/fsnotify/fsnotify v1.10.1 // direct github.com/go-acme/lego/v4 v4.35.2 github.com/golang-jwt/jwt/v5 v5.3.1 - github.com/google/go-containerregistry v0.22.0 + github.com/google/go-containerregistry v0.22.1 github.com/google/uuid v1.6.0 github.com/mattn/go-isatty v0.0.24 - github.com/mattn/go-runewidth v0.0.28 - github.com/moby/moby/api v1.55.0 - github.com/moby/moby/client v0.5.1 + github.com/mattn/go-runewidth v0.0.30 + github.com/moby/moby/api v1.56.0 + github.com/moby/moby/client v0.6.0 github.com/muesli/termenv v0.16.0 github.com/pelletier/go-toml/v2 v2.4.3 github.com/rivo/uniseg v0.4.7 @@ -32,21 +32,22 @@ require ( github.com/spf13/cobra v1.10.2 github.com/spf13/viper v1.21.0 github.com/stretchr/testify v1.12.1 + go.etcd.io/bbolt v1.5.0 go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.71.0 go.opentelemetry.io/otel v1.46.0 go.opentelemetry.io/otel/exporters/otlp/otlplog/otlploghttp v0.22.0 go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetrichttp v1.46.0 go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracehttp v1.46.0 + go.opentelemetry.io/otel/log v0.22.0 go.opentelemetry.io/otel/metric v1.46.0 go.opentelemetry.io/otel/sdk v1.46.0 go.opentelemetry.io/otel/sdk/log v0.22.0 go.opentelemetry.io/otel/sdk/metric v1.46.0 go.opentelemetry.io/otel/trace v1.46.0 go.uber.org/mock v0.6.0 // direct - golang.org/x/mod v0.40.0 - golang.org/x/sync v0.22.0 - golang.org/x/sys v0.47.0 - golang.org/x/time v0.15.0 + golang.org/x/sync v0.23.0 + golang.org/x/sys v0.48.0 + golang.org/x/time v0.16.0 gopkg.in/natefinch/lumberjack.v2 v2.2.1 ) @@ -54,71 +55,69 @@ require ( github.com/Microsoft/go-winio v0.6.2 // indirect github.com/atotto/clipboard v0.1.4 // indirect github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.20 // indirect - github.com/aws/aws-sdk-go-v2/credentials v1.20.1 // indirect - github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.19.1 // indirect - github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.1 // indirect - github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.1 // indirect - github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.1 // indirect + github.com/aws/aws-sdk-go-v2/credentials v1.20.6 // indirect + github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.1 // indirect + github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.4 // indirect + github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.4 // indirect + github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.4 // indirect github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19 // indirect - github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.11.1 // indirect - github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.1 // indirect - github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.20.1 // indirect - github.com/aws/aws-sdk-go-v2/service/signin v1.7.1 // indirect - github.com/aws/aws-sdk-go-v2/service/sso v1.35.1 // indirect - github.com/aws/aws-sdk-go-v2/service/ssooidc v1.40.1 // indirect - github.com/aws/aws-sdk-go-v2/service/sts v1.47.1 // indirect - github.com/aws/smithy-go v1.28.1 // indirect + github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.11.5 // indirect + github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.4 // indirect + github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.20.4 // indirect + github.com/aws/aws-sdk-go-v2/service/signin v1.10.1 // indirect + github.com/aws/aws-sdk-go-v2/service/sso v1.38.1 // indirect + github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.1 // indirect + github.com/aws/aws-sdk-go-v2/service/sts v1.51.1 // indirect + github.com/aws/smithy-go v1.28.2 // indirect github.com/aymanbagabas/go-osc52/v2 v2.0.1 // indirect github.com/cenkalti/backoff/v5 v5.0.3 // indirect github.com/cespare/xxhash/v2 v2.3.0 // indirect github.com/charmbracelet/colorprofile v0.4.3 // indirect - github.com/charmbracelet/x/ansi v0.11.6 // indirect + github.com/charmbracelet/x/ansi v0.11.8 // indirect github.com/charmbracelet/x/cellbuf v0.0.15 // indirect github.com/charmbracelet/x/term v0.2.2 // indirect github.com/clipperhouse/displaywidth v0.11.0 // indirect github.com/clipperhouse/uax29/v2 v2.7.0 // indirect github.com/containerd/errdefs/pkg v0.3.0 // indirect github.com/distribution/reference v0.6.0 // indirect - github.com/docker/cli v29.7.2+incompatible // indirect - github.com/docker/docker-credential-helpers v0.9.5 // indirect + github.com/docker/cli v29.8.1+incompatible // indirect + github.com/docker/docker-credential-helpers v0.9.9 // indirect github.com/docker/go-connections v0.8.1 // indirect github.com/erikgeiser/coninput v0.0.0-20211004153227-1c3628e74d0f // indirect github.com/felixge/httpsnoop v1.1.0 // indirect - github.com/go-jose/go-jose/v4 v4.1.4 // indirect + github.com/go-jose/go-jose/v4 v4.1.5 // indirect github.com/go-logr/logr v1.4.4 // indirect github.com/go-logr/stdr v1.2.2 // indirect github.com/go-viper/mapstructure/v2 v2.5.0 // indirect - github.com/grpc-ecosystem/grpc-gateway/v2 v2.30.0 // indirect + github.com/grpc-ecosystem/grpc-gateway/v2 v2.31.0 // indirect github.com/inconshreveable/mousetrap v1.1.0 // indirect - github.com/klauspost/compress v1.19.2 // indirect - github.com/lucasb-eyer/go-colorful v1.4.0 // indirect + github.com/klauspost/compress v1.20.1 // indirect + github.com/lucasb-eyer/go-colorful v1.4.1 // indirect github.com/mattn/go-colorable v0.1.15 // indirect github.com/mattn/go-localereader v0.0.1 // indirect - github.com/miekg/dns v1.1.72 // indirect + github.com/miekg/dns v1.1.73 // indirect github.com/moby/docker-image-spec v1.3.1 // indirect github.com/muesli/ansi v0.0.0-20230316100256-276c6243b2f6 // indirect github.com/muesli/cancelreader v0.2.2 // indirect github.com/opencontainers/go-digest v1.0.0 // indirect github.com/opencontainers/image-spec v1.1.1 // indirect github.com/sagikazarmark/locafero v0.12.0 // indirect - github.com/sirupsen/logrus v1.9.4 // indirect + github.com/sirupsen/logrus v1.10.2 // indirect github.com/spf13/cast v1.10.0 // indirect github.com/spf13/pflag v1.0.10 // indirect github.com/stretchr/objx v0.5.3 // indirect github.com/subosito/gotenv v1.6.0 // indirect - github.com/xo/terminfo v0.0.0-20220910002029-abceb7e1c41e // indirect + github.com/xo/terminfo v1.2.0 // indirect go.opentelemetry.io/auto/sdk v1.2.1 // indirect go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.46.0 // indirect - go.opentelemetry.io/otel/log v0.22.0 // indirect go.opentelemetry.io/proto/otlp v1.11.0 // indirect go.yaml.in/yaml/v3 v3.0.5 // indirect - golang.org/x/crypto v0.55.0 // indirect - golang.org/x/net v0.58.0 // indirect - golang.org/x/text v0.41.0 // indirect - golang.org/x/tools v0.49.0 // indirect - google.golang.org/genproto/googleapis/api v0.0.0-20260819154853-08b0e4226688 // indirect - google.golang.org/genproto/googleapis/rpc v0.0.0-20260819154853-08b0e4226688 // indirect - google.golang.org/grpc v1.83.1 // indirect + golang.org/x/crypto v0.57.0 // indirect + golang.org/x/net v0.59.0 // indirect + golang.org/x/text v0.42.0 // indirect + google.golang.org/genproto/googleapis/api v0.0.0-20260921155816-b14227669459 // indirect + google.golang.org/genproto/googleapis/rpc v0.0.0-20260921155816-b14227669459 // indirect + google.golang.org/grpc v1.84.0 // indirect google.golang.org/protobuf v1.36.12 // indirect howett.net/plist v1.0.1 // indirect ) diff --git a/go.sum b/go.sum index 217b094e5..a9088daa5 100644 --- a/go.sum +++ b/go.sum @@ -2,44 +2,44 @@ github.com/Microsoft/go-winio v0.6.2 h1:F2VQgta7ecxGYO8k3ZZz3RS8fVIXVxONVUPlNERo github.com/Microsoft/go-winio v0.6.2/go.mod h1:yd8OoFMLzJbo9gZq8j5qaps8bJ9aShtEA8Ipt1oGCvU= github.com/atotto/clipboard v0.1.4 h1:EH0zSVneZPSuFR11BlR9YppQTVDbh5+16AmcJi4g1z4= github.com/atotto/clipboard v0.1.4/go.mod h1:ZY9tmq7sm5xIbd9bOK4onWV4S6X0u6GY7Vn0Yu86PYI= -github.com/aws/aws-sdk-go-v2 v1.45.1 h1:iIoG3NaLhV6UZpPXyPXlDj2I9oS8tV/nMcMnITCC6Ks= -github.com/aws/aws-sdk-go-v2 v1.45.1/go.mod h1:bttEH6JqnUL8LepvDVfdrds/fZ5bCIxzpe3abyUrhDU= +github.com/aws/aws-sdk-go-v2 v1.47.1 h1:uOIZnp4PK3ZhKI0dNrJrhTEsLxbpXHTAJlwoS1pvAtw= +github.com/aws/aws-sdk-go-v2 v1.47.1/go.mod h1:bttEH6JqnUL8LepvDVfdrds/fZ5bCIxzpe3abyUrhDU= github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.20 h1:GPRlPwz40I2B2VrBEASOA3Bi77NyeqejNLkifosX0rs= github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.20/go.mod h1:g7PNzKcsOKWb4fkSRBA7BZVAS6Y8IcxzN+nRohhQ1Q8= -github.com/aws/aws-sdk-go-v2/config v1.33.1 h1:bq9jze1hQ5YTCLoVxNnbp0T7rglrlOE7N9YsHqjGkEw= -github.com/aws/aws-sdk-go-v2/config v1.33.1/go.mod h1:2A3HQwG4zaL5Tm80rc6RZj8LmWWv4WYT5v8raSz/L7A= -github.com/aws/aws-sdk-go-v2/credentials v1.20.1 h1:Z8GRNEx0u9sDkZOq4PUnN8mjGwbUQGRzMSXpvt3d8xQ= -github.com/aws/aws-sdk-go-v2/credentials v1.20.1/go.mod h1:uBIK00kFo95dnemqfFMTWx0X8YRqsh6ecIoCjjOkZqM= -github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.19.1 h1:YIEBqcqRnpi4Pfv0YHImtgi6czGCwKHANC7SwmUAVD0= -github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.19.1/go.mod h1:imEf0oufgAo8KAkCHhrOdqGEC0YWx1PPBQH82shSxGw= -github.com/aws/aws-sdk-go-v2/feature/s3/transfermanager v0.4.1 h1:I3mWvASaICc5c8vJ3ftYjroh7LT3jG0q/KtzSu5wW/s= -github.com/aws/aws-sdk-go-v2/feature/s3/transfermanager v0.4.1/go.mod h1:VwSN8piv62OyyxYWtJzj6j7gECyLc0WwMo30L5NQZlE= -github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.1 h1:pc138gM1CW+XPc60rEwUlwwuwWFQK16CI1T7v1F9Oec= -github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.1/go.mod h1:1+koxpPIbfBdfzP6vojm5/zTpTQ/micYwlxIiNB3TxI= -github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.1 h1:K0JsbZQj+1h208Ro1zHeA4l7bMp0NvRffHQ91q8Ol1s= -github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.1/go.mod h1:W3/vL6EtCIatICGy9ab29QhMuae+cOKPWcMxv02CO+Q= -github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.1 h1:yhw5KD1phVyP9vijxOUzDfEtJx+bt+L63k+VfuiYFAA= -github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.1/go.mod h1:ZW2e0d7DYlRxlS9hEiMXE47gTdX5KRN4byUiNbUpG+Q= +github.com/aws/aws-sdk-go-v2/config v1.33.6 h1:MBjkSTLczek/UgiK+EYPIoRTqE7gP8vtW3OFbFo7Nug= +github.com/aws/aws-sdk-go-v2/config v1.33.6/go.mod h1:grRAFzdAZJrwcbasJRg2MPvIrVjtlfXllHssN6+E1JE= +github.com/aws/aws-sdk-go-v2/credentials v1.20.6 h1:NpAFXCU7NzXNkdGK3zQTtsRJ+3v9tZQV0xcdRw8uBdw= +github.com/aws/aws-sdk-go-v2/credentials v1.20.6/go.mod h1:mcZCoiPnyMvP8VMNbygNX5lLqSlkYJIMPODylQMurOk= +github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.1 h1:8gALAAmacnIXh+z6VkdDanv4/IkG5APdg4DZLDTmLog= +github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.1/go.mod h1:Z7IJhJU+poOdJjUR2wpyY21ossQ1XS/R3Lk9Msq5kM4= +github.com/aws/aws-sdk-go-v2/feature/s3/transfermanager v0.4.10 h1:DaCvPvqeZCMqziWeKPiLCv/tUvpe4/Cq79bjE1ZGtHY= +github.com/aws/aws-sdk-go-v2/feature/s3/transfermanager v0.4.10/go.mod h1:eHZtvcgtQ2WQDWcIs/VUuM3zglCEGipTaxfaTmzZxpw= +github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.4 h1:CLq4+8UHCI+ZZYl/EuJxXovaIVN2xeeT8JV+dsApQ5E= +github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.4/go.mod h1:Wv4q5sAM04xAMkoOedxLx2inVf6K5FdxYp+A61L+q/0= +github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.4 h1:dD4MR81I7YkpEBRk6UP9rocC2QnT3qVuXwzlYTtfGEs= +github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.4/go.mod h1:EcXV1kAFd5XwSkDHlj94gnF3q5CkJyYiIJfH8N0VmrE= +github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.4 h1:7Wo47d/xn/7KttCSBd8EGYeZ7ULRFRkUHr6vkZPBzVQ= +github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.4/go.mod h1:tDB2IVC1xC3vX8o+6uRlzhTxP3g1b77CZXFX/oD2FnQ= github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19 h1:bAdDl/HkGCcGPoe25ToSHEw23VIxt6CT5fLcg111BKg= github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19/go.mod h1:KaUzbLxv4CeSxh6ZCl9B4m7CuFenS8kUEaDs+f/DQr4= -github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.11.1 h1:s67hBfG5t9rn1NCvDuB4E3QIep3UFhHPtaIqFDjV3N8= -github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.11.1/go.mod h1:FpvjBMXtSNMLPmDJsWwcY5cRnqJlpS2y1R6n4pvzs4k= -github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.1 h1:RmmWQPREQdk9U+PfqeHW3MqZaBaNK7TpV9W3RY+b+7g= -github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.1/go.mod h1:0A3W4F+68ZnNk5XcNL/e9HFMwnP8RlEicFfy6eOEDyw= -github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.20.1 h1:ZMbtPZZQRca+3+XYQne9PBvRiYpHZlNJJOZfE9WNfT0= -github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.20.1/go.mod h1:YAGWQdCYlVCoqrzvfv3RLxO6zKwti7gsAULOGWPLYv4= -github.com/aws/aws-sdk-go-v2/service/s3 v1.109.1 h1:kVpzaDBzOdRtOftmiSpTdQbWVqRg0kONLXijktiwXnk= -github.com/aws/aws-sdk-go-v2/service/s3 v1.109.1/go.mod h1:CUr46sCpGAg/rHaclRyhJX0LJAmH73uWSJPPSaMUrSk= -github.com/aws/aws-sdk-go-v2/service/signin v1.7.1 h1:mdMtSVKdQ3+mzBh+l0ogrFYZVQUCg6pJZOirA2ARsYE= -github.com/aws/aws-sdk-go-v2/service/signin v1.7.1/go.mod h1:9IqUlsJDbUPcg6cgx3WEzXdjrbWzLDQrak0aaSqlTcI= -github.com/aws/aws-sdk-go-v2/service/sso v1.35.1 h1:B6WFn91tobD6gG4724ONHaqrpKsoETGnv98LHe/yIGM= -github.com/aws/aws-sdk-go-v2/service/sso v1.35.1/go.mod h1:tWuiVBUtPBr8/rgRiYS8Uf85sHcAN+G7XS3D3CEoUh8= -github.com/aws/aws-sdk-go-v2/service/ssooidc v1.40.1 h1:6yeYCWFvgbI2TI3K6jr9LtBNhXgJ7g4xqD+DEiaDDmM= -github.com/aws/aws-sdk-go-v2/service/ssooidc v1.40.1/go.mod h1:naFe83jSMuYkH+QjQPX8n1MLhBkeCFM5Lsnh5m5wz3c= -github.com/aws/aws-sdk-go-v2/service/sts v1.47.1 h1:Sv2xPnRHlThSUtVujYuUBPI/Il8si6UPHXL8DMiB/F0= -github.com/aws/aws-sdk-go-v2/service/sts v1.47.1/go.mod h1:mKo/CzaCz8qytGW70NG4vIIGAx1HXTlb5lHNkC5k3lk= -github.com/aws/smithy-go v1.28.1 h1:R/nXH00c8qcfCzQVELtRw+eLQWtzv+VAIEFJ1/xxXlQ= -github.com/aws/smithy-go v1.28.1/go.mod h1:YE2RhdIuDbA5E5bTdciG9KrW3+TiEONeUWCqxX9i1Fc= +github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.11.5 h1:/TYsZXdA8UTa+WCtCYSAJIr1vwl0+eho6TUgJGwFFO8= +github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.11.5/go.mod h1:qPqp1Uwd/BqdhPufv6oem9j5J7HNsgc2V22dUiDPn+s= +github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.4 h1:29SvnfGhXjTl8ONxFwbj2rs6lbhiFXD2CgFQmbT/bXY= +github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.4/go.mod h1:wm04I5DMuNVvZHFe/dHnUxincvNbbK7AiNBbYsQivek= +github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.20.4 h1:pPiWfgeNxqluKEph7hvU88kuGKBPOWzO+Dk9t2zqqNs= +github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.20.4/go.mod h1:YlwGoIUDG/3kBQbdNOVs/xKZ9J01G8e/6D1mRBj9uTk= +github.com/aws/aws-sdk-go-v2/service/s3 v1.113.4 h1:n6kO3OlBvnDEksQpvBLbAldjHwGlu8kErvhHJkhlaRY= +github.com/aws/aws-sdk-go-v2/service/s3 v1.113.4/go.mod h1:9APRWGLFITKD+xzWSIyT9V7QV4bNlEuIieWlzXgGFlI= +github.com/aws/aws-sdk-go-v2/service/signin v1.10.1 h1:DzCCWLzcIRQ77F3DEUljud7bEjTgFOIKXP52NmVRyhU= +github.com/aws/aws-sdk-go-v2/service/signin v1.10.1/go.mod h1:xpo/geVldu8payT375WekctUzopG/hBU7miiqItMUlw= +github.com/aws/aws-sdk-go-v2/service/sso v1.38.1 h1:Umtl/0YZhng4xndfW3lKJrYYP7NLEjI6bGXVomwLcs0= +github.com/aws/aws-sdk-go-v2/service/sso v1.38.1/go.mod h1:rRD/dnm7q0HYE/I5TMaPgkWyyUGLcwuxHLABsLnQ3e0= +github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.1 h1:orIWdNiLgzrhu/11RcPPKO/SBzUUymbUQuZbSPImghg= +github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.1/go.mod h1:skwM/xsbR/1ReUTesv9BhpJp1VjajR7DWQnuVLwiXsQ= +github.com/aws/aws-sdk-go-v2/service/sts v1.51.1 h1:0HOqZXRvMytH6bFHVIc0oJX07sZjfhz0zXtjs6gdE8s= +github.com/aws/aws-sdk-go-v2/service/sts v1.51.1/go.mod h1:26zA0GhDrLo+yiLI2yXWxqB1PdsShfLikoI7GOEgugM= +github.com/aws/smithy-go v1.28.2 h1:myhcykQcatTul2B/zITjDk203G7t0awUAs1hVry5Bvg= +github.com/aws/smithy-go v1.28.2/go.mod h1:YE2RhdIuDbA5E5bTdciG9KrW3+TiEONeUWCqxX9i1Fc= github.com/aymanbagabas/go-osc52/v2 v2.0.1 h1:HwpRHbFMcZLEVr42D4p7XBqjyuxQH5SMiErDT4WkJ2k= github.com/aymanbagabas/go-osc52/v2 v2.0.1/go.mod h1:uYgXzlJ7ZpABp8OJ+exZzJJhRNQ2ASbcXHWsFqH8hp8= github.com/aymanbagabas/go-udiff v0.3.1 h1:LV+qyBQ2pqe0u42ZsUEtPiCaUoqgA9gYRDs3vj1nolY= @@ -58,8 +58,8 @@ github.com/charmbracelet/colorprofile v0.4.3 h1:QPa1IWkYI+AOB+fE+mg/5/4HRMZcaXex github.com/charmbracelet/colorprofile v0.4.3/go.mod h1:/zT4BhpD5aGFpqQQqw7a+VtHCzu+zrQtt1zhMt9mR4Q= github.com/charmbracelet/lipgloss v1.1.0 h1:vYXsiLHVkK7fp74RkV7b2kq9+zDLoEU4MZoFqR/noCY= github.com/charmbracelet/lipgloss v1.1.0/go.mod h1:/6Q8FR2o+kj8rz4Dq0zQc3vYf7X+B0binUUBwA0aL30= -github.com/charmbracelet/x/ansi v0.11.6 h1:GhV21SiDz/45W9AnV2R61xZMRri5NlLnl6CVF7ihZW8= -github.com/charmbracelet/x/ansi v0.11.6/go.mod h1:2JNYLgQUsyqaiLovhU2Rv/pb8r6ydXKS3NIttu3VGZQ= +github.com/charmbracelet/x/ansi v0.11.8 h1:JMFwp0CgDC2+jcOB162HH5k7I3FVbgFSMMYg7dSPBQQ= +github.com/charmbracelet/x/ansi v0.11.8/go.mod h1:ZNN+3mXny/516oTQPLMPIBeSINvNJJQ8uQXDgbeJxY0= github.com/charmbracelet/x/cellbuf v0.0.15 h1:ur3pZy0o6z/R7EylET877CBxaiE1Sp1GMxoFPAIztPI= github.com/charmbracelet/x/cellbuf v0.0.15/go.mod h1:J1YVbR7MUuEGIFPCaaZ96KDl5NoS0DAWkskup+mOY+Q= github.com/charmbracelet/x/exp/golden v0.0.0-20241011142426-46044092ad91 h1:payRxjMjKgx2PaCWLZ4p3ro9y97+TVLZNaRZgJwSVDQ= @@ -79,10 +79,10 @@ github.com/coreos/go-systemd/v22 v22.7.0/go.mod h1:xNUYtjHu2EDXbsxz1i41wouACIwT7 github.com/cpuguy83/go-md2man/v2 v2.0.6/go.mod h1:oOW0eioCTA6cOiMLiUPZOpcVxMig6NIQQ7OS05n1F4g= github.com/distribution/reference v0.6.0 h1:0IXCQ5g4/QMHHkarYzh5l+u8T3t73zM5QvfrDyIgxBk= github.com/distribution/reference v0.6.0/go.mod h1:BbU0aIcezP1/5jX/8MP0YiH4SdvB5Y4f/wlDRiLyi3E= -github.com/docker/cli v29.7.2+incompatible h1:dlkwallR8XqfeVnA2ELEhdwvb4lsSwuB4IgsG8Q9cLY= -github.com/docker/cli v29.7.2+incompatible/go.mod h1:JLrzqnKDaYBop7H2jaqPtU4hHvMKP+vjCwu2uszcLI8= -github.com/docker/docker-credential-helpers v0.9.5 h1:EFNN8DHvaiK8zVqFA2DT6BjXE0GzfLOZ38ggPTKePkY= -github.com/docker/docker-credential-helpers v0.9.5/go.mod h1:v1S+hepowrQXITkEfw6o4+BMbGot02wiKpzWhGUZK6c= +github.com/docker/cli v29.8.1+incompatible h1:qYL1bCp6cRw2SB1xmLlIOPyV171dilw9W2Jew38vy9c= +github.com/docker/cli v29.8.1+incompatible/go.mod h1:JLrzqnKDaYBop7H2jaqPtU4hHvMKP+vjCwu2uszcLI8= +github.com/docker/docker-credential-helpers v0.9.9 h1:BkydjIgZ46JnDbqyM2p2fc63KMw6y+KHL3Em/2AGJ7w= +github.com/docker/docker-credential-helpers v0.9.9/go.mod h1:v1S+hepowrQXITkEfw6o4+BMbGot02wiKpzWhGUZK6c= github.com/docker/go-connections v0.8.1 h1:JibmG5hULs5qXSr/cp/w3Pw5fZuStt4MOHMUExb29/M= github.com/docker/go-connections v0.8.1/go.mod h1:no1qkHdjq7kLMGUXYAduOhYPSJxxvgWBh7ogVvptn3Q= github.com/docker/go-units v0.5.0 h1:69rxXcBk27SvSaaxTtLh/8llcHD8vYHT7WSdRZ/jvr4= @@ -97,8 +97,8 @@ github.com/fsnotify/fsnotify v1.10.1 h1:b0/UzAf9yR5rhf3RPm9gf3ehBPpf0oZKIjtpKrx5 github.com/fsnotify/fsnotify v1.10.1/go.mod h1:TLheqan6HD6GBK6PrDWyDPBaEV8LspOxvPSjC+bVfgo= github.com/go-acme/lego/v4 v4.35.2 h1:uVQg+KC/yj9R2g7Q9W5wDqhvQvxV5SMu5eqFVoN5xZU= github.com/go-acme/lego/v4 v4.35.2/go.mod h1:pX2jN5n8OphMGY1IaMjYm5DAEzguBaKRt8AvJAgJXpc= -github.com/go-jose/go-jose/v4 v4.1.4 h1:moDMcTHmvE6Groj34emNPLs/qtYXRVcd6S7NHbHz3kA= -github.com/go-jose/go-jose/v4 v4.1.4/go.mod h1:x4oUasVrzR7071A4TnHLGSPpNOm2a21K9Kf04k1rs08= +github.com/go-jose/go-jose/v4 v4.1.5 h1:RjgjO2LOtWOJKUC5wpwY9LR3B3vwVAz6JS2YHfYU6eA= +github.com/go-jose/go-jose/v4 v4.1.5/go.mod h1:x4oUasVrzR7071A4TnHLGSPpNOm2a21K9Kf04k1rs08= github.com/go-logr/logr v1.2.2/go.mod h1:jdQByPbusPIv2/zmleS9BjJVeZ6kBagPoEUsqbVz/1A= github.com/go-logr/logr v1.4.4 h1:tG4xh9yMsRCAiodLVTxyrkzSZ9+o0L1Kg/+cPVcbP/8= github.com/go-logr/logr v1.4.4/go.mod h1:9T104GzyrTigFIr8wt5mBrctHMim0Nb2HLGrmQ40KvY= @@ -112,39 +112,39 @@ github.com/golang/protobuf v1.5.4 h1:i7eJL8qZTpSEXOPTxNKhASYpMn+8e5Q6AdndVa1dWek github.com/golang/protobuf v1.5.4/go.mod h1:lnTiLA8Wa4RWRcIUkrtSVa5nRhsEGBg48fD6rSs7xps= github.com/google/go-cmp v0.7.0 h1:wk8382ETsv4JYUZwIsn6YpYiWiBsYLSJiTsyBybVuN8= github.com/google/go-cmp v0.7.0/go.mod h1:pXiqmnSA92OHEEa9HXL2W4E7lf9JzCmGVUdgjX3N/iU= -github.com/google/go-containerregistry v0.22.0 h1:eGbCiPeYxAH/7WLLq6zTBALP0tUIFsoyRauhxXDJ53I= -github.com/google/go-containerregistry v0.22.0/go.mod h1:bJR35SK8XgisYmhg/FMQ/5RK0S/XrOAqLBV5/LR2XE0= +github.com/google/go-containerregistry v0.22.1 h1:RZuuSYhTvlDvtsK+NkutoCZ//C0X2ebLK8X8l3ULs84= +github.com/google/go-containerregistry v0.22.1/go.mod h1:bJR35SK8XgisYmhg/FMQ/5RK0S/XrOAqLBV5/LR2XE0= github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0= github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo= -github.com/grpc-ecosystem/grpc-gateway/v2 v2.30.0 h1:/Tnpcb2E0Pz/tN9s3bfEY2Q8ePCEX9iuS+cneUwncnw= -github.com/grpc-ecosystem/grpc-gateway/v2 v2.30.0/go.mod h1:zOBXOsUaBSjKgmH4OGzV1esUpR3oUSCPYVd2cUBjKYY= +github.com/grpc-ecosystem/grpc-gateway/v2 v2.31.0 h1:Bd7KaOxzULLxtZ/K5s1aLbWhR0+5RToO65TXHsf3bqQ= +github.com/grpc-ecosystem/grpc-gateway/v2 v2.31.0/go.mod h1:nN7ts3dFXKtCZWc//yfkpcQNKJABg16/uDVAZpLDalo= github.com/inconshreveable/mousetrap v1.1.0 h1:wN+x4NVGpMsO7ErUn/mUI3vEoE6Jt13X2s0bqwp9tc8= github.com/inconshreveable/mousetrap v1.1.0/go.mod h1:vpF70FUmC8bwa3OWnCshd2FqLfsEA9PFc4w1p2J65bw= github.com/jessevdk/go-flags v1.4.0/go.mod h1:4FA24M0QyGHXBuZZK/XkWh8h0e1EYbRYJSGM75WSRxI= -github.com/klauspost/compress v1.19.2 h1:hMRETovs/pu/dVWN7zIT1PGG8t509MwT6bO7XSi26R8= -github.com/klauspost/compress v1.19.2/go.mod h1:cwPg85FWrGar70rWktvGQj8/hthj3wpl0PGDogxkrSQ= +github.com/klauspost/compress v1.20.1 h1:T7kKElXUMXrUJ2E9QhQhxFtcK5rPyLdsGZvdbLMPdiQ= +github.com/klauspost/compress v1.20.1/go.mod h1:LUdAzn7YLVvxLpc7y3V1m40wESHTgc1422pwwBSKYuI= github.com/kr/pretty v0.3.1 h1:flRD4NNwYAUpkphVc1HcthR4KEIFJ65n8Mw5qdRn3LE= github.com/kr/pretty v0.3.1/go.mod h1:hoEshYVHaxMs3cyo3Yncou5ZscifuDolrwPKZanG3xk= github.com/kr/text v0.2.0 h1:5Nx0Ya0ZqY2ygV366QzturHI13Jq95ApcVaJBhpS+AY= github.com/kr/text v0.2.0/go.mod h1:eLer722TekiGuMkidMxC/pM04lWEeraHUUmBw8l2grE= -github.com/lucasb-eyer/go-colorful v1.4.0 h1:UtrWVfLdarDgc44HcS7pYloGHJUjHV/4FwW4TvVgFr4= -github.com/lucasb-eyer/go-colorful v1.4.0/go.mod h1:R4dSotOR9KMtayYi1e77YzuveK+i7ruzyGqttikkLy0= +github.com/lucasb-eyer/go-colorful v1.4.1 h1:1EO+WB73+EH8EVbzlrG3KLAfEypQWVHIBqlTf+2hNss= +github.com/lucasb-eyer/go-colorful v1.4.1/go.mod h1:R4dSotOR9KMtayYi1e77YzuveK+i7ruzyGqttikkLy0= github.com/mattn/go-colorable v0.1.15 h1:+u9SLTRGnXv73cEsnsmoZBom+dMU88B2M0aDcWy0/jY= github.com/mattn/go-colorable v0.1.15/go.mod h1:6LmQG8QLFO4G5z1gPvYEzlUgJ2wF+stgPZH1UqBm1s8= github.com/mattn/go-isatty v0.0.24 h1:tGZZoVgT/KiqK1c8ocVLeDS8BSWMRd47J3Lbz7vsReI= github.com/mattn/go-isatty v0.0.24/go.mod h1:nMCL3Zebbrt45jsMDgnfIwz6ydEQApk5oEI3HqDio6A= github.com/mattn/go-localereader v0.0.1 h1:ygSAOl7ZXTx4RdPYinUpg6W99U8jWvWi9Ye2JC/oIi4= github.com/mattn/go-localereader v0.0.1/go.mod h1:8fBrzywKY7BI3czFoHkuzRoWE9C+EiG4R1k4Cjx5p88= -github.com/mattn/go-runewidth v0.0.28 h1:rPyg2ybwEKPebvpzVWe1gKBkH8EQFkxO4Y0hjBeLaBU= -github.com/mattn/go-runewidth v0.0.28/go.mod h1:3qAiGCV4Koz/yuveO58qUefmUTRm8r0IGEXZ9jeHp/8= -github.com/miekg/dns v1.1.72 h1:vhmr+TF2A3tuoGNkLDFK9zi36F2LS+hKTRW0Uf8kbzI= -github.com/miekg/dns v1.1.72/go.mod h1:+EuEPhdHOsfk6Wk5TT2CzssZdqkmFhf8r+aVyDEToIs= +github.com/mattn/go-runewidth v0.0.30 h1:+KUuiDA4fF0R1p5FeueHefjDm+GIM+kWfFnDjybOPgk= +github.com/mattn/go-runewidth v0.0.30/go.mod h1:3qAiGCV4Koz/yuveO58qUefmUTRm8r0IGEXZ9jeHp/8= +github.com/miekg/dns v1.1.73 h1:uhT8nJxmTrPJYClxVxTCX+CVn6qnzSiybRk72Z6DgrE= +github.com/miekg/dns v1.1.73/go.mod h1:RW2Obtfd5NZHvOFe3zYG0W8koWOQtAzyHaLo8vASBuQ= github.com/moby/docker-image-spec v1.3.1 h1:jMKff3w6PgbfSa69GfNg+zN/XLhfXJGnEx3Nl2EsFP0= github.com/moby/docker-image-spec v1.3.1/go.mod h1:eKmb5VW8vQEh/BAr2yvVNvuiJuY6UIocYsFu/DxxRpo= -github.com/moby/moby/api v1.55.0 h1:2/sexvQyqIWS8pRSCFddBfpW2qE7vR7FCL+vN8pxwMc= -github.com/moby/moby/api v1.55.0/go.mod h1:+RQ6wluLwtYaTd1WnPLykIDPekkuyD/ROWQClE83pzs= -github.com/moby/moby/client v0.5.1 h1:tYNaJno4c0HXz12y5BiqEDy0rVTYkWzI26lGvnTMiJw= -github.com/moby/moby/client v0.5.1/go.mod h1:odLstlZ6uSnfvAgVxMpvgmb8SUdd+siH2T0GBuxVAlM= +github.com/moby/moby/api v1.56.0 h1:GQzua3NA599ASSIICx0iFgiJeO9YkdDARvQsm23ZZuQ= +github.com/moby/moby/api v1.56.0/go.mod h1:sZ+THbVWkjOmBPPfbnzdD/G1LuIexWhqlSHHPTDQ1Uk= +github.com/moby/moby/client v0.6.0 h1:AJjEB21QPbXSXjDsZorFBoDZPhMrfbpaPLgSMAW9Bgs= +github.com/moby/moby/client v0.6.0/go.mod h1:OCo00wNRyA3m4lmJ228W3JbyCN4ZNNYjpOXiJydBdcQ= github.com/muesli/ansi v0.0.0-20230316100256-276c6243b2f6 h1:ZK8zHtRHOkbHy6Mmr5D264iyp3TiX5OmNcI5cIARiQI= github.com/muesli/ansi v0.0.0-20230316100256-276c6243b2f6/go.mod h1:CJlz5H+gyd6CUWT45Oy4q24RdLyn7Md9Vj2/ldJBSIo= github.com/muesli/cancelreader v0.2.2 h1:3I4Kt4BQjOR54NavqnDogx/MIoWBFa0StPA8ELUXHmA= @@ -166,8 +166,8 @@ github.com/rs/zerolog v1.35.1/go.mod h1:EjML9kdfa/RMA7h/6z6pYmq1ykOuA8/mjWaEvGI+ github.com/russross/blackfriday/v2 v2.1.0/go.mod h1:+Rmxgy9KzJVeS9/2gXHxylqXiyQDYRxCVz55jmeOWTM= github.com/sagikazarmark/locafero v0.12.0 h1:/NQhBAkUb4+fH1jivKHWusDYFjMOOKU88eegjfxfHb4= github.com/sagikazarmark/locafero v0.12.0/go.mod h1:sZh36u/YSZ918v0Io+U9ogLYQJ9tLLBmM4eneO6WwsI= -github.com/sirupsen/logrus v1.9.4 h1:TsZE7l11zFCLZnZ+teH4Umoq5BhEIfIzfRDZ1Uzql2w= -github.com/sirupsen/logrus v1.9.4/go.mod h1:ftWc9WdOfJ0a92nsE2jF5u5ZwH8Bv2zdeOC42RjbV2g= +github.com/sirupsen/logrus v1.10.2 h1:G2SED73/qrAu6YwbdxOD6peLkCBI3z7L+ykJFTXJBBo= +github.com/sirupsen/logrus v1.10.2/go.mod h1:SLEg8TqYulVKKfIGHldVp2K2aYz2DKSVBq4g/H5bR7Q= github.com/smallstep/truststore v0.13.0 h1:90if9htAOblavbMeWlqNLnO9bsjjgVv2hQeQJCi/py4= github.com/smallstep/truststore v0.13.0/go.mod h1:3tmMp2aLKZ/OA/jnFUB0cYPcho402UG2knuJoPh4j7A= github.com/spf13/afero v1.15.0 h1:b/YBCLWAJdFWJTN9cLhiXXcD7mzKn9Dm86dNnfyQw1I= @@ -187,8 +187,10 @@ github.com/stretchr/testify v1.12.1 h1:EuwCh5fleGS7H32xRwO3wRGT7DxrDhLAT6FF8MpWD github.com/stretchr/testify v1.12.1/go.mod h1:MDEgiDPPsNp5cuIrHPPCyornHKgEVbtFUmoNlxoYthg= github.com/subosito/gotenv v1.6.0 h1:9NlTDc1FTs4qu0DDq7AEtTPNw6SVm7uBMsUCUjABIf8= github.com/subosito/gotenv v1.6.0/go.mod h1:Dk4QP5c2W3ibzajGcXpNraDfq2IrhjMIvMSWPKKo0FU= -github.com/xo/terminfo v0.0.0-20220910002029-abceb7e1c41e h1:JVG44RsyaB9T2KIHavMF/ppJZNG9ZpyihvCd0w101no= -github.com/xo/terminfo v0.0.0-20220910002029-abceb7e1c41e/go.mod h1:RbqR21r5mrJuqunuUZ/Dhy/avygyECGrLceyNeo4LiM= +github.com/xo/terminfo v1.2.0 h1:d0ZTOCpuGE0lwSAOs0zcJjwz3jWQyqcRt9XbGJpCOl4= +github.com/xo/terminfo v1.2.0/go.mod h1:lGzkSo8Fe7IRh/w+Gqz7n5mDog4FVXCj3gy/DYTBqio= +go.etcd.io/bbolt v1.5.0 h1:S7GAl7Fxv12yohbwFfIbQCGDWbQbtDGPET4P/bD4lxU= +go.etcd.io/bbolt v1.5.0/go.mod h1:mkltfYE5aUHQxUct9N9V+Kp7aSjFqjgrhcXIS70Lrdk= go.opentelemetry.io/auto/sdk v1.2.1 h1:jXsnJ4Lmnqd11kwkBV2LgLoFMZKizbCi5fNZ/ipaZ64= go.opentelemetry.io/auto/sdk v1.2.1/go.mod h1:KRTj+aOaElaLi+wW1kO/DZRXwkF4C5xPbEe3ZiIhN7Y= go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.71.0 h1:3g7B90UzBltIDKq1/5mrTGxTnOFDV0ICOhLoxiZ8jlg= @@ -228,33 +230,27 @@ go.uber.org/mock v0.6.0/go.mod h1:KiVJ4BqZJaMj4svdfmHM0AUx4NJYO8ZNpPnZn1Z+BBU= go.yaml.in/yaml/v3 v3.0.4/go.mod h1:DhzuOOF2ATzADvBadXxruRBLzYTpT36CKvDb3+aBEFg= go.yaml.in/yaml/v3 v3.0.5 h1:N6y/pJk8buWs9NY5ERU2HSMfm+IuD/OtfdAnq6kESPw= go.yaml.in/yaml/v3 v3.0.5/go.mod h1:HVTZu1O7/Vkt2N+BFy8Zza+lnLsABggaTM2ZpNIGuKg= -golang.org/x/crypto v0.55.0 h1:+KWHjbgOaAQ66dh/YlkZKHlz9ZUlq61AFirAR9ntP8M= -golang.org/x/crypto v0.55.0/go.mod h1:uq0V9dE/fzQuJtbnL+2EhWOE63vo164FY8xqEnV9xis= -golang.org/x/exp v0.0.0-20260410095643-746e56fc9e2f h1:W3F4c+6OLc6H2lb//N1q4WpJkhzJCK5J6kUi1NTVXfM= -golang.org/x/exp v0.0.0-20260410095643-746e56fc9e2f/go.mod h1:J1xhfL/vlindoeF/aINzNzt2Bket5bjo9sdOYzOsU80= -golang.org/x/mod v0.40.0 h1:hUv+3cXcdRHz08UmSiOob7sadHig73uo5bkXxQ/tvUs= -golang.org/x/mod v0.40.0/go.mod h1:0/weTWkPWGBikyTWAX3dkjVztMmBA5hM0DH6BElSupE= -golang.org/x/net v0.58.0 h1:ynWG7rqYi4ccpTEuPZ2QGWHktVEM9DMCj9yzDE0Q7To= -golang.org/x/net v0.58.0/go.mod h1:YwCddHnFlT7eLQqVprV19OnhLGtc5xOKgE0RyqgfWAU= -golang.org/x/sync v0.22.0 h1:SZjpbeLmrCk4xhRSZFNZW5gFUeCeFgjekvI/+gfScek= -golang.org/x/sync v0.22.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0= +golang.org/x/crypto v0.57.0 h1:3ZVCjf8Ggz7zneR/EHRVx68Ctf+2pmIMP2UFhh9cC6M= +golang.org/x/crypto v0.57.0/go.mod h1:Fdz0i5U6CoizGwLda9DttjSk6qlZo25zYNtR+ycvuZA= +golang.org/x/net v0.59.0 h1:5zfYln+w5XCxwrnMMJPufRgNoXEaGxl0wo5GqPXyues= +golang.org/x/net v0.59.0/go.mod h1:2DA/G1UfVbCpQPeWTmMPGY7Cs2PkBkwu743bVX5PIVg= +golang.org/x/sync v0.23.0 h1:KameEIfc1IkluZyXWLn39Wd4tURc6GbCiISGiZm2bQk= +golang.org/x/sync v0.23.0/go.mod h1:sUUOizhqBxiL6pEWpqNLUiaJn1ShEbZ6BBqskPbjZm0= golang.org/x/sys v0.0.0-20210809222454-d867a43fc93e/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= -golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs= -golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw= -golang.org/x/text v0.41.0 h1:vz/seA0lnX87Othu2f/0L24RcgrXD9/YFTSuGjj3rH8= -golang.org/x/text v0.41.0/go.mod h1:jvf1O8ajNzZqhSrQBPbutR/EB83Cc0CFrezNQIwbb5M= -golang.org/x/time v0.15.0 h1:bbrp8t3bGUeFOx08pvsMYRTCVSMk89u4tKbNOZbp88U= -golang.org/x/time v0.15.0/go.mod h1:Y4YMaQmXwGQZoFaVFk4YpCt4FLQMYKZe9oeV/f4MSno= -golang.org/x/tools v0.49.0 h1:3NI7VXzL9+1WZD52Dx2ttoPwD5DWrFGpl9mFZDlmisI= -golang.org/x/tools v0.49.0/go.mod h1:SJNXV9DBKT0UbdttsQjbfJlAE/q+y36++zo3uL3N0Oo= +golang.org/x/sys v0.48.0 h1:bbX/i/6MgT9BVLM9RT1thmxL04yeTAhbEz4SyadbXoo= +golang.org/x/sys v0.48.0/go.mod h1:hNLxWAXmnKAxqDtdwIYC4bM9oQPEecfsnNMuSxOs3og= +golang.org/x/text v0.42.0 h1:JbOZXgfeCPU9gacVtYliJqOhD+zhrEqK4LfdpmlUZqI= +golang.org/x/text v0.42.0/go.mod h1:ojzP1Z+2QtioaF8DTtO8K5q7JWVVYwZKenzujK0Zd0E= +golang.org/x/time v0.16.0 h1:vMb6ptszcQMkcwiRTAuNNU50gom6++Q/6gY2hDM6VDE= +golang.org/x/time v0.16.0/go.mod h1:rVKOqvZeKvrDKTQiAHJ7wmwP0RzleSphoEA9RcdLA0s= gonum.org/v1/gonum v0.17.0 h1:VbpOemQlsSMrYmn7T2OUvQ4dqxQXU+ouZFQsZOx50z4= gonum.org/v1/gonum v0.17.0/go.mod h1:El3tOrEuMpv2UdMrbNlKEh9vd86bmQ6vqIcDwxEOc1E= -google.golang.org/genproto/googleapis/api v0.0.0-20260819154853-08b0e4226688 h1:ax2KzoSRIZU/M0cIxri3pKxy99vniH1PVxWC6si/eZI= -google.golang.org/genproto/googleapis/api v0.0.0-20260819154853-08b0e4226688/go.mod h1:1RJ9BQGyNdZwkGc1eTqkErfRZ6RJyYPHZo73BZ1vQqI= -google.golang.org/genproto/googleapis/rpc v0.0.0-20260819154853-08b0e4226688 h1:cYNAzI2sUwhmCcoj9TxvihSrqsxt6uIkj3rDRhSDmW4= -google.golang.org/genproto/googleapis/rpc v0.0.0-20260819154853-08b0e4226688/go.mod h1:DjtHYE8FKJLivXcBEjGwndXfIC23G0VpXiXKqG179uA= -google.golang.org/grpc v1.83.1 h1:HIO0+BEtBP6soyqvqC8sNUjZ7bTs+0hFQuFF+RAy++Y= -google.golang.org/grpc v1.83.1/go.mod h1:kDyl6SKsiHKt0uylY5gtn5cEjkrIOhQOGDgIc4JGwzQ= +google.golang.org/genproto/googleapis/api v0.0.0-20260921155816-b14227669459 h1:GS9OIt/j7c8bvBjYNgnKQysVfmV7e4jM0H8ZK95G4t8= +google.golang.org/genproto/googleapis/api v0.0.0-20260921155816-b14227669459/go.mod h1:PX5/4vemwVoXtwEcRDWwcR1/r0qrosfx3qoVADMwnVE= +google.golang.org/genproto/googleapis/rpc v0.0.0-20260921155816-b14227669459 h1:b0xCahf3FK2m2Cv0p4vTozGPWncCvLfwV86UNg8xWU8= +google.golang.org/genproto/googleapis/rpc v0.0.0-20260921155816-b14227669459/go.mod h1:OaIUM3+LpYcK2GXM4FTmhWoIq371Owdr+Cc7/BsYHHc= +google.golang.org/grpc v1.84.0 h1:soMyaPJ8pAak5PIQ0DGBUir0XRo2fRoMqhNWMLlLxO0= +google.golang.org/grpc v1.84.0/go.mod h1:ljCht0DrxQrXBDRTZp52Qxh3Ffk8CdYm2sj4O2QN2C0= google.golang.org/protobuf v1.36.12 h1:pJOKDDOyeXErUroCihFAd5LQuwXBSpVnKGrj5o/fwxc= google.golang.org/protobuf v1.36.12/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco= gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= diff --git a/gordon.toml.example b/gordon.toml.example index 7583b13ae..f5ef6955c 100644 --- a/gordon.toml.example +++ b/gordon.toml.example @@ -55,21 +55,15 @@ per_ip_rps = 50 burst = 100 trusted_proxies = [] -[deploy] -pull_policy = "if-tag-changed" -readiness_mode = "auto" -health_timeout = "90s" -readiness_delay = "5s" -drain_mode = "auto" -drain_timeout = "30s" -drain_delay = "2s" - [containers] # "compat" preserves broad image compatibility. "strict" enables a read-only # root filesystem and narrower Linux capabilities. security_profile = "compat" [network_isolation] +# Installation network policy for Gordon-managed networks (prefix filter +# for `gordon daemon networks`). Per-app isolation is removed: apps declare +# `[[network.shared]]` membership in their app file instead. enabled = true network_prefix = "gordon" # Set true to create Docker internal networks without direct external egress. @@ -81,10 +75,10 @@ prefix = "gordon" preserve = true [images] -# Explicit external registries must be listed here. Gordon's own registry is -# always allowed; localhost/private/link-local registries are always rejected. +# Docker Hub, ghcr.io, quay.io, and Gordon's registry are always allowed. +# Add other registries as exact hostname+port entries; this does not configure credentials. allowed_registries = [] -# Require @sha256: for allowlisted external registries. +# Require @sha256:<64 hex chars> for images from every registry, including Gordon. require_digest = false [images.prune] @@ -92,9 +86,6 @@ enabled = false schedule = "daily" keep_last = 3 -[auto_route] -enabled = false - [logging] level = "info" format = "console" @@ -106,13 +97,6 @@ enabled = false # max_backups = 3 # max_age = 28 -[logging.container_logs] -enabled = true -# dir = "~/.gordon/logs/containers" -# max_size = 100 -# max_backups = 3 -# max_age = 28 - [logging.access_log] enabled = false format = "json" @@ -124,6 +108,40 @@ max_age = 28 exclude_health_checks = true syslog_identifier = "gordon-access" +# ============================================================================= +# ADMINISTRATIVE APP BIND MOUNTS +# ============================================================================= +# A named policy is the only way an app manifest may reference a host path. +# The manifest declares [[services..bind]] name = ""; direct host paths +# are always rejected. allowed_apps/allowed_services are exact and non-empty, +# and read_only/readonly force read-only: either side wins. +# [app_mounts.app-logs] +# source = "/srv/gordon/host-logs" # Required; absolute, normalized +# read_only = true # Either side forcing read-only wins +# allowed_apps = ["metrics-agent"] # Required; exact, non-empty +# allowed_services = ["web"] # Required; exact, non-empty +# root = "/srv/gordon" # Optional; defaults to the source's parent + +# ============================================================================= +# ADMINISTRATIVE APP DEVICES (CDI) +# ============================================================================= +# A named device grant is the only way an app manifest may request host +# devices. The manifest declares devices = [""]; raw /dev paths, +# privileged containers, and runtime-specific GPU flags are never used. +# Gordon resolves each name to explicit CDI IDs at activation time and +# encodes them as one native CDI DeviceRequest. allowed_apps and +# allowed_services are exact and non-empty: only that app+service pair +# receives the grant. Aggregate "=all" selectors are rejected: name every +# device explicitly. Changing a mapping never recreates a running +# container; the next deploy serves the new resolution. Revoking a grant +# fails subsequent deploys closed while the running service is untouched. +# Supported engines: Podman 5.4+ or Docker 28.3+ with native CDI +# configured (CDI specs, toolkit, device permissions on the host). +# [app_devices.transcode-gpu] +# cdi = ["example.com/gpu=GPU-device-uuid"] # Required; explicit, non-empty +# allowed_apps = ["video"] # Required; exact, non-empty +# allowed_services = ["transcoder"] # Required; exact, non-empty + [telemetry] enabled = false # endpoint = "http://localhost:4318" @@ -133,9 +151,6 @@ metrics = true logs = true trace_sample_rate = 1.0 -[env] -# dir = "~/.gordon/env" - [backups.databases] enabled = false schedule = "daily" @@ -168,14 +183,33 @@ prefix = "" [backups.volumes.retention] keep = 14 -[routes] +# [routes] is retired (see REMOVED list below): declaring it fails boot/reload closed. # "app.example.com" = { image = "myapp:latest" } [external_routes] # "status.example.com" = "127.0.0.1:9000" +# +# NOTE: the pre-v3 application keys below were removed by the +# declarative-apps cutover. Gordon fails boot/reload closed when any of +# them is present (see validateRetiredAppConfig in internal/app/run.go). +# Declare application workloads in standalone .toml files instead: +# one file per app, applied with `gordon apps apply --file .toml` +# and activated with `gordon apps deploy `. +# +# REMOVED: [routes] -> [[services..http]] in an app file +# REMOVED: [attachments] -> [services.] + volumes in an app file +# REMOVED: [network_groups] -> [[network.shared]] in an app file +# REMOVED: app-like [[services]] -> one [services.] table per app file +# (installation [[services]] below stays valid for L4 workloads) +# REMOVED: [service_routes] -> [[services..http]] in an app file +# REMOVED: [auto]/[auto_route] -> explicit interfaces in an app file +# REMOVED: [previews] -> staging is an ordinary app file # Gordon-managed standalone L4 services. Use service:: from -# traffic routers to expose published loopback ports. +# traffic routers to expose published loopback ports. Standalone services +# are installation-level: they live in gordon.toml, not in app files. +# (App services are declared with [services.] tables in .toml files and +# managed with `gordon apps`; see docs/cli/apps.md.) # [[services]] # name = "rust" # image = "registry.example.com:5000/rust:latest" @@ -213,9 +247,6 @@ keep = 14 # name = "rust-rcon" # entrypoint = "rcon" # service = "service:rust:rcon" - -[network_groups] -# "backend" = ["app.example.com", "api.example.com"] - -[attachments] -# "app.example.com" = ["postgres:18"] +# +# REMOVED application sections (see note under [external_routes] above): +# [routes], [network_groups], [attachments] were deleted with the cutover. diff --git a/install.sh b/install.sh index 35ef1a9c1..021c5a733 100755 --- a/install.sh +++ b/install.sh @@ -2,12 +2,55 @@ set -e # Gordon installer script -# Usage: curl -fsSL https://gordon.bnema.dev/install | bash -# Usage with pre-release: curl -fsSL https://gordon.bnema.dev/install | GORDON_PRERELEASE=1 bash +# Usage: curl -fsSL https://bnema.dev/gordon/install | sh +# Usage with pre-release: curl -fsSL https://bnema.dev/gordon/install | GORDON_PRERELEASE=1 sh +# Usage with next source build: curl -fsSL https://bnema.dev/gordon/install | GORDON_CHANNEL=next sh REPO="bnema/gordon" -INSTALL_DIR="/usr/local/bin" VERSION="${GORDON_VERSION:-latest}" +CHANNEL="${GORDON_CHANNEL:-stable}" + +case "${GORDON_UPDATE_PATH:-}" in + ''|0|1) ;; + *) echo "Error: GORDON_UPDATE_PATH must be 0 or 1"; exit 1 ;; +esac + +if [ -n "${GORDON_INSTALL_DIR+x}" ]; then + INSTALL_DIR=$GORDON_INSTALL_DIR + if [ -z "$INSTALL_DIR" ]; then + echo "Error: GORDON_INSTALL_DIR must not be empty" + exit 1 + fi +else + if [ -n "${SUDO_USER:-}" ] || [ "$(id -u)" -eq 0 ]; then + echo "Error: Refusing a default user-local install while running as root or through sudo." + echo "Run the installer as the target user, or set GORDON_INSTALL_DIR explicitly for a global install." + exit 1 + fi + if [ -z "${HOME:-}" ]; then + echo "Error: HOME is required for the default installation directory" + exit 1 + fi + INSTALL_DIR=$HOME/.local/bin +fi + +case "$INSTALL_DIR" in + /*) ;; + *) echo "Error: GORDON_INSTALL_DIR must be an absolute path"; exit 1 ;; +esac +if [ "$INSTALL_DIR" != "/" ]; then + while [ "${INSTALL_DIR%/}" != "$INSTALL_DIR" ]; do + INSTALL_DIR=${INSTALL_DIR%/} + done +fi +INSTALL_DIR_WITHOUT_CONTROLS=$(LC_ALL=C printf '%sX' "$INSTALL_DIR" | LC_ALL=C tr -d '\000-\037\177') +if [ "$INSTALL_DIR" = "/" ] || [ "$INSTALL_DIR_WITHOUT_CONTROLS" != "${INSTALL_DIR}X" ]; then + echo "Error: GORDON_INSTALL_DIR contains an unsafe control character or destination" + exit 1 +fi +case "$INSTALL_DIR" in + *:*) echo "Error: GORDON_INSTALL_DIR must not contain ':' because it separates PATH entries"; exit 1 ;; +esac echo "Installing Gordon..." @@ -43,6 +86,259 @@ esac echo "Detected: ${OS}/${ARCH}" +install_binary_atomically() { + binary=$1 + destination=$2 + destination_dir=${destination%/*} + staging="${destination}.install.$$" + + if ! mkdir -p "$destination_dir" 2>/dev/null; then + if [ -z "${GORDON_INSTALL_DIR+x}" ]; then + echo "Error: Could not create user installation directory ${destination_dir}" + return 1 + fi + echo "sudo required to create ${destination_dir}" + sudo mkdir -p "$destination_dir" + fi + + if [ -w "$destination_dir" ]; then + trap 'rm -f "$staging"; rm -rf "${TMP_DIR:-}"' EXIT HUP INT TERM + install -m 0755 "$binary" "$staging" + mv -f "$staging" "$destination" + else + if [ -z "${GORDON_INSTALL_DIR+x}" ]; then + echo "Error: User installation directory is not writable: ${destination_dir}" + return 1 + fi + echo "sudo required to install to ${destination_dir}" + sudo install -m 0755 "$binary" "$staging" + if ! sudo mv -f "$staging" "$destination"; then + sudo rm -f "$staging" + return 1 + fi + fi +} + +path_contains_install_dir() { + case ":${PATH:-}:" in + *:"$INSTALL_DIR":*) return 0 ;; + *) return 1 ;; + esac +} + +single_quote() { + printf "'%s'" "$(printf '%s' "$1" | sed "s/'/'\\\\''/g")" +} + +print_path_instructions() { + quoted_dir=$(single_quote "$INSTALL_DIR") + echo "Add Gordon to the current shell with:" + case "${SHELL##*/}" in + fish) echo " fish_add_path ${quoted_dir}" ;; + *) echo " export PATH=${quoted_dir}:\$PATH" ;; + esac +} + +update_shell_path() { + shell_name=${SHELL##*/} + marker_start="# >>> Gordon installer PATH >>>" + marker_end="# <<< Gordon installer PATH <<<" + + case "$shell_name" in + fish) + config_file=$HOME/.config/fish/config.fish + config_dir=${config_file%/*} + command_line="fish_add_path $(single_quote "$INSTALL_DIR")" + ;; + bash) + config_file=$HOME/.bashrc + config_dir=$HOME + command_line="export PATH=$(single_quote "$INSTALL_DIR"):\$PATH" + ;; + zsh) + config_file=$HOME/.zshrc + config_dir=$HOME + command_line="export PATH=$(single_quote "$INSTALL_DIR"):\$PATH" + ;; + *) + echo "Could not update PATH automatically: unsupported or missing SHELL (${SHELL:-unset})." + print_path_instructions + return 0 + ;; + esac + + if ! mkdir -p "$config_dir" 2>/dev/null || + { [ -e "$config_file" ] && [ ! -w "$config_file" ]; } || + { [ ! -e "$config_file" ] && [ ! -w "$config_dir" ]; }; then + echo "Could not update PATH automatically: ${config_file} is not writable." + print_path_instructions + return 0 + fi + + if [ -f "$config_file" ] && grep -F "$marker_start" "$config_file" >/dev/null 2>&1; then + echo "PATH configuration already exists in ${config_file}." + else + { + printf '\n%s\n' "$marker_start" + printf '%s\n' "$command_line" + printf '%s\n' "$marker_end" + } >>"$config_file" + echo "Added ${INSTALL_DIR} to PATH in ${config_file}." + fi + print_path_instructions +} + +post_install() { + echo "" + echo "Gordon installed successfully at ${INSTALL_DIR}/gordon." + + if path_contains_install_dir; then + "$INSTALL_DIR/gordon" version + return + fi + + case "${GORDON_UPDATE_PATH:-}" in + 1) update_shell_path ;; + 0) print_path_instructions ;; + '') + if [ -r /dev/tty ] && [ -w /dev/tty ]; then + printf 'Add %s to your PATH configuration? [y/N] ' "$INSTALL_DIR" >/dev/tty + if IFS= read -r answer /dev/null 2>&1; then + echo "Error: GORDON_CHANNEL=next requires Go" + exit 1 + fi + + echo "WARNING: UNVERIFIED DEVELOPMENT BUILD" + echo "The next channel builds source locally from the current next branch commit." + echo "It is not covered by the signed release checksum path and may be unstable." + echo "Resolving the current next commit..." + COMMIT_DATA=$(curl -fsSL -H "Accept: application/vnd.github+json" \ + "https://api.github.com/repos/${REPO}/commits/next" 2>/dev/null || echo "") + COMMIT=$(printf '%s\n' "$COMMIT_DATA" | sed -n 's/.*"sha"[[:space:]]*:[[:space:]]*"\([0-9a-fA-F]*\)".*/\1/p' | head -n 1) + case "$COMMIT" in + *[!0-9a-fA-F]*|'') COMMIT="" ;; + esac + if [ "${#COMMIT}" -ne 40 ]; then + echo "Error: Could not resolve an exact next commit SHA from the GitHub API" + exit 1 + fi + + TMP_DIR=$(mktemp -d) + trap 'rm -rf "$TMP_DIR"' EXIT HUP INT TERM + SOURCE_TARBALL="$TMP_DIR/source.tar.gz" + SOURCE_URL="https://codeload.github.com/${REPO}/tar.gz/${COMMIT}" + echo "Downloading source pinned to ${COMMIT}..." + if ! curl -fsSL "$SOURCE_URL" -o "$SOURCE_TARBALL"; then + echo "Error: Failed to download source for ${COMMIT}" + exit 1 + fi + + SOURCE_ROOT="gordon-${COMMIT}" + if ! tar -tzf "$SOURCE_TARBALL" >"$TMP_DIR/archive.list"; then + echo "Error: Invalid source archive" + exit 1 + fi + if [ ! -s "$TMP_DIR/archive.list" ] || + grep -Ev "^${SOURCE_ROOT}(/.*)?$" "$TMP_DIR/archive.list" >/dev/null || + grep -E '(^|/)\.\.?(/|$)' "$TMP_DIR/archive.list" >/dev/null || + tar -tvzf "$SOURCE_TARBALL" | grep -Ev '^[-d]' >/dev/null; then + echo "Error: Source archive contains an unsafe path or entry type" + exit 1 + fi + tar -xzf "$SOURCE_TARBALL" -C "$TMP_DIR" + SOURCE_DIR="$TMP_DIR/$SOURCE_ROOT" + if [ ! -f "$SOURCE_DIR/go.mod" ] || [ ! -f "$SOURCE_DIR/main.go" ]; then + echo "Error: Source archive is missing go.mod or the root command" + exit 1 + fi + + REQUIRED_GO=$(awk '$1 == "go" { print $2; exit }' "$SOURCE_DIR/go.mod") + case "$REQUIRED_GO" in + [0-9]*.[0-9]*) ;; + *) echo "Error: Source go.mod has no valid Go version"; exit 1 ;; + esac + INSTALLED_GO=$(go env GOVERSION 2>/dev/null || true) + INSTALLED_GO=${INSTALLED_GO#go} + case "$INSTALLED_GO" in + [0-9]*.[0-9]*) ;; + *) echo "Error: Could not determine installed Go version"; exit 1 ;; + esac + if ! version_at_least "$INSTALLED_GO" "$REQUIRED_GO"; then + echo "Error: Source requires Go ${REQUIRED_GO} or newer; found ${INSTALLED_GO}" + exit 1 + fi + + BUILD_DATE=$(date -u '+%Y-%m-%dT%H:%M:%SZ') + NEXT_VERSION="next-${COMMIT}" + BINARY="$TMP_DIR/gordon" + echo "Building ${NEXT_VERSION} for ${OS}/${ARCH}..." + # The source is a pristine tarball of the pinned commit, so it can + # never be a dirty checkout: dirty is reported as false. + if ! (cd "$SOURCE_DIR" && CGO_ENABLED=0 GOOS="$OS" GOARCH="$ARCH" GOTOOLCHAIN=local \ + go build -trimpath -ldflags "-s -w -X main.version=${NEXT_VERSION} -X main.commit=${COMMIT} -X main.date=${BUILD_DATE} -X main.dirty=false" \ + -o "$BINARY" .); then + echo "Error: Failed to build Gordon from source" + exit 1 + fi + if [ ! -f "$BINARY" ] || [ -L "$BINARY" ]; then + echo "Error: Build did not produce a regular Gordon binary" + exit 1 + fi + + echo "Installing to ${INSTALL_DIR}..." + install_binary_atomically "$BINARY" "$INSTALL_DIR/gordon" + echo "Installed unverified next build ${COMMIT}." +} + +case "$CHANNEL" in + next) + install_next + post_install + exit 0 + ;; + stable|'') ;; + *) echo "Error: Unsupported GORDON_CHANNEL: $CHANNEL"; exit 1 ;; +esac + # Construct download URL TARBALL="gordon_${OS}_${ARCH}.tar.gz" @@ -101,6 +397,8 @@ fi # Verify checksum echo "Verifying checksum..." +# The sed expression is intentionally literal; shell expansion would corrupt it. +# shellcheck disable=SC2016 TARBALL_ESCAPED=$(printf '%s\n' "$TARBALL" | sed 's/[.[\*^$()+?{|]/\\&/g') CHECKSUM_LINES=$(grep -E "^[0-9a-fA-F]{64}[[:space:]]+\*?${TARBALL_ESCAPED}\$" "$TMP_DIR/checksums.txt" || true) CHECKSUM_COUNT=$(printf '%s\n' "$CHECKSUM_LINES" | sed '/^$/d' | wc -l | tr -d ' ') @@ -147,19 +445,6 @@ if [ ! -f "$BINARY" ] || [ -L "$BINARY" ]; then fi echo "Installing to ${INSTALL_DIR}..." -if [ -w "$INSTALL_DIR" ]; then - install -m 0755 "$BINARY" "$INSTALL_DIR/gordon" -else - echo "sudo required to install to ${INSTALL_DIR}" - sudo install -m 0755 "$BINARY" "$INSTALL_DIR/gordon" -fi +install_binary_atomically "$BINARY" "$INSTALL_DIR/gordon" -# Verify installation -if command -v gordon >/dev/null 2>&1; then - echo "" - echo "Gordon installed successfully!" - gordon version -else - echo "" - echo "Installation complete. You may need to add ${INSTALL_DIR} to your PATH." -fi +post_install diff --git a/internal/adapters/dto/admin_apps.go b/internal/adapters/dto/admin_apps.go new file mode 100644 index 000000000..226e50de1 --- /dev/null +++ b/internal/adapters/dto/admin_apps.go @@ -0,0 +1,203 @@ +package dto + +import "time" + +// App admin DTOs use stable wire-facing field names. +// No secret values exist in any shape here: secret references carry +// paths and env keys only. The versioned error envelope is a deliberate +// change from ErrorResponse for app mutations; clients accept both +// during the mixed-version window. + +// AppError is the v3 app-mutation error envelope. Its stable fields are +// error/message/cause/hint. +// `logs` never appears on mutations. Clients accept the legacy +// single-field ErrorResponse during the mixed-version window. +type AppError struct { + Error string `json:"error"` + Message string `json:"message,omitempty"` + Cause string `json:"cause,omitempty"` + Hint string `json:"hint,omitempty"` +} + +// AppDiffSection carries normalized added/removed/changed paths. +type AppDiffSection struct { + Added []string `json:"added"` + Removed []string `json:"removed"` + Changed []string `json:"changed"` +} + +// AppApplyRequest asks the daemon to validate and persist desired state. +type AppApplyRequest struct { + ManifestTOML string `json:"manifest_toml"` + DryRun bool `json:"dry_run,omitempty"` +} + +// AppApplyResponse reports persistence success separately from deploy. +type AppApplyResponse struct { + App string `json:"app"` + FormerRevision string `json:"former_revision,omitempty"` + ResultingRevision string `json:"resulting_revision,omitempty"` + Pending bool `json:"pending"` + Noop bool `json:"noop,omitempty"` + Diff AppDiffSection `json:"diff"` + Intent string `json:"intent,omitempty"` +} + +// AppDeployRequest activates a captured revision. +type AppDeployRequest struct { + Revision string `json:"revision,omitempty"` + Service string `json:"service,omitempty"` + // All confirms an app-wide deploy of a multi-service app. + All bool `json:"all,omitempty"` +} + +// AppServiceResultDTO is one service's terminal deployment result. +type AppServiceResultDTO struct { + Result string `json:"result"` + EffectiveRevision string `json:"effective_revision"` + Before string `json:"before,omitempty"` + After string `json:"after,omitempty"` + RestartUnsafe bool `json:"restart_unsafe"` + Error string `json:"error,omitempty"` +} + +// AppStepDTO is one journaled operation step. +type AppStepDTO struct { + ID string `json:"id"` + State string `json:"state"` + Detail string `json:"detail,omitempty"` + Error string `json:"error,omitempty"` + Before string `json:"before,omitempty"` + After string `json:"after,omitempty"` + // Diagnostics is redacted failure output, populated only for callers + // holding the logs read scope. + Diagnostics []string `json:"diagnostics,omitempty"` +} + +// AppCleanupWarningDTO records post-publication leftovers. +type AppCleanupWarningDTO struct { + Service string `json:"service"` + Leftover string `json:"leftover"` + Detail string `json:"detail"` +} + +// AppEffectiveDTO maps services to effective revisions. +type AppEffectiveDTO struct { + Converged bool `json:"converged"` + ConvergedRevision string `json:"converged_revision,omitempty"` + Services map[string]string `json:"services"` +} + +// AppRetainedDTO lists the resources an app owns and would retain on +// removal (names and paths, never values). +type AppRetainedDTO struct { + Volumes []string `json:"volumes"` + Secrets []string `json:"secrets"` + Images []string `json:"images"` +} + +// AppSecretMetadataDTO reports registration metadata without secret values. +type AppSecretMetadataDTO struct { + Service string `json:"service"` + Key string `json:"key"` + Name string `json:"name"` + Source string `json:"source"` + Presence string `json:"presence"` +} + +// AppStatusRunning is the wire status of an operation whose effects are +// still executing in the background. A terminal operation reports its +// persisted outcome as the status instead. +const AppStatusRunning = "running" + +// AppDeployResponse reports deploy outcome with terminal results. +// Effective and Retained are filled from current app state when the app +// still exists; they are omitted when it does not. +type AppDeployResponse struct { + Op string `json:"op"` + App string `json:"app"` + Revision string `json:"revision"` + Status string `json:"status"` + Outcome string `json:"outcome"` + Services map[string]AppServiceResultDTO `json:"services"` + Steps []AppStepDTO `json:"steps"` + CleanupWarnings []AppCleanupWarningDTO `json:"cleanup_warnings,omitempty"` + Effective *AppEffectiveDTO `json:"effective,omitempty"` + Retained *AppRetainedDTO `json:"retained,omitempty"` +} + +// AppDesiredDTO summarizes desired state. +type AppDesiredDTO struct { + Revision string `json:"revision,omitempty"` + Status string `json:"status,omitempty"` + // Pending reports desired state ACTIVE has not reached yet. + Pending bool `json:"pending"` +} + +// AppActiveServiceDTO is one effective service for inspection. +type AppActiveServiceDTO struct { + EffectiveRevision string `json:"effective_revision"` + Digest string `json:"digest,omitempty"` + Container string `json:"container,omitempty"` + RestartUnsafe bool `json:"restart_unsafe"` +} + +// AppActiveDTO carries per-service effective state. +type AppActiveDTO struct { + Converged bool `json:"converged"` + ConvergedRevision string `json:"converged_revision,omitempty"` + Services map[string]AppActiveServiceDTO `json:"services"` +} + +// AppIntentDTO carries durable stopped/running intent. +type AppIntentDTO struct { + Stopped bool `json:"stopped"` +} + +// AppLastOpDTO references the latest operation. +type AppLastOpDTO struct { + Op string `json:"op"` + Kind string `json:"kind,omitempty"` + Outcome string `json:"outcome,omitempty"` + StartedAt time.Time `json:"started_at,omitzero"` +} + +// AppShowResponse inspects desired + active + intent + ownership + op ref. +type AppShowResponse struct { + App string `json:"app"` + Desired AppDesiredDTO `json:"desired"` + Active AppActiveDTO `json:"active"` + Intent AppIntentDTO `json:"intent"` + Retained AppRetainedDTO `json:"retained"` + LastOp *AppLastOpDTO `json:"last_op,omitempty"` +} + +// AppSummaryDTO is one row of the app list. +type AppSummaryDTO struct { + App string `json:"app"` + Desired string `json:"desired,omitempty"` + DesiredStatus string `json:"desired_status,omitempty"` + Active string `json:"active,omitempty"` + Converged bool `json:"converged"` + Pending bool `json:"pending"` + Stopped bool `json:"stopped"` + LastOutcome string `json:"last_outcome,omitempty"` +} + +// AppDiffResponse reports normalized desired-vs-active diff. +type AppDiffResponse struct { + App string `json:"app"` + Diff AppDiffSection `json:"diff"` +} + +// AppSecretSetRequest writes secret values (names pre-registered). +type AppSecretSetRequest struct { + Service string `json:"service"` + Secrets map[string]string `json:"secrets"` +} + +// AppSecretDeleteRequest removes one secret value. +type AppSecretDeleteRequest struct { + Service string `json:"service"` + Key string `json:"key"` +} diff --git a/internal/adapters/dto/admin_apps_test.go b/internal/adapters/dto/admin_apps_test.go new file mode 100644 index 000000000..7a26952c7 --- /dev/null +++ b/internal/adapters/dto/admin_apps_test.go @@ -0,0 +1,148 @@ +package dto + +// Frozen-wire tests for the v3 app admin DTOs pin the exact JSON field +// names the daemon and the CLI exchange, and prove no response shape can +// carry secret values. The single intentional exception is +// AppSecretSetRequest: the write path must transport values. + +import ( + "encoding/json" + "strings" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func dtoJSONKeys(t *testing.T, v any) map[string]any { + t.Helper() + raw, err := json.Marshal(v) + require.NoError(t, err) + var decoded map[string]any + require.NoError(t, json.Unmarshal(raw, &decoded)) + return decoded +} + +func TestAppErrorEnvelopeKeys(t *testing.T) { + envelope := AppError{Error: "secret-missing", Message: "db password missing", Cause: "pass lookup failed", Hint: "set it first"} + keys := dtoJSONKeys(t, envelope) + assert.Equal(t, "secret-missing", keys["error"]) + assert.Equal(t, "db password missing", keys["message"]) + assert.Equal(t, "pass lookup failed", keys["cause"]) + assert.Equal(t, "set it first", keys["hint"]) + assert.NotContains(t, keys, "logs", "mutations must never carry logs") + + // Mixed-version window: the legacy single-field shape still decodes. + var legacy AppError + require.NoError(t, json.Unmarshal([]byte(`{"error":"image-unresolvable"}`), &legacy)) + assert.Equal(t, "image-unresolvable", legacy.Error) + assert.Empty(t, legacy.Message) +} + +func TestAppApplyResponseKeys(t *testing.T) { + resp := AppApplyResponse{ + App: "blog", + FormerRevision: "rev-old", + ResultingRevision: "rev-new", + Pending: true, + Diff: AppDiffSection{Added: []string{"service.web"}, Removed: []string{}, Changed: []string{}}, + Intent: "apply-abc", + } + keys := dtoJSONKeys(t, resp) + for _, key := range []string{"app", "former_revision", "resulting_revision", "pending", "diff", "intent"} { + assert.Contains(t, keys, key) + } + diff := keys["diff"].(map[string]any) + for _, key := range []string{"added", "removed", "changed"} { + assert.Contains(t, diff, key) + } +} + +func TestAppDeployResponseKeys(t *testing.T) { + resp := AppDeployResponse{ + Op: "op-abc", + App: "blog", + Revision: "rev-new", + Status: AppStatusRunning, + Outcome: "success", + Services: map[string]AppServiceResultDTO{ + "web": {Result: "deployed", EffectiveRevision: "rev-new"}, + }, + Steps: []AppStepDTO{{ID: "service.web.replace", State: "succeeded"}}, + CleanupWarnings: []AppCleanupWarningDTO{ + {Service: "web", Leftover: "ctr-old", Detail: "retire failed after publish"}, + }, + Effective: &AppEffectiveDTO{Converged: true, Services: map[string]string{"web": "rev-new"}}, + Retained: &AppRetainedDTO{Volumes: []string{"blog-db"}, Secrets: []string{"db/password"}}, + } + keys := dtoJSONKeys(t, resp) + for _, key := range []string{"op", "app", "revision", "status", "outcome", "services", "steps", "cleanup_warnings", "effective", "retained"} { + assert.Contains(t, keys, key) + } + svc := keys["services"].(map[string]any)["web"].(map[string]any) + for _, key := range []string{"result", "effective_revision", "restart_unsafe"} { + assert.Contains(t, svc, key) + } +} + +func TestAppShowResponseKeys(t *testing.T) { + resp := AppShowResponse{ + App: "blog", + Desired: AppDesiredDTO{Revision: "rev-new", Status: "pending", Pending: true}, + Active: AppActiveDTO{Converged: false, Services: map[string]AppActiveServiceDTO{ + "web": {EffectiveRevision: "rev-old", Digest: "sha256:abc", Container: "ctr-old", RestartUnsafe: true}, + }}, + Intent: AppIntentDTO{Stopped: false}, + LastOp: &AppLastOpDTO{Op: "op-abc", Outcome: "success"}, + } + keys := dtoJSONKeys(t, resp) + for _, key := range []string{"app", "desired", "active", "intent", "retained", "last_op"} { + assert.Contains(t, keys, key) + } + web := keys["active"].(map[string]any)["services"].(map[string]any)["web"].(map[string]any) + for _, key := range []string{"effective_revision", "digest", "container", "restart_unsafe"} { + assert.Contains(t, web, key) + } + assert.NotContains(t, web, "env", "active services must not inline env values") +} + +// TestAppResponsesCarryNoSecretValues marshals representative responses and +// proves a sentinel secret value appears nowhere. The request that carries +// values (AppSecretSetRequest) is asserted as the single exception. +func TestAppResponsesCarryNoSecretValues(t *testing.T) { + const sentinel = "s3cr3t-value-sentinel" + + responses := []any{ + AppApplyResponse{App: "blog", ResultingRevision: "rev-x", Intent: "apply-x"}, + AppDeployResponse{ + Op: "op-x", App: "blog", Revision: "rev-x", Outcome: "failed", + Services: map[string]AppServiceResultDTO{ + "web": {Result: "failed", EffectiveRevision: "rev-x", Error: "boom"}, + }, + Effective: &AppEffectiveDTO{Services: map[string]string{"web": "rev-x"}}, + Retained: &AppRetainedDTO{Volumes: []string{"blog-db"}, Secrets: []string{"db/password"}}, + }, + AppShowResponse{ + App: "blog", + Desired: AppDesiredDTO{Revision: "rev-x", Status: "active"}, + Active: AppActiveDTO{Services: map[string]AppActiveServiceDTO{ + "web": {EffectiveRevision: "rev-x", Digest: "sha256:x", Container: "ctr-x"}, + }}, + }, + AppSummaryDTO{App: "blog", Desired: "rev-x", Active: "rev-x"}, + AppDiffResponse{App: "blog", Diff: AppDiffSection{Changed: []string{"secret.db/password"}}}, + } + for i, resp := range responses { + raw, err := json.Marshal(resp) + require.NoError(t, err, "response %d", i) + body := string(raw) + assert.NotContains(t, body, sentinel, "response %d leaks a secret value", i) + assert.NotContains(t, strings.ToLower(body), `"value"`, "response %d has a value field", i) + } + + // The one intentional exception: the set-secret request transports values. + setReq := AppSecretSetRequest{Service: "web", Secrets: map[string]string{"password": sentinel}} + raw, err := json.Marshal(setReq) + require.NoError(t, err) + assert.Contains(t, string(raw), sentinel) +} diff --git a/internal/adapters/dto/admin_attachments.go b/internal/adapters/dto/admin_attachments.go deleted file mode 100644 index 7cca53db5..000000000 --- a/internal/adapters/dto/admin_attachments.go +++ /dev/null @@ -1,29 +0,0 @@ -// Package dto provides shared data transfer objects for API responses. -package dto - -// AttachmentsConfigResponse represents all configured attachments. -type AttachmentsConfigResponse struct { - Attachments map[string][]string `json:"attachments"` -} - -// AttachmentConfigResponse represents attachments for a specific target. -type AttachmentConfigResponse struct { - Target string `json:"target"` - Images []string `json:"images"` -} - -// AttachmentTargetsByImageResponse represents attachment targets for a specific image. -type AttachmentTargetsByImageResponse struct { - Image string `json:"image"` - Targets []string `json:"targets"` -} - -// AttachmentAddRequest represents a request to add an attachment. -type AttachmentAddRequest struct { - Image string `json:"image"` -} - -// AttachmentStatusResponse represents attachment operation result. -type AttachmentStatusResponse struct { - Status string `json:"status"` -} diff --git a/internal/adapters/dto/admin_autoroute.go b/internal/adapters/dto/admin_autoroute.go deleted file mode 100644 index 08ebe2591..000000000 --- a/internal/adapters/dto/admin_autoroute.go +++ /dev/null @@ -1,16 +0,0 @@ -package dto - -// AutoRouteAllowedDomainsResponse represents the list of allowed domains. -type AutoRouteAllowedDomainsResponse struct { - Domains []string `json:"domains"` -} - -// AutoRouteAllowedDomainRequest represents a request to add/remove a domain pattern. -type AutoRouteAllowedDomainRequest struct { - Pattern string `json:"pattern"` -} - -// AutoRouteStatusResponse represents a status response for auto-route operations. -type AutoRouteStatusResponse struct { - Status string `json:"status"` -} diff --git a/internal/adapters/dto/admin_backups.go b/internal/adapters/dto/admin_backups.go index 33402a23e..48c2f7bbb 100644 --- a/internal/adapters/dto/admin_backups.go +++ b/internal/adapters/dto/admin_backups.go @@ -2,11 +2,13 @@ package dto import "time" -// BackupJob represents backup metadata in admin API responses. +// BackupJob represents backup metadata in admin API responses. App is the +// canonical backup identity: a domain is never used. type BackupJob struct { ID string `json:"id"` - Domain string `json:"domain"` - DBName string `json:"db_name"` + App string `json:"app"` + Service string `json:"service,omitempty"` + Database string `json:"database,omitempty"` Schedule string `json:"schedule,omitempty"` Type string `json:"type"` Status string `json:"status"` @@ -21,9 +23,12 @@ type BackupsResponse struct { Backups []BackupJob `json:"backups"` } -// BackupRunRequest triggers a backup run. +// BackupRunRequest triggers one declared database backup. Service and +// Database are explicit selectors; an omitted selector only succeeds when +// exactly one compatible target exists. type BackupRunRequest struct { - DB string `json:"db,omitempty"` + Service string `json:"service"` + Database string `json:"database,omitempty"` } // BackupRunResponse is returned after triggering a backup. @@ -32,38 +37,23 @@ type BackupRunResponse struct { Backup *BackupJob `json:"backup,omitempty"` } -// DatabaseInfo represents a detected database attachment. -type DatabaseInfo struct { - Type string `json:"type"` - Name string `json:"name"` - Version string `json:"version,omitempty"` - Host string `json:"host"` - Port int `json:"port"` - ContainerID string `json:"container_id"` - ImageName string `json:"image_name"` -} - -// BackupDetectResponse is returned by detect endpoints. -type BackupDetectResponse struct { - Databases []DatabaseInfo `json:"databases"` -} - -// VolumeBackupJob represents volume backup metadata in admin API responses. +// VolumeBackupJob represents volume backup metadata in admin API +// responses. App, service, and the declared volume name are the identity. type VolumeBackupJob struct { - ID string `json:"id"` - Domain string `json:"domain"` - ContainerName string `json:"container_name,omitempty"` - ContainerID string `json:"container_id,omitempty"` - VolumeName string `json:"volume_name"` - MountPath string `json:"mount_path,omitempty"` - Compression string `json:"compression,omitempty"` - Type string `json:"type"` - Status string `json:"status"` - StartedAt *time.Time `json:"started_at,omitempty"` - CompletedAt *time.Time `json:"completed_at,omitempty"` - SizeBytes int64 `json:"size_bytes"` - ArtifactRef string `json:"artifact_ref,omitempty"` - Error string `json:"error,omitempty"` + ID string `json:"id"` + App string `json:"app"` + Service string `json:"service,omitempty"` + VolumeName string `json:"volume"` + RuntimeVolumeName string `json:"runtime_volume,omitempty"` + MountPath string `json:"mount_path,omitempty"` + Compression string `json:"compression,omitempty"` + Type string `json:"type"` + Status string `json:"status"` + StartedAt *time.Time `json:"started_at,omitempty"` + CompletedAt *time.Time `json:"completed_at,omitempty"` + SizeBytes int64 `json:"size_bytes"` + ArtifactRef string `json:"artifact_ref,omitempty"` + Error string `json:"error,omitempty"` } // VolumeBackupsResponse is returned by volume backup listing endpoints. @@ -71,12 +61,13 @@ type VolumeBackupsResponse struct { Backups []VolumeBackupJob `json:"backups"` } -// VolumeBackupRunRequest triggers volume backups. +// VolumeBackupRunRequest triggers one declared volume backup. type VolumeBackupRunRequest struct { - Volume string `json:"volume,omitempty"` + Service string `json:"service"` + Volume string `json:"volume,omitempty"` } -// VolumeBackupRunResponse is returned after triggering volume backups. +// VolumeBackupRunResponse is returned after triggering a volume backup. type VolumeBackupRunResponse struct { Status string `json:"status"` Backups []VolumeBackupJob `json:"backups,omitempty"` diff --git a/internal/adapters/dto/admin_backups_test.go b/internal/adapters/dto/admin_backups_test.go new file mode 100644 index 000000000..99c2febb4 --- /dev/null +++ b/internal/adapters/dto/admin_backups_test.go @@ -0,0 +1,37 @@ +package dto + +import ( + "encoding/json" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// TestBackupDTOsUseAppIdentity proves the wire shape identifies a backup +// by app, service, and database or volume: a domain is never part of it. +func TestBackupDTOsUseAppIdentity(t *testing.T) { + job := BackupJob{ID: "job-1", App: "shop", Service: "api", Database: "orders", Schedule: "daily", Status: "completed"} + volume := VolumeBackupJob{ID: "job-2", App: "shop", Service: "api", VolumeName: "data", RuntimeVolumeName: "gordon-shop--api--vol--data", Status: "completed"} + + for name, payload := range map[string]any{"database job": job, "volume job": volume} { + t.Run(name, func(t *testing.T) { + raw, err := json.Marshal(payload) + require.NoError(t, err) + var decoded map[string]any + require.NoError(t, json.Unmarshal(raw, &decoded)) + for _, key := range []string{"id", "app", "service"} { + assert.Contains(t, decoded, key) + } + assert.NotContains(t, decoded, "domain", "a domain is never a backup identity") + }) + } + + runRequest, err := json.Marshal(BackupRunRequest{Service: "api", Database: "orders"}) + require.NoError(t, err) + assert.JSONEq(t, `{"service":"api","database":"orders"}`, string(runRequest)) + + volumeRequest, err := json.Marshal(VolumeBackupRunRequest{Service: "api", Volume: "data"}) + require.NoError(t, err) + assert.JSONEq(t, `{"service":"api","volume":"data"}`, string(volumeRequest)) +} diff --git a/internal/adapters/dto/admin_bootstrap.go b/internal/adapters/dto/admin_bootstrap.go deleted file mode 100644 index 24f788338..000000000 --- a/internal/adapters/dto/admin_bootstrap.go +++ /dev/null @@ -1,25 +0,0 @@ -package dto - -// BootstrapRequest represents a bootstrap operation request. -type BootstrapRequest struct { - Domain string `json:"domain"` - Image string `json:"image"` - Attachments []string `json:"attachments,omitempty"` - Env map[string]string `json:"env,omitempty"` - AttachmentEnv map[string]map[string]string `json:"attachment_env,omitempty"` -} - -// BootstrapStep represents the result of one bootstrap step. -type BootstrapStep struct { - Name string `json:"name"` - Status string `json:"status"` // "created", "configured", "updated", "noop", "failed" -} - -// BootstrapResponse represents the result of a bootstrap operation. -type BootstrapResponse struct { - Domain string `json:"domain"` - Image string `json:"image"` - Steps []BootstrapStep `json:"steps"` - Warnings []string `json:"warnings,omitempty"` - Next string `json:"next"` -} diff --git a/internal/adapters/dto/admin_config.go b/internal/adapters/dto/admin_config.go index 17fe4c364..68ca3feb7 100644 --- a/internal/adapters/dto/admin_config.go +++ b/internal/adapters/dto/admin_config.go @@ -1,12 +1,12 @@ package dto // ConfigResponse represents server configuration. +// ConfigResponse represents installation configuration. App routes live +// under apps show, never here. type ConfigResponse struct { Server ServerConfig `json:"server"` - AutoRoute AutoRouteConfig `json:"auto_route"` NetworkIsolation NetworkIsolationConfig `json:"network_isolation"` Volumes VolumesConfig `json:"volumes"` - Routes []Route `json:"routes"` ExternalRoutes []ExternalRoute `json:"external_routes"` } @@ -18,11 +18,6 @@ type ServerConfig struct { DataDir string `json:"data_dir,omitempty"` } -// AutoRouteConfig represents auto-route config details. -type AutoRouteConfig struct { - Enabled bool `json:"enabled"` -} - // NetworkIsolationConfig represents network isolation settings. type NetworkIsolationConfig struct { Enabled bool `json:"enabled"` diff --git a/internal/adapters/dto/admin_deploy.go b/internal/adapters/dto/admin_deploy.go deleted file mode 100644 index a6b088406..000000000 --- a/internal/adapters/dto/admin_deploy.go +++ /dev/null @@ -1,8 +0,0 @@ -package dto - -// DeployResponse represents a deployment response. -type DeployResponse struct { - Status string `json:"status"` - ContainerID string `json:"container_id"` - Domain string `json:"domain"` -} diff --git a/internal/adapters/dto/admin_images.go b/internal/adapters/dto/admin_images.go index 2517490de..c7b108656 100644 --- a/internal/adapters/dto/admin_images.go +++ b/internal/adapters/dto/admin_images.go @@ -17,11 +17,13 @@ type ImagesResponse struct { Images []Image `json:"images"` } -// ImagePruneRequest triggers image pruning. +// ImagePruneRequest triggers image pruning. DryRun plans the full +// operation and deletes nothing. type ImagePruneRequest struct { KeepLast *int `json:"keep_last,omitempty"` PruneDangling *bool `json:"prune_dangling,omitempty"` PruneRegistry *bool `json:"prune_registry,omitempty"` + DryRun *bool `json:"dry_run,omitempty"` } // RuntimePruneResult represents runtime prune results. @@ -37,8 +39,10 @@ type RegistryPruneResult struct { SpaceReclaimed int64 `json:"space_reclaimed"` } -// ImagePruneResponse is returned by image prune endpoints. +// ImagePruneResponse is returned by image prune endpoints. Plan is the +// same shape for dry runs and executions. type ImagePruneResponse struct { Runtime RuntimePruneResult `json:"runtime"` Registry RegistryPruneResult `json:"registry"` + Plan PruneSummary `json:"plan"` } diff --git a/internal/adapters/dto/admin_restart.go b/internal/adapters/dto/admin_restart.go deleted file mode 100644 index b142a0c41..000000000 --- a/internal/adapters/dto/admin_restart.go +++ /dev/null @@ -1,7 +0,0 @@ -package dto - -// RestartResponse represents a restart response. -type RestartResponse struct { - Status string `json:"status"` - Domain string `json:"domain"` -} diff --git a/internal/adapters/dto/admin_secrets.go b/internal/adapters/dto/admin_secrets.go deleted file mode 100644 index eeeda9f6c..000000000 --- a/internal/adapters/dto/admin_secrets.go +++ /dev/null @@ -1,19 +0,0 @@ -package dto - -// AttachmentSecretsResponse represents secrets for an attachment container. -type AttachmentSecretsResponse struct { - Service string `json:"service"` - Keys []string `json:"keys"` -} - -// SecretsListResponse represents a list of secret keys for a domain. -type SecretsListResponse struct { - Domain string `json:"domain"` - Keys []string `json:"keys"` - Attachments []AttachmentSecretsResponse `json:"attachments,omitempty"` -} - -// SecretsStatusResponse represents a secrets write response. -type SecretsStatusResponse struct { - Status string `json:"status"` -} diff --git a/internal/adapters/dto/admin_status.go b/internal/adapters/dto/admin_status.go index 317cdb026..94790538d 100644 --- a/internal/adapters/dto/admin_status.go +++ b/internal/adapters/dto/admin_status.go @@ -2,11 +2,10 @@ package dto // StatusResponse represents admin status information. type StatusResponse struct { - Routes int `json:"routes"` + Apps int `json:"apps"` RegistryDomain string `json:"registry_domain"` RegistryPort int `json:"registry_port"` ServerPort int `json:"server_port"` - AutoRoute bool `json:"auto_route"` NetworkIsolation bool `json:"network_isolation"` ContainerStatuses map[string]string `json:"container_status"` } diff --git a/internal/adapters/dto/prune.go b/internal/adapters/dto/prune.go new file mode 100644 index 000000000..10908073f --- /dev/null +++ b/internal/adapters/dto/prune.go @@ -0,0 +1,102 @@ +package dto + +import "github.com/bnema/gordon/internal/domain" + +// PruneCandidate is one planned prune candidate with its verdict and the +// stable reason codes behind it. +type PruneCandidate struct { + Kind string `json:"kind"` + Ref string `json:"ref"` + Verdict string `json:"verdict"` + Reasons []string `json:"reasons"` +} + +// PruneGap is one inventory that could not be read in full. A gap never +// fails the operation; it makes the dependent candidates unknown. +type PruneGap struct { + Source string `json:"source"` + Reason string `json:"reason"` + Detail string `json:"detail,omitempty"` +} + +// PruneFailure is one deletion that failed without aborting the run. +type PruneFailure struct { + Kind string `json:"kind"` + Ref string `json:"ref"` + Error string `json:"error"` +} + +// PruneSummary is the shared prune report shape: dry-run and execution +// return the same structure, distinguished by Applied. +type PruneSummary struct { + // Applied is false for a dry run, where nothing was deleted. + Applied bool `json:"applied"` + // Eligible, Protected, and Unknown count the planned candidates. + Eligible int `json:"eligible"` + Protected int `json:"protected"` + Unknown int `json:"unknown"` + // Candidates lists every planned candidate with its verdict, so an + // operator can see why each resource was kept or skipped. + Candidates []PruneCandidate `json:"candidates"` + // Deleted lists exactly the identities that were removed. + Deleted []PruneCandidate `json:"deleted"` + // Failures lists deletions that failed without aborting the run. + Failures []PruneFailure `json:"failures,omitempty"` + // Gaps lists every inventory that could not be read in full. + Gaps []PruneGap `json:"gaps,omitempty"` + // ReclaimedBytes is set only when the adapter can attribute exact + // bytes to the deletions it performed. + ReclaimedBytes int64 `json:"reclaimed_bytes"` + ReclaimedKnown bool `json:"reclaimed_known"` +} + +// CountByKind counts the eligible candidates of one resource kind. +func (s PruneSummary) CountByKind(kind domain.PruneResourceKind) int { + count := 0 + for _, candidate := range s.Candidates { + if candidate.Kind == string(kind) && candidate.Verdict == string(domain.PruneVerdictEligible) { + count++ + } + } + return count +} + +// PruneSummaryFromDomain maps the domain prune report to its wire form. +// Dry-run and execution share the shape, distinguished by Applied. +func PruneSummaryFromDomain(report domain.PruneReport) PruneSummary { + summary := PruneSummary{ + Applied: report.Applied, + Eligible: report.CountByVerdict(domain.PruneVerdictEligible), + Protected: report.CountByVerdict(domain.PruneVerdictProtected), + Unknown: report.CountByVerdict(domain.PruneVerdictUnknown), + ReclaimedBytes: report.ReclaimedBytes, + ReclaimedKnown: report.ReclaimedKnown, + } + for _, candidate := range report.Candidates { + summary.Candidates = append(summary.Candidates, pruneCandidateFromDomain(candidate)) + } + for _, candidate := range report.Deleted { + summary.Deleted = append(summary.Deleted, pruneCandidateFromDomain(candidate)) + } + for _, failure := range report.Failures { + summary.Failures = append(summary.Failures, PruneFailure{ + Kind: string(failure.Kind), Ref: failure.Ref, Error: failure.Err, + }) + } + for _, gap := range report.Gaps { + summary.Gaps = append(summary.Gaps, PruneGap{ + Source: string(gap.Source), Reason: string(gap.Reason), Detail: gap.Detail, + }) + } + return summary +} + +func pruneCandidateFromDomain(candidate domain.PruneCandidateReport) PruneCandidate { + reasons := make([]string, 0, len(candidate.Reasons)) + for _, reason := range candidate.Reasons { + reasons = append(reasons, string(reason)) + } + return PruneCandidate{ + Kind: string(candidate.Kind), Ref: candidate.Ref, Verdict: string(candidate.Verdict), Reasons: reasons, + } +} diff --git a/internal/adapters/dto/volumes.go b/internal/adapters/dto/volumes.go index 83631a033..b3404c6fb 100644 --- a/internal/adapters/dto/volumes.go +++ b/internal/adapters/dto/volumes.go @@ -20,8 +20,10 @@ type VolumePruneRequest struct { } // VolumePruneResponse contains the results of a volume prune operation. +// Plan is the same shape for dry runs and executions. type VolumePruneResponse struct { - VolumesRemoved int `json:"volumes_removed"` - SpaceReclaimed int64 `json:"space_reclaimed"` - Volumes []Volume `json:"volumes,omitempty"` + VolumesRemoved int `json:"volumes_removed"` + SpaceReclaimed int64 `json:"space_reclaimed"` + Volumes []Volume `json:"volumes,omitempty"` + Plan PruneSummary `json:"plan"` } diff --git a/internal/adapters/in/appmanifest/parse.go b/internal/adapters/in/appmanifest/parse.go new file mode 100644 index 000000000..d7387fec3 --- /dev/null +++ b/internal/adapters/in/appmanifest/parse.go @@ -0,0 +1,418 @@ +// Package appmanifest decodes standalone app TOML files into +// normalized domain.AppSpec values. It performs SHAPE decoding only +// (strict unknown fields, type mapping, defaults); all semantic rules +// live in domain.AppSpec.Validate. It never reads files, environment, +// or secrets beyond the manifest bytes given to it. +package appmanifest + +import ( + "bytes" + "errors" + "fmt" + "maps" + "slices" + "strconv" + "strings" + "time" + + "github.com/pelletier/go-toml/v2" + + "github.com/bnema/gordon/internal/domain" +) + +// rawManifest mirrors the frozen TOML schema for strict decoding. +// Services is keyed by service name: [services.]. +type rawManifest struct { + Name string `toml:"name"` + Env map[string]string `toml:"env"` + Services map[string]rawService `toml:"services"` + Network rawNetwork `toml:"network"` + Telemetry rawTelemetry `toml:"telemetry"` +} + +// rawTelemetry mirrors [telemetry] and [services..telemetry]. +// A nil Logs means "inherit": service inherits the app, the app +// inherits the default (export on). +type rawTelemetry struct { + Logs *bool `toml:"logs"` +} + +// rawNetwork mirrors [network] and its [[network.shared]] children. +type rawNetwork struct { + Shared []rawSharedNetwork `toml:"shared"` +} + +// rawService mirrors one [services.] table. The service name is +// the table key, so there is no name field inside the table. +type rawService struct { + Image string `toml:"image"` + Command []string `toml:"command"` + StopGrace string `toml:"stop_grace"` + Env map[string]string `toml:"env"` + Readiness rawReadiness `toml:"readiness"` + HTTP []rawHTTP `toml:"http"` + TCP []rawTCP `toml:"tcp"` + UDP []rawUDP `toml:"udp"` + Secrets map[string]string `toml:"secrets"` + Volumes []rawVolume `toml:"volume"` + Binds []rawBind `toml:"bind"` + // Devices lists logical device names from `devices = [...]`. + Devices []string `toml:"devices"` + Databases []rawDatabase `toml:"database"` + Backup rawBackup `toml:"backup"` + Telemetry rawTelemetry `toml:"telemetry"` +} + +// rawReadiness mirrors [services..readiness]. +type rawReadiness struct { + Type string `toml:"type"` + Path string `toml:"path"` + Contains string `toml:"contains"` + Port int `toml:"port"` + Timeout string `toml:"timeout"` +} + +// rawHTTP mirrors [[services..http]]. +type rawHTTP struct { + Host string `toml:"host"` + Port int `toml:"port"` + // TLS uses presence tracking because internal interfaces reject every + // explicit value, including an empty string. + TLS *string `toml:"tls"` + // Visibility is empty when unset; the parser normalizes it to public. + Visibility string `toml:"visibility"` +} + +// rawTCP mirrors [[services..tcp]]. +type rawTCP struct { + Entrypoint string `toml:"entrypoint"` + Port int `toml:"port"` + Publish string `toml:"publish"` +} + +// rawUDP mirrors [[services..udp]]. +type rawUDP struct { + Entrypoint string `toml:"entrypoint"` + Port int `toml:"port"` + Publish string `toml:"publish"` +} + +// rawVolume mirrors [[services..volume]]. +type rawVolume struct { + Name string `toml:"name"` + Path string `toml:"path"` + ReadOnly bool `toml:"readonly"` +} + +// rawBind mirrors [[services..bind]]. +type rawBind struct { + Name string `toml:"name"` + Path string `toml:"path"` + ReadOnly bool `toml:"readonly"` +} + +// rawDatabase mirrors [[services..database]]. +type rawDatabase struct { + Name string `toml:"name"` + Type string `toml:"type"` + Schedule string `toml:"schedule"` +} + +// rawBackup mirrors [services..backup]. +type rawBackup struct { + Postgres []string `toml:"postgres"` + Volume []string `toml:"volume"` +} + +// rawSharedNetwork mirrors [[network.shared]]. +type rawSharedNetwork struct { + Network string `toml:"network"` + Services []string `toml:"services"` + Aliases []string `toml:"aliases"` +} + +// Parse decodes manifest TOML bytes into a validated, normalized AppSpec. +// sourceName is used for mismatch warnings only; the name field is authoritative. +func Parse(data []byte, sourceName string) (domain.AppSpec, []string, error) { + var raw rawManifest + decoder := toml.NewDecoder(bytes.NewReader(data)).DisallowUnknownFields() + if err := decoder.Decode(&raw); err != nil { + if hint := legacyKeyedSchemaHint(data); hint != "" { + return domain.AppSpec{}, nil, fmt.Errorf("%w: %s", domain.ErrInvalidAppSpec, hint) + } + return domain.AppSpec{}, nil, fmt.Errorf("%w: %s", domain.ErrInvalidAppSpec, formatDecodeError(err)) + } + spec, err := toDomain(raw) + if err != nil { + return domain.AppSpec{}, nil, err + } + var warnings []string + if sourceName != "" && sourceName != spec.Name && sourceName != spec.Name+".toml" { + warnings = append(warnings, fmt.Sprintf("file name %q does not match app name %q", sourceName, spec.Name)) + } + if err := spec.Validate(); err != nil { + return domain.AppSpec{}, warnings, err + } + return spec, warnings, nil +} + +func formatDecodeError(err error) string { + var strictErr *toml.StrictMissingError + if !errors.As(err, &strictErr) { + return err.Error() + } + unknown := make([]string, 0, len(strictErr.Errors)) + for i := range strictErr.Errors { + key := strictErr.Errors[i].Key() + if len(key) > 0 { + unknown = append(unknown, strings.Join(key, ".")) + } + } + if len(unknown) == 0 { + return strictErr.Error() + } + return "unknown TOML fields or tables: " + strings.Join(unknown, ", ") +} + +// legacyKeyedSchemaHint recognizes the retired array-of-tables app shapes +// ([[service]] and [[services]]) and returns a stable, actionable +// diagnostic. It inspects the loosely decoded document rather than +// matching go-toml error wording, which is not a stable contract. +func legacyKeyedSchemaHint(data []byte) string { + var doc map[string]any + if err := toml.Unmarshal(data, &doc); err != nil { + return "" + } + const fix = "array-of-tables is a retired app shape; use one [services.] table per service (keyed schema), e.g. [services.web]" + switch { + case isArrayTable(doc["service"]): + return "[[service]] " + fix + case isArrayTable(doc["services"]): + return "[[services]] " + fix + default: + return "" + } +} + +// isArrayTable reports whether a loosely decoded value is the [[name]] +// TOML shape: a non-empty array whose elements are all tables. Scalar +// values, string arrays, and keyed tables are not array tables, so they +// keep the plain unknown-field diagnostic. +func isArrayTable(v any) bool { + tables, ok := v.([]any) + if !ok || len(tables) == 0 { + return false + } + for _, table := range tables { + if _, ok := table.(map[string]any); !ok { + return false + } + } + return true +} + +// toDomain maps raw TOML onto domain types with normalization and defaults. +func toDomain(raw rawManifest) (domain.AppSpec, error) { + spec := domain.AppSpec{ + Name: raw.Name, + Env: map[string]string{}, + Services: make([]domain.AppService, 0, len(raw.Services)), + Networks: make([]domain.AppSharedNetwork, 0, len(raw.Network.Shared)), + } + for key, value := range raw.Env { + spec.Env[key] = value + } + // Iterate the service map in sorted key order so that domain + // conversion and every downstream projection are deterministic + // regardless of TOML declaration order. + for _, name := range slices.Sorted(maps.Keys(raw.Services)) { + svc, err := toDomainService(name, raw.Services[name]) + if err != nil { + return domain.AppSpec{}, err + } + svc.LogExportDisabled = !resolveLogExport(raw.Telemetry.Logs, raw.Services[name].Telemetry.Logs) + spec.Services = append(spec.Services, svc) + } + for _, net := range raw.Network.Shared { + spec.Networks = append(spec.Networks, domain.AppSharedNetwork{ + Network: net.Network, + Services: append([]string(nil), net.Services...), + Aliases: append([]string(nil), net.Aliases...), + }) + } + return spec, nil +} + +// resolveLogExport applies the precedence service > app > default (on). +func resolveLogExport(app, service *bool) bool { + if service != nil { + return *service + } + if app != nil { + return *app + } + return true +} + +func toDomainHTTP(service string, h rawHTTP) (domain.AppHTTPInterface, error) { + visibility := h.Visibility + if visibility == "" { + visibility = domain.AppVisibilityPublic + } + if visibility == domain.AppVisibilityInternal && h.TLS != nil { + return domain.AppHTTPInterface{}, fmt.Errorf("%w: service %q internal http port %d must not declare tls", domain.ErrInvalidAppSpec, service, h.Port) + } + tls := "" + if h.TLS != nil { + tls = *h.TLS + } else if visibility == domain.AppVisibilityPublic { + tls = domain.AppTLSAuto + } + return domain.AppHTTPInterface{ + Host: domain.CanonicalHTTPHost(h.Host), Port: h.Port, TLS: tls, Visibility: visibility, + }, nil +} + +// toDomainService maps one raw service with defaults. serviceName is the +// [services.] table key and becomes the service identity. +func toDomainService(serviceName string, raw rawService) (domain.AppService, error) { + svc := domain.AppService{ + Name: serviceName, + Image: normalizeImage(raw.Image), + Command: append([]string(nil), raw.Command...), + Secrets: map[string]string{}, + Backup: domain.AppBackup{ + Postgres: append([]string(nil), raw.Backup.Postgres...), + Volume: append([]string(nil), raw.Backup.Volume...), + }, + } + stopGrace := domain.AppDefaultStopGrace + if raw.StopGrace != "" { + parsed, err := time.ParseDuration(raw.StopGrace) + if err != nil { + return domain.AppService{}, fmt.Errorf("%w: service %q stop_grace %q is invalid: %v", domain.ErrInvalidAppSpec, serviceName, raw.StopGrace, err) + } + stopGrace = parsed + } + svc.StopGrace = stopGrace + readiness, err := toDomainReadiness(serviceName, raw.Readiness) + if err != nil { + return domain.AppService{}, err + } + svc.Readiness = readiness + for _, h := range raw.HTTP { + iface, err := toDomainHTTP(serviceName, h) + if err != nil { + return domain.AppService{}, err + } + svc.HTTP = append(svc.HTTP, iface) + } + for _, t := range raw.TCP { + svc.TCP = append(svc.TCP, domain.AppTCPInterface{ + Entrypoint: t.Entrypoint, + Port: t.Port, + Publish: t.Publish, + }) + } + for _, u := range raw.UDP { + svc.UDP = append(svc.UDP, domain.AppUDPInterface{ + Entrypoint: u.Entrypoint, + Port: u.Port, + Publish: u.Publish, + }) + } + for key, value := range raw.Secrets { + svc.Secrets[key] = value + } + for _, v := range raw.Volumes { + svc.Volumes = append(svc.Volumes, domain.AppVolume{ + Name: v.Name, + Path: v.Path, + ReadOnly: v.ReadOnly, + }) + } + for _, b := range raw.Binds { + svc.Binds = append(svc.Binds, domain.AppBind{ + Name: b.Name, + Path: b.Path, + ReadOnly: b.ReadOnly, + }) + } + // Preserve manifest order; domain equalDevices treats the slice as a + // set, so reorder-only input is a no-op in diffs. + svc.Devices = append([]string(nil), raw.Devices...) + for _, db := range raw.Databases { + svc.Databases = append(svc.Databases, domain.AppDatabase{ + Name: db.Name, + Type: db.Type, + Schedule: db.Schedule, + }) + } + if len(raw.Env) > 0 { + return domain.AppService{}, fmt.Errorf("%w: service %q [services.%s.env] is not allowed, use secrets", domain.ErrInvalidAppSpec, serviceName, tomlKey(serviceName)) + } + return svc, nil +} + +// toDomainReadiness maps readiness with defaults. +func toDomainReadiness(service string, raw rawReadiness) (domain.AppReadiness, error) { + readiness := domain.AppReadiness{ + Type: raw.Type, + Path: raw.Path, + Contains: raw.Contains, + Port: raw.Port, + Timeout: domain.AppDefaultReadinessTimeout, + } + if readiness.Type == "" { + readiness.Type = domain.AppReadinessNone + } + if raw.Timeout != "" { + parsed, err := time.ParseDuration(raw.Timeout) + if err != nil { + return domain.AppReadiness{}, fmt.Errorf("%w: service %q readiness timeout %q is invalid: %v", domain.ErrInvalidAppSpec, service, raw.Timeout, err) + } + readiness.Timeout = parsed + } + return readiness, nil +} + +// tomlKey renders a service name as a TOML key, quoting it when it is +// not a bare key (service names may contain dots, e.g. web.api). +func tomlKey(name string) string { + if isBareTOMLKey(name) { + return name + } + return strconv.Quote(name) +} + +// isBareTOMLKey reports whether name is a TOML bare key (A-Za-z0-9_-). +func isBareTOMLKey(name string) bool { + if name == "" { + return false + } + for _, r := range name { + switch { + case r >= 'a' && r <= 'z', r >= 'A' && r <= 'Z', r >= '0' && r <= '9', r == '_', r == '-': + default: + return false + } + } + return true +} + +// normalizeImage records a missing tag as explicit :latest. +func normalizeImage(image string) string { + image = strings.TrimSpace(image) + if image == "" { + return image + } + if strings.Contains(image, "@") { + return image + } + lastSlash := strings.LastIndex(image, "/") + lastColon := strings.LastIndex(image, ":") + if lastColon <= lastSlash { + return image + ":latest" + } + return image +} diff --git a/internal/adapters/in/appmanifest/parse_test.go b/internal/adapters/in/appmanifest/parse_test.go new file mode 100644 index 000000000..82655e1e1 --- /dev/null +++ b/internal/adapters/in/appmanifest/parse_test.go @@ -0,0 +1,605 @@ +package appmanifest_test + +import ( + "strings" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/in/appmanifest" + "github.com/bnema/gordon/internal/domain" +) + +const validWeb = ` +name = "blog" + +[env] +APP_ENV = "production" + +[services.web] +image = "registry.example.com/blog/web:1.4.2" + +[services.web.readiness] +type = "http" +path = "/healthz" + +[[services.web.http]] +host = "Blog.Example.COM." +port = 8080 + +[services.web.secrets] +DATABASE_URL = "database-url" +` + +func TestParse_ValidWeb(t *testing.T) { + spec, warnings, err := appmanifest.Parse([]byte(validWeb), "blog.toml") + require.NoError(t, err) + assert.Empty(t, warnings) + assert.Equal(t, "blog", spec.Name) + assert.Equal(t, "production", spec.Env["APP_ENV"]) + require.Len(t, spec.Services, 1) + svc := spec.Services[0] + assert.Equal(t, "web", svc.Name) + assert.Equal(t, "registry.example.com/blog/web:1.4.2", svc.Image) + assert.Equal(t, "blog.example.com", svc.HTTP[0].Host) + assert.Equal(t, "auto", svc.HTTP[0].TLS) + assert.Equal(t, "http", svc.Readiness.Type) +} + +func TestParse_SharedNetwork(t *testing.T) { + doc := ` +name = "media" +[services.web] +image = "registry.example.com/media/web:1" +[[network.shared]] +network = "backend" +services = ["web"] +aliases = ["media-web"] +` + + spec, warnings, err := appmanifest.Parse([]byte(doc), "media.toml") + require.NoError(t, err) + assert.Empty(t, warnings) + require.Len(t, spec.Networks, 1) + assert.Equal(t, domain.AppSharedNetwork{ + Network: "backend", Services: []string{"web"}, Aliases: []string{"media-web"}, + }, spec.Networks[0]) +} + +func TestParse_FileNameMismatchIsWarning(t *testing.T) { + _, warnings, err := appmanifest.Parse([]byte(validWeb), "other.toml") + require.NoError(t, err) + require.Len(t, warnings, 1) + assert.Contains(t, warnings[0], "other.toml") +} + +func TestParse_MissingTagNormalizesToLatest(t *testing.T) { + doc := ` +name = "blog" +[services.web] +image = "registry.example.com/blog/web" +[[services.web.http]] +host = "blog.example.com" +port = 8080 +` + spec, _, err := appmanifest.Parse([]byte(doc), "blog.toml") + require.NoError(t, err) + assert.Equal(t, "registry.example.com/blog/web:latest", spec.Services[0].Image) +} + +func TestParse_UnknownFieldsRejected(t *testing.T) { + cases := []struct { + name string + doc string + }{ + {"top level", "name = \"blog\"\nbogus = 1\n[services.web]\nimage = \"img:1\"\n"}, + {"service", "name = \"blog\"\n[services.web]\nimage = \"img:1\"\nbogus = 1\n"}, + {"readiness", "name = \"blog\"\n[services.web]\nimage = \"img:1\"\n[services.web.readiness]\nbogus = 1\n"}, + {"http", "name = \"blog\"\n[services.web]\nimage = \"img:1\"\n[[services.web.http]]\nhost = \"blog.example.com\"\nport = 8080\nbogus = 1\n"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + _, _, err := appmanifest.Parse([]byte(tc.doc), "blog.toml") + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrInvalidAppSpec) + }) + } +} + +// TestParse_OldSyntaxRejected proves the retired array-of-tables schema +// and its name field are no longer accepted, with no compatibility shim. +func TestParse_OldSyntaxRejected(t *testing.T) { + cases := []struct { + name string + doc string + wantErr string + }{ + { + "legacy service array", + "name = \"blog\"\n[[service]]\nname = \"web\"\nimage = \"img:1\"\n", + "[[service]] array-of-tables is a retired app shape", + }, + { + "legacy plural service array", + "name = \"blog\"\n[[services]]\nname = \"web\"\nimage = \"img:1\"\n", + "[[services]] array-of-tables is a retired app shape", + }, + { + "service name as field", + "name = \"blog\"\n[services.web]\nname = \"web\"\nimage = \"img:1\"\n", + "unknown TOML fields or tables: services.web.name", + }, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + _, _, err := appmanifest.Parse([]byte(tc.doc), "blog.toml") + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrInvalidAppSpec) + assert.ErrorContains(t, err, tc.wantErr) + }) + } +} + +// TestParse_RetiredArraySyntaxHintIsActionable proves the rejected +// [[service]] and [[services]] shapes carry the same stable keyed-schema +// remedy instead of leaking go-toml decoding wording. +func TestParse_RetiredArraySyntaxHintIsActionable(t *testing.T) { + docs := map[string]string{ + "singular": "name = \"blog\"\n[[service]]\nname = \"web\"\nimage = \"img:1\"\n", + "plural": "name = \"blog\"\n[[services]]\nname = \"web\"\nimage = \"img:1\"\n", + } + for name, doc := range docs { + t.Run(name, func(t *testing.T) { + _, _, err := appmanifest.Parse([]byte(doc), "blog.toml") + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrInvalidAppSpec) + assert.ErrorContains(t, err, "keyed schema") + assert.ErrorContains(t, err, "[services.]") + assert.ErrorContains(t, err, "[services.web]") + }) + } +} + +// TestParse_NonArrayTablesKeepPlainDiagnostic proves the keyed-schema hint +// stays reserved for real array-of-tables shapes: a [service] table, a +// scalar service value, and a string array under services are not array +// tables, so they keep the plain unknown-field diagnostic instead of +// being described as [[service]]/[[services]]. +func TestParse_NonArrayTablesKeepPlainDiagnostic(t *testing.T) { + cases := []struct { + name string + doc string + wantErr string + }{ + { + "service table", + "name = \"blog\"\n[service]\nimage = \"img:1\"\n", + "unknown TOML fields or tables: service", + }, + { + "service scalar", + "name = \"blog\"\nservice = \"web\"\n", + "unknown TOML fields or tables: service", + }, + { + // services is a keyed table, so an array value fails to decode. + // Assert only that the diagnostic names the key, never go-toml's + // internal wording. + "services string array", + "name = \"blog\"\nservices = [\"web\"]\n", + "services", + }, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + _, _, err := appmanifest.Parse([]byte(tc.doc), "blog.toml") + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrInvalidAppSpec) + assert.Contains(t, strings.ToLower(err.Error()), strings.ToLower(tc.wantErr)) + assert.NotContains(t, err.Error(), "array-of-tables is a retired app shape") + }) + } +} + +// TestParse_UnknownNestedFieldRejected proves strict decoding still +// reports the dotted path of an unknown table or field under a service. +func TestParse_UnknownNestedFieldRejected(t *testing.T) { + cases := []struct { + name string + doc string + wantErr string + }{ + { + "unknown protocol table", + "name = \"blog\"\n[services.web]\nimage = \"img:1\"\n[[services.web.rcon]]\nport = 28016\n", + "unknown TOML fields or tables: services.web.rcon", + }, + { + "unknown nested table", + "name = \"blog\"\n[services.web]\nimage = \"img:1\"\n[services.web.extra]\nkey = \"value\"\n", + "unknown TOML fields or tables: services.web.extra", + }, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + _, _, err := appmanifest.Parse([]byte(tc.doc), "blog.toml") + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrInvalidAppSpec) + assert.ErrorContains(t, err, tc.wantErr) + }) + } +} + +// TestParse_QuotedDottedServiceKey proves a service name containing a +// dot must be quoted, decodes as one key, and is usable by nested tables. +func TestParse_QuotedDottedServiceKey(t *testing.T) { + doc := ` +name = "blog" +[services."web.api"] +image = "registry.example.com/blog/api:1" +[[services."web.api".http]] +host = "api.example.com" +port = 8080 +` + spec, _, err := appmanifest.Parse([]byte(doc), "blog.toml") + require.NoError(t, err) + require.Len(t, spec.Services, 1) + assert.Equal(t, "web.api", spec.Services[0].Name) + require.Len(t, spec.Services[0].HTTP, 1) + assert.Equal(t, "api.example.com", spec.Services[0].HTTP[0].Host) +} + +// TestParse_UnquotedDottedKeyRejected proves an unquoted dotted key is +// read as nested tables, not as one service name, and is rejected. +func TestParse_UnquotedDottedKeyRejected(t *testing.T) { + doc := ` +name = "blog" +[services.web.api] +image = "registry.example.com/blog/api:1" +` + _, _, err := appmanifest.Parse([]byte(doc), "blog.toml") + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrInvalidAppSpec) + assert.ErrorContains(t, err, "unknown TOML fields or tables: services.web.api") +} + +// TestParse_DuplicateServiceTableRejected proves the TOML decoder rejects +// a repeated [services.] table before domain validation runs. +func TestParse_DuplicateServiceTableRejected(t *testing.T) { + doc := ` +name = "blog" +[services.web] +image = "img:1" +[services.web] +image = "img:2" +` + _, _, err := appmanifest.Parse([]byte(doc), "blog.toml") + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrInvalidAppSpec) + assert.ErrorContains(t, err, "already exists") +} + +// TestParse_ServiceOrderDeterministic proves map keys are sorted before +// domain conversion so spec.Services order never depends on TOML order. +func TestParse_ServiceOrderDeterministic(t *testing.T) { + doc := ` +name = "blog" +[services.zeta] +image = "img:1" +[services.alpha] +image = "img:2" +[services.mid] +image = "img:3" +` + for i := 0; i < 3; i++ { + spec, _, err := appmanifest.Parse([]byte(doc), "blog.toml") + require.NoError(t, err) + require.Len(t, spec.Services, 3) + assert.Equal(t, []string{"alpha", "mid", "zeta"}, []string{ + spec.Services[0].Name, spec.Services[1].Name, spec.Services[2].Name, + }) + } +} + +func TestParse_ServiceEnvRejected(t *testing.T) { + cases := []struct { + name string + doc string + wantErr string + }{ + { + "bare service name", + "name = \"blog\"\n[services.web]\nimage = \"img:1\"\n[services.web.env]\nFOO = \"bar\"\n", + "[services.web.env]", + }, + { + "dotted service name is quoted", + "name = \"blog\"\n[services.\"web.api\"]\nimage = \"img:1\"\n[services.\"web.api\".env]\nFOO = \"bar\"\n", + `[services."web.api".env]`, + }, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + _, _, err := appmanifest.Parse([]byte(tc.doc), "blog.toml") + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrInvalidAppSpec) + require.ErrorContains(t, err, tc.wantErr) + }) + } +} + +func TestParse_Names(t *testing.T) { + cases := []struct { + name string + doc string + wantErr string + }{ + {"bad app", "name = \"Blog!\"\n[services.web]\nimage = \"img:1\"\n", "app name"}, + {"double dash app", "name = \"a--b\"\n[services.web]\nimage = \"img:1\"\n", "--"}, + {"reserved app", "name = \"gordon\"\n[services.web]\nimage = \"img:1\"\n", "reserved"}, + {"no services", "name = \"blog\"\n", "at least one service"}, + {"dup service", "name = \"blog\"\n[services.web]\nimage = \"img:1\"\n[services.web]\nimage = \"img:2\"\n", "already exists"}, + {"dup service case", "name = \"blog\"\n[services.Web]\nimage = \"img:1\"\n", "must match"}, + {"normalized collision", "name = \"blog\"\n[services.\"a.b\"]\nimage = \"img:1\"\n[services.\"a-b\"]\nimage = \"img:2\"\n", "same runtime identifier"}, + {"replicas rejected", "name = \"blog\"\n[services.web]\nimage = \"img:1\"\nreplicas = 2\n", "unknown TOML fields or tables: services.web.replicas"}, + {"volume double dash", "name = \"blog\"\n[services.web]\nimage = \"img:1\"\n[[services.web.volume]]\nname = \"a--b\"\npath = \"/data\"\n", "--"}, + {"shared volume", "name = \"blog\"\n[services.a]\nimage = \"img:1\"\n[[services.a.volume]]\nname = \"d\"\npath = \"/data\"\n[services.b]\nimage = \"img:1\"\n[[services.b.volume]]\nname = \"d\"\npath = \"/data\"\n", "claimed by both"}, + {"env secret key collision", "name = \"blog\"\n[env]\nDB_PASSWORD = \"x\"\n[services.web]\nimage = \"img:1\"\n[services.web.secrets]\nDB_PASSWORD = \"db-password\"\n", "collides with a secret key"}, + {"secret ref in env", "name = \"blog\"\n[env]\nFOO = \"${pass:x}\"\n[services.web]\nimage = \"img:1\"\n", "secret references"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + _, _, err := appmanifest.Parse([]byte(tc.doc), "blog.toml") + require.ErrorContains(t, err, tc.wantErr) + }) + } +} + +func TestParse_Readiness(t *testing.T) { + cases := []struct { + name string + doc string + wantErr string + }{ + {"bad type", "name = \"b\"\n[services.w]\nimage = \"i:1\"\n[services.w.readiness]\ntype = \"exec\"\n", "none|tcp|http|log"}, + {"log needs path", "name = \"b\"\n[services.w]\nimage = \"i:1\"\n[services.w.readiness]\ntype = \"log\"\ncontains = \"x\"\n", "path and contains"}, + {"http needs path", "name = \"b\"\n[services.w]\nimage = \"i:1\"\n[[services.w.http]]\nhost = \"b.example.com\"\nport = 8080\n[services.w.readiness]\ntype = \"http\"\n", "http readiness requires path"}, + {"udp tcp rejected", "name = \"b\"\n[services.w]\nimage = \"i:1\"\n[[services.w.udp]]\nentrypoint = \"udp\"\nport = 9000\npublish = \"0.0.0.0:9000\"\n[services.w.readiness]\ntype = \"tcp\"\n", "UDP-only"}, + {"multi tcp needs port", "name = \"b\"\n[services.w]\nimage = \"i:1\"\n[[services.w.http]]\nhost = \"b.example.com\"\nport = 8080\n[[services.w.tcp]]\nentrypoint = \"tcp\"\nport = 9000\npublish = \"9000\"\n[services.w.readiness]\ntype = \"tcp\"\n", "readiness.port is required"}, + {"bad port ref", "name = \"b\"\n[services.w]\nimage = \"i:1\"\n[[services.w.http]]\nhost = \"b.example.com\"\nport = 8080\n[services.w.readiness]\ntype = \"tcp\"\nport = 9999\n", "matches no declared container port"}, + {"bad timeout", "name = \"b\"\n[services.w]\nimage = \"i:1\"\n[services.w.readiness]\ntimeout = \"99h\"\n", "timeout must be within"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + _, _, err := appmanifest.Parse([]byte(tc.doc), "b.toml") + require.ErrorContains(t, err, tc.wantErr) + }) + } +} + +func TestParse_Interfaces(t *testing.T) { + cases := []struct { + name string + doc string + wantErr string + }{ + {"tcp needs entrypoint", "name = \"b\"\n[services.w]\nimage = \"i:1\"\n[[services.w.tcp]]\nport = 9000\npublish = \"9000\"\n", "entrypoint is required"}, + {"publish hostname", "name = \"b\"\n[services.w]\nimage = \"i:1\"\n[[services.w.tcp]]\nentrypoint = \"tcp\"\nport = 9000\npublish = \"example.com:9000\"\n", "literal IP"}, + {"rcon table rejected", "name = \"b\"\n[services.w]\nimage = \"i:1\"\n[[services.w.rcon]]\nentrypoint = \"tcp\"\nport = 28016\npublish = \"0.0.0.0:28016\"\npublic = true\n", "unknown TOML fields or tables: services.w.rcon"}, + {"bad host", "name = \"b\"\n[services.w]\nimage = \"i:1\"\n[[services.w.http]]\nhost = \"localhost\"\nport = 8080\n", "not a valid public hostname"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + _, _, err := appmanifest.Parse([]byte(tc.doc), "b.toml") + require.ErrorContains(t, err, tc.wantErr) + }) + } +} + +func TestParse_Backups(t *testing.T) { + doc := ` +name = "shop" +[services.api] +image = "registry.example.com/shop/api:2.0.0" +[[services.api.http]] +host = "shop.example.com" +port = 8080 +[[services.api.database]] +name = "main" +type = "postgres" +schedule = "daily" +[services.api.backup] +postgres = ["main"] +` + spec, _, err := appmanifest.Parse([]byte(doc), "shop.toml") + require.NoError(t, err) + assert.Equal(t, "daily", spec.Services[0].Databases[0].Schedule) + + badRef := strings.Replace(doc, "postgres = [\"main\"]", "postgres = [\"ghost\"]", 1) + _, _, err = appmanifest.Parse([]byte(badRef), "shop.toml") + require.ErrorContains(t, err, "unknown database") + + badSchedule := strings.Replace(doc, "schedule = \"daily\"", "schedule = \"minutely\"", 1) + _, _, err = appmanifest.Parse([]byte(badSchedule), "shop.toml") + require.ErrorContains(t, err, "schedule must be") + + badEngine := strings.Replace(doc, "type = \"postgres\"", "type = \"mysql\"", 1) + _, _, err = appmanifest.Parse([]byte(badEngine), "shop.toml") + require.ErrorContains(t, err, "must be postgres") +} + +func TestParse_GameAndStagingExamples(t *testing.T) { + game := ` +name = "rust" +[services.server] +image = "registry.example.com/games/rust:2026.09" +[services.server.readiness] +type = "log" +path = "/data/logs/server.log" +contains = "Server startup complete" +timeout = "5m" +[[services.server.udp]] +entrypoint = "udp" +port = 28015 +publish = "0.0.0.0:28015" +[[services.server.tcp]] +entrypoint = "tcp" +port = 28016 +publish = "127.0.0.1:28016" +[[services.server.volume]] +name = "rust-data" +path = "/data" +` + spec, _, err := appmanifest.Parse([]byte(game), "rust.toml") + require.NoError(t, err) + assert.Equal(t, "log", spec.Services[0].Readiness.Type) + require.Len(t, spec.Services[0].TCP, 1) + assert.Equal(t, 28016, spec.Services[0].TCP[0].Port) + + staging := ` +name = "blog-staging" +[env] +APP_ENV = "staging" +[services.web] +image = "registry.example.com/blog/web:1.4.2-rc.1" +[[services.web.http]] +host = "staging-blog.example.com" +port = 8080 +` + spec, _, err = appmanifest.Parse([]byte(staging), "blog-staging.toml") + require.NoError(t, err) + assert.Equal(t, "blog-staging", spec.Name) +} + +func TestParse_NoFileEnvSecretReads(t *testing.T) { + // Parser input is bytes only; there is no path parameter that could + // trigger file, env, or secret reads beyond the manifest input. + spec, _, err := appmanifest.Parse([]byte(validWeb), "") + require.NoError(t, err) + assert.Equal(t, "blog", spec.Name) +} + +func TestParse_Binds(t *testing.T) { + doc := ` +name = "blog" +[services.web] +image = "registry.example.com/blog/web:1.4.2" +[[services.web.bind]] +name = "config" +path = "/etc/app.conf" +readonly = true +[[services.web.bind]] +name = "data" +path = "/var/lib/app" +` + spec, _, err := appmanifest.Parse([]byte(doc), "blog.toml") + require.NoError(t, err) + require.Len(t, spec.Services[0].Binds, 2) + assert.Equal(t, domain.AppBind{Name: "config", Path: "/etc/app.conf", ReadOnly: true}, spec.Services[0].Binds[0]) + assert.Equal(t, domain.AppBind{Name: "data", Path: "/var/lib/app"}, spec.Services[0].Binds[1]) +} + +func TestParse_BindDestinationRejected(t *testing.T) { + doc := ` +name = "blog" +[services.web] +image = "registry.example.com/blog/web:1.4.2" +[[services.web.bind]] +name = "dev" +path = "/dev" +` + _, _, err := appmanifest.Parse([]byte(doc), "blog.toml") + require.ErrorIs(t, err, domain.ErrInvalidAppSpec) + assert.Contains(t, err.Error(), "sensitive container path") +} + +func TestParse_Devices(t *testing.T) { + doc := ` +name = "demo" +[services.worker] +image = "registry.example.com/demo/worker:1" +devices = ["test_gpu"] +` + spec, _, err := appmanifest.Parse([]byte(doc), "demo.toml") + require.NoError(t, err) + require.Len(t, spec.Services, 1) + assert.Equal(t, []string{"test_gpu"}, spec.Services[0].Devices) +} + +func TestParse_DevicesWrongType(t *testing.T) { + doc := ` +name = "demo" +[services.worker] +image = "registry.example.com/demo/worker:1" +devices = "test_gpu" +` + _, _, err := appmanifest.Parse([]byte(doc), "demo.toml") + require.Error(t, err) +} + +func TestParse_DevicesMalformedRejected(t *testing.T) { + doc := ` +name = "demo" +[services.worker] +image = "registry.example.com/demo/worker:1" +devices = ["Bad Name!"] +` + _, _, err := appmanifest.Parse([]byte(doc), "demo.toml") + require.ErrorIs(t, err, domain.ErrInvalidAppSpec) +} + +func TestParse_DevicesDuplicateRejected(t *testing.T) { + doc := ` +name = "demo" +[services.worker] +image = "registry.example.com/demo/worker:1" +devices = ["test_gpu", "test_gpu"] +` + _, _, err := appmanifest.Parse([]byte(doc), "demo.toml") + require.ErrorIs(t, err, domain.ErrInvalidAppSpec) + assert.Contains(t, err.Error(), "duplicate device") +} + +func TestParse_UnknownGPUFieldRejected(t *testing.T) { + doc := ` +name = "demo" +[services.worker] +image = "registry.example.com/demo/worker:1" +gpus = "all" +` + _, _, err := appmanifest.Parse([]byte(doc), "demo.toml") + require.Error(t, err) +} + +func TestParse_TelemetryLogsPrecedence(t *testing.T) { + const manifest = ` +name = "blog" + +[telemetry] +logs = false + +[services.web] +image = "registry.example.com/blog/web:1" + +[services.web.telemetry] +logs = true + +[services.db] +image = "postgres:17" +` + spec, _, err := appmanifest.Parse([]byte(manifest), "blog.toml") + require.NoError(t, err) + byName := map[string]domain.AppService{} + for _, svc := range spec.Services { + byName[svc.Name] = svc + } + assert.False(t, byName["web"].LogExportDisabled, "service override wins") + assert.True(t, byName["db"].LogExportDisabled, "service inherits app") +} + +func TestParse_TelemetryLogsDefaultOn(t *testing.T) { + spec, _, err := appmanifest.Parse([]byte(validWeb), "blog.toml") + require.NoError(t, err) + assert.False(t, spec.Services[0].LogExportDisabled) +} diff --git a/internal/adapters/in/appmanifest/parse_visibility_test.go b/internal/adapters/in/appmanifest/parse_visibility_test.go new file mode 100644 index 000000000..1bc04eb78 --- /dev/null +++ b/internal/adapters/in/appmanifest/parse_visibility_test.go @@ -0,0 +1,148 @@ +package appmanifest_test + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/in/appmanifest" + "github.com/bnema/gordon/internal/domain" +) + +// TestParse_VisibilityNormalization proves absent visibility becomes +// explicit public, an internal interface keeps its TLS absent, and a +// public interface still defaults an absent TLS to auto. +func TestParse_VisibilityNormalization(t *testing.T) { + t.Run("absent visibility stays public with auto tls", func(t *testing.T) { + doc := ` +name = "blog" +[services.web] +image = "registry.example.com/blog/web:1.0.0" +[[services.web.http]] +host = "blog.example.com" +port = 8080 +` + spec, _, err := appmanifest.Parse([]byte(doc), "blog.toml") + require.NoError(t, err) + require.Len(t, spec.Services[0].HTTP, 1) + iface := spec.Services[0].HTTP[0] + assert.Equal(t, domain.AppVisibilityPublic, iface.Visibility) + assert.Equal(t, domain.AppTLSAuto, iface.TLS) + }) + + t.Run("explicit internal keeps tls absent and has no host", func(t *testing.T) { + doc := ` +name = "blog" +[services.api] +image = "registry.example.com/blog/api:1.0.0" +[services.api.readiness] +type = "http" +path = "/healthz" +[[services.api.http]] +visibility = "internal" +port = 8080 +` + spec, _, err := appmanifest.Parse([]byte(doc), "blog.toml") + require.NoError(t, err) + require.Len(t, spec.Services[0].HTTP, 1) + iface := spec.Services[0].HTTP[0] + assert.Equal(t, domain.AppVisibilityInternal, iface.Visibility) + assert.Empty(t, iface.Host) + assert.Empty(t, iface.TLS, "an internal interface must not inherit the public auto tls default") + assert.True(t, spec.Services[0].InternallyOnlyPort(8080)) + }) + + t.Run("explicit public normalizes to auto tls", func(t *testing.T) { + doc := ` +name = "blog" +[services.web] +image = "registry.example.com/blog/web:1.0.0" +[[services.web.http]] +host = "blog.example.com" +port = 8080 +visibility = "public" +` + spec, _, err := appmanifest.Parse([]byte(doc), "blog.toml") + require.NoError(t, err) + assert.Equal(t, domain.AppVisibilityPublic, spec.Services[0].HTTP[0].Visibility) + assert.Equal(t, domain.AppTLSAuto, spec.Services[0].HTTP[0].TLS) + }) +} + +// TestParse_VisibilityRejections proves invalid combinations fail at +// validation with the sentinel error. +func TestParse_VisibilityRejections(t *testing.T) { + tests := map[string]string{ + "internal forbids host": ` +name = "blog" +[services.api] +image = "registry.example.com/blog/api:1.0.0" +[[services.api.http]] +host = "blog.example.com" +port = 8080 +visibility = "internal" +`, + "internal forbids tls": ` +name = "blog" +[services.api] +image = "registry.example.com/blog/api:1.0.0" +[[services.api.http]] +port = 8080 +visibility = "internal" +tls = "auto" +`, + "internal forbids explicitly empty tls": ` +name = "blog" +[services.api] +image = "registry.example.com/blog/api:1.0.0" +[[services.api.http]] +port = 8080 +visibility = "internal" +tls = "" +`, + "unknown visibility value": ` +name = "blog" +[services.web] +image = "registry.example.com/blog/web:1.0.0" +[[services.web.http]] +host = "blog.example.com" +port = 8080 +visibility = "vpn" +`, + "internal port also published as tcp": ` +name = "blog" +[services.api] +image = "registry.example.com/blog/api:1.0.0" +[[services.api.http]] +port = 8080 +visibility = "internal" +[[services.api.tcp]] +port = 8080 +`, + } + for name, doc := range tests { + t.Run(name, func(t *testing.T) { + _, _, err := appmanifest.Parse([]byte(doc), "blog.toml") + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrInvalidAppSpec) + }) + } +} + +// TestParse_VisibilityUnknownFieldStillRejected proves the parser stays +// strict: a near-miss key is still a decode error. +func TestParse_VisibilityUnknownFieldStillRejected(t *testing.T) { + doc := ` +name = "blog" +[services.web] +image = "registry.example.com/blog/web:1.0.0" +[[services.web.http]] +host = "blog.example.com" +port = 8080 +visibilities = "internal" +` + _, _, err := appmanifest.Parse([]byte(doc), "blog.toml") + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrInvalidAppSpec) +} diff --git a/internal/adapters/in/cli/apps.go b/internal/adapters/in/cli/apps.go new file mode 100644 index 000000000..d05fec229 --- /dev/null +++ b/internal/adapters/in/cli/apps.go @@ -0,0 +1,1221 @@ +package cli + +// App CLI surface for declarative app operations. +// +// Conventions (per AGENTS.md): cmd.Context(), cmd.OutOrStdout(), errors +// from RunE, --json via writeJSON with equivalent semantics, stable +// sorted ordering, and NO secret values or sensitive output anywhere. + +import ( + "context" + "errors" + "fmt" + "io" + "os" + "sort" + "strings" + + "github.com/spf13/cobra" + + "github.com/bnema/gordon/internal/adapters/dto" + "github.com/bnema/gordon/internal/adapters/in/cli/remote" + "github.com/bnema/gordon/internal/domain" +) + +// appApplyMaxBytes bounds the apply manifest body (05-api-cli.md §2). +const appApplyMaxBytes = 1 << 20 + +// newLocalAppClient constructs the owner-only local daemon client. It is a +// variable so command tests can inject a socket-backed or unavailable client. +var newLocalAppClient = remote.NewLocalClient + +// resolveAppClient returns the daemon client for app operations. An explicit +// remote (flag or GORDON_REMOTE) is authoritative and never falls back to the +// local socket; otherwise the owner-only local admin socket is used. +func resolveAppClient() (*remote.Client, error) { + client, isRemote, err := GetRemoteClient() + if err != nil { + return nil, err + } + if isRemote { + return client, nil + } + + localClient, err := newLocalAppClient() + if err != nil { + return nil, fmt.Errorf( + "daemon-unavailable: no daemon endpoint is reachable; app operations "+ + "are daemon-owned and have no local-write fallback "+ + "(start the daemon or pass --remote): %w", err) + } + return localClient, nil +} + +// resolveAppPlane returns the daemon-backed control plane for app commands. +// App operations are daemon-owned for BOTH local and remote paths +// (05-api-cli.md §1): the explicit remote when one is selected, otherwise +// the owner-only local admin socket. When neither is reachable the command +// fails with daemon-unavailable — there is no local-write fallback. The +// returned handle owns the client and must be closed by the caller. +func resolveAppPlane() (*controlPlaneHandle, error) { + client, err := resolveAppClient() + if err != nil { + return nil, err + } + return &controlPlaneHandle{plane: client}, nil +} + +// appMutationError translates ambiguous transport outcomes into the +// outcome-unknown guidance: never blindly retry, re-query by key first. +func appMutationError(op, app, key string, err error) error { + var unknown *remote.OutcomeUnknownError + if errors.As(err, &unknown) { + return fmt.Errorf( + "outcome-unknown: %s %s may have executed; run the by-key lookup "+ + "gordon apps operations show %s --key %s before retrying with the same key: %w", + op, app, app, key, err) + } + return err +} + +// renderAppOpConflict renders the journaled 409 deploy/lifecycle response +// carried by remote.AppOpConflictError, then returns the terminal failure +// (nonzero exit) with the operation key and by-key recovery guidance. It +// reports false when err is not a journaled conflict so callers fall +// through to appMutationError. +func renderAppOpConflict(out io.Writer, op, app, key string, err error, jsonOut bool) (error, bool) { + var conflict *remote.AppOpConflictError + if !errors.As(err, &conflict) { + return nil, false + } + if jsonOut { + if werr := writeJSON(out, conflict.Response); werr != nil { + return werr, true + } + } else if rerr := renderAppDeployResponse(out, &conflict.Response); rerr != nil { + return rerr, true + } + return appOpConflictMessage(op, app, key, conflict), true +} + +// appOpConflictMessage is the by-key recovery guidance shared by the human +// and JSON renderings of a journaled 409 conflict. +func appOpConflictMessage(op, app, key string, conflict *remote.AppOpConflictError) error { + return fmt.Errorf( + "%s of %s did not succeed (outcome %s, op %s); journal rendered above; run the by-key lookup "+ + "gordon apps operations show %s --key %s before retrying with the same key: %w", + op, app, conflict.Response.Outcome, conflict.Response.Op, app, key, conflict) +} + +// newAppsCmd creates the `apps` parent command. +func newAppsCmd() *cobra.Command { + cmd := &cobra.Command{ + Use: "apps", + Short: "Manage applications (daemon-owned)", + Long: `Validate, persist, inspect, and operate applications. + +All mutations are executed by the daemon through the admin API; +without a reachable daemon the commands fail instead of writing locally. + +Examples: + gordon apps apply --file blog.toml + gordon apps list + gordon apps show blog + gordon apps diff blog`, + } + cmd.AddCommand( + newAppsApplyCmd(), + newAppsOperationsCmd(), + newAppsListCmd(), + newAppsShowCmd(), + newAppsDiffCmd(), + newAppsSecretsCmd(), + newAppDeployCmd(), + newAppRestartCmd(), + newAppStopCmd(), + newAppStartCmd(), + newAppRemoveCmd(), + newAppStatusCmd(), + newAppLogsCmd(), + ) + return cmd +} + +// newAppsApplyCmd creates `apps apply`. +func newAppsApplyCmd() *cobra.Command { + var file string + var dryRun bool + var chainDeploy bool + var jsonOut bool + cmd := &cobra.Command{ + Use: "apply", + Short: "Validate and persist an app manifest", + Long: `Validates a manifest file and persists it as desired state (or dry-runs). + +With --deploy, chains exactly the accepted revision into a deploy after +persistence succeeds; the two outcomes are reported separately because a +deploy may fail after the apply succeeded.`, + RunE: func(cmd *cobra.Command, args []string) error { + handle, err := resolveAppPlane() + if err != nil { + return err + } + defer handle.close() + plane := handle.plane + return runAppsApply(cmd.Context(), plane, cmd.InOrStdin(), cmd.OutOrStdout(), cmd.ErrOrStderr(), file, dryRun, chainDeploy, jsonOut) + }, + } + cmd.Flags().StringVar(&file, "file", "", "Path to the app manifest file (required)") + cmd.Flags().BoolVar(&dryRun, "dry-run", false, "Validate without persisting") + cmd.Flags().BoolVar(&chainDeploy, "deploy", false, "Deploy the accepted revision after applying") + cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") + return cmd +} + +func runAppsApply(ctx context.Context, plane ControlPlane, _ io.Reader, out, errOut io.Writer, file string, dryRun, chainDeploy, jsonOut bool) error { + if dryRun && chainDeploy { + return fmt.Errorf("cannot combine --dry-run with --deploy: dry-run persists nothing to deploy") + } + if file == "" { + return fmt.Errorf("missing required flag --file") + } + data, err := os.ReadFile(file) + if err != nil { + return fmt.Errorf("failed to read manifest file: %w", err) + } + if len(data) > appApplyMaxBytes { + return fmt.Errorf("manifest exceeds the 1 MiB apply bound (%d bytes)", len(data)) + } + resp, err := plane.ApplyApp(ctx, dto.AppApplyRequest{ManifestTOML: string(data), DryRun: dryRun}) + if err != nil { + return err + } + if !chainDeploy { + if jsonOut { + return writeJSON(out, resp) + } + return renderAppApply(out, resp, dryRun) + } + if jsonOut { + return runAppsApplyDeployJSON(ctx, plane, resp, out, errOut) + } + if err := renderAppApply(out, resp, dryRun); err != nil { + return err + } + return runAppsApplyDeploy(ctx, plane, resp, out, errOut) +} + +// appApplyDeployDocument is the single machine-readable apply+deploy result. +// The apply is always included so a failed chained deploy never hides that +// the persist itself succeeded. +type appApplyDeployDocument struct { + Apply *dto.AppApplyResponse `json:"apply"` + Deploy *dto.AppDeployResponse `json:"deploy"` +} + +// runAppsApplyDeploy follows a chained deploy in human mode. The apply outcome +// is already rendered, then an accepted (202 running) deploy reuses the shared +// by-key watch: progress only on change, terminal journal, and Ctrl-C guidance +// that the daemon-side operation keeps running. A terminal partial/failed +// outcome exits nonzero without hiding the successful apply. +func runAppsApplyDeploy(ctx context.Context, plane ControlPlane, apply *dto.AppApplyResponse, out, errOut io.Writer) error { + deployResp, key, err := plane.DeployApp(ctx, apply.App, dto.AppDeployRequest{Revision: apply.ResultingRevision, All: true}) + if err != nil { + if conflictErr, ok := renderAppOpConflict(out, "deploy", apply.App, key, err, false); ok { + return applySucceededError(apply, conflictErr) + } + return applySucceededError(apply, appMutationError("deploy", apply.App, key, err)) + } + if deployResp.Status == dto.AppStatusRunning { + if werr := watchOperation(ctx, plane, apply.App, key, deployResp, out, errOut, false); werr != nil { + return applySucceededError(apply, werr) + } + return nil + } + return renderAppDeployResponse(out, deployResp) +} + +// runAppsApplyDeployJSON follows a chained deploy and writes exactly one final +// {apply,deploy} document once the operation is terminal. The mutation is +// issued once and observed through the existing by-key endpoint; no initial +// running document is emitted, and progress, transient warnings, and Ctrl-C +// resume guidance go to errOut so stdout stays machine-readable. +func runAppsApplyDeployJSON(ctx context.Context, plane ControlPlane, apply *dto.AppApplyResponse, out, errOut io.Writer) error { + deployResp, key, err := plane.DeployApp(ctx, apply.App, dto.AppDeployRequest{Revision: apply.ResultingRevision, All: true}) + if err != nil { + var conflict *remote.AppOpConflictError + if errors.As(err, &conflict) { + if werr := writeJSON(out, appApplyDeployDocument{Apply: apply, Deploy: &conflict.Response}); werr != nil { + return werr + } + return applySucceededError(apply, appOpConflictMessage("deploy", apply.App, key, conflict)) + } + return applySucceededError(apply, appMutationError("deploy", apply.App, key, err)) + } + terminal := deployResp + if deployResp.Status == dto.AppStatusRunning { + op, werr := awaitOperation(ctx, plane, apply.App, key, deployResp, errOut) + if werr != nil { + return applySucceededError(apply, werr) + } + terminal = op + } + if werr := writeJSON(out, appApplyDeployDocument{Apply: apply, Deploy: terminal}); werr != nil { + return werr + } + if terminal.Outcome == domain.AppOutcomeSuccess { + return nil + } + return applySucceededError(apply, &OperationFailedError{App: apply.App, Op: terminal.Op, Outcome: terminal.Outcome}) +} + +// applySucceededError preserves the apply outcome in every chained-deploy +// failure path: the desired state was already persisted, so the caller must +// not be told the apply failed. +func applySucceededError(apply *dto.AppApplyResponse, cause error) error { + return fmt.Errorf("apply of %s succeeded (%s); %w", apply.App, apply.ResultingRevision, cause) +} + +func renderAppApply(out io.Writer, resp *dto.AppApplyResponse, dryRun bool) error { + if resp.Noop { + return cliWriteLine(out, cliRenderMuted(fmt.Sprintf("No changes for %s%s", resp.App, appRevisionSuffix(resp.ResultingRevision)))) + } + verb := "Applied" + if dryRun { + verb = "Validated (dry-run)" + } + if err := cliWriteLine(out, cliRenderSuccess(fmt.Sprintf("%s %s%s", verb, resp.App, appRevisionSuffix(resp.ResultingRevision)))); err != nil { + return err + } + if err := renderAppDiffSection(out, resp.Diff); err != nil { + return err + } + if resp.Pending { + return cliWriteLine(out, cliRenderMeta("intent:", resp.Intent)) + } + return nil +} + +// appRevisionSuffix omits the parenthesized revision when empty (dry-run +// validation may not produce a revision). +func appRevisionSuffix(revision string) string { + if revision == "" { + return "" + } + return " (" + revision + ")" +} + +// renderAppDiffSection renders normalized diff paths using the same +// added/removed/changed sections as `apps diff`; empty diffs render nothing. +func renderAppDiffSection(out io.Writer, diff dto.AppDiffSection) error { + if len(diff.Added) == 0 && len(diff.Removed) == 0 && len(diff.Changed) == 0 { + return nil + } + for _, section := range []struct { + title string + paths []string + }{ + {"added:", diff.Added}, + {"removed:", diff.Removed}, + {"changed:", diff.Changed}, + } { + paths := append([]string(nil), section.paths...) + sort.Strings(paths) + for _, p := range paths { + if err := cliWriteLine(out, cliRenderMeta(section.title, p)); err != nil { + return err + } + } + } + return nil +} + +// newAppsListCmd creates `apps list`. +func newAppsListCmd() *cobra.Command { + var jsonOut bool + cmd := &cobra.Command{ + Use: "list", + Short: "List applications", + RunE: func(cmd *cobra.Command, args []string) error { + handle, err := resolveAppPlane() + if err != nil { + return err + } + defer handle.close() + plane := handle.plane + return runAppsList(cmd.Context(), plane, cmd.OutOrStdout(), jsonOut) + }, + } + cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") + return cmd +} + +func runAppsList(ctx context.Context, plane ControlPlane, out io.Writer, jsonOut bool) error { + apps, err := plane.ListApps(ctx) + if err != nil { + return err + } + if jsonOut { + if apps == nil { + apps = []dto.AppSummaryDTO{} + } + return writeJSON(out, apps) + } + if len(apps) == 0 { + return cliWriteLine(out, cliRenderEmptyState("No apps.")) + } + names := make([]string, 0, len(apps)) + byName := make(map[string]dto.AppSummaryDTO, len(apps)) + for _, a := range apps { + names = append(names, a.App) + byName[a.App] = a + } + sort.Strings(names) + for _, name := range names { + a := byName[name] + state := "active" + switch { + case a.Stopped: + state = "stopped" + case a.Pending: + state = "pending" + case a.Active == "": + state = "applied" + case a.LastOutcome != "" && a.LastOutcome != "success": + state += " (" + a.LastOutcome + ")" + } + if err := cliWriteLine(out, cliRenderListItem(fmt.Sprintf("%s (%s)", name, state))); err != nil { + return err + } + } + return nil +} + +// newAppsShowCmd creates `apps show`. +func newAppsShowCmd() *cobra.Command { + var jsonOut bool + cmd := &cobra.Command{ + Use: "show APP", + Short: "Show desired and active state for an app", + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + handle, err := resolveAppPlane() + if err != nil { + return err + } + defer handle.close() + plane := handle.plane + return runAppsShow(cmd.Context(), plane, args[0], cmd.OutOrStdout(), jsonOut) + }, + } + cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") + return cmd +} + +func runAppsShow(ctx context.Context, plane ControlPlane, app string, out io.Writer, jsonOut bool) error { + resp, err := plane.ShowApp(ctx, app) + if err != nil { + return err + } + if jsonOut { + return writeJSON(out, resp) + } + if err := cliWriteLine(out, cliRenderTitle(resp.App)); err != nil { + return err + } + if err := cliWriteLine(out, cliRenderMeta("desired:", resp.Desired.Revision+" ("+resp.Desired.Status+")")); err != nil { + return err + } + for _, line := range renderActiveServices(resp.Active) { + if err := cliWriteLine(out, line); err != nil { + return err + } + } + if resp.Intent.Stopped { + if err := cliWriteLine(out, cliRenderWarning("stopped intent is set")); err != nil { + return err + } + } + if resp.LastOp != nil && resp.LastOp.Op != "" { + if err := cliWriteLine(out, cliRenderMeta("last-op:", resp.LastOp.Op+" ("+resp.LastOp.Outcome+")")); err != nil { + return err + } + } + owned := append(append([]string(nil), resp.Retained.Volumes...), resp.Retained.Images...) + owned = append(owned, resp.Retained.Secrets...) + if len(owned) > 0 { + sort.Strings(owned) + return cliWriteLine(out, cliRenderMeta("retained:", strings.Join(owned, " "))) + } + return nil +} + +// renderActiveServices renders per-service effective state, sorted by name. +// Only ids and digests appear — never secret values. +func renderActiveServices(active dto.AppActiveDTO) []string { + names := make([]string, 0, len(active.Services)) + for name := range active.Services { + names = append(names, name) + } + sort.Strings(names) + lines := make([]string, 0, len(names)+1) + converged := "converged" + if !active.Converged { + converged = "diverged" + } + lines = append(lines, cliRenderMeta("active:", converged)) + for _, name := range names { + svc := active.Services[name] + detail := svc.EffectiveRevision + if svc.Container != "" { + detail += " container=" + svc.Container + } + if svc.RestartUnsafe { + detail += " restart_unsafe" + } + lines = append(lines, cliRenderMeta(" "+name+":", detail)) + } + return lines +} + +// newAppsDiffCmd creates `apps diff`. +func newAppsDiffCmd() *cobra.Command { + var jsonOut bool + cmd := &cobra.Command{ + Use: "diff APP", + Short: "Show the normalized desired-vs-active diff", + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + handle, err := resolveAppPlane() + if err != nil { + return err + } + defer handle.close() + plane := handle.plane + return runAppsDiff(cmd.Context(), plane, args[0], cmd.OutOrStdout(), jsonOut) + }, + } + cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") + return cmd +} + +func runAppsDiff(ctx context.Context, plane ControlPlane, app string, out io.Writer, jsonOut bool) error { + resp, err := plane.DiffApp(ctx, app) + if err != nil { + return err + } + if jsonOut { + return writeJSON(out, resp) + } + if len(resp.Diff.Added) == 0 && len(resp.Diff.Removed) == 0 && len(resp.Diff.Changed) == 0 { + return cliWriteLine(out, cliRenderMuted(fmt.Sprintf("No differences for %s.", app))) + } + for _, section := range []struct { + title string + paths []string + }{ + {"added:", resp.Diff.Added}, + {"removed:", resp.Diff.Removed}, + {"changed:", resp.Diff.Changed}, + } { + paths := append([]string(nil), section.paths...) + sort.Strings(paths) + for _, p := range paths { + if err := cliWriteLine(out, cliRenderMeta(section.title, p)); err != nil { + return err + } + } + } + return nil +} + +// newAppsSecretsCmd creates the `apps secrets` parent command. +func newAppsSecretsCmd() *cobra.Command { + cmd := &cobra.Command{ + Use: "secrets", + Short: "Manage app secret values", + Long: `Values are accepted via KEY=VALUE arguments (discouraged: shell history), +--stdin with KEY=VALUE lines, or --stdin --key KEY for one raw value. Names +must already exist in desired or active state. Only key names are ever echoed +back — never values.`, + } + cmd.AddCommand(newAppsSecretsListCmd(), newAppsSecretsSetCmd(), newAppsSecretsDeleteCmd()) + return cmd +} + +func newAppsSecretsListCmd() *cobra.Command { + var service string + var jsonOut bool + cmd := &cobra.Command{Use: "list APP", Short: "List app secret registration metadata", Args: cobra.ExactArgs(1), RunE: func(cmd *cobra.Command, args []string) error { + handle, err := resolveAppPlane() + if err != nil { + return err + } + defer handle.close() + return runAppsSecretsList(cmd.Context(), handle.plane, cmd.OutOrStdout(), args[0], service, jsonOut) + }} + cmd.Flags().StringVar(&service, "service", "", "Filter by service") + cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") + return cmd +} + +// runAppsSecretsList renders metadata-only registrations for one app, +// optionally filtered to one service. Secret values never appear. +func runAppsSecretsList(ctx context.Context, plane ControlPlane, out io.Writer, app, service string, jsonOut bool) error { + entries, err := plane.ListAppSecrets(ctx, app, service) + if err != nil { + return fmt.Errorf("list app secrets for %s: %w", app, err) + } + if jsonOut { + return writeJSON(out, entries) + } + for _, entry := range entries { + if err := cliWriteLine(out, cliRenderMeta(entry.Service+"/"+entry.Key+":", entry.Name+" "+entry.Source+" "+entry.Presence)); err != nil { + return err + } + } + if len(entries) == 0 { + return cliWriteLine(out, cliRenderMuted("No registered secrets")) + } + return nil +} + +// newAppsSecretsSetCmd creates `apps secrets set`. +func newAppsSecretsSetCmd() *cobra.Command { + var service string + var key string + var fromStdin bool + var jsonOut bool + cmd := &cobra.Command{ + Use: "set APP KEY=VALUE…", + Short: "Write app secret values", + Args: cobra.MinimumNArgs(0), + RunE: func(cmd *cobra.Command, args []string) error { + handle, err := resolveAppPlane() + if err != nil { + return err + } + defer handle.close() + plane := handle.plane + if len(args) == 0 { + return fmt.Errorf("missing APP argument") + } + return runAppsSecretsSetMode(cmd.Context(), plane, cmd.InOrStdin(), cmd.OutOrStdout(), args[0], args[1:], service, key, fromStdin, jsonOut) + }, + } + cmd.Flags().StringVar(&service, "service", "", "Service the secrets belong to (required)") + cmd.Flags().StringVar(&key, "key", "", "Secret key for single-value stdin mode") + cmd.Flags().BoolVar(&fromStdin, "stdin", false, "Read KEY=VALUE lines, or one raw value with --key") + cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") + return cmd +} + +func runAppsSecretsSet(ctx context.Context, plane ControlPlane, stdin io.Reader, out io.Writer, app string, pairs []string, service string, fromStdin, jsonOut bool) error { + return runAppsSecretsSetMode(ctx, plane, stdin, out, app, pairs, service, "", fromStdin, jsonOut) +} + +func runAppsSecretsSetMode(ctx context.Context, plane ControlPlane, stdin io.Reader, out io.Writer, app string, pairs []string, service, key string, fromStdin, jsonOut bool) error { + if app == "" { + return fmt.Errorf("missing APP argument") + } + if service == "" { + return fmt.Errorf("missing required flag --service: secrets are service-scoped") + } + if key != "" { + if !fromStdin || len(pairs) != 0 { + return fmt.Errorf("--key requires --stdin and forbids KEY=VALUE arguments") + } + value, err := readRawSecretStdin(stdin) + if err != nil { + return err + } + pairs = []string{key + "=" + value} + } else if fromStdin { + stdinPairs, err := readSecretStdin(stdin) + if err != nil { + return err + } + pairs = append(pairs, stdinPairs...) + } + values, err := parseSecretPairs(pairs) + if err != nil { + return err + } + if len(values) == 0 { + return fmt.Errorf("no secrets provided: pass KEY=VALUE arguments or --stdin") + } + if err := plane.SetAppSecrets(ctx, app, dto.AppSecretSetRequest{Service: service, Secrets: values}); err != nil { + return err + } + keys := make([]string, 0, len(values)) + for k := range values { + keys = append(keys, k) + } + sort.Strings(keys) + if jsonOut { + return writeJSON(out, map[string]any{"app": app, "service": service, "keys": keys}) + } + if err := cliWriteLine(out, cliRenderSuccess(fmt.Sprintf("Set %d secret(s) for %s/%s: %s", len(keys), app, service, strings.Join(keys, ", ")))); err != nil { + return err + } + return cliWriteLine(out, cliRenderMuted(fmt.Sprintf("Running containers keep their old values. Apply them with: gordon apps deploy %s --service %s", app, service))) +} + +// newAppsSecretsDeleteCmd creates `apps secrets delete`. +func newAppsSecretsDeleteCmd() *cobra.Command { + var service string + var jsonOut bool + cmd := &cobra.Command{ + Use: "delete APP KEY", + Short: "Delete an app secret value", + Args: cobra.ExactArgs(2), + RunE: func(cmd *cobra.Command, args []string) error { + handle, err := resolveAppPlane() + if err != nil { + return err + } + defer handle.close() + plane := handle.plane + return runAppsSecretsDelete(cmd.Context(), plane, cmd.OutOrStdout(), args[0], args[1], service, jsonOut) + }, + } + cmd.Flags().StringVar(&service, "service", "", "Service the secret belongs to (required)") + cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") + return cmd +} + +func runAppsSecretsDelete(ctx context.Context, plane ControlPlane, out io.Writer, app, key, service string, jsonOut bool) error { + if service == "" { + return fmt.Errorf("missing required flag --service: secrets are service-scoped") + } + if key == "" { + return fmt.Errorf("missing KEY argument") + } + if err := plane.DeleteAppSecret(ctx, app, dto.AppSecretDeleteRequest{Service: service, Key: key}); err != nil { + return err + } + if jsonOut { + return writeJSON(out, map[string]any{"app": app, "service": service, "deleted": key}) + } + return cliWriteLine(out, cliRenderSuccess(fmt.Sprintf("Deleted secret %s for %s/%s", key, app, service))) +} + +// parseSecretPairs parses KEY=VALUE arguments. Keys must be non-empty and +// each value must pass validateSecretValue (1-MaxAppEnvValueLen bytes on a +// single line); an empty value is rejected. +func parseSecretPairs(pairs []string) (map[string]string, error) { + values := make(map[string]string, len(pairs)) + for index, pair := range pairs { + key, value, ok := strings.Cut(pair, "=") + if !ok || key == "" { + return nil, fmt.Errorf("invalid secret input %d: expected KEY=VALUE", index+1) + } + if err := validateSecretValue(value); err != nil { + return nil, fmt.Errorf("invalid secret input %d: %w", index+1, err) + } + values[key] = value + } + return values, nil +} + +func validateSecretValue(value string) error { + if value == "" || len(value) > domain.MaxAppEnvValueLen { + return fmt.Errorf("value must be 1-%d bytes", domain.MaxAppEnvValueLen) + } + if strings.ContainsAny(value, "\r\n") { + return fmt.Errorf("value must be a single line") + } + return nil +} + +func readRawSecretStdin(in io.Reader) (string, error) { + data, err := io.ReadAll(io.LimitReader(in, domain.MaxAppEnvValueLen+3)) + if err != nil { + return "", fmt.Errorf("failed to read secret from stdin: %w", err) + } + if strings.HasSuffix(string(data), "\r\n") { + data = data[:len(data)-2] + } else if strings.HasSuffix(string(data), "\n") { + data = data[:len(data)-1] + } + value := string(data) + if err := validateSecretValue(value); err != nil { + return "", fmt.Errorf("invalid stdin secret: %w", err) + } + return value, nil +} + +const maxSecretStdinBytes = 1 << 20 + +// readSecretStdin reads bounded KEY=VALUE lines. Whitespace-only lines are +// ignored, while every byte in nonblank values is preserved. +func readSecretStdin(in io.Reader) ([]string, error) { + data, err := io.ReadAll(io.LimitReader(in, maxSecretStdinBytes+1)) + if err != nil { + return nil, fmt.Errorf("failed to read secrets from stdin: %w", err) + } + if len(data) > maxSecretStdinBytes { + return nil, fmt.Errorf("secret input exceeds %d bytes", maxSecretStdinBytes) + } + var pairs []string + for _, line := range strings.Split(string(data), "\n") { + line = strings.TrimSuffix(line, "\r") + if strings.TrimSpace(line) != "" { + pairs = append(pairs, line) + } + } + return pairs, nil +} + +// newAppDeployCmd creates `deploy APP`. At cutover this replaces the +// domain-based deploy for apps; domain-based deploy keeps working for +// non-app resources until then (05-api-cli.md §5). +func newAppDeployCmd() *cobra.Command { + var revision string + var service string + var all bool + var jsonOut bool + cmd := &cobra.Command{ + Use: "deploy APP", + Short: "Activate an app revision", + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + handle, err := resolveAppPlane() + if err != nil { + return err + } + defer handle.close() + plane := handle.plane + req := dto.AppDeployRequest{Revision: revision, Service: service, All: all} + return runAppDeploy(cmd.Context(), plane, args[0], req, cmd.OutOrStdout(), cmd.ErrOrStderr(), jsonOut) + }, + } + cmd.Flags().StringVar(&revision, "revision", "", "Revision to activate (default: desired head)") + cmd.Flags().StringVar(&service, "service", "", "Deploy a single service only") + cmd.Flags().BoolVar(&all, "all", false, "Deploy every service (required for multi-service apps without --service)") + cmd.MarkFlagsMutuallyExclusive("service", "all") + cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") + return cmd +} + +// runAppDeploy issues one deploy mutation and, when the daemon answers 202 +// with a running journal, polls the existing by-key endpoint to terminal +// rather than reissuing the mutation or hiding the outcome. +func runAppDeploy(ctx context.Context, plane ControlPlane, app string, req dto.AppDeployRequest, out, errOut io.Writer, jsonOut bool) error { + resp, key, err := plane.DeployApp(ctx, app, req) + if err != nil { + if conflictErr, ok := renderAppOpConflict(out, "deploy", app, key, err, jsonOut); ok { + return conflictErr + } + return appMutationError("deploy", app, key, err) + } + if resp.Status == dto.AppStatusRunning { + return watchOperation(ctx, plane, app, key, resp, out, errOut, jsonOut) + } + if jsonOut { + return writeJSON(out, resp) + } + return renderAppDeployResponse(out, resp) +} + +// newAppRestartCmd creates `restart APP`: pinned digests, no re-resolve. +func newAppRestartCmd() *cobra.Command { + var service string + var all bool + var jsonOut bool + cmd := &cobra.Command{ + Use: "restart APP", + Short: "Restart an app from pinned digests", + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + handle, err := resolveAppPlane() + if err != nil { + return err + } + defer handle.close() + plane := handle.plane + return runAppRestart(cmd.Context(), plane, args[0], service, all, cmd.OutOrStdout(), jsonOut) + }, + } + cmd.Flags().StringVar(&service, "service", "", "Restart a single service only") + cmd.Flags().BoolVar(&all, "all", false, "Restart every service (required for multi-service apps without --service)") + cmd.MarkFlagsMutuallyExclusive("service", "all") + cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") + return cmd +} + +func runAppRestart(ctx context.Context, plane ControlPlane, app, service string, all bool, out io.Writer, jsonOut bool) error { + resp, key, err := plane.RestartApp(ctx, app, service, all) + if err != nil { + if conflictErr, ok := renderAppOpConflict(out, "restart", app, key, err, jsonOut); ok { + return conflictErr + } + return appMutationError("restart", app, key, err) + } + if jsonOut { + return writeJSON(out, resp) + } + return renderAppDeployResponse(out, resp) +} + +// newAppStopCmd creates `stop APP`: durable stopped intent, data preserved. +func newAppStopCmd() *cobra.Command { + var jsonOut bool + cmd := &cobra.Command{ + Use: "stop APP", + Short: "Stop an app (preserves all data)", + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + handle, err := resolveAppPlane() + if err != nil { + return err + } + defer handle.close() + plane := handle.plane + return runAppLifecycle(cmd.Context(), plane.StopApp, "stop", args[0], cmd.OutOrStdout(), jsonOut) + }, + } + cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") + return cmd +} + +// newAppStartCmd creates `start APP`: clears intent, ensures running. +func newAppStartCmd() *cobra.Command { + var jsonOut bool + cmd := &cobra.Command{ + Use: "start APP", + Short: "Start a stopped app from active state", + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + handle, err := resolveAppPlane() + if err != nil { + return err + } + defer handle.close() + plane := handle.plane + return runAppLifecycle(cmd.Context(), plane.StartApp, "start", args[0], cmd.OutOrStdout(), jsonOut) + }, + } + cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") + return cmd +} + +// newAppRemoveCmd creates `remove APP`: withdraws workloads, retains data. +func newAppRemoveCmd() *cobra.Command { + var jsonOut bool + cmd := &cobra.Command{ + Use: "remove APP", + Short: "Remove app workloads (volumes and secrets are retained)", + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + handle, err := resolveAppPlane() + if err != nil { + return err + } + defer handle.close() + plane := handle.plane + return runAppLifecycle(cmd.Context(), plane.RemoveApp, "remove", args[0], cmd.OutOrStdout(), jsonOut) + }, + } + cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") + return cmd +} + +// appLifecycleFunc is one body-less app lifecycle mutation. +type appLifecycleFunc func(ctx context.Context, app string) (*dto.AppDeployResponse, string, error) + +func runAppLifecycle(ctx context.Context, fn appLifecycleFunc, op, app string, out io.Writer, jsonOut bool) error { + resp, key, err := fn(ctx, app) + if err != nil { + if conflictErr, ok := renderAppOpConflict(out, op, app, key, err, jsonOut); ok { + return conflictErr + } + return appMutationError(op, app, key, err) + } + if jsonOut { + return writeJSON(out, resp) + } + return renderAppDeployResponse(out, resp) +} + +// renderAppDeployResponse renders the deploy outcome with terminal +// per-service results, cleanup warnings, and the effective/retained +// summaries. Sorts services for stable output. +func renderAppDeployResponse(out io.Writer, resp *dto.AppDeployResponse) error { + if err := renderDeployHeadline(out, resp); err != nil { + return err + } + if err := renderDeployServices(out, resp); err != nil { + return err + } + if err := renderDeploySteps(out, resp.Steps); err != nil { + return err + } + return renderDeploySummaries(out, resp) +} + +func renderDeploySteps(out io.Writer, steps []dto.AppStepDTO) error { + for _, step := range steps { + detail := step.State + if step.Detail != "" { + detail += ": " + sanitizeTerminalText(step.Detail) + } + if step.Error != "" { + detail += ": " + sanitizeTerminalText(step.Error) + } + if err := cliWriteLine(out, cliRenderMeta(" step "+step.ID+":", detail)); err != nil { + return err + } + for _, diagnostic := range step.Diagnostics { + if err := cliWriteLine(out, cliRenderMuted(" "+sanitizeTerminalText(diagnostic))); err != nil { + return err + } + } + } + return nil +} + +func sanitizeTerminalText(value string) string { + return strings.Map(func(r rune) rune { + if r == '\n' || r == '\t' || r >= ' ' && r != '\x7f' { + return r + } + return '�' + }, value) +} + +func renderDeployHeadline(out io.Writer, resp *dto.AppDeployResponse) error { + headline := fmt.Sprintf("%s %s: %s", resp.Op, resp.App, resp.Outcome) + if resp.Outcome == "success" { + return cliWriteLine(out, cliRenderSuccess(headline)) + } + return cliWriteLine(out, cliRenderWarning(headline)) +} + +func renderDeployServices(out io.Writer, resp *dto.AppDeployResponse) error { + names := make([]string, 0, len(resp.Services)) + for name := range resp.Services { + names = append(names, name) + } + sort.Strings(names) + for _, name := range names { + svc := resp.Services[name] + detail := svc.Result + " " + svc.EffectiveRevision + if svc.Result == domain.AppServiceUnchanged { + detail += " (already running this image, config, and secrets)" + } + if svc.RestartUnsafe { + detail += " restart_unsafe" + } + if svc.Error != "" { + detail += ": " + svc.Error + } + if err := cliWriteLine(out, cliRenderMeta(" "+name+":", detail)); err != nil { + return err + } + } + for _, w := range resp.CleanupWarnings { + if err := cliWriteLine(out, cliRenderWarning(fmt.Sprintf("cleanup: %s left %s (%s)", w.Service, w.Leftover, w.Detail))); err != nil { + return err + } + } + return nil +} + +func renderDeploySummaries(out io.Writer, resp *dto.AppDeployResponse) error { + if resp.Effective != nil && len(resp.Effective.Services) > 0 { + pairs := make([]string, 0, len(resp.Effective.Services)) + for name, rev := range resp.Effective.Services { + pairs = append(pairs, name+"="+rev) + } + sort.Strings(pairs) + if err := cliWriteLine(out, cliRenderMeta("effective:", strings.Join(pairs, " "))); err != nil { + return err + } + } + if resp.Retained == nil { + return nil + } + owned := append(append([]string(nil), resp.Retained.Volumes...), resp.Retained.Images...) + owned = append(owned, resp.Retained.Secrets...) + if len(owned) > 0 { + sort.Strings(owned) + if err := cliWriteLine(out, cliRenderMeta("retained:", strings.Join(owned, " "))); err != nil { + return err + } + } + return nil +} + +// newAppStatusCmd creates `status APP`: effective vs observed, per service. +func newAppStatusCmd() *cobra.Command { + var jsonOut bool + cmd := &cobra.Command{ + Use: "status APP", + Short: "Show effective vs observed state for an app", + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + handle, err := resolveAppPlane() + if err != nil { + return err + } + defer handle.close() + plane := handle.plane + return runAppStatus(cmd.Context(), plane, args[0], cmd.OutOrStdout(), jsonOut) + }, + } + cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") + return cmd +} + +func runAppStatus(ctx context.Context, plane ControlPlane, app string, out io.Writer, jsonOut bool) error { + resp, err := plane.ShowApp(ctx, app) + if err != nil { + return err + } + if jsonOut { + return writeJSON(out, resp) + } + if err := cliWriteLine(out, cliRenderTitle(resp.App)); err != nil { + return err + } + for _, line := range renderActiveServices(resp.Active) { + if err := cliWriteLine(out, line); err != nil { + return err + } + } + var running []string + for _, svc := range resp.Active.Services { + if svc.Container != "" { + running = append(running, svc.Container) + } + } + sort.Strings(running) + observed := "none" + if len(running) > 0 { + observed = strings.Join(running, " ") + } + if err := cliWriteLine(out, cliRenderMeta("observed:", observed)); err != nil { + return err + } + if resp.Intent.Stopped { + return cliWriteLine(out, cliRenderWarning("stopped intent is set")) + } + return nil +} + +// appLogReader streams logs by Gordon app/service reference. The daemon +// resolves the reference through its authoritative ACTIVE record. +type appLogReader interface { + GetContainerLogs(ctx context.Context, ref string, lines int) ([]string, error) + StreamContainerLogs(ctx context.Context, ref string, lines int) (<-chan string, error) +} + +// remoteAppLogReader adapts *remote.Client to appLogReader. +type remoteAppLogReader struct { + client *remote.Client +} + +func (r *remoteAppLogReader) GetContainerLogs(ctx context.Context, ref string, lines int) ([]string, error) { + return r.client.GetContainerLogs(ctx, ref, lines) +} + +func (r *remoteAppLogReader) StreamContainerLogs(ctx context.Context, ref string, lines int) (<-chan string, error) { + return r.client.StreamContainerLogs(ctx, ref, lines) +} + +// newAppLogsCmd creates `logs APP`. +func newAppLogsCmd() *cobra.Command { + var service string + var follow bool + var tail int + var jsonOut bool + cmd := &cobra.Command{ + Use: "logs APP", + Short: "Show logs for an app service", + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + client, err := resolveAppClient() + if err != nil { + return err + } + return runAppLogs(cmd.Context(), client, &remoteAppLogReader{client: client}, args[0], service, follow, tail, cmd.OutOrStdout(), jsonOut) + }, + } + cmd.Flags().StringVar(&service, "service", "", "Service to read logs from (required when the app has several)") + cmd.Flags().BoolVarP(&follow, "follow", "f", false, "Follow log output") + cmd.Flags().IntVarP(&tail, "tail", "n", 50, "Number of lines to show") + cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") + return cmd +} + +func runAppLogs(ctx context.Context, plane ControlPlane, reader appLogReader, app, service string, follow bool, tail int, out io.Writer, jsonOut bool) error { + if follow && jsonOut { + return fmt.Errorf("cannot combine --json with --follow: follow streams plain log lines") + } + show, err := plane.ShowApp(ctx, app) + if err != nil { + return err + } + ref, err := appLogRef(show, service) + if err != nil { + return err + } + if follow { + stream, err := reader.StreamContainerLogs(ctx, ref, tail) + if err != nil { + return err + } + for line := range stream { + if err := cliWriteLine(out, line); err != nil { + return err + } + } + return nil + } + lines, err := reader.GetContainerLogs(ctx, ref, tail) + if err != nil { + return err + } + if jsonOut { + return writeJSON(out, map[string]any{"app": app, "service": service, "lines": lines}) + } + for _, line := range lines { + if err := cliWriteLine(out, line); err != nil { + return err + } + } + return nil +} + +// appLogRef resolves the container ref for one app service. When service +// is empty and the app has exactly one service, that service is used; +// otherwise --service is required. A service with no recorded container +// has nothing to stream yet. +func appLogRef(show *dto.AppShowResponse, service string) (string, error) { + if len(show.Active.Services) == 0 { + return "", fmt.Errorf("app %s has no active services yet", show.App) + } + if service == "" { + if len(show.Active.Services) == 1 { + for name := range show.Active.Services { + service = name + } + } else { + names := make([]string, 0, len(show.Active.Services)) + for name := range show.Active.Services { + names = append(names, name) + } + sort.Strings(names) + return "", fmt.Errorf("app %s has several services (%s): pass --service", show.App, strings.Join(names, ", ")) + } + } + svc, ok := show.Active.Services[service] + if !ok { + return "", fmt.Errorf("app %s has no service %q", show.App, service) + } + if svc.Container == "" { + return "", fmt.Errorf("service %s has no recorded container yet", service) + } + return show.App + "/" + service, nil +} diff --git a/internal/adapters/in/cli/apps_apply_watch_test.go b/internal/adapters/in/cli/apps_apply_watch_test.go new file mode 100644 index 000000000..1294ff27e --- /dev/null +++ b/internal/adapters/in/cli/apps_apply_watch_test.go @@ -0,0 +1,187 @@ +package cli + +// Tests for the `apps apply --deploy` chain over a 202 accepted deploy: +// it must reuse the shared by-key watch (progress only on change, terminal +// journal, Ctrl-C resume) and must never reissue the deploy POST. JSON mode +// keeps stdout to exactly one final combined {apply,deploy} document while +// progress and resume guidance go to stderr. + +import ( + "bytes" + "context" + "encoding/json" + "io" + "strings" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/dto" +) + +func acceptedApply() *dto.AppApplyResponse { + return &dto.AppApplyResponse{App: "blog", ResultingRevision: "rev-b", Pending: true, Intent: "apply-1"} +} + +func TestRunAppsApply_DeployAcceptedPollsToSuccess(t *testing.T) { + shortenOperationPoll(t) + plane := appPlane(t) + plane.EXPECT().ApplyApp(mock.Anything, mock.Anything).Return(acceptedApply(), nil).Once() + plane.EXPECT().DeployApp(mock.Anything, "blog", mock.Anything).Return( + runningOp("op-1", dto.AppStepDTO{ID: "service.web.start", State: "pending"}), "key-1", nil).Once() + plane.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return( + &dto.AppDeployResponse{Op: "op-1", App: "blog", Status: "success", Outcome: "success"}, nil).Once() + + var out, errOut bytes.Buffer + require.NoError(t, runAppsApply(context.Background(), plane, strings.NewReader(""), &out, &errOut, + writeManifest(t, "[app]\nname = \"blog\"\n"), false, true, false)) + text := out.String() + assert.Contains(t, text, "Applied blog (rev-b)", "the apply outcome stays visible") + assert.Contains(t, text, "op op-1:", "the accepted deploy reports progress through the watch path") + assert.Contains(t, text, "op-1 blog: success", "the terminal journal is rendered") +} + +func TestRunAppsApply_DeployAcceptedNoSecondMutation(t *testing.T) { + shortenOperationPoll(t) + plane := appPlane(t) + plane.EXPECT().ApplyApp(mock.Anything, mock.Anything).Return(acceptedApply(), nil).Once() + // .Once() proves the accepted deploy is observed by key, never re-POSTed. + plane.EXPECT().DeployApp(mock.Anything, "blog", mock.Anything).Return(runningOp("op-1"), "key-1", nil).Once() + plane.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return( + &dto.AppDeployResponse{Op: "op-1", App: "blog", Status: "success", Outcome: "success"}, nil).Once() + + var out, errOut bytes.Buffer + require.NoError(t, runAppsApply(context.Background(), plane, strings.NewReader(""), &out, &errOut, + writeManifest(t, "x"), false, true, false)) +} + +func TestRunAppsApply_DeployAcceptedUnchangedStateNotReprinted(t *testing.T) { + shortenOperationPoll(t) + plane := appPlane(t) + step := dto.AppStepDTO{ID: "service.web.start", State: "pending"} + advanced := runningOp("op-1", + dto.AppStepDTO{ID: "service.web.start", State: "succeeded"}, + dto.AppStepDTO{ID: "service.web.health", State: "pending"}) + + plane.EXPECT().ApplyApp(mock.Anything, mock.Anything).Return(acceptedApply(), nil).Once() + plane.EXPECT().DeployApp(mock.Anything, "blog", mock.Anything).Return(runningOp("op-1", step), "key-1", nil).Once() + plane.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return(runningOp("op-1", step), nil).Once() + plane.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return(advanced, nil).Once() + plane.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return( + &dto.AppDeployResponse{Op: "op-1", App: "blog", Status: "success", Outcome: "success"}, nil).Once() + + var out, errOut bytes.Buffer + require.NoError(t, runAppsApply(context.Background(), plane, strings.NewReader(""), &out, &errOut, + writeManifest(t, "x"), false, true, false)) + assert.Equal(t, 2, strings.Count(out.String(), "op op-1:"), + "initial and one advanced poll render progress; the unchanged poll is not reprinted") +} + +func TestRunAppsApply_DeployAcceptedPartialFailsButKeepsApply(t *testing.T) { + shortenOperationPoll(t) + plane := appPlane(t) + plane.EXPECT().ApplyApp(mock.Anything, mock.Anything).Return(acceptedApply(), nil).Once() + plane.EXPECT().DeployApp(mock.Anything, "blog", mock.Anything).Return(runningOp("op-1"), "key-1", nil).Once() + plane.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return(&dto.AppDeployResponse{ + Op: "op-1", App: "blog", Status: "partial", Outcome: "partial", + Services: map[string]dto.AppServiceResultDTO{ + "web": {Result: "failed", EffectiveRevision: "rev-b", Error: "boom"}, + }, + }, nil).Once() + + var out, errOut bytes.Buffer + err := runAppsApply(context.Background(), plane, strings.NewReader(""), &out, &errOut, + writeManifest(t, "x"), false, true, false) + require.Error(t, err, "a terminal partial deploy must exit nonzero") + assert.Contains(t, err.Error(), "apply of blog succeeded (rev-b)") + assert.Contains(t, err.Error(), "partial") + assert.Contains(t, out.String(), "Applied blog (rev-b)", "the apply outcome stays visible") + assert.Contains(t, out.String(), "boom", "the terminal journal is rendered before failing") +} + +func TestRunAppsApply_DeployAcceptedJSONKeepsStdoutSingleCombinedDocument(t *testing.T) { + shortenOperationPoll(t) + plane := appPlane(t) + plane.EXPECT().ApplyApp(mock.Anything, mock.Anything).Return(acceptedApply(), nil).Once() + plane.EXPECT().DeployApp(mock.Anything, "blog", mock.Anything).Return( + runningOp("op-1", dto.AppStepDTO{ID: "service.web.start", State: "pending"}), "key-1", nil).Once() + plane.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return( + runningOp("op-1", dto.AppStepDTO{ID: "service.web.start", State: "succeeded"}), nil).Once() + plane.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return( + &dto.AppDeployResponse{Op: "op-1", App: "blog", Status: "success", Outcome: "success"}, nil).Once() + + var out, errOut bytes.Buffer + require.NoError(t, runAppsApply(context.Background(), plane, strings.NewReader(""), &out, &errOut, + writeManifest(t, "x"), false, true, true)) + + dec := json.NewDecoder(bytes.NewReader(out.Bytes())) + var got appApplyDeployDocument + require.NoError(t, dec.Decode(&got)) + assert.Equal(t, "apply-1", got.Apply.Intent) + assert.Equal(t, "op-1", got.Deploy.Op) + assert.Equal(t, "success", got.Deploy.Outcome) + var extra json.RawMessage + assert.ErrorIs(t, dec.Decode(&extra), io.EOF, "stdout carries exactly one JSON document") + assert.NotContains(t, out.String(), "op op-1:", "progress must not pollute JSON stdout") + assert.NotContains(t, out.String(), dto.AppStatusRunning, "no initial running JSON is emitted") + assert.Contains(t, errOut.String(), "op op-1:", "progress goes to stderr in JSON mode") +} + +func TestRunAppsApply_DeployAcceptedJSONFailureKeepsApplyDocument(t *testing.T) { + shortenOperationPoll(t) + plane := appPlane(t) + plane.EXPECT().ApplyApp(mock.Anything, mock.Anything).Return(acceptedApply(), nil).Once() + plane.EXPECT().DeployApp(mock.Anything, "blog", mock.Anything).Return(runningOp("op-1"), "key-1", nil).Once() + plane.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return( + &dto.AppDeployResponse{Op: "op-1", App: "blog", Status: "failed", Outcome: "failed"}, nil).Once() + + var out, errOut bytes.Buffer + err := runAppsApply(context.Background(), plane, strings.NewReader(""), &out, &errOut, + writeManifest(t, "x"), false, true, true) + require.Error(t, err) + assert.Contains(t, err.Error(), "apply of blog succeeded (rev-b)") + + dec := json.NewDecoder(bytes.NewReader(out.Bytes())) + var got appApplyDeployDocument + require.NoError(t, dec.Decode(&got)) + assert.Equal(t, "rev-b", got.Apply.ResultingRevision, "the failed deploy must not hide the successful apply") + assert.Equal(t, "failed", got.Deploy.Outcome) + var extra json.RawMessage + assert.ErrorIs(t, dec.Decode(&extra), io.EOF, "stdout carries exactly one JSON document") +} + +func TestRunAppsApply_DeployAcceptedCancelPrintsResume(t *testing.T) { + plane := appPlane(t) + ctx, cancel := context.WithCancel(context.Background()) + cancel() + plane.EXPECT().ApplyApp(mock.Anything, mock.Anything).Return(acceptedApply(), nil).Once() + plane.EXPECT().DeployApp(mock.Anything, "blog", mock.Anything).Return(runningOp("op-1"), "key-1", nil).Once() + + var out, errOut bytes.Buffer + err := runAppsApply(ctx, plane, strings.NewReader(""), &out, &errOut, + writeManifest(t, "x"), false, true, false) + require.Error(t, err) + assert.Contains(t, err.Error(), "apply of blog succeeded (rev-b)") + text := out.String() + assert.Contains(t, text, "not cancelled", "Ctrl-C must not imply a server-side abort") + assert.Contains(t, text, "gordon apps operations watch blog --key key-1") +} + +func TestRunAppsApply_DeployAcceptedCancelKeepsJSONStdoutClean(t *testing.T) { + plane := appPlane(t) + ctx, cancel := context.WithCancel(context.Background()) + cancel() + plane.EXPECT().ApplyApp(mock.Anything, mock.Anything).Return(acceptedApply(), nil).Once() + plane.EXPECT().DeployApp(mock.Anything, "blog", mock.Anything).Return(runningOp("op-1"), "key-1", nil).Once() + + var out, errOut bytes.Buffer + err := runAppsApply(ctx, plane, strings.NewReader(""), &out, &errOut, + writeManifest(t, "x"), false, true, true) + require.Error(t, err) + assert.Contains(t, err.Error(), "apply of blog succeeded (rev-b)") + assert.Empty(t, out.String(), "JSON stdout stays empty when interrupted before terminal") + assert.Contains(t, errOut.String(), "not cancelled") + assert.Contains(t, errOut.String(), "gordon apps operations watch blog --key key-1") +} diff --git a/internal/adapters/in/cli/apps_findings_test.go b/internal/adapters/in/cli/apps_findings_test.go new file mode 100644 index 000000000..97a13465c --- /dev/null +++ b/internal/adapters/in/cli/apps_findings_test.go @@ -0,0 +1,75 @@ +package cli + +import ( + "bytes" + "context" + "strings" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/dto" + climocks "github.com/bnema/gordon/internal/adapters/in/cli/mocks" + "github.com/bnema/gordon/internal/adapters/in/cli/remote" +) + +func TestRunAppsApply_DryRunOmitsEmptyRevisionAndRendersDiff(t *testing.T) { + plane := appPlane(t) + plane.EXPECT().ApplyApp(mock.Anything, mock.Anything).Return(&dto.AppApplyResponse{ + App: "blog", + Noop: false, + Diff: dto.AppDiffSection{Added: []string{"service.web"}}, + Intent: "apply-7", + }, nil).Once() + var out bytes.Buffer + require.NoError(t, runAppsApply(context.Background(), plane, strings.NewReader(""), &out, &bytes.Buffer{}, + writeManifest(t, "[app]\nname = \"blog\"\n"), true, false, false)) + text := out.String() + assert.Contains(t, text, "Validated (dry-run) blog") + assert.NotContains(t, text, "()") + assert.Contains(t, text, "added:") + assert.Contains(t, text, "service.web") +} + +func TestRunAppsApply_NoopOmitsEmptyRevision(t *testing.T) { + plane := appPlane(t) + plane.EXPECT().ApplyApp(mock.Anything, mock.Anything). + Return(&dto.AppApplyResponse{App: "blog", Noop: true}, nil).Once() + var out bytes.Buffer + require.NoError(t, runAppsApply(context.Background(), plane, strings.NewReader(""), &out, &bytes.Buffer{}, + writeManifest(t, "x"), true, false, false)) + assert.Contains(t, out.String(), "No changes for blog") + assert.NotContains(t, out.String(), "()") +} + +func TestRunAppLogs_RejectsJSONWithFollow(t *testing.T) { + // The JSON/follow rejection happens before any read. + plane := appPlane(t) + err := runAppLogs(context.Background(), plane, &fakeAppLogReader{}, "blog", "web", true, 50, &bytes.Buffer{}, true) + require.Error(t, err) + assert.Contains(t, err.Error(), "--json") + assert.Contains(t, err.Error(), "--follow") +} + +func TestAppLogRef_ZeroServicesAccurateError(t *testing.T) { + empty := &dto.AppShowResponse{App: "new"} + _, err := appLogRef(empty, "") + require.Error(t, err) + assert.Contains(t, err.Error(), "no active services") +} + +func TestRunStatusCmd_AppsLabelAndSortedHelpers(t *testing.T) { + plane := climocks.NewMockControlPlane(t) + plane.EXPECT().GetStatus(mock.Anything).Return(&remote.Status{ + Apps: 2, RegistryDomain: "g.example.com", + ContainerStatus: map[string]string{"zeta": "active", "alpha": "stopped"}, + }, nil).Once() + var out bytes.Buffer + require.NoError(t, runStatusCmd(context.Background(), plane, &out)) + text := out.String() + assert.Contains(t, text, "Apps:") + assert.NotContains(t, text, "Routes:") + assert.Less(t, strings.Index(text, "alpha"), strings.Index(text, "zeta")) +} diff --git a/internal/adapters/in/cli/apps_local_integration_test.go b/internal/adapters/in/cli/apps_local_integration_test.go new file mode 100644 index 000000000..8bbd6ad65 --- /dev/null +++ b/internal/adapters/in/cli/apps_local_integration_test.go @@ -0,0 +1,155 @@ +package cli + +import ( + "bytes" + "context" + "net/http" + "path/filepath" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/dto" + "github.com/bnema/gordon/internal/adapters/in/cli/remote" + adminhttp "github.com/bnema/gordon/internal/adapters/in/http/admin" + "github.com/bnema/gordon/internal/adapters/localadmin" + in "github.com/bnema/gordon/internal/boundaries/in" + inmocks "github.com/bnema/gordon/internal/boundaries/in/mocks" + "github.com/bnema/gordon/internal/domain" +) + +// validLocalAppManifest is a minimal manifest accepted by the admin apply +// surface (one service, one HTTP host). +const validLocalAppManifest = ` +name = "blog" +[services.web] +image = "img:1" +[[services.web.http]] +host = "blog.example.com" +port = 8080 +` + +// TestSharedCommandLocalAdminComposition_Status executes a shared command +// through socket discovery, the Unix transport, and the local authority. +func TestSharedCommandLocalAdminComposition_Status(t *testing.T) { + xdg := t.TempDir() + t.Setenv("XDG_RUNTIME_DIR", xdg) + withRemoteTarget(t, "") + + configSvc := inmocks.NewMockConfigService(t) + configSvc.EXPECT().GetRegistryDomain().Return("registry.example.test").Once() + configSvc.EXPECT().GetRegistryPort().Return(5000).Once() + configSvc.EXPECT().GetServerPort().Return(8080).Once() + configSvc.EXPECT().IsNetworkIsolationEnabled().Return(true).Once() + appSvc := inmocks.NewMockAppService(t) + appSvc.EXPECT().List(mock.Anything).Return([]in.AppSummary{{App: "blog", Active: "rev-1", Converged: true}}, nil).Once() + + handler := adminhttp.NewHandler(adminhttp.HandlerDeps{ConfigSvc: configSvc, AppSvc: appSvc, Log: zerowrap.Default()}) + shutdown := startLocalAuthority(t, filepath.Join(xdg, "gordon"), handler.LocalAuthority()) + defer shutdown() + stubLocalAppClient(t, remote.NewLocalClient) + + cmd := newStatusCmd() + var out bytes.Buffer + cmd.SetOut(&out) + require.NoError(t, cmd.ExecuteContext(context.Background())) + assert.Contains(t, out.String(), "registry.example.test") + assert.Contains(t, out.String(), "blog") +} + +// TestAppLocalAdminComposition_EndToEnd composes the real local authority, +// Unix listener, discovery, transport, and app control plane. +func TestAppLocalAdminComposition_EndToEnd(t *testing.T) { + xdg := t.TempDir() + t.Setenv("XDG_RUNTIME_DIR", xdg) + withRemoteTarget(t, "") + + appSvc := inmocks.NewMockAppService(t) + appSvc.EXPECT().Apply(mock.Anything, mock.Anything, mock.Anything, false).Return( + &in.AppApplyResult{App: "blog", ResultingRevision: "rev-2", Pending: true, IntentID: "apply-1"}, + nil, nil, + ).Once() + appSvc.EXPECT().List(mock.Anything).Return([]in.AppSummary{ + {App: "blog", Desired: "rev-2", Active: "rev-1"}, + }, nil).Once() + appSvc.EXPECT().Show(mock.Anything, "blog").Return(&in.AppDetail{ + App: "blog", + DesiredRevision: "rev-2", + DesiredStatus: "pending", + Services: map[string]in.AppServiceView{"web": {Container: "ctr-web"}}, + }, nil).Once() + appSvc.EXPECT().Diff(mock.Anything, "blog").Return(domain.AppDiff{}, nil).Once() + appSvc.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return(&domain.AppOperation{ + Op: "apply", + App: "blog", + Outcome: "succeeded", + }, nil).Once() + + handler := adminhttp.NewHandler(adminhttp.HandlerDeps{AppSvc: appSvc, Log: zerowrap.Default()}) + dir := filepath.Join(xdg, "gordon") + shutdown := startLocalAuthority(t, dir, handler.LocalAuthority()) + + // The CLI resolves the socket through real discovery and the Unix transport. + stubLocalAppClient(t, remote.NewLocalClient) + + handle, err := resolveAppPlane() + require.NoError(t, err) + defer handle.close() + plane := handle.plane + + ctx := context.Background() + + apply, err := plane.ApplyApp(ctx, dto.AppApplyRequest{ManifestTOML: validLocalAppManifest}) + require.NoError(t, err) + assert.Equal(t, "rev-2", apply.ResultingRevision) + + apps, err := plane.ListApps(ctx) + require.NoError(t, err) + require.Len(t, apps, 1) + assert.Equal(t, "blog", apps[0].App) + + show, err := plane.ShowApp(ctx, "blog") + require.NoError(t, err) + assert.Equal(t, "ctr-web", show.Active.Services["web"].Container) + + diff, err := plane.DiffApp(ctx, "blog") + require.NoError(t, err) + assert.Equal(t, "blog", diff.App) + + // Idempotent-mutation recovery path: re-query the same key. + op, err := plane.OperationByKey(ctx, "blog", "key-1") + require.NoError(t, err) + assert.Equal(t, "succeeded", op.Outcome) + + // Shutdown removes only the daemon-owned socket; restart recreates it. + shutdown() + assert.NoFileExists(t, localadmin.SocketPath(dir)) + + restartShutdown := startLocalAuthority(t, dir, handler.LocalAuthority()) + require.FileExists(t, localadmin.SocketPath(dir)) + restartShutdown() + assert.NoFileExists(t, localadmin.SocketPath(dir)) +} + +// startLocalAuthority serves the given handler on an owner-only Unix socket +// and returns a shutdown function that stops the server and removes the +// socket, mirroring daemon lifecycle. +func startLocalAuthority(t *testing.T, dir string, handler http.Handler) func() { + t.Helper() + ln, info, err := localadmin.Listen(dir) + require.NoError(t, err) + + srv := &http.Server{Handler: handler, ReadHeaderTimeout: 5 * time.Second} + go func() { _ = srv.Serve(ln) }() + + return func() { + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + _ = srv.Shutdown(ctx) + require.NoError(t, localadmin.RemoveOwnedSocket(localadmin.SocketPath(dir), info)) + } +} diff --git a/internal/adapters/in/cli/apps_local_test.go b/internal/adapters/in/cli/apps_local_test.go new file mode 100644 index 000000000..2445608af --- /dev/null +++ b/internal/adapters/in/cli/apps_local_test.go @@ -0,0 +1,222 @@ +package cli + +import ( + "context" + "encoding/json" + "net/http" + "net/http/httptest" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/dto" + "github.com/bnema/gordon/internal/adapters/in/cli/remote" + "github.com/bnema/gordon/internal/adapters/localadmin" +) + +// appPlaneFixtureHandler serves the same app DTO payloads a daemon would, +// independent of transport, so local and remote seams can be compared. +func appPlaneFixtureHandler(t *testing.T) http.Handler { + t.Helper() + mux := http.NewServeMux() + mux.HandleFunc("/admin/apps", func(w http.ResponseWriter, _ *http.Request) { + writeFixtureJSON(t, w, map[string]any{ + "apps": []dto.AppSummaryDTO{{App: "blog", Desired: "rev-2", Active: "rev-1", Converged: false}}, + }) + }) + mux.HandleFunc("/admin/apps/blog", func(w http.ResponseWriter, _ *http.Request) { + writeFixtureJSON(t, w, dto.AppShowResponse{ + App: "blog", + Desired: dto.AppDesiredDTO{Revision: "rev-2", Status: "pending"}, + Active: dto.AppActiveDTO{ + Converged: true, + Services: map[string]dto.AppActiveServiceDTO{ + "web": {EffectiveRevision: "rev-1", Container: "ctr-web"}, + }, + }, + }) + }) + mux.HandleFunc("/admin/apps/blog/diff", func(w http.ResponseWriter, _ *http.Request) { + writeFixtureJSON(t, w, dto.AppDiffResponse{App: "blog"}) + }) + mux.HandleFunc("/admin/apps/apply", func(w http.ResponseWriter, r *http.Request) { + if r.Method != http.MethodPost { + w.WriteHeader(http.StatusMethodNotAllowed) + return + } + writeFixtureJSON(t, w, dto.AppApplyResponse{App: "blog", ResultingRevision: "rev-3"}) + }) + return mux +} + +func writeFixtureJSON(t *testing.T, w http.ResponseWriter, payload any) { + t.Helper() + w.Header().Set("Content-Type", "application/json") + require.NoError(t, json.NewEncoder(w).Encode(payload)) +} + +// startUnixAppPlane serves the fixture over an owner-only Unix socket. +func startUnixAppPlane(t *testing.T) string { + t.Helper() + dir := t.TempDir() + ln, _, err := localadmin.Listen(dir) + require.NoError(t, err) + + srv := &http.Server{Handler: appPlaneFixtureHandler(t), ReadHeaderTimeout: 5 * time.Second} + go func() { _ = srv.Serve(ln) }() + t.Cleanup(func() { _ = srv.Close() }) + + return localadmin.SocketPath(dir) +} + +// withRemoteTarget pins the CLI remote flags and clears the environment. +func withRemoteTarget(t *testing.T, target string) { + t.Helper() + origRemote, origToken, origInsecure := remoteFlag, tokenFlag, insecureTLSFlag + t.Cleanup(func() { remoteFlag, tokenFlag, insecureTLSFlag = origRemote, origToken, origInsecure }) + remoteFlag, tokenFlag, insecureTLSFlag = target, "", false + + t.Setenv("GORDON_REMOTE", "") + t.Setenv("GORDON_TOKEN", "") + t.Setenv("XDG_CONFIG_HOME", t.TempDir()) +} + +func stubLocalAppClient(t *testing.T, fn func() (*remote.Client, error)) { + t.Helper() + restore := newLocalAppClient + newLocalAppClient = fn + t.Cleanup(func() { newLocalAppClient = restore }) +} + +func TestResolveAppPlane_UsesLocalAdminSocket(t *testing.T) { + socketPath := startUnixAppPlane(t) + withRemoteTarget(t, "") + stubLocalAppClient(t, func() (*remote.Client, error) { + return remote.NewLocalClientForSocket(socketPath), nil + }) + + handle, err := resolveAppPlane() + require.NoError(t, err) + defer handle.close() + plane := handle.plane + + apps, err := plane.ListApps(context.Background()) + require.NoError(t, err) + require.Len(t, apps, 1) + assert.Equal(t, "blog", apps[0].App) + + show, err := plane.ShowApp(context.Background(), "blog") + require.NoError(t, err) + assert.Equal(t, "rev-2", show.Desired.Revision) + + diff, err := plane.DiffApp(context.Background(), "blog") + require.NoError(t, err) + assert.Equal(t, "blog", diff.App) + + apply, err := plane.ApplyApp(context.Background(), dto.AppApplyRequest{ManifestTOML: "name = \"blog\"\n"}) + require.NoError(t, err) + assert.Equal(t, "rev-3", apply.ResultingRevision) +} + +// TestResolveAppControlPlane_LocalAndRemoteDTOParity asserts the shared DTO +// seam yields identical values over both transports. +func TestControlPlane_LocalAndRemoteDTOParity(t *testing.T) { + fixture := appPlaneFixtureHandler(t) + + tcp := httptest.NewServer(fixture) + t.Cleanup(tcp.Close) + + dir := t.TempDir() + ln, _, err := localadmin.Listen(dir) + require.NoError(t, err) + unixSrv := &http.Server{Handler: fixture, ReadHeaderTimeout: 5 * time.Second} + go func() { _ = unixSrv.Serve(ln) }() + t.Cleanup(func() { _ = unixSrv.Close() }) + + localPlane := remote.NewLocalClientForSocket(localadmin.SocketPath(dir)) + remotePlane := remote.NewClient(tcp.URL) + + ctx := context.Background() + + localApps, err := localPlane.ListApps(ctx) + require.NoError(t, err) + remoteApps, err := remotePlane.ListApps(ctx) + require.NoError(t, err) + assert.Equal(t, remoteApps, localApps) + + localShow, err := localPlane.ShowApp(ctx, "blog") + require.NoError(t, err) + remoteShow, err := remotePlane.ShowApp(ctx, "blog") + require.NoError(t, err) + assert.Equal(t, remoteShow, localShow) + + localDiff, err := localPlane.DiffApp(ctx, "blog") + require.NoError(t, err) + remoteDiff, err := remotePlane.DiffApp(ctx, "blog") + require.NoError(t, err) + assert.Equal(t, remoteDiff, localDiff) +} + +// TestResolveAppControlPlane_ExplicitRemoteNeverProbesLocal pins that an +// explicit remote is authoritative: the local socket is never consulted, and +// a remote failure never falls back to it. +func TestResolveAppPlane_ExplicitRemoteNeverProbesLocal(t *testing.T) { + t.Run("remote succeeds", func(t *testing.T) { + tcp := httptest.NewServer(appPlaneFixtureHandler(t)) + t.Cleanup(tcp.Close) + withRemoteTarget(t, tcp.URL) + + probed := false + stubLocalAppClient(t, func() (*remote.Client, error) { + probed = true + return nil, remote.ErrDaemonUnavailable + }) + + handle, err := resolveAppPlane() + require.NoError(t, err) + defer handle.close() + plane := handle.plane + + apps, err := plane.ListApps(context.Background()) + require.NoError(t, err) + require.Len(t, apps, 1) + assert.False(t, probed, "local socket must not be probed when a remote is selected") + }) + + t.Run("remote fails without local fallback", func(t *testing.T) { + tcp := httptest.NewServer(http.HandlerFunc(func(http.ResponseWriter, *http.Request) {})) + url := tcp.URL + tcp.Close() + + withRemoteTarget(t, url) + + probed := false + stubLocalAppClient(t, func() (*remote.Client, error) { + probed = true + return remote.NewLocalClientForSocket("/nonexistent/admin.sock"), nil + }) + + handle, err := resolveAppPlane() + require.NoError(t, err, "client construction succeeds; the failure must surface on the request") + defer handle.close() + plane := handle.plane + + _, err = plane.ListApps(context.Background()) + require.Error(t, err) + assert.False(t, probed, "a failed remote must never fall back to the local socket") + }) +} + +func TestResolveAppPlane_ReportsDaemonUnavailable(t *testing.T) { + withRemoteTarget(t, "") + stubLocalAppClient(t, func() (*remote.Client, error) { + return nil, remote.ErrDaemonUnavailable + }) + + _, err := resolveAppPlane() + require.Error(t, err) + assert.ErrorContains(t, err, "daemon-unavailable") + assert.ErrorIs(t, err, remote.ErrDaemonUnavailable) +} diff --git a/internal/adapters/in/cli/apps_operations.go b/internal/adapters/in/cli/apps_operations.go new file mode 100644 index 000000000..0ee219974 --- /dev/null +++ b/internal/adapters/in/cli/apps_operations.go @@ -0,0 +1,51 @@ +package cli + +import ( + "context" + "fmt" + "io" + + "github.com/spf13/cobra" +) + +func newAppsOperationsCmd() *cobra.Command { + cmd := &cobra.Command{Use: "operations", Short: "Inspect app operation journals"} + cmd.AddCommand(newAppsOperationsShowCmd(), newAppsOperationsWatchCmd()) + return cmd +} + +func newAppsOperationsShowCmd() *cobra.Command { + var key string + var jsonOut bool + cmd := &cobra.Command{ + Use: "show APP", + Short: "Show an app operation by request key", + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + handle, err := resolveAppPlane() + if err != nil { + return err + } + defer handle.close() + return runAppsOperationsShow(cmd.Context(), handle.plane, cmd.OutOrStdout(), args[0], key, jsonOut) + }, + } + cmd.Flags().StringVar(&key, "key", "", "Request key (required)") + cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") + _ = cmd.MarkFlagRequired("key") + return cmd +} + +func runAppsOperationsShow(ctx context.Context, plane ControlPlane, out io.Writer, app, key string, jsonOut bool) error { + if app == "" || key == "" { + return fmt.Errorf("APP and --key are required") + } + operation, err := plane.OperationByKey(ctx, app, key) + if err != nil { + return fmt.Errorf("lookup operation for app %s: %w", app, err) + } + if jsonOut { + return writeJSON(out, operation) + } + return renderAppDeployResponse(out, operation) +} diff --git a/internal/adapters/in/cli/apps_operations_watch.go b/internal/adapters/in/cli/apps_operations_watch.go new file mode 100644 index 000000000..38d468566 --- /dev/null +++ b/internal/adapters/in/cli/apps_operations_watch.go @@ -0,0 +1,46 @@ +package cli + +import ( + "context" + "fmt" + "io" + + "github.com/spf13/cobra" +) + +// newAppsOperationsWatchCmd creates `operations watch APP --key KEY`: it +// resumes observing an existing operation through the by-key endpoint until +// it is terminal, never reissuing the mutation. +func newAppsOperationsWatchCmd() *cobra.Command { + var key string + var jsonOut bool + cmd := &cobra.Command{ + Use: "watch APP", + Short: "Poll an app operation until it is terminal", + Long: `Polls the by-key operation journal until the operation is terminal. + +Use it to resume watching an operation from an earlier interrupted run +without reissuing the mutation. Ctrl-C stops only local polling; the +daemon-side operation keeps running and the command prints how to resume.`, + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + handle, err := resolveAppPlane() + if err != nil { + return err + } + defer handle.close() + return runAppsOperationsWatch(cmd.Context(), handle.plane, cmd.OutOrStdout(), cmd.ErrOrStderr(), args[0], key, jsonOut) + }, + } + cmd.Flags().StringVar(&key, "key", "", "Request key (required)") + cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") + _ = cmd.MarkFlagRequired("key") + return cmd +} + +func runAppsOperationsWatch(ctx context.Context, plane ControlPlane, out, errOut io.Writer, app, key string, jsonOut bool) error { + if app == "" || key == "" { + return fmt.Errorf("APP and --key are required") + } + return watchOperation(ctx, plane, app, key, nil, out, errOut, jsonOut) +} diff --git a/internal/adapters/in/cli/apps_test.go b/internal/adapters/in/cli/apps_test.go new file mode 100644 index 000000000..32ead5eb0 --- /dev/null +++ b/internal/adapters/in/cli/apps_test.go @@ -0,0 +1,549 @@ +package cli + +// Tests for the app CLI surface (apps.go). The commands go through one +// ControlPlane served by the generated mock: JSON parity (text and --json +// carry equivalent semantics), no secret values in output, plan-contract +// errors, and outcome-unknown guidance. + +import ( + "bytes" + "context" + "encoding/json" + "errors" + "net/http" + "os" + "path/filepath" + "strings" + "testing" + + "github.com/spf13/cobra" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/dto" + climocks "github.com/bnema/gordon/internal/adapters/in/cli/mocks" + "github.com/bnema/gordon/internal/adapters/in/cli/remote" +) + +// appPlane returns the generated control plane mock for one test. +func appPlane(t *testing.T) *climocks.MockControlPlane { + t.Helper() + return climocks.NewMockControlPlane(t) +} + +func writeManifest(t *testing.T, content string) string { + t.Helper() + path := filepath.Join(t.TempDir(), "app.toml") + require.NoError(t, os.WriteFile(path, []byte(content), 0o600)) + return path +} + +func TestRunAppsApply_RejectsDryRunWithDeploy(t *testing.T) { + plane := appPlane(t) + err := runAppsApply(context.Background(), plane, strings.NewReader(""), &bytes.Buffer{}, &bytes.Buffer{}, "x.toml", true, true, false) + require.Error(t, err) + assert.Contains(t, err.Error(), "--dry-run") + assert.Contains(t, err.Error(), "--deploy") +} + +func TestRunAppsApply_RequiresFile(t *testing.T) { + plane := appPlane(t) + err := runAppsApply(context.Background(), plane, strings.NewReader(""), &bytes.Buffer{}, &bytes.Buffer{}, "", false, false, false) + require.Error(t, err) + assert.Contains(t, err.Error(), "--file") +} + +func TestRunAppsApply_JSONParity(t *testing.T) { + want := &dto.AppApplyResponse{ + App: "blog", FormerRevision: "rev-a", ResultingRevision: "rev-b", + Pending: true, Intent: "apply-123", + Diff: dto.AppDiffSection{Added: []string{"service.web"}}, + } + plane := appPlane(t) + plane.EXPECT().ApplyApp(mock.Anything, mock.Anything).Return(want, nil).Once() + var out bytes.Buffer + require.NoError(t, runAppsApply(context.Background(), plane, strings.NewReader(""), &out, &bytes.Buffer{}, + writeManifest(t, "[app]\nname = \"blog\"\n"), false, false, true)) + var got dto.AppApplyResponse + require.NoError(t, json.Unmarshal(out.Bytes(), &got)) + assert.Equal(t, *want, got) +} + +func TestRunAppsApply_ChainsAcceptedRevision(t *testing.T) { + plane := appPlane(t) + plane.EXPECT().ApplyApp(mock.Anything, mock.Anything).Return( + &dto.AppApplyResponse{App: "blog", ResultingRevision: "rev-b", Pending: true, Intent: "apply-1"}, nil).Once() + plane.EXPECT().DeployApp(mock.Anything, "blog", mock.Anything).Return( + &dto.AppDeployResponse{Op: "op-1", App: "blog", Revision: "rev-b", Outcome: "success"}, "key-1", nil).Once() + var out bytes.Buffer + require.NoError(t, runAppsApply(context.Background(), plane, strings.NewReader(""), &out, &bytes.Buffer{}, + writeManifest(t, "x"), false, true, false)) + assert.Contains(t, out.String(), "rev-b") +} + +func TestRunAppsApply_ChainedJSONIsOneDocument(t *testing.T) { + plane := appPlane(t) + plane.EXPECT().ApplyApp(mock.Anything, mock.Anything).Return( + &dto.AppApplyResponse{App: "blog", ResultingRevision: "rev-b", Pending: true, Intent: "apply-1"}, nil).Once() + plane.EXPECT().DeployApp(mock.Anything, "blog", mock.Anything).Return( + &dto.AppDeployResponse{Op: "op-1", App: "blog", Revision: "rev-b", Outcome: "success"}, "key-1", nil).Once() + var out bytes.Buffer + require.NoError(t, runAppsApply(context.Background(), plane, strings.NewReader(""), &out, &bytes.Buffer{}, + writeManifest(t, "x"), false, true, true)) + var got struct { + Apply dto.AppApplyResponse `json:"apply"` + Deploy dto.AppDeployResponse `json:"deploy"` + } + require.NoError(t, json.Unmarshal(out.Bytes(), &got)) + assert.Equal(t, "apply-1", got.Apply.Intent) + assert.Equal(t, "op-1", got.Deploy.Op) +} + +func TestRunAppsList_JSONParityAndSorting(t *testing.T) { + plane := appPlane(t) + plane.EXPECT().ListApps(mock.Anything).Return([]dto.AppSummaryDTO{ + {App: "zeta", Converged: true}, + {App: "alpha", Stopped: true}, + }, nil) + var out bytes.Buffer + require.NoError(t, runAppsList(context.Background(), plane, &out, true)) + var got []dto.AppSummaryDTO + require.NoError(t, json.Unmarshal(out.Bytes(), &got)) + require.Len(t, got, 2) + + out.Reset() + require.NoError(t, runAppsList(context.Background(), plane, &out, false)) + text := out.String() + assert.Less(t, strings.Index(text, "alpha"), strings.Index(text, "zeta"), "text lists apps sorted") + assert.Contains(t, text, "stopped") +} + +func TestRunAppsList_EmptyJSONIsArray(t *testing.T) { + plane := appPlane(t) + plane.EXPECT().ListApps(mock.Anything).Return(nil, nil).Once() + var out bytes.Buffer + require.NoError(t, runAppsList(context.Background(), plane, &out, true)) + assert.JSONEq(t, `[]`, strings.TrimSpace(out.String())) +} + +func TestRunAppsShow_JSONParity(t *testing.T) { + want := &dto.AppShowResponse{ + App: "blog", + Desired: dto.AppDesiredDTO{Revision: "rev-b", Status: "pending"}, + Active: dto.AppActiveDTO{Converged: false, Services: map[string]dto.AppActiveServiceDTO{ + "web": {EffectiveRevision: "rev-a", Container: "ctr-a", RestartUnsafe: true}, + }}, + Intent: dto.AppIntentDTO{Stopped: true}, + LastOp: &dto.AppLastOpDTO{Op: "op-9", Outcome: "failed"}, + } + plane := appPlane(t) + plane.EXPECT().ShowApp(mock.Anything, "blog").Return(want, nil) + var out bytes.Buffer + require.NoError(t, runAppsShow(context.Background(), plane, "blog", &out, true)) + var got dto.AppShowResponse + require.NoError(t, json.Unmarshal(out.Bytes(), &got)) + assert.Equal(t, *want, got) + + out.Reset() + require.NoError(t, runAppsShow(context.Background(), plane, "blog", &out, false)) + assert.Contains(t, out.String(), "restart_unsafe", "text shows WHY recovery is blocked") +} + +func TestRunAppsSecretsSet_NeverEchoesValues(t *testing.T) { + plane := appPlane(t) + var gotApp string + var gotReq dto.AppSecretSetRequest + plane.EXPECT().SetAppSecrets(mock.Anything, mock.Anything, mock.Anything). + Run(func(_ context.Context, app string, req dto.AppSecretSetRequest) { + gotApp, gotReq = app, req + }).Return(nil).Once() + var out bytes.Buffer + require.NoError(t, runAppsSecretsSet(context.Background(), plane, strings.NewReader(""), + &out, "blog", []string{"password=s3cr3t-hunter2", "token=abc"}, "web", false, false)) + text := out.String() + assert.NotContains(t, text, "s3cr3t-hunter2") + assert.NotContains(t, text, "abc") + assert.Contains(t, text, "password") + assert.Contains(t, text, "token") + assert.Equal(t, "blog", gotApp) + assert.Equal(t, "web", gotReq.Service) + assert.Equal(t, map[string]string{"password": "s3cr3t-hunter2", "token": "abc"}, gotReq.Secrets) +} + +func TestRunAppsSecretsSet_RequiresService(t *testing.T) { + plane := appPlane(t) + err := runAppsSecretsSet(context.Background(), plane, strings.NewReader(""), + &bytes.Buffer{}, "blog", []string{"k=v"}, "", false, false) + require.Error(t, err) + assert.Contains(t, err.Error(), "--service") +} + +func TestRunAppsSecretsSet_StdinAndValidation(t *testing.T) { + plane := appPlane(t) + var gotReq dto.AppSecretSetRequest + plane.EXPECT().SetAppSecrets(mock.Anything, "blog", mock.Anything). + Run(func(_ context.Context, _ string, req dto.AppSecretSetRequest) { gotReq = req }).Return(nil).Once() + var out bytes.Buffer + stdin := strings.NewReader("from_stdin=stdin-value\n\nspaced= preserved \n") + require.NoError(t, runAppsSecretsSet(context.Background(), plane, stdin, + &out, "blog", []string{"from_flag=flag-value"}, "web", true, false)) + assert.Equal(t, map[string]string{ + "from_stdin": "stdin-value", + "spaced": " preserved ", + "from_flag": "flag-value", + }, gotReq.Secrets) + + err := runAppsSecretsSet(context.Background(), plane, strings.NewReader(""), + &bytes.Buffer{}, "blog", []string{"no-equals-here"}, "web", false, false) + require.Error(t, err) + assert.Contains(t, err.Error(), "KEY=VALUE") +} + +func TestRunAppDeploy_OutcomeUnknownMentionsKey(t *testing.T) { + plane := appPlane(t) + plane.EXPECT().DeployApp(mock.Anything, "blog", mock.Anything).Return( + nil, "key-abc", &remote.OutcomeUnknownError{Method: "POST", Path: "/apps/blog/deploy", Err: errors.New("boom")}).Once() + err := runAppDeploy(context.Background(), plane, "blog", dto.AppDeployRequest{}, &bytes.Buffer{}, &bytes.Buffer{}, false) + require.Error(t, err) + assert.Contains(t, err.Error(), "outcome-unknown") + assert.Contains(t, err.Error(), "key-abc") + assert.Contains(t, err.Error(), "by-key") +} + +func TestRunAppDeploy_ConflictRendersJournal(t *testing.T) { + plane := appPlane(t) + plane.EXPECT().DeployApp(mock.Anything, "blog", mock.Anything).Return(nil, "key-9", &remote.AppOpConflictError{ + StatusCode: http.StatusConflict, Status: "409 Conflict", + Response: dto.AppDeployResponse{ + Op: "op-9", App: "blog", Revision: "rev-b", Outcome: "failed", + Services: map[string]dto.AppServiceResultDTO{ + "web": {Result: "failed", EffectiveRevision: "rev-b", Error: "nope"}, + }, + }, + }).Once() + var out bytes.Buffer + err := runAppDeploy(context.Background(), plane, "blog", dto.AppDeployRequest{}, &out, &bytes.Buffer{}, false) + require.Error(t, err, "conflict must retain failure semantics") + text := out.String() + assert.Contains(t, text, "web:") + assert.Contains(t, text, "nope") + assert.Contains(t, err.Error(), "failed") + assert.Contains(t, err.Error(), "op-9") + assert.Contains(t, err.Error(), "key-9") + assert.Contains(t, err.Error(), "by-key") +} + +func TestRunAppDeploy_ConflictJSONRendersJournal(t *testing.T) { + plane := appPlane(t) + plane.EXPECT().DeployApp(mock.Anything, "blog", mock.Anything).Return(nil, "key-9", &remote.AppOpConflictError{ + StatusCode: http.StatusConflict, Status: "409 Conflict", + Response: dto.AppDeployResponse{ + Op: "op-9", App: "blog", Revision: "rev-b", Outcome: "failed", + Services: map[string]dto.AppServiceResultDTO{ + "web": {Result: "failed", EffectiveRevision: "rev-b", Error: "nope"}, + }, + }, + }).Once() + var out bytes.Buffer + err := runAppDeploy(context.Background(), plane, "blog", dto.AppDeployRequest{}, &out, &bytes.Buffer{}, true) + require.Error(t, err) + var got dto.AppDeployResponse + require.NoError(t, json.Unmarshal(out.Bytes(), &got)) + assert.Equal(t, "op-9", got.Op) + assert.Equal(t, "failed", got.Services["web"].Result) +} + +func TestRunAppLifecycle_ConflictRendersJournal(t *testing.T) { + plane := appPlane(t) + plane.EXPECT().StopApp(mock.Anything, "blog").Return(nil, "key-7", &remote.AppOpConflictError{ + StatusCode: http.StatusConflict, Status: "409 Conflict", + Response: dto.AppDeployResponse{ + Op: "op-7", App: "blog", Outcome: "partial", + Services: map[string]dto.AppServiceResultDTO{ + "web": {Result: "failed", EffectiveRevision: "rev-b", Error: "busy"}, + }, + }, + }).Once() + var out bytes.Buffer + err := runAppLifecycle(context.Background(), plane.StopApp, "stop", "blog", &out, false) + require.Error(t, err) + assert.Contains(t, out.String(), "busy") + assert.Contains(t, err.Error(), "key-7") + assert.Contains(t, err.Error(), "by-key") +} + +func TestRenderAppDeployResponse_SortedServices(t *testing.T) { + resp := &dto.AppDeployResponse{ + Op: "op-1", App: "blog", Outcome: "partial", + Services: map[string]dto.AppServiceResultDTO{ + "web": {Result: "deployed", EffectiveRevision: "rev-b"}, + "db": {Result: "failed", EffectiveRevision: "rev-a", Error: "nope", RestartUnsafe: true}, + }, + CleanupWarnings: []dto.AppCleanupWarningDTO{{Service: "web", Leftover: "ctr-old", Detail: "retire failed"}}, + Effective: &dto.AppEffectiveDTO{Services: map[string]string{"web": "rev-b", "db": "rev-a"}}, + Retained: &dto.AppRetainedDTO{Volumes: []string{"blog-db"}}, + } + var out bytes.Buffer + require.NoError(t, renderAppDeployResponse(&out, resp)) + text := out.String() + assert.Less(t, strings.Index(text, "db:"), strings.Index(text, "web:")) + assert.Contains(t, text, "restart_unsafe") + assert.Contains(t, text, "cleanup:") + assert.Contains(t, text, "blog-db") +} + +func TestAppLogRef_Selection(t *testing.T) { + multi := &dto.AppShowResponse{App: "blog", Active: dto.AppActiveDTO{Services: map[string]dto.AppActiveServiceDTO{ + "web": {Container: "ctr-web"}, + "db": {Container: "ctr-db"}, + }}} + _, err := appLogRef(multi, "") + require.Error(t, err) + assert.Contains(t, err.Error(), "--service") + + ref, err := appLogRef(multi, "db") + require.NoError(t, err) + assert.Equal(t, "blog/db", ref) + + _, err = appLogRef(multi, "missing") + require.Error(t, err) + + single := &dto.AppShowResponse{App: "solo", Active: dto.AppActiveDTO{Services: map[string]dto.AppActiveServiceDTO{ + "web": {Container: "ctr-solo"}, + }}} + ref, err = appLogRef(single, "") + require.NoError(t, err) + assert.Equal(t, "solo/web", ref) + + empty := &dto.AppShowResponse{App: "new", Active: dto.AppActiveDTO{Services: map[string]dto.AppActiveServiceDTO{ + "web": {}, + }}} + _, err = appLogRef(empty, "web") + require.Error(t, err) + assert.Contains(t, err.Error(), "no recorded container") +} + +func TestResolveAppPlane_FailsWithoutDaemon(t *testing.T) { + t.Setenv("XDG_CONFIG_HOME", t.TempDir()) + t.Setenv("GORDON_REMOTE", "") + t.Setenv("GORDON_TOKEN", "") + origRemote, origToken, origInsecure := remoteFlag, tokenFlag, insecureTLSFlag + t.Cleanup(func() { remoteFlag, tokenFlag, insecureTLSFlag = origRemote, origToken, origInsecure }) + remoteFlag, tokenFlag, insecureTLSFlag = "", "", false + + restore := newLocalAppClient + newLocalAppClient = func() (*remote.Client, error) { return nil, remote.ErrDaemonUnavailable } + t.Cleanup(func() { newLocalAppClient = restore }) + + _, err := resolveAppPlane() + require.Error(t, err) + assert.Contains(t, err.Error(), "daemon-unavailable") +} + +// TestAppCommandSurface pins the frozen CLI surface (05-api-cli.md §5): +// every command exists with its frozen Use string and flags. The commands +// stay unregistered until cutover; this test is what keeps them referenced. +func TestAppCommandSurface(t *testing.T) { + root := newAppsCmd() + require.Equal(t, "apps", root.Use) + subs := map[string]bool{} + for _, sub := range root.Commands() { + subs[sub.Name()] = true + } + for _, name := range []string{"apply", "list", "show", "diff", "secrets", "deploy", "restart", "stop", "start", "remove", "status", "logs"} { + assert.True(t, subs[name], "apps subcommand %s", name) + } + + secrets := newAppsSecretsCmd() + secretSubs := map[string]bool{} + for _, sub := range secrets.Commands() { + secretSubs[sub.Name()] = true + } + assert.True(t, secretSubs["set"]) + assert.True(t, secretSubs["delete"]) + + flagged := map[*cobra.Command][]string{ + newAppsApplyCmd(): {"file", "dry-run", "deploy", "json"}, + newAppsListCmd(): {"json"}, + newAppsShowCmd(): {"json"}, + newAppsDiffCmd(): {"json"}, + newAppsSecretsSetCmd(): {"service", "stdin", "json"}, + newAppsSecretsDeleteCmd(): {"service", "json"}, + newAppDeployCmd(): {"revision", "service", "json"}, + newAppRestartCmd(): {"service", "json"}, + newAppStopCmd(): {"json"}, + newAppStartCmd(): {"json"}, + newAppRemoveCmd(): {"json"}, + newAppStatusCmd(): {"json"}, + newAppLogsCmd(): {"service", "follow", "tail", "json"}, + } + for cmd, flags := range flagged { + for _, flag := range flags { + assert.NotNil(t, cmd.Flags().Lookup(flag), "%s must define --%s", cmd.Use, flag) + } + } + + assert.Equal(t, "deploy APP", newAppDeployCmd().Use) + assert.Equal(t, "restart APP", newAppRestartCmd().Use) + assert.Equal(t, "stop APP", newAppStopCmd().Use) + assert.Equal(t, "start APP", newAppStartCmd().Use) + assert.Equal(t, "remove APP", newAppRemoveCmd().Use) + assert.Equal(t, "status APP", newAppStatusCmd().Use) + assert.Equal(t, "logs APP", newAppLogsCmd().Use) +} + +func TestRunAppsDiff_TextAndJSON(t *testing.T) { + diff := &dto.AppDiffResponse{ + App: "blog", + Diff: dto.AppDiffSection{Added: []string{"b"}, Removed: []string{"a"}, Changed: []string{"c"}}, + } + plane := appPlane(t) + plane.EXPECT().DiffApp(mock.Anything, "blog").Return(diff, nil).Times(2) + var out bytes.Buffer + require.NoError(t, runAppsDiff(context.Background(), plane, "blog", &out, false)) + text := out.String() + assert.Contains(t, text, "added:") + assert.Contains(t, text, "removed:") + assert.Contains(t, text, "changed:") + + out.Reset() + require.NoError(t, runAppsDiff(context.Background(), plane, "blog", &out, true)) + var got dto.AppDiffResponse + require.NoError(t, json.Unmarshal(out.Bytes(), &got)) + assert.Equal(t, *diff, got) + + plane.EXPECT().DiffApp(mock.Anything, "blog").Return(&dto.AppDiffResponse{App: "blog"}, nil).Once() + out.Reset() + require.NoError(t, runAppsDiff(context.Background(), plane, "blog", &out, false)) + assert.Contains(t, out.String(), "No differences") +} + +func TestRunAppsSecretsDelete(t *testing.T) { + plane := appPlane(t) + plane.EXPECT().DeleteAppSecret(mock.Anything, "blog", dto.AppSecretDeleteRequest{Service: "web", Key: "password"}).Return(nil).Times(2) + var out bytes.Buffer + require.NoError(t, runAppsSecretsDelete(context.Background(), plane, &out, "blog", "password", "web", false)) + assert.Contains(t, out.String(), "password") + + out.Reset() + require.NoError(t, runAppsSecretsDelete(context.Background(), plane, &out, "blog", "password", "web", true)) + var got map[string]string + require.NoError(t, json.Unmarshal(out.Bytes(), &got)) + assert.Equal(t, "password", got["deleted"]) + + err := runAppsSecretsDelete(context.Background(), plane, &bytes.Buffer{}, "blog", "password", "", false) + require.Error(t, err) + assert.Contains(t, err.Error(), "--service") +} + +func TestRunAppRestart_AndLifecycle(t *testing.T) { + resp := &dto.AppDeployResponse{Op: "op-2", App: "blog", Revision: "rev-b", Outcome: "success"} + plane := appPlane(t) + plane.EXPECT().RestartApp(mock.Anything, "blog", "", false).Return(resp, "key-lc", nil).Times(2) + plane.EXPECT().StopApp(mock.Anything, "blog").Return(resp, "key-lc", nil).Once() + plane.EXPECT().StartApp(mock.Anything, "blog").Return(resp, "key-lc", nil).Once() + plane.EXPECT().RemoveApp(mock.Anything, "blog").Return(resp, "key-lc", nil).Once() + var out bytes.Buffer + require.NoError(t, runAppRestart(context.Background(), plane, "blog", "", false, &out, false)) + assert.Contains(t, out.String(), "success") + + out.Reset() + require.NoError(t, runAppRestart(context.Background(), plane, "blog", "", false, &out, true)) + var got dto.AppDeployResponse + require.NoError(t, json.Unmarshal(out.Bytes(), &got)) + assert.Equal(t, *resp, got) + + for _, fn := range []appLifecycleFunc{plane.StopApp, plane.StartApp, plane.RemoveApp} { + out.Reset() + require.NoError(t, runAppLifecycle(context.Background(), fn, "stop", "blog", &out, false)) + assert.Contains(t, out.String(), "op-2") + } + + unknown := appPlane(t) + unknown.EXPECT().StopApp(mock.Anything, "blog").Return( + nil, "key-x", &remote.OutcomeUnknownError{Method: "POST", Path: "/x", Err: errors.New("boom")}).Once() + err := runAppLifecycle(context.Background(), unknown.StopApp, "stop", "blog", &bytes.Buffer{}, false) + require.Error(t, err) + assert.Contains(t, err.Error(), "key-x") +} + +func TestRunAppStatus_Observed(t *testing.T) { + show := &dto.AppShowResponse{ + App: "blog", + Active: dto.AppActiveDTO{Converged: true, Services: map[string]dto.AppActiveServiceDTO{ + "web": {EffectiveRevision: "rev-b", Container: "ctr-b"}, + }}, + } + plane := appPlane(t) + plane.EXPECT().ShowApp(mock.Anything, "blog").Return(show, nil).Times(2) + var out bytes.Buffer + require.NoError(t, runAppStatus(context.Background(), plane, "blog", &out, false)) + assert.Contains(t, out.String(), "ctr-b") + + out.Reset() + require.NoError(t, runAppStatus(context.Background(), plane, "blog", &out, true)) + var got dto.AppShowResponse + require.NoError(t, json.Unmarshal(out.Bytes(), &got)) + assert.Equal(t, *show, got) +} + +// fakeAppLogReader scripts container log reads for app log tests. +type fakeAppLogReader struct { + lines []string + stream []string +} + +func (f *fakeAppLogReader) GetContainerLogs(_ context.Context, _ string, _ int) ([]string, error) { + return f.lines, nil +} + +func (f *fakeAppLogReader) StreamContainerLogs(_ context.Context, _ string, _ int) (<-chan string, error) { + ch := make(chan string, len(f.stream)) + for _, line := range f.stream { + ch <- line + } + close(ch) + return ch, nil +} + +func TestRunAppLogs_TextJSONFollow(t *testing.T) { + show := &dto.AppShowResponse{App: "blog", Active: dto.AppActiveDTO{Services: map[string]dto.AppActiveServiceDTO{ + "web": {Container: "ctr-web"}, + }}} + plane := appPlane(t) + plane.EXPECT().ShowApp(mock.Anything, "blog").Return(show, nil).Times(3) + reader := &fakeAppLogReader{lines: []string{"l1", "l2"}, stream: []string{"s1"}} + + var out bytes.Buffer + require.NoError(t, runAppLogs(context.Background(), plane, reader, "blog", "", false, 50, &out, false)) + assert.Equal(t, "l1\nl2\n", out.String()) + + out.Reset() + require.NoError(t, runAppLogs(context.Background(), plane, reader, "blog", "web", false, 50, &out, true)) + var got map[string]any + require.NoError(t, json.Unmarshal(out.Bytes(), &got)) + assert.Equal(t, "blog", got["app"]) + + out.Reset() + require.NoError(t, runAppLogs(context.Background(), plane, reader, "blog", "web", true, 50, &out, false)) + assert.Equal(t, "s1\n", out.String()) +} + +func TestRunAppsSecretsList_RendersMetadataAndWrapsErrors(t *testing.T) { + plane := appPlane(t) + plane.EXPECT().ListAppSecrets(mock.Anything, "blog", "web").Return([]dto.AppSecretMetadataDTO{ + {Service: "web", Key: "DATABASE_URL", Name: "database-url", Source: "desired", Presence: "unknown"}, + }, nil).Once() + var out bytes.Buffer + require.NoError(t, runAppsSecretsList(context.Background(), plane, &out, "blog", "web", false)) + assert.Contains(t, out.String(), "web/DATABASE_URL") + assert.Contains(t, out.String(), "desired") + + errPlane := appPlane(t) + errPlane.EXPECT().ListAppSecrets(mock.Anything, "blog", "").Return(nil, errors.New("boom")).Once() + err := runAppsSecretsList(context.Background(), errPlane, &bytes.Buffer{}, "blog", "", false) + require.Error(t, err) + assert.Contains(t, err.Error(), "list app secrets for blog") + assert.Contains(t, err.Error(), "boom") +} diff --git a/internal/adapters/in/cli/apps_watch.go b/internal/adapters/in/cli/apps_watch.go new file mode 100644 index 000000000..060c15525 --- /dev/null +++ b/internal/adapters/in/cli/apps_watch.go @@ -0,0 +1,225 @@ +package cli + +// Operation watch: a 202 mutation (or an explicit resume) is observed +// through the existing by-key journal endpoint until it reaches a terminal +// state. Polling is intentionally local: the CLI never reissues the +// mutation and never asks the daemon to cancel the operation, so Ctrl-C +// only stops this process from watching. + +import ( + "context" + "fmt" + "io" + "strings" + "time" + + "github.com/bnema/gordon/internal/adapters/dto" + "github.com/bnema/gordon/internal/domain" +) + +// operationPollInterval is the pause between by-key polls of a running +// operation. It is a package variable (no flag) so command tests can +// shorten it; production uses a modest one-second cadence. +var operationPollInterval = time.Second + +// operationPollMaxTransientErrors bounds consecutive transient polling +// failures so a broken connection fails loudly instead of spinning or +// hiding the error. +const operationPollMaxTransientErrors = 5 + +// OperationFailedError reports a terminal non-success watch outcome. The +// terminal journal is rendered before this error is returned so the +// command exits nonzero without hiding the terminal detail. +type OperationFailedError struct { + App string + Op string + Outcome string +} + +func (e *OperationFailedError) Error() string { + return fmt.Sprintf("operation %s for app %s did not succeed (outcome %s)", e.Op, e.App, e.Outcome) +} + +// operationResumeCommand is the concise command that resumes polling an +// existing operation without reissuing the mutation. +func operationResumeCommand(app, key string) string { + cmd := fmt.Sprintf("gordon apps operations watch %s --key %s", app, key) + if remoteFlag != "" { + cmd += " --remote " + remoteFlag + } + return cmd +} + +// watchOperation polls the by-key journal until the operation is terminal, +// reports progress while it runs, then renders the terminal journal. In +// JSON mode stdout carries exactly one final document and progress goes to +// errOut. A terminal non-success outcome renders the journal and returns +// OperationFailedError. +func watchOperation(ctx context.Context, plane ControlPlane, app, key string, initial *dto.AppDeployResponse, out, errOut io.Writer, jsonOut bool) error { + if app == "" || key == "" { + return fmt.Errorf("APP and --key are required") + } + progressOut := out + if jsonOut { + progressOut = errOut + } + op, err := awaitOperation(ctx, plane, app, key, initial, progressOut) + if err != nil { + return err + } + if jsonOut { + if err := writeJSON(out, op); err != nil { + return err + } + } else if err := renderAppDeployResponse(out, op); err != nil { + return err + } + if op.Outcome == domain.AppOutcomeSuccess { + return nil + } + return &OperationFailedError{App: app, Op: op.Op, Outcome: op.Outcome} +} + +// awaitOperation polls OperationByKey until the journal is terminal. The +// optional initial response (the 202 mutation body) is reported before the +// first poll so its state is not lost. Progress is written only when the +// operation/step fingerprint changes. +func awaitOperation(ctx context.Context, plane ControlPlane, app, key string, initial *dto.AppDeployResponse, progressOut io.Writer) (*dto.AppDeployResponse, error) { + lastProgress := "" + transient := 0 + for { + op := initial + initial = nil + if op == nil { + got, err := plane.OperationByKey(ctx, app, key) + if err != nil { + if perr := handlePollError(ctx, progressOut, app, key, err, &transient); perr != nil { + return nil, perr + } + continue + } + transient = 0 + op = got + } + if op.Status != dto.AppStatusRunning { + return op, nil + } + if sig := operationProgressSignature(op); sig != lastProgress { + lastProgress = sig + if err := renderOperationProgress(progressOut, op); err != nil { + return nil, err + } + } + if err := waitOperationPoll(ctx); err != nil { + return nil, stopOperationWatch(progressOut, app, key, err) + } + } +} + +// handlePollError classifies one by-key lookup error. It returns nil when +// the caller should poll again and a terminal error otherwise: 404 is +// terminal, cancellation stops local polling, and any other failure is +// retried up to operationPollMaxTransientErrors consecutive times. +func handlePollError(ctx context.Context, progressOut io.Writer, app, key string, err error, transient *int) error { + if ctxErr := ctx.Err(); ctxErr != nil { + return stopOperationWatch(progressOut, app, key, ctxErr) + } + if isRemoteNotFoundError(err) { + return fmt.Errorf("operation %s for app %s is no longer recorded; nothing to resume: %w", key, app, err) + } + *transient++ + if *transient > operationPollMaxTransientErrors { + return fmt.Errorf( + "gave up watching operation %s for app %s after %d consecutive polling errors: %w; resume with %s", + key, app, *transient, err, operationResumeCommand(app, key)) + } + if werr := cliWriteLine(progressOut, cliRenderWarning(fmt.Sprintf( + "operation watch: transient polling error (%d/%d): %v", *transient, operationPollMaxTransientErrors, err))); werr != nil { + return werr + } + if werr := waitOperationPoll(ctx); werr != nil { + return stopOperationWatch(progressOut, app, key, werr) + } + return nil +} + +// waitOperationPoll sleeps for one polling interval, aborting early when +// the command context is cancelled. +func waitOperationPoll(ctx context.Context) error { + if err := ctx.Err(); err != nil { + return err + } + timer := time.NewTimer(operationPollInterval) + defer timer.Stop() + select { + case <-ctx.Done(): + return ctx.Err() + case <-timer.C: + return nil + } +} + +// stopOperationWatch reports an interrupted watch without implying any +// server-side cancellation, then returns the interruption as an error. The +// resume command is printed so the user can continue watching later. +func stopOperationWatch(progressOut io.Writer, app, key string, cause error) error { + if err := cliWriteLine(progressOut, cliRenderWarning("polling stopped; the operation is not cancelled and keeps running on the daemon")); err != nil { + return err + } + if err := cliWriteLine(progressOut, cliRenderMeta("resume:", operationResumeCommand(app, key))); err != nil { + return err + } + return fmt.Errorf("operation watch interrupted: %w", cause) +} + +// operationProgressSignature fingerprints the visible operation/step state +// so unchanged polls are not reprinted. +func operationProgressSignature(op *dto.AppDeployResponse) string { + var b strings.Builder + b.WriteString(op.Status) + b.WriteByte(0) + b.WriteString(op.Outcome) + for _, step := range op.Steps { + b.WriteByte(0) + b.WriteString(step.ID) + b.WriteByte('=') + b.WriteString(step.State) + b.WriteByte(':') + b.WriteString(step.Detail) + b.WriteByte(':') + b.WriteString(step.Error) + } + return b.String() +} + +// renderOperationProgress writes one concise progress line for the current +// operation/step state. +func renderOperationProgress(out io.Writer, op *dto.AppDeployResponse) error { + detail := op.Status + if step, ok := activeOperationStep(op.Steps); ok { + detail = fmt.Sprintf("%s %s: %s", op.Status, step.ID, step.State) + if step.Error != "" { + detail += ": " + sanitizeTerminalText(step.Error) + } else if step.Detail != "" { + detail += ": " + sanitizeTerminalText(step.Detail) + } + } + return cliWriteLine(out, cliRenderMeta("op "+op.Op+":", detail)) +} + +// activeOperationStep returns the step still to run (pending) when one +// exists, otherwise the last journaled step. +func activeOperationStep(steps []dto.AppStepDTO) (dto.AppStepDTO, bool) { + if len(steps) == 0 { + return dto.AppStepDTO{}, false + } + for _, step := range steps { + switch step.State { + case domain.AppStepSucceeded, domain.AppStepFailed, domain.AppStepNotRun: + continue + default: + return step, true + } + } + return steps[len(steps)-1], true +} diff --git a/internal/adapters/in/cli/apps_watch_test.go b/internal/adapters/in/cli/apps_watch_test.go new file mode 100644 index 000000000..a84283f50 --- /dev/null +++ b/internal/adapters/in/cli/apps_watch_test.go @@ -0,0 +1,208 @@ +package cli + +// Tests for 202 deploy polling and `apps operations watch`: state-change +// progress, machine-readable JSON, terminal rendering/exit, bounded +// transient failures, fast 404, and Ctrl-C resume without server abort. + +import ( + "bytes" + "context" + "encoding/json" + "errors" + "io" + "net/http" + "strings" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/dto" + "github.com/bnema/gordon/internal/adapters/in/cli/remote" +) + +// shortenOperationPoll makes the polling cadence near-instant for tests. +func shortenOperationPoll(t *testing.T) { + t.Helper() + prev := operationPollInterval + operationPollInterval = time.Millisecond + t.Cleanup(func() { operationPollInterval = prev }) +} + +func runningOp(op string, steps ...dto.AppStepDTO) *dto.AppDeployResponse { + return &dto.AppDeployResponse{Op: op, App: "blog", Status: dto.AppStatusRunning, Steps: steps} +} + +func TestRunAppDeploy_PollsAcceptedToTerminal(t *testing.T) { + shortenOperationPoll(t) + plane := appPlane(t) + plane.EXPECT().DeployApp(mock.Anything, "blog", mock.Anything).Return( + runningOp("op-1"), "key-1", nil).Once() + plane.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return( + runningOp("op-1", dto.AppStepDTO{ID: "service.web.start", State: "pending"}), nil).Once() + plane.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return( + &dto.AppDeployResponse{Op: "op-1", App: "blog", Status: "success", Outcome: "success"}, nil).Once() + + var out, errOut bytes.Buffer + require.NoError(t, runAppDeploy(context.Background(), plane, "blog", dto.AppDeployRequest{}, &out, &errOut, false)) + text := out.String() + assert.Contains(t, text, "op op-1:") + assert.Contains(t, text, "running") + assert.Contains(t, text, "success") +} + +func TestRunAppDeploy_UnchangedStateNotReprinted(t *testing.T) { + shortenOperationPoll(t) + plane := appPlane(t) + step := dto.AppStepDTO{ID: "service.web.start", State: "pending"} + unchanged := runningOp("op-1", step) + advanced := runningOp("op-1", + dto.AppStepDTO{ID: "service.web.start", State: "succeeded"}, + dto.AppStepDTO{ID: "service.web.health", State: "pending"}) + + plane.EXPECT().DeployApp(mock.Anything, "blog", mock.Anything).Return(runningOp("op-1", step), "key-1", nil).Once() + plane.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return(unchanged, nil).Once() + plane.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return(advanced, nil).Once() + plane.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return( + &dto.AppDeployResponse{Op: "op-1", App: "blog", Status: "success", Outcome: "success"}, nil).Once() + + var out, errOut bytes.Buffer + require.NoError(t, runAppDeploy(context.Background(), plane, "blog", dto.AppDeployRequest{}, &out, &errOut, false)) + assert.Equal(t, 2, strings.Count(out.String(), "op op-1:"), + "initial and one advanced poll render progress; the unchanged poll is not reprinted") +} + +func TestRunAppDeploy_AcceptedIssuesNoSecondMutation(t *testing.T) { + shortenOperationPoll(t) + plane := appPlane(t) + // .Once() on DeployApp proves the accepted path never re-POSTs; the + // by-key lookups are the only way the outcome is observed. + plane.EXPECT().DeployApp(mock.Anything, "blog", mock.Anything).Return(runningOp("op-1"), "key-1", nil).Once() + plane.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return( + &dto.AppDeployResponse{Op: "op-1", App: "blog", Status: "success", Outcome: "success"}, nil).Once() + + var out, errOut bytes.Buffer + require.NoError(t, runAppDeploy(context.Background(), plane, "blog", dto.AppDeployRequest{}, &out, &errOut, false)) +} + +func TestRunAppDeploy_TerminalPartialRendersAndFails(t *testing.T) { + shortenOperationPoll(t) + plane := appPlane(t) + plane.EXPECT().DeployApp(mock.Anything, "blog", mock.Anything).Return(runningOp("op-1"), "key-1", nil).Once() + plane.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return(&dto.AppDeployResponse{ + Op: "op-1", App: "blog", Status: "partial", Outcome: "partial", + Services: map[string]dto.AppServiceResultDTO{ + "web": {Result: "failed", EffectiveRevision: "rev-b", Error: "boom"}, + }, + }, nil).Once() + + var out, errOut bytes.Buffer + err := runAppDeploy(context.Background(), plane, "blog", dto.AppDeployRequest{}, &out, &errOut, false) + require.Error(t, err, "a terminal partial outcome must exit nonzero") + assert.Contains(t, err.Error(), "partial") + assert.Contains(t, out.String(), "boom", "the terminal journal is rendered before failing") +} + +func TestRunAppDeploy_JSONKeepsStdoutMachineReadable(t *testing.T) { + shortenOperationPoll(t) + plane := appPlane(t) + plane.EXPECT().DeployApp(mock.Anything, "blog", mock.Anything).Return( + runningOp("op-1", dto.AppStepDTO{ID: "service.web.start", State: "pending"}), "key-1", nil).Once() + plane.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return( + runningOp("op-1", dto.AppStepDTO{ID: "service.web.start", State: "succeeded"}), nil).Once() + plane.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return( + &dto.AppDeployResponse{Op: "op-1", App: "blog", Status: "success", Outcome: "success"}, nil).Once() + + var out, errOut bytes.Buffer + require.NoError(t, runAppDeploy(context.Background(), plane, "blog", dto.AppDeployRequest{}, &out, &errOut, true)) + + dec := json.NewDecoder(bytes.NewReader(out.Bytes())) + var got dto.AppDeployResponse + require.NoError(t, dec.Decode(&got)) + assert.Equal(t, "op-1", got.Op) + assert.Equal(t, "success", got.Outcome) + var extra json.RawMessage + assert.ErrorIs(t, dec.Decode(&extra), io.EOF, "stdout carries exactly one JSON document") + assert.NotContains(t, out.String(), "op op-1:", "progress must not pollute JSON stdout") + assert.Contains(t, errOut.String(), "op op-1:", "progress goes to stderr in JSON mode") +} + +func TestWatchOperation_CancelPrintsResumeWithoutAbort(t *testing.T) { + plane := appPlane(t) + ctx, cancel := context.WithCancel(context.Background()) + cancel() + + var out, errOut bytes.Buffer + err := watchOperation(ctx, plane, "blog", "key-1", runningOp("op-1"), &out, &errOut, false) + require.Error(t, err) + assert.ErrorIs(t, err, context.Canceled) + text := out.String() + assert.Contains(t, text, "gordon apps operations watch blog --key key-1") + assert.Contains(t, text, "not cancelled") + assert.NotContains(t, text, "cancel op") +} + +func TestWatchOperation_CancelKeepsJSONStdoutClean(t *testing.T) { + plane := appPlane(t) + ctx, cancel := context.WithCancel(context.Background()) + cancel() + + var out, errOut bytes.Buffer + err := watchOperation(ctx, plane, "blog", "key-1", runningOp("op-1"), &out, &errOut, true) + require.Error(t, err) + assert.Empty(t, out.String(), "JSON stdout stays empty on interruption") + assert.Contains(t, errOut.String(), "gordon apps operations watch blog --key key-1") +} + +func TestRunAppsOperationsWatch_ResumesExisting(t *testing.T) { + shortenOperationPoll(t) + plane := appPlane(t) + plane.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return( + runningOp("op-1", dto.AppStepDTO{ID: "service.web.start", State: "pending"}), nil).Once() + plane.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return( + &dto.AppDeployResponse{Op: "op-1", App: "blog", Status: "success", Outcome: "success"}, nil).Once() + + var out, errOut bytes.Buffer + require.NoError(t, runAppsOperationsWatch(context.Background(), plane, &out, &errOut, "blog", "key-1", false)) + assert.Contains(t, out.String(), "success") +} + +func TestAwaitOperation_TransientErrorsAreBounded(t *testing.T) { + shortenOperationPoll(t) + plane := appPlane(t) + boom := errors.New("connection refused") + plane.EXPECT().OperationByKey(mock.Anything, "blog", "key-1").Return(nil, boom).Times(operationPollMaxTransientErrors + 1) + + var progress bytes.Buffer + _, err := awaitOperation(context.Background(), plane, "blog", "key-1", nil, &progress) + require.Error(t, err) + assert.Contains(t, err.Error(), "gave up watching") + assert.Contains(t, err.Error(), "gordon apps operations watch blog --key key-1") + assert.Contains(t, progress.String(), "transient polling error") +} + +func TestAwaitOperation_NotFoundFailsFast(t *testing.T) { + plane := appPlane(t) + plane.EXPECT().OperationByKey(mock.Anything, "blog", "gone").Return(nil, + &remote.HTTPError{StatusCode: http.StatusNotFound, Status: "404 Not Found", Body: "unknown operation"}).Once() + + _, err := awaitOperation(context.Background(), plane, "blog", "gone", nil, &bytes.Buffer{}) + require.Error(t, err) + assert.Contains(t, err.Error(), "no longer recorded") +} + +func TestAppsOperationsWatchCmd_Flags(t *testing.T) { + cmd := newAppsOperationsWatchCmd() + assert.Equal(t, "watch APP", cmd.Use) + assert.NotNil(t, cmd.Flags().Lookup("key")) + assert.NotNil(t, cmd.Flags().Lookup("json")) + + subs := map[string]bool{} + for _, sub := range newAppsOperationsCmd().Commands() { + subs[sub.Name()] = true + } + assert.True(t, subs["show"]) + assert.True(t, subs["watch"]) +} diff --git a/internal/adapters/in/cli/attachments.go b/internal/adapters/in/cli/attachments.go deleted file mode 100644 index f2cb4c3b1..000000000 --- a/internal/adapters/in/cli/attachments.go +++ /dev/null @@ -1,516 +0,0 @@ -package cli - -import ( - "context" - "fmt" - "io" - "sort" - - "github.com/bnema/gordon/internal/adapters/in/cli/remote" - "github.com/bnema/gordon/internal/adapters/in/cli/ui/components" - "github.com/bnema/gordon/internal/adapters/in/cli/ui/styles" - "github.com/bnema/gordon/internal/domain" - - "github.com/spf13/cobra" -) - -// newAttachmentsCmd creates the attachments command group. -func newAttachmentsCmd() *cobra.Command { - cmd := &cobra.Command{ - Use: "attachments", - Aliases: []string{"attach"}, - Short: "Manage container attachments", - Long: `Manage container attachments (databases, caches, queues) in the configuration. - -Attachments are service dependencies that run alongside your application containers. -They are defined per-domain or per-network-group in the configuration. - -When targeting a remote Gordon instance (via --remote flag or GORDON_REMOTE env var), -these commands operate on the remote server. Otherwise, they require access to -the local Gordon configuration.`, - } - - cmd.AddCommand(newAttachmentsListCmd()) - cmd.AddCommand(newAttachmentsAddCmd()) - cmd.AddCommand(newAttachmentsRemoveCmd()) - cmd.AddCommand(newAttachmentsOrphansCmd()) - cmd.AddCommand(newAttachmentsPruneCmd()) - cmd.AddCommand(newAttachmentsPushCmd()) - - return cmd -} - -// newAttachmentsListCmd creates the attachments list command. -func newAttachmentsListCmd() *cobra.Command { - var jsonOut bool - - cmd := &cobra.Command{ - Use: "list [domain-or-group]", - Short: "List configured attachments", - Long: `List all configured attachments, or attachments for a specific domain/group. - -Examples: - gordon attachments list # List all attachments - gordon attachments list app.example.com # List attachments for domain - gordon attachments list backend # List attachments for network group`, - Args: cobra.MaximumNArgs(1), - RunE: func(cmd *cobra.Command, args []string) error { - ctx := cmd.Context() - target := "" - if len(args) > 0 { - target = args[0] - } - - if target != "" { - handle, err := resolveControlPlaneForAttachmentTarget(ctx, target) - if err != nil { - return err - } - defer handle.close() - return runAttachmentsList(ctx, handle.plane, target, jsonOut, cmd.OutOrStdout()) - } - - client, isRemote, err := GetRemoteClient() - if err != nil { - return err - } - if isRemote { - return runAttachmentsListRemote(ctx, client, target, jsonOut, cmd.OutOrStdout()) - } - return runAttachmentsListLocal(ctx, configPath, target, jsonOut, cmd.OutOrStdout()) - }, - } - - cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") - - return cmd -} - -func runAttachmentsList(ctx context.Context, cp ControlPlane, target string, jsonOut bool, out io.Writer) error { - if target != "" { - images, err := cp.GetAttachmentsConfig(ctx, target) - if err != nil { - return fmt.Errorf("failed to get attachments: %w", err) - } - if len(images) == 0 { - _, err := fmt.Fprintf(out, "No attachments configured for '%s'\n", target) - return err - } - if jsonOut { - return writeJSON(out, map[string]any{"target": target, "images": images}) - } - return renderAttachmentTargetList(out, fmt.Sprintf("Attachments for %s", target), images) - } - - attachments, err := cp.GetAllAttachmentsConfig(ctx) - if err != nil { - return fmt.Errorf("failed to list attachments: %w", err) - } - if len(attachments) == 0 { - return cliWriteLine(out, styles.Theme.Muted.Render("No attachments configured")) - } - if jsonOut { - return writeJSON(out, map[string]any{"attachments": attachments}) - } - return renderAllAttachmentsList(out, "Attachments", attachments) -} - -func renderAttachmentTargetList(out io.Writer, title string, images []string) error { - if err := cliWriteLine(out, cliRenderTitle(title)); err != nil { - return err - } - if err := cliWriteLine(out, ""); err != nil { - return err - } - for _, img := range images { - if err := cliWritef(out, " %s\n", img); err != nil { - return err - } - } - return nil -} - -func renderAllAttachmentsList(out io.Writer, title string, attachments map[string][]string) error { - targets := make([]string, 0, len(attachments)) - for t := range attachments { - targets = append(targets, t) - } - sort.Strings(targets) - - if err := cliWriteLine(out, cliRenderTitle(title)); err != nil { - return err - } - if err := cliWriteLine(out, ""); err != nil { - return err - } - for _, t := range targets { - images := attachments[t] - if err := cliWritef(out, "%s\n", styles.Theme.Bold.Render(t)); err != nil { - return err - } - for _, img := range images { - if err := cliWritef(out, " %s\n", img); err != nil { - return err - } - } - } - return nil -} - -// runAttachmentsListRemote lists attachments from a remote Gordon instance. -func runAttachmentsListRemote(ctx context.Context, client *remote.Client, target string, jsonOut bool, out io.Writer) error { - if target != "" { - // List for specific target - images, err := client.GetAttachmentsConfig(ctx, target) - if err != nil { - return fmt.Errorf("failed to get attachments: %w", err) - } - - if len(images) == 0 { - fmt.Printf("No attachments configured for '%s'\n", target) - return nil - } - - if jsonOut { - return writeJSON(out, map[string]any{"target": target, "images": images}) - } - - fmt.Println(styles.Theme.Title.Render(fmt.Sprintf("Attachments for %s", target))) - fmt.Println() - for _, img := range images { - fmt.Printf(" %s\n", img) - } - return nil - } - - // List all attachments - attachments, err := client.GetAllAttachmentsConfig(ctx) - if err != nil { - return fmt.Errorf("failed to list attachments: %w", err) - } - - if len(attachments) == 0 { - fmt.Println(styles.Theme.Muted.Render("No attachments configured")) - return nil - } - - if jsonOut { - return writeJSON(out, map[string]any{"attachments": attachments}) - } - - // Sort targets for consistent output - targets := make([]string, 0, len(attachments)) - for t := range attachments { - targets = append(targets, t) - } - sort.Strings(targets) - - fmt.Println(styles.Theme.Title.Render("Attachments")) - fmt.Println() - - for _, t := range targets { - images := attachments[t] - fmt.Printf("%s\n", styles.Theme.Bold.Render(t)) - for _, img := range images { - fmt.Printf(" %s\n", img) - } - } - - return nil -} - -// runAttachmentsListLocal lists attachments from local configuration. -func runAttachmentsListLocal(ctx context.Context, cfgPath string, target string, jsonOut bool, out io.Writer) error { - local, err := GetLocalServices(cfgPath) - if err != nil { - return fmt.Errorf("failed to initialize local services: %w", err) - } - - if target != "" { - // List for specific target - images, err := local.GetConfigService().GetAttachmentsFor(ctx, target) - if err != nil { - return fmt.Errorf("failed to get attachments: %w", err) - } - - if len(images) == 0 { - fmt.Printf("No attachments configured for '%s'\n", target) - return nil - } - - if jsonOut { - return writeJSON(out, map[string]any{"target": target, "images": images}) - } - - fmt.Println(styles.Theme.Title.Render(fmt.Sprintf("Attachments for %s (local)", target))) - fmt.Println() - for _, img := range images { - fmt.Printf(" %s\n", img) - } - return nil - } - - // List all attachments - attachments := local.GetConfigService().GetAllAttachments(ctx) - - if len(attachments) == 0 { - fmt.Println(styles.Theme.Muted.Render("No attachments configured")) - return nil - } - - if jsonOut { - return writeJSON(out, map[string]any{"attachments": attachments}) - } - - // Sort targets for consistent output - targets := make([]string, 0, len(attachments)) - for t := range attachments { - targets = append(targets, t) - } - sort.Strings(targets) - - fmt.Println(styles.Theme.Title.Render("Attachments (local)")) - fmt.Println() - - for _, t := range targets { - images := attachments[t] - fmt.Printf("%s\n", styles.Theme.Bold.Render(t)) - for _, img := range images { - fmt.Printf(" %s\n", img) - } - } - - return nil -} - -func newAttachmentsOrphansCmd() *cobra.Command { - var jsonOut bool - cmd := &cobra.Command{ - Use: "orphans", - Short: "List orphaned attachment containers", - Args: cobra.NoArgs, - RunE: func(cmd *cobra.Command, args []string) error { - handle, err := resolveControlPlane(configPath) - if err != nil { - return err - } - defer handle.close() - lister, ok := handle.plane.(interface { - ListOrphanedAttachments(context.Context) ([]domain.CleanupAttachment, error) - }) - if !ok { - return fmt.Errorf("attachment orphan listing is unavailable") - } - attachments, err := lister.ListOrphanedAttachments(cmd.Context()) - if err != nil { - return fmt.Errorf("failed to list orphaned attachments: %w", err) - } - if jsonOut { - return writeJSON(cmd.OutOrStdout(), map[string]any{"attachments": attachments}) - } - return writeAttachmentOrphansText(cmd.OutOrStdout(), attachments) - }, - } - cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") - return cmd -} - -func newAttachmentsPruneCmd() *cobra.Command { - var jsonOut bool - var stop bool - cmd := &cobra.Command{ - Use: "prune", - Short: "Review or stop orphaned attachment containers", - Args: cobra.NoArgs, - RunE: func(cmd *cobra.Command, args []string) error { - handle, err := resolveControlPlane(configPath) - if err != nil { - return err - } - defer handle.close() - cleaner, ok := handle.plane.(interface { - CleanupOrphanedAttachments(context.Context, string, bool) (*domain.CleanupReport, error) - }) - if !ok { - return fmt.Errorf("attachment orphan cleanup is unavailable") - } - report, err := cleaner.CleanupOrphanedAttachments(cmd.Context(), "", stop) - if err != nil { - return fmt.Errorf("failed to cleanup orphaned attachments: %w", err) - } - if jsonOut { - return writeJSON(cmd.OutOrStdout(), report) - } - return writeAttachmentPruneText(cmd.OutOrStdout(), report, stop) - }, - } - cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") - cmd.Flags().BoolVar(&stop, "stop", false, "Stop and remove orphaned attachment containers while preserving volumes") - return cmd -} - -func writeAttachmentOrphansText(out io.Writer, attachments []domain.CleanupAttachment) error { - if len(attachments) == 0 { - return cliWriteLine(out, styles.Theme.Muted.Render("No orphaned attachments found")) - } - if err := cliWriteLine(out, cliRenderTitle("Orphaned attachments")); err != nil { - return err - } - for _, a := range attachments { - label := a.Name - if label == "" { - label = a.ContainerID - } - if a.Image != "" { - label += " -> " + a.Image - } - if err := cliWriteLine(out, " "+label); err != nil { - return err - } - } - return nil -} - -func writeAttachmentPruneText(out io.Writer, report *domain.CleanupReport, stop bool) error { - if report == nil { - return nil - } - if !stop { - if err := writeAttachmentOrphansText(out, report.PreservedAttachments); err != nil { - return err - } - return writeCleanupMessages(out, report.Warnings, report.Hints) - } - if len(report.RemovedContainers) == 0 { - if err := cliWriteLine(out, styles.Theme.Muted.Render("No orphaned attachments removed")); err != nil { - return err - } - } else { - if err := cliWriteLine(out, cliRenderSuccess("Removed orphaned attachment containers:")); err != nil { - return err - } - for _, c := range report.RemovedContainers { - label := c.Name - if label == "" { - label = c.ID - } - if err := cliWriteLine(out, " "+label); err != nil { - return err - } - } - } - for _, failure := range report.PartialFailures { - if err := cliWriteLine(out, cliRenderWarning(fmt.Sprintf("%s %s failed: %s", failure.Action, failure.Name, failure.Error))); err != nil { - return err - } - } - return writeCleanupMessages(out, report.Warnings, report.Hints) -} - -// newAttachmentsAddCmd creates the attachments add command. -func newAttachmentsAddCmd() *cobra.Command { - cmd := &cobra.Command{ - Use: "add ", - Short: "Add an attachment", - Long: `Add a container attachment to a domain or network group. - -Note: Attachments require network_isolation.enabled = true in your configuration. -Without network isolation, containers use Docker's default bridge network which -does not provide DNS resolution - your app won't be able to reach attachments -by hostname (e.g., 'postgres:5432'). - -Examples: - gordon attachments add app.example.com postgres:18 - gordon attachments add backend redis:7-alpine - gordon --remote https://gordon.mydomain.com attachments add api.mydomain.com memcached:latest`, - Args: cobra.ExactArgs(2), - RunE: func(cmd *cobra.Command, args []string) error { - ctx := cmd.Context() - target := args[0] - image := args[1] - - handle, err := resolveControlPlaneForAttachmentTarget(ctx, target) - if err != nil { - return err - } - defer handle.close() - - if !handle.isRemote { - local, err := GetLocalServices(configPath) - if err != nil { - return fmt.Errorf("failed to initialize local services: %w", err) - } - - // Warn if network isolation is disabled - if !local.GetConfigService().IsNetworkIsolationEnabled() { - fmt.Println(styles.RenderWarning("network_isolation.enabled = false")) - fmt.Println(styles.Theme.Muted.Render(" Attachments require network isolation for DNS resolution.")) - fmt.Println(styles.Theme.Muted.Render(" Without it, your app cannot reach attachments by hostname.")) - fmt.Println(styles.Theme.Muted.Render(" Enable with: [network_isolation] enabled = true")) - fmt.Println() - } - } - - if err := handle.plane.AddAttachment(ctx, target, image); err != nil { - return fmt.Errorf("failed to add attachment: %w", err) - } - - fmt.Println(styles.RenderSuccess(fmt.Sprintf("Attachment added: %s -> %s", target, image))) - return nil - }, - } - - return cmd -} - -// newAttachmentsRemoveCmd creates the attachments remove command. -func newAttachmentsRemoveCmd() *cobra.Command { - var force bool - - cmd := &cobra.Command{ - Use: "remove ", - Short: "Remove an attachment", - Long: `Remove a container attachment from a domain or network group. - -Examples: - gordon attachments remove app.example.com postgres:15 - gordon attachments remove backend redis:7-alpine --force`, - Args: cobra.ExactArgs(2), - RunE: func(cmd *cobra.Command, args []string) error { - ctx := cmd.Context() - target := args[0] - image := args[1] - - // Confirm unless --force - if !force { - confirmed, err := components.RunConfirm( - fmt.Sprintf("Remove attachment '%s' from '%s'?", image, target), - components.WithDescription("This will remove the attachment from the configuration. The container will be stopped on next reload."), - ) - if err != nil { - return err - } - if !confirmed { - fmt.Println(styles.Theme.Muted.Render("Cancelled")) - return nil - } - } - - handle, err := resolveControlPlaneForAttachmentTarget(ctx, target) - if err != nil { - return err - } - defer handle.close() - if err := handle.plane.RemoveAttachment(ctx, target, image); err != nil { - return fmt.Errorf("failed to remove attachment: %w", err) - } - - fmt.Println(styles.RenderSuccess(fmt.Sprintf("Attachment removed: %s -> %s", target, image))) - return nil - }, - } - - cmd.Flags().BoolVarP(&force, "force", "f", false, "Skip confirmation") - - return cmd -} diff --git a/internal/adapters/in/cli/attachments_push.go b/internal/adapters/in/cli/attachments_push.go deleted file mode 100644 index ef5752361..000000000 --- a/internal/adapters/in/cli/attachments_push.go +++ /dev/null @@ -1,221 +0,0 @@ -package cli - -import ( - "context" - "fmt" - "io" - "sort" - "strings" - - "github.com/spf13/cobra" - - "github.com/bnema/gordon/internal/adapters/in/cli/ui/styles" - "github.com/bnema/gordon/pkg/validation" -) - -var ( - resolveControlPlaneFn = resolveControlPlane - determineVersionFn = determineVersion -) - -// attachmentPushRequest holds all inputs for the attachments push command. -type attachmentPushRequest struct { - ImageArg string - Tag string - Build buildConfig -} - -func newAttachmentsPushCmd() *cobra.Command { - var ( - build bool - platform string - tag string - dockerfile string - buildArgs []string - ) - - cmd := &cobra.Command{ - Use: "push ", - Short: "Build, tag, and push an attachment image", - Args: cobra.ExactArgs(1), - RunE: func(cmd *cobra.Command, args []string) error { - return runAttachmentsPush(cmd.Context(), attachmentPushRequest{ - ImageArg: args[0], - Tag: tag, - Build: buildConfig{Enabled: build, Platform: platform, Dockerfile: dockerfile, BuildArgs: buildArgs}, - }, cmd.OutOrStdout()) - }, - } - - cmd.Flags().BoolVar(&build, "build", false, "Build the image first using docker buildx") - cmd.Flags().StringVar(&platform, "platform", "linux/amd64", "Target platform (used with --build)") - cmd.Flags().StringVarP(&dockerfile, "file", "f", "", "Path to Dockerfile (default: ./Dockerfile, used with --build)") - cmd.Flags().StringVar(&tag, "tag", "", "Override version tag (default: CI tag ref or git describe)") - cmd.Flags().StringArrayVar(&buildArgs, "build-arg", nil, "Additional build args (used with --build)") - - return cmd -} - -func runAttachmentsPush(ctx context.Context, req attachmentPushRequest, out io.Writer) error { - dockerfile, err := resolveDockerfile(req.Build.Dockerfile, req.Build.Enabled) - if err != nil { - return err - } - - inferredRemote, err := inferRemoteForAttachmentImage(ctx, req.ImageArg) - if err != nil { - return err - } - - var handle *controlPlaneHandle - if inferredRemote != nil { - handle = newRemoteControlPlaneHandle(inferredRemote) - if err := cliWritef(out, "Remote: %s %s\n", styles.Theme.Bold.Render(inferredRemote.DisplayName()), styles.Theme.Muted.Render("(auto-detected)")); err != nil { - return err - } - } else { - handle, err = resolveControlPlaneFn(configPath) - if err != nil { - return err - } - } - defer handle.close() - - for _, ba := range req.Build.BuildArgs { - if err := validateBuildArg(ba); err != nil { - return err - } - } - - registry, imageName, targets, err := resolveAttachmentImage(ctx, handle.plane, req.ImageArg) - if err != nil { - return err - } - - version, err := resolveVersionWithFn(ctx, req.Tag) - if err != nil { - return err - } - - img := imagePush{ - Registry: registry, - ImageName: imageName, - Version: version, - } - img.VersionRef, img.LatestRef = resolveImageRefs(registry, imageName, version) - - var imageOps pushImageOps - if inferredRemote != nil { - imageOps, err = newImageOpsForResolvedRemote(inferredRemote) - } else { - imageOps, err = newImageOpsFn() - } - if err != nil { - return err - } - - if err := printAttachmentPushInfo(out, req.ImageArg, targets, img.VersionRef, img.LatestRef, version); err != nil { - return err - } - - build := buildConfig{ - Enabled: req.Build.Enabled, - Platform: req.Build.Platform, - Dockerfile: dockerfile, - BuildArgs: req.Build.BuildArgs, - } - - if err := performAttachmentPush(ctx, imageOps, build, img); err != nil { - return err - } - - return cliWriteLine(out, styles.RenderSuccess("Push complete")) -} - -func printAttachmentPushInfo(out io.Writer, imageArg string, targets []string, versionRef, latestRef, version string) error { - if err := cliWritef(out, "Attachment image: %s\n", styles.Theme.Bold.Render(imageArg)); err != nil { - return err - } - if err := cliWritef(out, "Targets: %s\n", styles.Theme.Bold.Render(formatAttachmentTargets(targets))); err != nil { - return err - } - if err := cliWritef(out, "Image: %s\n", styles.Theme.Bold.Render(versionRef)); err != nil { - return err - } - if version != "latest" { - if err := cliWritef(out, "Also: %s\n", styles.Theme.Bold.Render(latestRef)); err != nil { - return err - } - } - return nil -} - -func performAttachmentPush(ctx context.Context, ops pushImageOps, build buildConfig, img imagePush) error { - if build.Enabled { - return buildAndPush(ctx, ops, build, img) - } - return tagAndPush(ctx, ops, img) -} - -func resolveAttachmentImage(ctx context.Context, cp ControlPlane, imageArg string) (registry, imageName string, targets []string, err error) { - imageName = normalizeAttachmentImageName(imageArg) - - targets, err = cp.FindAttachmentTargetsByImage(ctx, imageArg) - if err != nil { - return "", "", nil, fmt.Errorf("failed to find attachment targets for image %q: %w", imageArg, err) - } - if len(targets) == 0 { - return "", "", nil, fmt.Errorf("image %q is not configured as an attachment", imageArg) - } - - status, err := cp.GetStatus(ctx) - if err != nil { - return "", "", nil, fmt.Errorf("failed to get status: %w", err) - } - - return status.RegistryDomain, imageName, targets, nil -} - -func normalizeAttachmentImageName(imageArg string) string { - name := imageArg - if idx := strings.Index(name, "@"); idx != -1 { - name = name[:idx] - } - if idx := strings.LastIndex(name, ":"); idx != -1 { - slashIdx := strings.LastIndex(name, "/") - if idx > slashIdx { - name = name[:idx] - } - } - parts := strings.SplitN(name, "/", 2) - if len(parts) == 2 { - host := parts[0] - if strings.ContainsAny(host, ".:") || host == "localhost" { - return parts[1] - } - } - return name -} - -func formatAttachmentTargets(targets []string) string { - cloned := append([]string(nil), targets...) - sort.Strings(cloned) - return fmt.Sprintf("%v", cloned) -} - -func resolveVersionWithFn(ctx context.Context, tag string) (string, error) { - version := determineVersionFn(ctx, tag) - if version != "latest" { - if err := validateVersionTag(version); err != nil { - return "", err - } - } - return version, nil -} - -func validateVersionTag(version string) error { - if err := validation.ValidateReference(version); err != nil { - return fmt.Errorf("invalid version tag %q: %w", version, err) - } - return nil -} diff --git a/internal/adapters/in/cli/attachments_push_test.go b/internal/adapters/in/cli/attachments_push_test.go deleted file mode 100644 index 3ee721512..000000000 --- a/internal/adapters/in/cli/attachments_push_test.go +++ /dev/null @@ -1,145 +0,0 @@ -package cli - -import ( - "bytes" - "context" - "testing" - - climocks "github.com/bnema/gordon/internal/adapters/in/cli/mocks" - "github.com/bnema/gordon/internal/adapters/in/cli/remote" - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/mock" - "github.com/stretchr/testify/require" -) - -func TestResolveAttachmentImage_OneTarget(t *testing.T) { - cpMock := climocks.NewMockControlPlane(t) - cpMock.EXPECT().FindAttachmentTargetsByImage(context.Background(), "postgres").Return([]string{"app.example.com"}, nil).Once() - cpMock.EXPECT().GetStatus(context.Background()).Return(&remote.Status{RegistryDomain: "registry.example.com"}, nil).Once() - - registry, imageName, targets, err := resolveAttachmentImage(context.Background(), cpMock, "postgres") - - require.NoError(t, err) - assert.Equal(t, "registry.example.com", registry) - assert.Equal(t, "postgres", imageName) - assert.Equal(t, []string{"app.example.com"}, targets) -} - -func TestResolveAttachmentImage_MultipleTargets(t *testing.T) { - cpMock := climocks.NewMockControlPlane(t) - cpMock.EXPECT().FindAttachmentTargetsByImage(context.Background(), "redis").Return([]string{"app.example.com", "backend"}, nil).Once() - cpMock.EXPECT().GetStatus(context.Background()).Return(&remote.Status{RegistryDomain: "registry.example.com"}, nil).Once() - - registry, imageName, targets, err := resolveAttachmentImage(context.Background(), cpMock, "redis") - - require.NoError(t, err) - assert.Equal(t, "registry.example.com", registry) - assert.Equal(t, "redis", imageName) - assert.Equal(t, []string{"app.example.com", "backend"}, targets) -} - -func TestResolveAttachmentImage_NotConfigured(t *testing.T) { - cpMock := climocks.NewMockControlPlane(t) - cpMock.EXPECT().FindAttachmentTargetsByImage(context.Background(), "postgres").Return(nil, nil).Once() - - _, _, _, err := resolveAttachmentImage(context.Background(), cpMock, "postgres") - - require.Error(t, err) - assert.Contains(t, err.Error(), "not configured as an attachment") -} - -func TestAttachmentPushCmd_NoDeploy(t *testing.T) { - origResolveControlPlane := resolveControlPlaneFn - origDetermineVersion := determineVersionFn - origNewImageOps := newImageOpsFn - t.Cleanup(func() { - resolveControlPlaneFn = origResolveControlPlane - determineVersionFn = origDetermineVersion - newImageOpsFn = origNewImageOps - }) - - cpMock := climocks.NewMockControlPlane(t) - cpMock.EXPECT().FindAttachmentTargetsByImage(context.Background(), "postgres").Return([]string{"app.example.com"}, nil).Once() - cpMock.EXPECT().GetStatus(context.Background()).Return(&remote.Status{RegistryDomain: "registry.example.com"}, nil).Once() - - resolveControlPlaneFn = func(string) (*controlPlaneHandle, error) { - return &controlPlaneHandle{plane: cpMock}, nil - } - determineVersionFn = func(context.Context, string) string { return "v1.2.3" } - pushCalled := false - newImageOpsFn = func() (pushImageOps, error) { - m := climocks.NewMockpushImageOps(t) - m.On("Tag", mock.Anything, mock.Anything, mock.Anything).Return(nil) - m.On("Exists", mock.Anything, mock.Anything).Return(true, nil) - m.On("Push", mock.Anything, mock.Anything).Run(func(mock.Arguments) { - pushCalled = true - }).Return(nil) - return m, nil - } - - cmd := newAttachmentsPushCmd() - out := new(bytes.Buffer) - cmd.SetOut(out) - cmd.SetErr(out) - cmd.SetArgs([]string{"postgres"}) - - err := cmd.Execute() - - require.NoError(t, err) - assert.True(t, pushCalled) - assert.Contains(t, out.String(), "Push complete") -} - -func TestAttachmentPushCmd_TaggedImageInputBuildsValidRefs(t *testing.T) { - origResolveControlPlane := resolveControlPlaneFn - origDetermineVersion := determineVersionFn - origNewImageOps := newImageOpsFn - t.Cleanup(func() { - resolveControlPlaneFn = origResolveControlPlane - determineVersionFn = origDetermineVersion - newImageOpsFn = origNewImageOps - }) - - cpMock := climocks.NewMockControlPlane(t) - cpMock.EXPECT().FindAttachmentTargetsByImage(context.Background(), "postgres:18").Return([]string{"app.example.com"}, nil).Once() - cpMock.EXPECT().GetStatus(context.Background()).Return(&remote.Status{RegistryDomain: "registry.example.com"}, nil).Once() - - resolveControlPlaneFn = func(string) (*controlPlaneHandle, error) { - return &controlPlaneHandle{plane: cpMock}, nil - } - determineVersionFn = func(context.Context, string) string { return "v1.2.3" } - - var gotVersionRef string - var gotLatestRef string - newImageOpsFn = func() (pushImageOps, error) { - m := climocks.NewMockpushImageOps(t) - m.On("Tag", mock.Anything, mock.Anything, mock.Anything).Return(nil) - m.On("Exists", mock.Anything, mock.Anything).Return(true, nil) - m.On("Push", mock.Anything, mock.Anything).Run(func(args mock.Arguments) { - ref := args.Get(1).(string) - if gotVersionRef == "" { - gotVersionRef = ref - } else if gotLatestRef == "" { - gotLatestRef = ref - } - }).Return(nil) - return m, nil - } - - cmd := newAttachmentsPushCmd() - cmd.SetOut(new(bytes.Buffer)) - cmd.SetErr(new(bytes.Buffer)) - cmd.SetArgs([]string{"postgres:18"}) - - err := cmd.Execute() - - require.NoError(t, err) - assert.Equal(t, "registry.example.com/postgres:v1.2.3", gotVersionRef) - assert.Equal(t, "registry.example.com/postgres:latest", gotLatestRef) -} - -func TestNormalizeAttachmentImageName_PreservesNamespaceWithoutRegistryHost(t *testing.T) { - got := normalizeAttachmentImageName("myorg/postgres:18") - - assert.Equal(t, "myorg/postgres", got) -} diff --git a/internal/adapters/in/cli/auth.go b/internal/adapters/in/cli/auth.go index 0e9a87020..1da75b4de 100644 --- a/internal/adapters/in/cli/auth.go +++ b/internal/adapters/in/cli/auth.go @@ -60,11 +60,13 @@ func newAuthTokenCmd() *cobra.Command { // newTokenGenerateCmd creates the token generate command. func newTokenGenerateCmd() *cobra.Command { var ( - subject string - scopes string - expiry string - configPath string - repo string + subject string + scopes string + expiry string + cliConfigPath string + repo string + rawOut bool + jsonOut bool ) cmd := &cobra.Command{ @@ -84,12 +86,13 @@ Registry scopes (for Docker push/pull): Admin scopes (for remote CLI access): Format: admin:: - Resources: routes, secrets, config, status, logs, volumes, * (all) + Resources: apps, secrets, config, status, logs, volumes, * (all) Actions: read, write, * (all) Examples: admin:*:* Full admin access (recommended for CLI) - admin:routes:read Read-only routes access + admin:apps:read Read-only apps access + admin:apps:write App mutations (apply, deploy, lifecycle) admin:status:read Read-only status access admin:logs:read Read-only log access admin:volumes:read Read-only volume access @@ -102,8 +105,8 @@ Repository scoping: Combined examples: --scopes "push,pull" All repos, registry only (default) --scopes "push" --repo myapp Push to myapp only - --scopes "push,pull,admin:routes:read" Minimum CI scope (all repos) - --scopes "push,admin:routes:read" --repo app Minimum CI scope (specific repo) + --scopes "push,pull,admin:apps:read" Minimum CI scope (all repos) + --scopes "push,admin:apps:read" --repo app Minimum CI scope (specific repo) EXPIRY: @@ -114,15 +117,18 @@ Supports human-friendly durations: Examples: 1y, 30d, 2w, 6M, 1y6M, 2w3d Use --expiry=0 for a token that never expires (useful for CI).`, RunE: func(cmd *cobra.Command, args []string) error { - return runTokenGenerate(subject, scopes, expiry, configPath, repo) + return runTokenGenerate(subject, scopes, expiry, cliConfigPath, repo, cmd.OutOrStdout(), rawOut, jsonOut) }, } cmd.Flags().StringVar(&subject, "subject", "", "Subject/username for the token (required)") cmd.Flags().StringVar(&scopes, "scopes", "push,pull", "Comma-separated list of scopes") cmd.Flags().StringVar(&expiry, "expiry", "30d", "Token expiry duration (e.g., 1y, 30d, 2w, 24h, 0 for never)") - cmd.Flags().StringVarP(&configPath, "config", "c", "", "Path to config file") + cmd.Flags().StringVarP(&cliConfigPath, "config", "c", "", "Path to config file") cmd.Flags().StringVar(&repo, "repo", "*", "Repository to scope the token to (default: * for all repositories)") + cmd.Flags().BoolVar(&rawOut, "raw", false, "Output only the raw token") + cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") + cmd.MarkFlagsMutuallyExclusive("raw", "json") _ = cmd.MarkFlagRequired("subject") @@ -132,8 +138,8 @@ Use --expiry=0 for a token that never expires (useful for CI).`, // newTokenListCmd creates the token list command. func newTokenListCmd() *cobra.Command { var ( - configPath string - jsonOut bool + cliConfigPath string + jsonOut bool ) cmd := &cobra.Command{ @@ -141,11 +147,11 @@ func newTokenListCmd() *cobra.Command { Short: "List all authentication tokens", Long: `List all stored authentication tokens with their subjects and expiry information.`, RunE: func(cmd *cobra.Command, args []string) error { - return runTokenList(configPath, cmd.OutOrStdout(), jsonOut) + return runTokenList(cliConfigPath, cmd.OutOrStdout(), jsonOut) }, } - cmd.Flags().StringVarP(&configPath, "config", "c", "", "Path to config file") + cmd.Flags().StringVarP(&cliConfigPath, "config", "c", "", "Path to config file") cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") return cmd @@ -154,8 +160,8 @@ func newTokenListCmd() *cobra.Command { // newTokenRevokeCmd creates the token revoke command. func newTokenRevokeCmd() *cobra.Command { var ( - configPath string - all bool + cliConfigPath string + all bool ) cmd := &cobra.Command{ @@ -169,16 +175,16 @@ Examples: Args: cobra.MaximumNArgs(1), RunE: func(cmd *cobra.Command, args []string) error { if all { - return runTokenRevokeAll(configPath) + return runTokenRevokeAll(cliConfigPath) } if len(args) == 0 { return fmt.Errorf("token ID required (or use --all to revoke all tokens)") } - return runTokenRevoke(args[0], configPath) + return runTokenRevoke(args[0], cliConfigPath) }, } - cmd.Flags().StringVarP(&configPath, "config", "c", "", "Path to config file") + cmd.Flags().StringVarP(&cliConfigPath, "config", "c", "", "Path to config file") cmd.Flags().BoolVar(&all, "all", false, "Revoke all tokens") return cmd @@ -188,18 +194,21 @@ Examples: func newAuthInternalCmd() *cobra.Command { return &cobra.Command{ Use: "internal", - Short: "Show internal registry credentials for manual recovery", - Long: `Display the auto-generated internal registry credentials. + Short: "Show server-local registry recovery credentials", + Long: `Display the server's auto-generated internal registry credentials. + +This sensitive recovery command is available only when Gordon runs locally on +the server. The daemon normally uses these ephemeral credentials automatically +to pull from its registry; remote clients and ordinary pushes use scoped tokens. -These credentials are used by Gordon for pulling images from the local registry -(localhost:5000) during internal deploys. You can use them for manual recovery: +For local recovery only: - docker login localhost:5000 -u gordon-internal -p + gordon auth internal + docker login localhost: -u gordon-internal --password-stdin -Note: Credentials are regenerated each time Gordon starts, and are only -available while Gordon is running.`, +Credentials are regenerated whenever the Gordon daemon starts.`, RunE: func(cmd *cobra.Command, args []string) error { - return runShowInternalAuth() + return runShowInternalAuth(cmd.OutOrStdout()) }, } } @@ -402,21 +411,31 @@ func runAuthLogout(out io.Writer) error { } // runShowInternalAuth displays the internal registry credentials. -func runShowInternalAuth() error { +func runShowInternalAuth(out io.Writer) error { + if _, isRemote, err := remote.ResolveStrict(remoteFlag, tokenFlag, insecureTLSFlag); err != nil { + return err + } else if isRemote { + return fmt.Errorf("auth internal is server-local only; run it directly on the Gordon server") + } + creds, err := app.GetInternalCredentials() if err != nil { return err } - fmt.Println("Internal Registry Credentials") - fmt.Println("==============================") - fmt.Printf("Username: %s\n", creds.Username) - fmt.Printf("Password: %s\n", creds.Password) - fmt.Println() - fmt.Println("Usage:") - fmt.Printf(" docker login localhost:5000 -u %s -p %s\n", creds.Username, creds.Password) - - return nil + if err := cliWriteLine(out, "Internal Registry Credentials"); err != nil { + return err + } + if err := cliWriteLine(out, "=============================="); err != nil { + return err + } + if err := cliWritef(out, "Username: %s\n", creds.Username); err != nil { + return err + } + if err := cliWritef(out, "Password: %s\n", creds.Password); err != nil { + return err + } + return cliWriteLine(out, "\nThese credentials are sensitive and intended only for server-local recovery.") } // runTokenGenerate generates a new authentication token. @@ -463,8 +482,8 @@ func parseAndConvertScopes(scopesStr, repo string) ([]string, error) { return scopes, nil } -func runTokenGenerate(subject, scopesStr, expiryStr, configPath, repo string) error { - cfg, err := loadAuthConfig(configPath) +func runTokenGenerate(subject, scopesStr, expiryStr, cliConfigPath, repo string, out io.Writer, rawOut, jsonOut bool) error { + cfg, err := loadAuthConfig(cliConfigPath) if err != nil { return err } @@ -497,26 +516,34 @@ func runTokenGenerate(subject, scopesStr, expiryStr, configPath, repo string) er return fmt.Errorf("failed to generate token: %w", err) } - fmt.Println("Token generated successfully!") - fmt.Printf("Subject: %s\n", subject) - fmt.Printf("Scopes: %s\n", strings.Join(scopes, ", ")) - if expiry == 0 { - fmt.Println("Expiry: never") - } else { - fmt.Printf("Expiry: %s\n", expiry) + if rawOut { + return cliWriteLine(out, token) + } + if jsonOut { + return writeJSON(out, map[string]any{ + "subject": subject, + "scopes": scopes, + "expiry": expiry.String(), + "token": token, + }) } - fmt.Println() - fmt.Println("Token (use as password with docker login):") - fmt.Println(token) - fmt.Println() - fmt.Printf("Usage: docker login -u %s -p \n", subject) - return nil + if err := cliWriteLine(out, "Token generated successfully!"); err != nil { + return err + } + if err := cliWritef(out, "Subject: %s\nScopes: %s\n", subject, strings.Join(scopes, ", ")); err != nil { + return err + } + expiryLabel := expiry.String() + if expiry == 0 { + expiryLabel = "never" + } + return cliWritef(out, "Expiry: %s\n\nToken (use as password with docker login):\n%s\n\nUsage: docker login -u %s -p \n", expiryLabel, token, subject) } // runTokenList lists all stored tokens. -func runTokenList(configPath string, out io.Writer, jsonOut bool) error { - cfg, err := loadAuthConfig(configPath) +func runTokenList(cliConfigPath string, out io.Writer, jsonOut bool) error { + cfg, err := loadAuthConfig(cliConfigPath) if err != nil { return err } @@ -579,8 +606,8 @@ func runTokenList(configPath string, out io.Writer, jsonOut bool) error { } // runTokenRevoke revokes a token by ID. -func runTokenRevoke(tokenID, configPath string) error { - cfg, err := loadAuthConfig(configPath) +func runTokenRevoke(tokenID, cliConfigPath string) error { + cfg, err := loadAuthConfig(cliConfigPath) if err != nil { return err } @@ -601,8 +628,8 @@ func runTokenRevoke(tokenID, configPath string) error { } // runTokenRevokeAll revokes all tokens. -func runTokenRevokeAll(configPath string) error { - cfg, err := loadAuthConfig(configPath) +func runTokenRevokeAll(cliConfigPath string) error { + cfg, err := loadAuthConfig(cliConfigPath) if err != nil { return err } @@ -635,14 +662,14 @@ type cliConfig struct { } // loadAuthConfig loads the configuration needed for auth CLI commands. -func loadAuthConfig(configPath string) (*cliConfig, error) { +func loadAuthConfig(cliConfigPath string) (*cliConfig, error) { v := viper.New() // Set defaults v.SetDefault("server.data_dir", app.DefaultDataDir()) v.SetDefault("auth.secrets_backend", "unsafe") - app.ConfigureViper(v, configPath) + app.ConfigureViper(v, cliConfigPath) if err := v.ReadInConfig(); err != nil { if _, ok := err.(viper.ConfigFileNotFoundError); !ok { diff --git a/internal/adapters/in/cli/auth_scopes_test.go b/internal/adapters/in/cli/auth_scopes_test.go index 31500cffc..680a31a57 100644 --- a/internal/adapters/in/cli/auth_scopes_test.go +++ b/internal/adapters/in/cli/auth_scopes_test.go @@ -26,9 +26,9 @@ func TestParseAndConvertScopes_SpecificRepoPushOnly(t *testing.T) { } func TestParseAndConvertScopes_MixedWithAdminScopes(t *testing.T) { - got, err := parseAndConvertScopes("push,pull,admin:routes:read", "myapp") + got, err := parseAndConvertScopes("push,pull,admin:apps:read", "myapp") require.NoError(t, err) - assert.ElementsMatch(t, []string{"repository:myapp:push,pull", "admin:routes:read"}, got) + assert.ElementsMatch(t, []string{"repository:myapp:push,pull", "admin:apps:read"}, got) } func TestParseAndConvertScopes_AlreadyV2Format_RepoIgnored(t *testing.T) { diff --git a/internal/adapters/in/cli/autoroute_allow.go b/internal/adapters/in/cli/autoroute_allow.go deleted file mode 100644 index fa9293c78..000000000 --- a/internal/adapters/in/cli/autoroute_allow.go +++ /dev/null @@ -1,112 +0,0 @@ -package cli - -import ( - "context" - "fmt" - "io" - - "github.com/spf13/cobra" -) - -func newAutorouteCmd() *cobra.Command { - cmd := &cobra.Command{ - Use: "autoroute", - Short: "Manage auto-route settings", - } - cmd.AddCommand(newAutorouteAllowCmd()) - return cmd -} - -func newAutorouteAllowCmd() *cobra.Command { - cmd := &cobra.Command{ - Use: "allow", - Short: "Manage auto-route allowed domains", - } - cmd.AddCommand(newAutorouteAllowAddCmd()) - cmd.AddCommand(newAutorouteAllowListCmd()) - cmd.AddCommand(newAutorouteAllowRemoveCmd()) - return cmd -} - -func newAutorouteAllowAddCmd() *cobra.Command { - return &cobra.Command{ - Use: "add ", - Short: "Allow an auto-route domain pattern", - Args: cobra.ExactArgs(1), - RunE: func(cmd *cobra.Command, args []string) error { - return runAutorouteAllowAdd(cmd.Context(), args, cmd.OutOrStdout()) - }, - } -} - -func newAutorouteAllowListCmd() *cobra.Command { - var jsonOut bool - cmd := &cobra.Command{ - Use: "list", - Short: "List auto-route allowed domains", - RunE: func(cmd *cobra.Command, args []string) error { - return runAutorouteAllowList(cmd.Context(), cmd.OutOrStdout(), jsonOut) - }, - } - cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") - return cmd -} - -func newAutorouteAllowRemoveCmd() *cobra.Command { - return &cobra.Command{ - Use: "remove ", - Short: "Remove an auto-route domain pattern", - Args: cobra.ExactArgs(1), - RunE: func(cmd *cobra.Command, args []string) error { - return runAutorouteAllowRemove(cmd.Context(), args, cmd.OutOrStdout()) - }, - } -} - -func runAutorouteAllowAdd(ctx context.Context, args []string, out io.Writer) error { - handle, err := resolveControlPlane(configPath) - if err != nil { - return err - } - defer handle.close() - if err := handle.plane.AddAutoRouteAllowedDomain(ctx, args[0]); err != nil { - return fmt.Errorf("failed to add auto-route allowed domain: %w", err) - } - return cliWriteLine(out, cliRenderSuccess("Allowed domain added")) -} - -func runAutorouteAllowList(ctx context.Context, out io.Writer, jsonOut bool) error { - handle, err := resolveControlPlane(configPath) - if err != nil { - return err - } - defer handle.close() - domains, err := handle.plane.GetAutoRouteAllowedDomains(ctx) - if err != nil { - return fmt.Errorf("failed to list auto-route allowed domains: %w", err) - } - if jsonOut { - return writeJSON(out, map[string]any{"domains": domains}) - } - if len(domains) == 0 { - return cliWriteLine(out, cliRenderMuted("No allowed domains configured")) - } - for _, domain := range domains { - if err := cliWriteLine(out, domain); err != nil { - return err - } - } - return nil -} - -func runAutorouteAllowRemove(ctx context.Context, args []string, out io.Writer) error { - handle, err := resolveControlPlane(configPath) - if err != nil { - return err - } - defer handle.close() - if err := handle.plane.RemoveAutoRouteAllowedDomain(ctx, args[0]); err != nil { - return fmt.Errorf("failed to remove auto-route allowed domain: %w", err) - } - return cliWriteLine(out, cliRenderSuccess("Allowed domain removed")) -} diff --git a/internal/adapters/in/cli/backup.go b/internal/adapters/in/cli/backup.go index 09a249343..f0d27db59 100644 --- a/internal/adapters/in/cli/backup.go +++ b/internal/adapters/in/cli/backup.go @@ -4,7 +4,6 @@ import ( "context" "fmt" "io" - "os" "text/tabwriter" "time" @@ -14,54 +13,35 @@ import ( "github.com/bnema/gordon/pkg/bytesize" ) -var backupResolveControlPlane = func(ctx context.Context, configPath, domainName string) (*controlPlaneHandle, error) { - if domainName != "" { - return resolveControlPlaneForRouteDomain(ctx, domainName) - } - return resolveControlPlane(configPath) -} - -// newBackupCmd creates the backup command group. +// newBackupCmd creates the backup command group. Backups are identified by +// app, service, and the declared database or volume: a domain is never an +// identity. func newBackupCmd() *cobra.Command { cmd := &cobra.Command{ Use: "backups", Aliases: []string{"backup"}, - Short: "Manage backups", - Long: `Manage backups. + Short: "Manage app backups", + Long: `Manage app backups. -Runs locally via in-process services by default, or against a remote Gordon -instance when --remote targeting is configured.`, +Backup targets come from the app manifest: a service declares its databases +and volumes, and which of them are backed up. Runs locally through the daemon's +authenticated Unix socket, or against a remote Gordon instance when --remote +targeting is configured.`, } - cmd.AddCommand(newBackupDatabasesCmd()) - cmd.AddCommand(newBackupVolumesCmd()) - - // Compatibility aliases for the original database backup command shape. cmd.AddCommand(newBackupListCmd()) cmd.AddCommand(newBackupRunCmd()) - cmd.AddCommand(newBackupDetectCmd()) cmd.AddCommand(newBackupStatusCmd()) + cmd.AddCommand(newBackupVolumeCmd()) return cmd } -func newBackupDatabasesCmd() *cobra.Command { - cmd := &cobra.Command{ - Use: "databases", - Aliases: []string{"database", "db"}, - Short: "Manage database backups", - } - cmd.AddCommand(newBackupListCmd()) - cmd.AddCommand(newBackupRunCmd()) - cmd.AddCommand(newBackupDetectCmd()) - cmd.AddCommand(newBackupStatusCmd()) - return cmd -} - -func newBackupVolumesCmd() *cobra.Command { +// newBackupVolumeCmd creates the `backup volume` group. +func newBackupVolumeCmd() *cobra.Command { cmd := &cobra.Command{ - Use: "volumes", - Short: "Manage volume backups", + Use: "volume", + Short: "Manage app volume backups", } cmd.AddCommand(newVolumeBackupListCmd()) cmd.AddCommand(newVolumeBackupRunCmd()) @@ -73,44 +53,36 @@ func newBackupListCmd() *cobra.Command { var jsonOut bool cmd := &cobra.Command{ - Use: "list [domain]", - Short: "List backups", - Long: cliRenderMuted("List backups for all domains or a specific domain."), + Use: "list [APP]", + Short: "List database backups", + Long: cliRenderMuted("List stored backups for all apps, or for one app."), Args: cobra.MaximumNArgs(1), RunE: func(cmd *cobra.Command, args []string) error { - domainName := "" + app := "" if len(args) == 1 { - domainName = args[0] - } - - var ( - handle *controlPlaneHandle - err error - ) - if domainName != "" { - handle, err = resolveControlPlaneForRouteDomain(cmd.Context(), domainName) - } else { - handle, err = resolveControlPlane(configPath) + app = args[0] } + handle, err := resolveControlPlane(cliConfigPath) if err != nil { return err } defer handle.close() - - jobs, err := handle.plane.ListBackups(cmd.Context(), domainName) - if err != nil { - return fmt.Errorf("failed to list backups: %w", err) - } - - return printBackupJobs(cmd.OutOrStdout(), jobs, jsonOut) + return runBackupList(cmd.Context(), handle.plane, app, cmd.OutOrStdout(), jsonOut) }, } cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") - return cmd } +func runBackupList(ctx context.Context, plane ControlPlane, app string, out io.Writer, jsonOut bool) error { + jobs, err := plane.ListBackups(ctx, app) + if err != nil { + return fmt.Errorf("failed to list backups: %w", err) + } + return printBackupJobs(out, jobs, jsonOut) +} + func printBackupJobs(out io.Writer, jobs []dto.BackupJob, jsonOut bool) error { if len(jobs) == 0 { if jsonOut { @@ -127,106 +99,192 @@ func printBackupJobs(out io.Writer, jobs []dto.BackupJob, jsonOut bool) error { return err } w := tabwriter.NewWriter(out, 0, 0, 2, ' ', 0) - if _, err := fmt.Fprintln(w, "DOMAIN\tDB\tSTATUS\tSTARTED_AT\tBACKUP_ID"); err != nil { + if _, err := fmt.Fprintln(w, "APP\tSERVICE\tDATABASE\tSTATUS\tSTARTED_AT\tBACKUP_ID"); err != nil { return err } - for _, job := range jobs { - if _, err := fmt.Fprintf(w, "%s\t%s\t%s\t%s\t%s\n", job.Domain, job.DBName, job.Status, formatBackupTime(job.StartedAt), job.ID); err != nil { + if _, err := fmt.Fprintf(w, "%s\t%s\t%s\t%s\t%s\t%s\n", + job.App, job.Service, job.Database, job.Status, formatBackupTime(job.StartedAt), job.ID); err != nil { return err } } return w.Flush() } -func newVolumeBackupListCmd() *cobra.Command { +func newBackupRunCmd() *cobra.Command { + var service string + var database string var jsonOut bool cmd := &cobra.Command{ - Use: "list [domain]", - Short: "List volume backups", - Args: cobra.MaximumNArgs(1), + Use: "run APP", + Short: "Run a database backup now", + Long: cliRenderMuted(`Run the declared database backup of one app service. + +--service and --database select the target. Omitting a selector is only +allowed when exactly one compatible target exists.`), + Args: cobra.ExactArgs(1), RunE: func(cmd *cobra.Command, args []string) error { - domainName := "" - if len(args) == 1 { - domainName = args[0] - } - handle, err := backupResolveControlPlane(cmd.Context(), configPath, domainName) + handle, err := resolveControlPlane(cliConfigPath) if err != nil { return err } defer handle.close() + return runBackupRun(cmd.Context(), handle.plane, args[0], service, database, cmd.OutOrStdout(), jsonOut) + }, + } + + cmd.Flags().StringVar(&service, "service", "", "Service that declares the database") + cmd.Flags().StringVar(&database, "database", "", "Declared database name (optional)") + cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") + return cmd +} + +func runBackupRun(ctx context.Context, plane ControlPlane, app, service, database string, out io.Writer, jsonOut bool) error { + result, err := plane.RunBackup(ctx, app, service, database) + if err != nil { + return fmt.Errorf("failed to run backup: %w", err) + } + if result.Backup == nil { + return fmt.Errorf("backup run completed without a backup payload") + } + if jsonOut { + return writeJSON(out, result) + } + return printBackupJobs(out, []dto.BackupJob{*result.Backup}, false) +} - jobs, err := handle.plane.ListVolumeBackups(cmd.Context(), domainName) +func newBackupStatusCmd() *cobra.Command { + var jsonOut bool + cmd := &cobra.Command{ + Use: "status", + Short: "Show database backup status", + Long: cliRenderMuted("Show stored backups plus declared targets that have no completed backup yet."), + RunE: func(cmd *cobra.Command, args []string) error { + handle, err := resolveControlPlane(cliConfigPath) if err != nil { - return fmt.Errorf("failed to list volume backups: %w", err) + return err } - return printVolumeBackupJobs(cmd.OutOrStdout(), jobs, jsonOut) + defer handle.close() + return runBackupStatus(cmd.Context(), handle.plane, cmd.OutOrStdout(), jsonOut) }, } cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") return cmd } -func newVolumeBackupRunCmd() *cobra.Command { - var volumeName string +func runBackupStatus(ctx context.Context, plane ControlPlane, out io.Writer, jsonOut bool) error { + jobs, err := plane.BackupStatus(ctx) + if err != nil { + return fmt.Errorf("failed to get backup status: %w", err) + } + return printBackupJobs(out, jobs, jsonOut) +} + +func newVolumeBackupListCmd() *cobra.Command { var jsonOut bool cmd := &cobra.Command{ - Use: "run [domain]", - Short: "Run volume backups now", + Use: "list [APP]", + Short: "List volume backups", Args: cobra.MaximumNArgs(1), RunE: func(cmd *cobra.Command, args []string) error { - domainName := "" + app := "" if len(args) == 1 { - domainName = args[0] + app = args[0] } - handle, err := backupResolveControlPlane(cmd.Context(), configPath, domainName) + handle, err := resolveControlPlane(cliConfigPath) if err != nil { return err } defer handle.close() + return runVolumeBackupList(cmd.Context(), handle.plane, app, cmd.OutOrStdout(), jsonOut) + }, + } + cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") + return cmd +} - result, err := handle.plane.RunVolumeBackups(cmd.Context(), domainName, volumeName) +func runVolumeBackupList(ctx context.Context, plane ControlPlane, app string, out io.Writer, jsonOut bool) error { + jobs, err := plane.ListVolumeBackups(ctx, app) + if err != nil { + return fmt.Errorf("failed to list volume backups: %w", err) + } + return printVolumeBackupJobs(out, jobs, jsonOut) +} + +func newVolumeBackupRunCmd() *cobra.Command { + var service string + var volume string + var jsonOut bool + + cmd := &cobra.Command{ + Use: "run APP", + Short: "Run a volume backup now", + Long: cliRenderMuted(`Run the declared volume backup of one app service. + +--service and --volume select the target. Omitting a selector is only +allowed when exactly one compatible target exists.`), + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + handle, err := resolveControlPlane(cliConfigPath) if err != nil { - if result != nil && len(result.Backups) > 0 { - if printErr := printVolumeBackupJobs(cmd.OutOrStdout(), result.Backups, jsonOut); printErr != nil { - return printErr - } - } - return fmt.Errorf("failed to run volume backups: %w", err) + return err } - return printVolumeBackupJobs(cmd.OutOrStdout(), result.Backups, jsonOut) + defer handle.close() + return runVolumeBackupRun(cmd.Context(), handle.plane, args[0], service, volume, cmd.OutOrStdout(), jsonOut) }, } - cmd.Flags().StringVar(&volumeName, "volume", "", "Volume name (optional)") + cmd.Flags().StringVar(&service, "service", "", "Service that declares the volume") + cmd.Flags().StringVar(&volume, "volume", "", "Declared volume name (optional)") cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") return cmd } +func runVolumeBackupRun(ctx context.Context, plane ControlPlane, app, service, volume string, out io.Writer, jsonOut bool) error { + result, err := plane.RunVolumeBackups(ctx, app, service, volume) + if err != nil { + // Partial results are reported before the error: an operator must + // see which volumes completed. + if result != nil && len(result.Backups) > 0 { + if printErr := printVolumeBackupJobs(out, result.Backups, jsonOut); printErr != nil { + return printErr + } + } + return fmt.Errorf("failed to run volume backup: %w", err) + } + if jsonOut { + return writeJSON(out, result) + } + return printVolumeBackupJobs(out, result.Backups, false) +} + func newVolumeBackupStatusCmd() *cobra.Command { var jsonOut bool cmd := &cobra.Command{ Use: "status", Short: "Show volume backup status", RunE: func(cmd *cobra.Command, args []string) error { - handle, err := backupResolveControlPlane(cmd.Context(), configPath, "") + handle, err := resolveControlPlane(cliConfigPath) if err != nil { return err } defer handle.close() - - jobs, err := handle.plane.VolumeBackupStatus(cmd.Context()) - if err != nil { - return fmt.Errorf("failed to get volume backup status: %w", err) - } - return printVolumeBackupJobs(cmd.OutOrStdout(), jobs, jsonOut) + return runVolumeBackupStatus(cmd.Context(), handle.plane, cmd.OutOrStdout(), jsonOut) }, } cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") return cmd } +func runVolumeBackupStatus(ctx context.Context, plane ControlPlane, out io.Writer, jsonOut bool) error { + jobs, err := plane.VolumeBackupStatus(ctx) + if err != nil { + return fmt.Errorf("failed to get volume backup status: %w", err) + } + return printVolumeBackupJobs(out, jobs, jsonOut) +} + func printVolumeBackupJobs(out io.Writer, jobs []dto.VolumeBackupJob, jsonOut bool) error { if len(jobs) == 0 { if jsonOut { @@ -241,150 +299,19 @@ func printVolumeBackupJobs(out io.Writer, jobs []dto.VolumeBackupJob, jsonOut bo return err } w := tabwriter.NewWriter(out, 0, 0, 2, ' ', 0) - if _, err := fmt.Fprintln(w, "DOMAIN\tVOLUME\tCONTAINER\tCOMPRESSION\tSTATUS\tSTARTED_AT\tSIZE\tARTIFACT"); err != nil { + if _, err := fmt.Fprintln(w, "APP\tSERVICE\tVOLUME\tCOMPRESSION\tSTATUS\tSTARTED_AT\tSIZE\tARTIFACT"); err != nil { return err } for _, job := range jobs { - if _, err := fmt.Fprintf(w, "%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\n", job.Domain, job.VolumeName, job.ContainerName, job.Compression, job.Status, formatBackupTime(job.StartedAt), bytesize.Format(job.SizeBytes), job.ArtifactRef); err != nil { + if _, err := fmt.Fprintf(w, "%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\n", + job.App, job.Service, job.VolumeName, job.Compression, job.Status, + formatBackupTime(job.StartedAt), bytesize.Format(job.SizeBytes), job.ArtifactRef); err != nil { return err } } return w.Flush() } -func newBackupRunCmd() *cobra.Command { - var dbName string - - cmd := &cobra.Command{ - Use: "run ", - Short: "Run backup now", - Args: cobra.ExactArgs(1), - RunE: func(cmd *cobra.Command, args []string) error { - handle, err := resolveControlPlaneForRouteDomain(cmd.Context(), args[0]) - if err != nil { - return err - } - defer handle.close() - - result, err := handle.plane.RunBackup(cmd.Context(), args[0], dbName) - if err != nil { - return fmt.Errorf("failed to run backup: %w", err) - } - - if result.Backup == nil { - return fmt.Errorf("backup run completed without backup payload") - } - if err := cliWriteLine(os.Stdout, cliRenderTitle("Backup Result")); err != nil { - return err - } - - w := tabwriter.NewWriter(os.Stdout, 0, 0, 2, ' ', 0) - if _, err := fmt.Fprintln(w, "DOMAIN\tDB\tSTATUS\tSTARTED_AT\tBACKUP_ID\tSIZE"); err != nil { - return err - } - if _, err := fmt.Fprintf(w, "%s\t%s\t%s\t%s\t%s\t%s\n", result.Backup.Domain, result.Backup.DBName, result.Backup.Status, formatBackupTime(result.Backup.StartedAt), result.Backup.ID, bytesize.Format(result.Backup.SizeBytes)); err != nil { - return err - } - if err := w.Flush(); err != nil { - return err - } - return nil - }, - } - - cmd.Flags().StringVar(&dbName, "db", "", "Database attachment name (optional)") - return cmd -} - -func newBackupDetectCmd() *cobra.Command { - return &cobra.Command{ - Use: "detect ", - Short: "Detect databases for domain", - Args: cobra.ExactArgs(1), - RunE: func(cmd *cobra.Command, args []string) error { - handle, err := resolveControlPlaneForRouteDomain(cmd.Context(), args[0]) - if err != nil { - return err - } - defer handle.close() - - dbs, err := handle.plane.DetectDatabases(cmd.Context(), args[0]) - if err != nil { - return fmt.Errorf("failed to detect databases: %w", err) - } - - if len(dbs) == 0 { - if err := cliWriteLine(os.Stdout, cliRenderMuted("No supported databases detected")); err != nil { - return err - } - return nil - } - if err := cliWriteLine(os.Stdout, cliRenderTitle("Detected Databases")); err != nil { - return err - } - w := tabwriter.NewWriter(os.Stdout, 0, 0, 2, ' ', 0) - if _, err := fmt.Fprintln(w, "NAME\tTYPE\tHOST\tPORT\tIMAGE"); err != nil { - return err - } - - for _, db := range dbs { - if _, err := fmt.Fprintf(w, "%s\t%s\t%s\t%d\t%s\n", db.Name, db.Type, db.Host, db.Port, db.ImageName); err != nil { - return err - } - } - if err := w.Flush(); err != nil { - return err - } - - return nil - }, - } -} - -func newBackupStatusCmd() *cobra.Command { - return &cobra.Command{ - Use: "status", - Short: "Show backup status", - RunE: func(cmd *cobra.Command, args []string) error { - handle, err := resolveControlPlane(configPath) - if err != nil { - return err - } - defer handle.close() - - jobs, err := handle.plane.BackupStatus(cmd.Context()) - if err != nil { - return fmt.Errorf("failed to get backup status: %w", err) - } - - if len(jobs) == 0 { - if err := cliWriteLine(os.Stdout, cliRenderMuted("No backup status available")); err != nil { - return err - } - return nil - } - if err := cliWriteLine(os.Stdout, cliRenderTitle("Backup Status")); err != nil { - return err - } - w := tabwriter.NewWriter(os.Stdout, 0, 0, 2, ' ', 0) - if _, err := fmt.Fprintln(w, "DOMAIN\tDB\tSTATUS\tSTARTED_AT"); err != nil { - return err - } - - for _, job := range jobs { - if _, err := fmt.Fprintf(w, "%s\t%s\t%s\t%s\n", job.Domain, job.DBName, job.Status, formatBackupTime(job.StartedAt)); err != nil { - return err - } - } - if err := w.Flush(); err != nil { - return err - } - - return nil - }, - } -} - func formatBackupTime(t *time.Time) string { if t == nil { return "" diff --git a/internal/adapters/in/cli/backup_volume_test.go b/internal/adapters/in/cli/backup_volume_test.go index 9e54c55db..deba73c3e 100644 --- a/internal/adapters/in/cli/backup_volume_test.go +++ b/internal/adapters/in/cli/backup_volume_test.go @@ -13,86 +13,36 @@ import ( climocks "github.com/bnema/gordon/internal/adapters/in/cli/mocks" ) -func withBackupControlPlane(t *testing.T, plane ControlPlane) { +// backupRunPlane builds the control plane double the runner functions +// take directly: no resolver indirection is needed. +func backupRunPlane(t *testing.T) *climocks.MockControlPlane { t.Helper() - old := backupResolveControlPlane - backupResolveControlPlane = func(context.Context, string, string) (*controlPlaneHandle, error) { - return &controlPlaneHandle{plane: plane}, nil - } - t.Cleanup(func() { backupResolveControlPlane = old }) -} - -func TestVolumeBackupRun_UsesDomainAwareResolver(t *testing.T) { - plane := climocks.NewMockControlPlane(t) - old := backupResolveControlPlane - var gotDomain string - backupResolveControlPlane = func(_ context.Context, _ string, domainName string) (*controlPlaneHandle, error) { - gotDomain = domainName - return &controlPlaneHandle{plane: plane}, nil - } - t.Cleanup(func() { backupResolveControlPlane = old }) - - plane.EXPECT().RunVolumeBackups(mock.Anything, "app.example.com", "").Return(&dto.VolumeBackupRunResponse{Status: "ok"}, nil) - - cmd := newVolumeBackupRunCmd() - cmd.SetArgs([]string{"app.example.com"}) - - require.NoError(t, cmd.ExecuteContext(context.Background())) - require.Equal(t, "app.example.com", gotDomain) -} - -func TestVolumeBackupRun_JSONFlag(t *testing.T) { - plane := climocks.NewMockControlPlane(t) - withBackupControlPlane(t, plane) - - plane.EXPECT().RunVolumeBackups(mock.Anything, "app.example.com", "gordon-app-data").Return(&dto.VolumeBackupRunResponse{ - Status: "ok", - Backups: []dto.VolumeBackupJob{{Domain: "app.example.com", VolumeName: "gordon-app-data", Status: "completed"}}, - }, nil) - - cmd := newVolumeBackupRunCmd() - var out bytes.Buffer - cmd.SetOut(&out) - cmd.SetArgs([]string{"app.example.com", "--volume", "gordon-app-data", "--json"}) - - require.NoError(t, cmd.ExecuteContext(context.Background())) - require.JSONEq(t, `[{"id":"","domain":"app.example.com","volume_name":"gordon-app-data","type":"","status":"completed","size_bytes":0}]`, out.String()) + return climocks.NewMockControlPlane(t) } func TestVolumeBackupStatus_JSONFlag(t *testing.T) { - plane := climocks.NewMockControlPlane(t) - withBackupControlPlane(t, plane) + plane := backupRunPlane(t) - plane.EXPECT().VolumeBackupStatus(mock.Anything).Return([]dto.VolumeBackupJob{{Domain: "app.example.com", VolumeName: "gordon-app-data", Status: "running"}}, nil) + plane.EXPECT().VolumeBackupStatus(mock.Anything).Return([]dto.VolumeBackupJob{{App: "shop", Service: "api", VolumeName: "data", Status: "running"}}, nil) - cmd := newVolumeBackupStatusCmd() var out bytes.Buffer - cmd.SetOut(&out) - cmd.SetArgs([]string{"--json"}) - - require.NoError(t, cmd.ExecuteContext(context.Background())) - require.JSONEq(t, `[{"id":"","domain":"app.example.com","volume_name":"gordon-app-data","type":"","status":"running","size_bytes":0}]`, out.String()) + require.NoError(t, runVolumeBackupStatus(context.Background(), plane, &out, true)) + require.JSONEq(t, `[{"id":"","app":"shop","service":"api","volume":"data","type":"","status":"running","size_bytes":0}]`, out.String()) } func TestVolumeBackupRun_PrintsPartialJobsBeforeReturningError(t *testing.T) { - plane := climocks.NewMockControlPlane(t) - withBackupControlPlane(t, plane) + plane := backupRunPlane(t) runErr := errors.New("one backup failed") - plane.EXPECT().RunVolumeBackups(mock.Anything, "app.example.com", "").Return(&dto.VolumeBackupRunResponse{ + plane.EXPECT().RunVolumeBackups(mock.Anything, "shop", "api", "data").Return(&dto.VolumeBackupRunResponse{ Status: "partial", - Backups: []dto.VolumeBackupJob{{Domain: "app.example.com", VolumeName: "gordon-app-data", Status: "completed"}}, + Backups: []dto.VolumeBackupJob{{App: "shop", Service: "api", VolumeName: "data", Status: "completed"}}, Error: runErr.Error(), }, runErr) - cmd := newVolumeBackupRunCmd() - cmd.SilenceUsage = true var out bytes.Buffer - cmd.SetOut(&out) - cmd.SetArgs([]string{"app.example.com", "--json"}) - - err := cmd.ExecuteContext(context.Background()) + err := runVolumeBackupRun(context.Background(), plane, "shop", "api", "data", &out, true) require.Error(t, err) require.ErrorIs(t, err, runErr) - require.JSONEq(t, `[{"id":"","domain":"app.example.com","volume_name":"gordon-app-data","type":"","status":"completed","size_bytes":0}]`, out.String()) + require.JSONEq(t, `[{"id":"","app":"shop","service":"api","volume":"data","type":"","status":"completed","size_bytes":0}]`, out.String()) } diff --git a/internal/adapters/in/cli/bootstrap.go b/internal/adapters/in/cli/bootstrap.go deleted file mode 100644 index 338fc8ad7..000000000 --- a/internal/adapters/in/cli/bootstrap.go +++ /dev/null @@ -1,136 +0,0 @@ -package cli - -import ( - "context" - "fmt" - "io" - "sort" - "strings" - - "github.com/spf13/cobra" - - "github.com/bnema/gordon/internal/adapters/dto" -) - -func newBootstrapCmd() *cobra.Command { - var attachments []string - var envPairs []string - var attachmentEnvPairs []string - var configPath string - - cmd := &cobra.Command{ - Use: "bootstrap ", - Short: "Create a route, attachments, and secrets together", - Args: cobra.ExactArgs(2), - RunE: func(cmd *cobra.Command, args []string) error { - req, err := parseBootstrapRequest(args, attachments, envPairs, attachmentEnvPairs) - if err != nil { - return err - } - return runBootstrap(cmd.Context(), req, configPath, cmd.OutOrStdout()) - }, - } - - cmd.Flags().StringVarP(&configPath, "config", "c", "", "Path to config file") - cmd.Flags().StringArrayVar(&attachments, "attachment", nil, "Attachment image (repeatable)") - cmd.Flags().StringArrayVar(&envPairs, "env", nil, "Environment variable KEY=VALUE (repeatable)") - cmd.Flags().StringArrayVar(&attachmentEnvPairs, "attachment-env", nil, "Attachment env service:KEY=VALUE (repeatable)") - - return cmd -} - -func parseBootstrapRequest(args []string, attachments, envPairs, attachmentEnvPairs []string) (dto.BootstrapRequest, error) { - req := dto.BootstrapRequest{ - Domain: args[0], - Image: args[1], - Attachments: append([]string(nil), attachments...), - Env: map[string]string{}, - AttachmentEnv: map[string]map[string]string{}, - } - - for _, pair := range envPairs { - parts := strings.SplitN(pair, "=", 2) - if len(parts) != 2 { - return dto.BootstrapRequest{}, fmt.Errorf("invalid --env value %q (expected KEY=VALUE)", pair) - } - req.Env[parts[0]] = parts[1] - } - - for _, pair := range attachmentEnvPairs { - serviceAndKV := strings.SplitN(pair, ":", 2) - if len(serviceAndKV) != 2 { - return dto.BootstrapRequest{}, fmt.Errorf("invalid --attachment-env value %q (expected service:KEY=VALUE)", pair) - } - keyValue := strings.SplitN(serviceAndKV[1], "=", 2) - if len(keyValue) != 2 { - return dto.BootstrapRequest{}, fmt.Errorf("invalid --attachment-env value %q (expected service:KEY=VALUE)", pair) - } - service := serviceAndKV[0] - if req.AttachmentEnv[service] == nil { - req.AttachmentEnv[service] = map[string]string{} - } - req.AttachmentEnv[service][keyValue[0]] = keyValue[1] - } - - if len(req.Env) == 0 { - req.Env = nil - } - if len(req.AttachmentEnv) == 0 { - req.AttachmentEnv = nil - } - - return req, nil -} - -func runBootstrap(ctx context.Context, req dto.BootstrapRequest, configPath string, out io.Writer) error { - handle, err := resolveControlPlane(configPath) - if err != nil { - return err - } - defer handle.close() - - resp, err := handle.plane.Bootstrap(ctx, req) - if resp != nil { - if writeErr := printBootstrapSummary(out, resp); writeErr != nil { - return writeErr - } - } - if err != nil { - return fmt.Errorf("failed to bootstrap: %w", err) - } - if err := cliWriteLine(out, cliRenderSuccess("Bootstrap complete")); err != nil { - return err - } - nextImage := req.Image - if resp != nil && resp.Image != "" { - nextImage = resp.Image - } - return cliWriteLine(out, fmt.Sprintf("Next: gordon push %s --build", nextImage)) -} - -func printBootstrapSummary(out io.Writer, resp *dto.BootstrapResponse) error { - if err := cliWriteLine(out, cliRenderTitle("Bootstrap")); err != nil { - return err - } - if err := cliWriteLine(out, cliRenderMeta("Domain:", resp.Domain)); err != nil { - return err - } - if err := cliWriteLine(out, cliRenderMeta("Image:", resp.Image)); err != nil { - return err - } - for _, step := range resp.Steps { - if err := cliWriteLine(out, fmt.Sprintf("- %s: %s", step.Name, step.Status)); err != nil { - return err - } - } - if len(resp.Warnings) > 0 { - sorted := append([]string(nil), resp.Warnings...) - sort.Strings(sorted) - for _, warning := range sorted { - if err := cliWriteLine(out, cliRenderWarning(warning)); err != nil { - return err - } - } - } - return nil -} diff --git a/internal/adapters/in/cli/ca.go b/internal/adapters/in/cli/ca.go index 1b216795d..273c156bc 100644 --- a/internal/adapters/in/cli/ca.go +++ b/internal/adapters/in/cli/ca.go @@ -96,7 +96,7 @@ func newCAInfoCmd() *cobra.Command { } func resolveCADataDir() (string, error) { - local, err := GetLocalServices(configPath) + local, err := GetLocalServices(cliConfigPath) if err != nil { return "", err } diff --git a/internal/adapters/in/cli/config.go b/internal/adapters/in/cli/config.go index 4c37bf3b6..b901d66cc 100644 --- a/internal/adapters/in/cli/config.go +++ b/internal/adapters/in/cli/config.go @@ -17,7 +17,7 @@ func newConfigCmd() *cobra.Command { Short: "Inspect Gordon configuration", } - cmd.AddCommand(newConfigShowCmd()) + cmd.AddCommand(newConfigShowCmd(), newConfigValidateCmd()) return cmd } @@ -32,12 +32,12 @@ func newConfigShowCmd() *cobra.Command { auto-route, network isolation, routes, and external routes. Examples: - gordon config show - gordon config show --json - gordon config show --remote https://gordon.mydomain.com --token $TOKEN`, + gordon daemon config show + gordon daemon config show --json + gordon daemon config show --remote prod`, RunE: func(cmd *cobra.Command, _ []string) error { ctx := cmd.Context() - handle, err := resolveControlPlane(configPath) + handle, err := resolveControlPlane(cliConfigPath) if err != nil { return err } @@ -72,9 +72,6 @@ func renderConfigTable(out io.Writer, config *remote.Config) error { if err := renderConfigSummary(out, config); err != nil { return err } - if err := renderConfigRoutes(out, config); err != nil { - return err - } return renderConfigExternalRoutes(out, config) } @@ -99,9 +96,6 @@ func renderConfigSummary(out io.Writer, config *remote.Config) error { return err } } - if err := cliWriteLine(out, cliRenderMeta("Auto-Route:", fmt.Sprintf("%v", config.AutoRoute.Enabled))); err != nil { - return err - } if err := cliWriteLine(out, cliRenderMeta("Network Isolation:", fmt.Sprintf("%v", config.NetworkIsolation.Enabled))); err != nil { return err } @@ -113,36 +107,6 @@ func renderConfigSummary(out io.Writer, config *remote.Config) error { return nil } -func renderConfigRoutes(out io.Writer, config *remote.Config) error { - if err := cliWriteLine(out, ""); err != nil { - return err - } - if len(config.Routes) == 0 { - if err := cliWriteLine(out, cliRenderMuted("No routes configured")); err != nil { - return err - } - } else { - if err := cliWriteLine(out, cliRenderTitle("Routes")); err != nil { - return err - } - rows := make([][]string, 0, len(config.Routes)) - for _, route := range config.Routes { - rows = append(rows, []string{route.Domain, route.Image}) - } - table := components.NewTable( - components.WithColumns([]components.TableColumn{ - {Title: "Domain", Width: 30}, - {Title: "Image", Width: 45}, - }), - components.WithRows(rows), - ) - if err := cliWriteLine(out, table.View()); err != nil { - return err - } - } - return nil -} - func renderConfigExternalRoutes(out io.Writer, config *remote.Config) error { if err := cliWriteLine(out, ""); err != nil { return err diff --git a/internal/adapters/in/cli/config_validate.go b/internal/adapters/in/cli/config_validate.go new file mode 100644 index 000000000..08f1bef8f --- /dev/null +++ b/internal/adapters/in/cli/config_validate.go @@ -0,0 +1,60 @@ +package cli + +import ( + "fmt" + "io" + + "github.com/spf13/cobra" + + "github.com/bnema/gordon/internal/app" +) + +type configDiagnostic struct { + Code string `json:"code"` + Key string `json:"key"` + Message string `json:"message"` +} + +type configValidationResult struct { + Valid bool `json:"valid"` + Diagnostics []configDiagnostic `json:"diagnostics"` + Scope string `json:"scope"` +} + +func newConfigValidateCmd() *cobra.Command { + var file string + var jsonOut bool + cmd := &cobra.Command{ + Use: "validate", + Short: "Statically validate a local configuration file", + Args: cobra.NoArgs, + RunE: func(cmd *cobra.Command, _ []string) error { + if remoteFlag != "" { + return fmt.Errorf("daemon config validate is local-only; --remote is not supported") + } + return runConfigValidate(cmd.OutOrStdout(), file, jsonOut) + }, + } + cmd.Flags().StringVar(&file, "file", "", "Local candidate configuration file (required)") + cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") + _ = cmd.MarkFlagRequired("file") + return cmd +} + +func runConfigValidate(out io.Writer, file string, jsonOut bool) error { + result := configValidationResult{Valid: true, Diagnostics: []configDiagnostic{}, Scope: "static"} + if err := app.ValidateConfigFile(file); err != nil { + result.Valid = false + result.Diagnostics = []configDiagnostic{{Code: "config-invalid", Message: "configuration failed static validation"}} + if jsonOut { + if writeErr := writeJSON(out, result); writeErr != nil { + return writeErr + } + } + return fmt.Errorf("configuration failed static validation: %w", err) + } + if jsonOut { + return writeJSON(out, result) + } + return cliWriteLine(out, cliRenderSuccess("Configuration is statically valid; runtime, ACTIVE-state, secret, pull, and listener checks were not performed.")) +} diff --git a/internal/adapters/in/cli/controlplane.go b/internal/adapters/in/cli/controlplane.go index b8596c083..3d0c0ff7b 100644 --- a/internal/adapters/in/cli/controlplane.go +++ b/internal/adapters/in/cli/controlplane.go @@ -9,35 +9,24 @@ import ( ) // ControlPlane defines command operations available to CLI execution paths. -// -// Remote implementations call admin HTTP APIs. -// Local implementations call services directly in-process. +// *remote.Client implements it for both the explicit remote and the +// owner-only local admin socket; tests substitute the generated mock. type ControlPlane interface { - ListRoutesWithDetails(ctx context.Context) ([]remote.RouteInfo, error) - GetHealth(ctx context.Context) (map[string]*remote.RouteHealth, error) - GetRoute(ctx context.Context, routeDomain string) (*domain.Route, error) - FindRoutesByImage(ctx context.Context, imageName string) ([]domain.Route, error) - AddRoute(ctx context.Context, route domain.Route) error - UpdateRoute(ctx context.Context, route domain.Route) error - RemoveRoute(ctx context.Context, routeDomain string) error - Bootstrap(ctx context.Context, req dto.BootstrapRequest) (*dto.BootstrapResponse, error) - - ListSecretsWithAttachments(ctx context.Context, secretDomain string) (*remote.SecretsListResult, error) - SetSecrets(ctx context.Context, secretDomain string, secrets map[string]string) error - DeleteSecret(ctx context.Context, secretDomain, key string) error - SetAttachmentSecrets(ctx context.Context, domain, service string, secrets map[string]string) error - DeleteAttachmentSecret(ctx context.Context, domain, service, key string) error - - GetAllAttachmentsConfig(ctx context.Context) (map[string][]string, error) - GetAttachmentsConfig(ctx context.Context, domainOrGroup string) ([]string, error) - ListOrphanedAttachments(ctx context.Context) ([]domain.CleanupAttachment, error) - CleanupOrphanedAttachments(ctx context.Context, owner string, stop bool) (*domain.CleanupReport, error) - FindAttachmentTargetsByImage(ctx context.Context, imageName string) ([]string, error) - AddAttachment(ctx context.Context, domainOrGroup, image string) error - RemoveAttachment(ctx context.Context, domainOrGroup, image string) error - GetAutoRouteAllowedDomains(ctx context.Context) ([]string, error) - AddAutoRouteAllowedDomain(ctx context.Context, pattern string) error - RemoveAutoRouteAllowedDomain(ctx context.Context, pattern string) error + // App lifecycle and reads. App mutations are daemon-owned for both + // the explicit remote and the owner-only local admin socket. + ApplyApp(ctx context.Context, req dto.AppApplyRequest) (*dto.AppApplyResponse, error) + ListApps(ctx context.Context) ([]dto.AppSummaryDTO, error) + ShowApp(ctx context.Context, app string) (*dto.AppShowResponse, error) + DiffApp(ctx context.Context, app string) (*dto.AppDiffResponse, error) + DeployApp(ctx context.Context, app string, req dto.AppDeployRequest) (*dto.AppDeployResponse, string, error) + StopApp(ctx context.Context, app string) (*dto.AppDeployResponse, string, error) + StartApp(ctx context.Context, app string) (*dto.AppDeployResponse, string, error) + RestartApp(ctx context.Context, app, service string, all bool) (*dto.AppDeployResponse, string, error) + RemoveApp(ctx context.Context, app string) (*dto.AppDeployResponse, string, error) + OperationByKey(ctx context.Context, app, key string) (*dto.AppDeployResponse, error) + ListAppSecrets(ctx context.Context, app, service string) ([]dto.AppSecretMetadataDTO, error) + SetAppSecrets(ctx context.Context, app string, req dto.AppSecretSetRequest) error + DeleteAppSecret(ctx context.Context, app string, req dto.AppSecretDeleteRequest) error GetStatus(ctx context.Context) (*remote.Status, error) GetTLSStatus(ctx context.Context) (*dto.TLSStatusResponse, error) @@ -45,24 +34,20 @@ type ControlPlane interface { Reload(ctx context.Context) error ListNetworks(ctx context.Context) ([]*domain.NetworkInfo, error) GetConfig(ctx context.Context) (*remote.Config, error) - DeployIntent(ctx context.Context, imageName string) error - Deploy(ctx context.Context, deployDomain string) (*remote.DeployResult, error) - Restart(ctx context.Context, restartDomain string, withAttachments bool) (*remote.RestartResult, error) ListTags(ctx context.Context, repository string) ([]string, error) - ListBackups(ctx context.Context, backupDomain string) ([]dto.BackupJob, error) + ListBackups(ctx context.Context, app string) ([]dto.BackupJob, error) BackupStatus(ctx context.Context) ([]dto.BackupJob, error) - RunBackup(ctx context.Context, backupDomain, dbName string) (*dto.BackupRunResponse, error) - DetectDatabases(ctx context.Context, backupDomain string) ([]dto.DatabaseInfo, error) - ListVolumeBackups(ctx context.Context, backupDomain string) ([]dto.VolumeBackupJob, error) + RunBackup(ctx context.Context, app, service, database string) (*dto.BackupRunResponse, error) + ListVolumeBackups(ctx context.Context, app string) ([]dto.VolumeBackupJob, error) VolumeBackupStatus(ctx context.Context) ([]dto.VolumeBackupJob, error) - RunVolumeBackups(ctx context.Context, backupDomain, volumeName string) (*dto.VolumeBackupRunResponse, error) + RunVolumeBackups(ctx context.Context, app, service, volume string) (*dto.VolumeBackupRunResponse, error) GetProcessLogs(ctx context.Context, lines int) ([]string, error) - GetContainerLogs(ctx context.Context, logDomain string, lines int) ([]string, error) StreamProcessLogs(ctx context.Context, lines int) (<-chan string, error) - StreamContainerLogs(ctx context.Context, logDomain string, lines int) (<-chan string, error) ListVolumes(ctx context.Context) ([]dto.Volume, error) PruneVolumes(ctx context.Context, req dto.VolumePruneRequest) (*dto.VolumePruneResponse, error) } + +var _ ControlPlane = (*remote.Client)(nil) diff --git a/internal/adapters/in/cli/controlplane_remote_test.go b/internal/adapters/in/cli/controlplane_client_test.go similarity index 62% rename from internal/adapters/in/cli/controlplane_remote_test.go rename to internal/adapters/in/cli/controlplane_client_test.go index 68e47be7d..fa2dac8e6 100644 --- a/internal/adapters/in/cli/controlplane_remote_test.go +++ b/internal/adapters/in/cli/controlplane_client_test.go @@ -15,49 +15,39 @@ import ( "github.com/bnema/gordon/internal/adapters/in/cli/remote" ) -var _ ControlPlane = (*remoteControlPlane)(nil) - -func TestRemoteControlPlane_ImplementsInterface(t *testing.T) { - client := remote.NewClient("https://gordon.example.com") - if NewRemoteControlPlane(client) == nil { - t.Fatal("expected non-nil remote control-plane") - } -} - func TestRemoteControlPlane_RunVolumeBackupsPreservesPartialResult(t *testing.T) { server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { w.Header().Set("Content-Type", "application/json") w.WriteHeader(http.StatusPartialContent) require.NoError(t, json.NewEncoder(w).Encode(dto.VolumeBackupRunResponse{ Status: "partial", - Backups: []dto.VolumeBackupJob{{ID: "v1", Domain: "app.example.com", VolumeName: "gordon-app-data"}}, + Backups: []dto.VolumeBackupJob{{ID: "v1", App: "shop", Service: "api", VolumeName: "data"}}, Error: "one volume failed", })) })) t.Cleanup(server.Close) - cp := NewRemoteControlPlane(remote.NewClient(server.URL)) - result, err := cp.RunVolumeBackups(context.Background(), "app.example.com", "") + cp := remote.NewClient(server.URL) + result, err := cp.RunVolumeBackups(context.Background(), "shop", "api", "data") require.Error(t, err) - assert.Contains(t, err.Error(), "run volume backups") + assert.Contains(t, err.Error(), "one volume failed") require.NotNil(t, result) assert.Equal(t, "partial", result.Status) require.Len(t, result.Backups, 1) assert.Equal(t, "v1", result.Backups[0].ID) } -func TestRemoteControlPlane_VolumeBackupErrorsAreWrapped(t *testing.T) { +func TestRemoteControlPlane_VolumeBackupErrorsKeepHTTPStatus(t *testing.T) { server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { http.Error(w, "boom", http.StatusBadGateway) })) t.Cleanup(server.Close) - cp := NewRemoteControlPlane(remote.NewClient(server.URL)) + cp := remote.NewClient(server.URL) _, err := cp.VolumeBackupStatus(context.Background()) require.Error(t, err) - assert.Contains(t, err.Error(), "get volume backup status") var httpErr *remote.HTTPError require.True(t, errors.As(err, &httpErr)) assert.Equal(t, http.StatusBadGateway, httpErr.StatusCode) diff --git a/internal/adapters/in/cli/controlplane_local.go b/internal/adapters/in/cli/controlplane_local.go deleted file mode 100644 index deaab7e3f..000000000 --- a/internal/adapters/in/cli/controlplane_local.go +++ /dev/null @@ -1,872 +0,0 @@ -package cli - -import ( - "context" - "errors" - "fmt" - "sort" - "time" - - "github.com/bnema/gordon/internal/adapters/dto" - "github.com/bnema/gordon/internal/adapters/in/cli/remote" - "github.com/bnema/gordon/internal/app" - "github.com/bnema/gordon/internal/boundaries/in" - "github.com/bnema/gordon/internal/domain" - "github.com/bnema/gordon/internal/usecase/config" -) - -type localControlPlane struct { - configSvc in.ConfigService - secretSvc in.SecretService - containerSvc in.ContainerService - backupSvc in.BackupService - volumeBackupSvc in.VolumeBackupService - registrySvc in.RegistryService - deployCoord in.DeployCoordinator - healthSvc in.HealthService - logSvc in.LogService - volumeSvc in.VolumeService - publicTLSSvc in.PublicTLSService -} - -func NewLocalControlPlane(kernel *app.Kernel) ControlPlane { - if kernel == nil { - return &localControlPlane{} - } - - registrySvc := kernel.Registry() - var deployCoord in.DeployCoordinator - if registrySvc != nil { - if coordinator, ok := any(registrySvc).(in.DeployCoordinator); ok { - deployCoord = coordinator - } - } - - return &localControlPlane{ - configSvc: kernel.Config(), - secretSvc: kernel.Secrets(), - containerSvc: kernel.Container(), - backupSvc: kernel.Backup(), - volumeBackupSvc: kernel.VolumeBackup(), - registrySvc: registrySvc, - deployCoord: deployCoord, - healthSvc: kernel.Health(), - logSvc: kernel.Logs(), - volumeSvc: kernel.Volumes(), - publicTLSSvc: kernel.PublicTLS(), - } -} - -func (l *localControlPlane) ListRoutesWithDetails(ctx context.Context) ([]remote.RouteInfo, error) { - if l.containerSvc != nil { - if err := l.containerSvc.SyncContainers(ctx); err != nil { - return nil, err - } - detailed := l.containerSvc.ListRoutesWithDetails(ctx) - if l.configSvc == nil { - return toRemoteRouteInfos(detailed), nil - } - return mergeConfiguredRemoteRouteInfos(l.configSvc.GetRoutes(ctx), detailed), nil - } - - if l.configSvc == nil { - return nil, fmt.Errorf("local config service unavailable") - } - - routes := l.configSvc.GetRoutes(ctx) - infos := make([]remote.RouteInfo, 0, len(routes)) - for _, route := range routes { - infos = append(infos, remote.RouteInfo{Domain: route.Domain, Image: route.Image}) - } - return infos, nil -} - -func (l *localControlPlane) GetHealth(ctx context.Context) (map[string]*remote.RouteHealth, error) { - if l.healthSvc == nil { - return map[string]*remote.RouteHealth{}, nil - } - - health := l.healthSvc.CheckAllRoutes(ctx) - result := make(map[string]*remote.RouteHealth, len(health)) - for domainName, h := range health { - if h == nil { - continue - } - result[domainName] = &remote.RouteHealth{ - ContainerStatus: h.ContainerStatus, - HTTPStatus: h.HTTPStatus, - ResponseTimeMs: h.ResponseTimeMs, - Healthy: h.Healthy, - Error: h.Error, - } - } - return result, nil -} - -func (l *localControlPlane) GetRoute(ctx context.Context, routeDomain string) (*domain.Route, error) { - if l.configSvc == nil { - return nil, fmt.Errorf("local config service unavailable") - } - return l.configSvc.GetRoute(ctx, routeDomain) -} - -func (l *localControlPlane) FindRoutesByImage(ctx context.Context, imageName string) ([]domain.Route, error) { - if l.configSvc == nil { - return nil, fmt.Errorf("local config service unavailable") - } - return l.configSvc.FindRoutesByImage(ctx, imageName), nil -} - -func (l *localControlPlane) AddRoute(ctx context.Context, route domain.Route) error { - if l.configSvc == nil { - return fmt.Errorf("local config service unavailable") - } - return l.configSvc.AddRoute(ctx, route) -} - -func (l *localControlPlane) UpdateRoute(ctx context.Context, route domain.Route) error { - if l.configSvc == nil { - return fmt.Errorf("local config service unavailable") - } - return l.configSvc.UpdateRoute(ctx, route) -} - -func (l *localControlPlane) RemoveRoute(ctx context.Context, routeDomain string) error { - _, err := l.RemoveRouteWithCleanup(ctx, routeDomain) - return err -} - -func (l *localControlPlane) RemoveRouteWithCleanup(ctx context.Context, routeDomain string) (*dto.RouteDeleteResponse, error) { - if l.configSvc == nil { - return nil, fmt.Errorf("local config service unavailable") - } - if err := l.configSvc.RemoveRoute(ctx, routeDomain); err != nil && !errors.Is(err, domain.ErrRouteNotFound) { - return nil, fmt.Errorf("remove route: %w", err) - } - resp := &dto.RouteDeleteResponse{Status: "removed"} - if l.containerSvc == nil { - return resp, nil - } - report, err := l.containerSvc.ReconcileRemovedRoute(ctx, routeDomain) - if err != nil { - return nil, fmt.Errorf("failed to cleanup removed route runtime state: %w", err) - } - resp.Cleanup = dto.CleanupReportFromDomain(report) - return resp, nil -} - -func (l *localControlPlane) GetRouteCleanupPreview(ctx context.Context, routeDomain string) (*domain.CleanupReport, error) { - previewer, ok := any(l.containerSvc).(interface { - PreviewRemovedRouteCleanup(context.Context, string) (*domain.CleanupReport, error) - }) - if !ok { - return nil, fmt.Errorf("route cleanup preview unavailable") - } - - report, err := previewer.PreviewRemovedRouteCleanup(ctx, routeDomain) - if err != nil { - return nil, err - } - if report == nil { - return nil, nil - } - if l.volumeSvc == nil { - return report, nil - } - - volumes, err := l.ListVolumes(ctx) - if err != nil { - return nil, fmt.Errorf("list volumes: %w", err) - } - if len(volumes) == 0 { - return report, nil - } - - scope := newRouteResourceScope(routeDomain, l.routeCleanupVolumePrefix(), nil, report.PreservedAttachments) - for _, volume := range volumesForRouteDiagnosis(volumes, scope) { - report.PreservedVolumes = append(report.PreservedVolumes, domain.CleanupVolume{ - Name: volume.Name, - Reason: "volume preserved for explicit cleanup review", - }) - } - return report, nil -} - -func (l *localControlPlane) routeCleanupVolumePrefix() string { - const defaultPrefix = "gordon" - if l.configSvc == nil { - return defaultPrefix - } - if volumeCfg, ok := any(l.configSvc).(interface{ GetVolumeConfig() (bool, string, bool) }); ok { - _, prefix, _ := volumeCfg.GetVolumeConfig() - if prefix != "" { - return prefix - } - } - return defaultPrefix -} - -func (l *localControlPlane) Bootstrap(ctx context.Context, req dto.BootstrapRequest) (*dto.BootstrapResponse, error) { - registryDomain := "" - if l.configSvc != nil { - registryDomain = l.configSvc.GetRegistryDomain() - } - normalizedImage, err := config.NormalizeBootstrapImage(req.Image, registryDomain) - if err != nil { - return nil, fmt.Errorf("invalid image: %w", err) - } - - resp := &dto.BootstrapResponse{ - Domain: req.Domain, - Image: normalizedImage, - Next: fmt.Sprintf("push %s to trigger deployment", normalizedImage), - } - - if l.configSvc == nil { - return resp, fmt.Errorf("local config service unavailable") - } - needsSecrets := len(req.Env) > 0 || len(req.AttachmentEnv) > 0 - if needsSecrets && l.secretSvc == nil { - return resp, fmt.Errorf("local secret service unavailable") - } - - addStep := func(name, status string) { - resp.Steps = append(resp.Steps, dto.BootstrapStep{Name: name, Status: status}) - } - - err = l.configSvc.AddRoute(ctx, domain.Route{Domain: req.Domain, Image: normalizedImage}) - switch err { - case nil: - addStep("route", "configured") - default: - addStep("route", "failed") - return resp, err - } - - if err := l.bootstrapAttachments(ctx, req, addStep); err != nil { - return resp, err - } - - if err := l.bootstrapSecrets(ctx, req, addStep); err != nil { - return resp, err - } - - if err := l.bootstrapAttachmentSecrets(ctx, req, addStep); err != nil { - return resp, err - } - - return resp, nil -} - -func (l *localControlPlane) bootstrapAttachments(ctx context.Context, req dto.BootstrapRequest, addStep func(name, status string)) error { - for _, attachment := range req.Attachments { - err := l.configSvc.AddAttachment(ctx, req.Domain, attachment) - if err == nil { - addStep("attachment:"+attachment, "created") - continue - } - if errors.Is(err, domain.ErrAttachmentExists) { - addStep("attachment:"+attachment, "noop") - continue - } - addStep("attachment:"+attachment, "failed") - return err - } - - return nil -} - -func (l *localControlPlane) bootstrapSecrets(ctx context.Context, req dto.BootstrapRequest, addStep func(name, status string)) error { - if len(req.Env) == 0 { - return nil - } - - if err := l.secretSvc.Set(ctx, req.Domain, req.Env); err != nil { - addStep("env", "failed") - return err - } - - addStep("env", "updated") - return nil -} - -func (l *localControlPlane) bootstrapAttachmentSecrets(ctx context.Context, req dto.BootstrapRequest, addStep func(name, status string)) error { - for service, env := range req.AttachmentEnv { - if err := l.secretSvc.SetAttachment(ctx, req.Domain, service, env); err != nil { - addStep("attachment_env:"+service, "failed") - return err - } - addStep("attachment_env:"+service, "updated") - } - - return nil -} - -func (l *localControlPlane) ListSecretsWithAttachments(ctx context.Context, secretDomain string) (*remote.SecretsListResult, error) { - if l.secretSvc == nil { - return nil, fmt.Errorf("local secret service unavailable") - } - - keys, attachments, err := l.secretSvc.ListKeysWithAttachments(ctx, secretDomain) - if err != nil { - return nil, fmt.Errorf("list secret keys with attachments: %w", err) - } - - result := &remote.SecretsListResult{ - Domain: secretDomain, - Keys: keys, - } - for _, att := range attachments { - result.Attachments = append(result.Attachments, remote.AttachmentSecrets{ - Service: att.Service, - Keys: att.Keys, - }) - } - - return result, nil -} - -func (l *localControlPlane) SetSecrets(ctx context.Context, secretDomain string, secrets map[string]string) error { - if l.secretSvc == nil { - return fmt.Errorf("local secret service unavailable") - } - return l.secretSvc.Set(ctx, secretDomain, secrets) -} - -func (l *localControlPlane) DeleteSecret(ctx context.Context, secretDomain, key string) error { - if l.secretSvc == nil { - return fmt.Errorf("local secret service unavailable") - } - return l.secretSvc.Delete(ctx, secretDomain, key) -} - -func (l *localControlPlane) SetAttachmentSecrets(ctx context.Context, domainName, service string, secrets map[string]string) error { - if l.secretSvc == nil { - return fmt.Errorf("local secret service unavailable") - } - return l.secretSvc.SetAttachment(ctx, domainName, service, secrets) -} - -func (l *localControlPlane) DeleteAttachmentSecret(ctx context.Context, domainName, service, key string) error { - if l.secretSvc == nil { - return fmt.Errorf("local secret service unavailable") - } - return l.secretSvc.DeleteAttachment(ctx, domainName, service, key) -} - -func (l *localControlPlane) ListOrphanedAttachments(ctx context.Context) ([]domain.CleanupAttachment, error) { - if l.containerSvc == nil { - return nil, fmt.Errorf("local container service unavailable") - } - attachments, err := l.containerSvc.ListOrphanedAttachments(ctx) - if err != nil { - return nil, fmt.Errorf("list orphaned attachments: %w", err) - } - return attachments, nil -} - -func (l *localControlPlane) CleanupOrphanedAttachments(ctx context.Context, owner string, stop bool) (*domain.CleanupReport, error) { - if l.containerSvc == nil { - return nil, fmt.Errorf("local container service unavailable") - } - report, err := l.containerSvc.CleanupOrphanedAttachments(ctx, owner, stop) - if err != nil { - return nil, fmt.Errorf("cleanup orphaned attachments: %w", err) - } - return report, nil -} - -func (l *localControlPlane) GetAllAttachmentsConfig(ctx context.Context) (map[string][]string, error) { - if l.configSvc == nil { - return nil, fmt.Errorf("local config service unavailable") - } - return l.configSvc.GetAllAttachments(ctx), nil -} - -func (l *localControlPlane) GetAttachmentsConfig(ctx context.Context, domainOrGroup string) ([]string, error) { - if l.configSvc == nil { - return nil, fmt.Errorf("local config service unavailable") - } - return l.configSvc.GetAttachmentsFor(ctx, domainOrGroup) -} - -func (l *localControlPlane) FindAttachmentTargetsByImage(ctx context.Context, imageName string) ([]string, error) { - if l.configSvc == nil { - return nil, fmt.Errorf("local config service unavailable") - } - return l.configSvc.FindAttachmentTargetsByImage(ctx, imageName), nil -} - -func (l *localControlPlane) AddAttachment(ctx context.Context, domainOrGroup, image string) error { - if l.configSvc == nil { - return fmt.Errorf("local config service unavailable") - } - return l.configSvc.AddAttachment(ctx, domainOrGroup, image) -} - -func (l *localControlPlane) RemoveAttachment(ctx context.Context, domainOrGroup, image string) error { - if l.configSvc == nil { - return fmt.Errorf("local config service unavailable") - } - return l.configSvc.RemoveAttachment(ctx, domainOrGroup, image) -} - -func (l *localControlPlane) GetAutoRouteAllowedDomains(ctx context.Context) ([]string, error) { - if l.configSvc == nil { - return nil, fmt.Errorf("local config service unavailable") - } - return l.configSvc.GetAutoRouteAllowedDomains(ctx) -} - -func (l *localControlPlane) AddAutoRouteAllowedDomain(ctx context.Context, pattern string) error { - if l.configSvc == nil { - return fmt.Errorf("local config service unavailable") - } - return l.configSvc.AddAutoRouteAllowedDomain(ctx, pattern) -} - -func (l *localControlPlane) RemoveAutoRouteAllowedDomain(ctx context.Context, pattern string) error { - if l.configSvc == nil { - return fmt.Errorf("local config service unavailable") - } - return l.configSvc.RemoveAutoRouteAllowedDomain(ctx, pattern) -} - -func (l *localControlPlane) GetTLSStatus(ctx context.Context) (*dto.TLSStatusResponse, error) { - if l.publicTLSSvc == nil { - return &dto.TLSStatusResponse{ - ACMEEnabled: false, - SelectionReason: "public TLS service not configured", - }, nil - } - - status := l.publicTLSSvc.Status(ctx) - result := dto.TLSStatusFromDomain(status) - return &result, nil -} - -func (l *localControlPlane) GetTrafficStatus(_ context.Context) (*dto.TrafficStatusResponse, error) { - return nil, fmt.Errorf("local traffic status is unavailable from the in-process CLI control plane; query the running Gordon daemon with --remote or set GORDON_REMOTE to its admin URL: %w", domain.ErrTrafficStatusUnavailable) -} - -func (l *localControlPlane) GetStatus(ctx context.Context) (*remote.Status, error) { - if l.configSvc == nil { - return nil, fmt.Errorf("local config service unavailable") - } - - status := &remote.Status{ - Routes: len(l.configSvc.GetRoutes(ctx)), - RegistryDomain: l.configSvc.GetRegistryDomain(), - RegistryPort: l.configSvc.GetRegistryPort(), - ServerPort: l.configSvc.GetServerPort(), - AutoRoute: l.configSvc.IsAutoRouteEnabled(), - NetworkIsolation: l.configSvc.IsNetworkIsolationEnabled(), - ContainerStatus: map[string]string{}, - } - - if l.containerSvc != nil { - for domainName, container := range l.containerSvc.List(ctx) { - if container == nil { - continue - } - status.ContainerStatus[domainName] = container.Status - } - } - - return status, nil -} - -func (l *localControlPlane) Reload(_ context.Context) error { - return app.SendReloadSignal() -} - -func (l *localControlPlane) ListNetworks(ctx context.Context) ([]*domain.NetworkInfo, error) { - if l.containerSvc == nil { - return nil, fmt.Errorf("container service unavailable") - } - return l.containerSvc.ListNetworks(ctx) -} - -func (l *localControlPlane) GetConfig(ctx context.Context) (*remote.Config, error) { - if l.configSvc == nil { - return nil, fmt.Errorf("config service unavailable") - } - externalRoutes := l.configSvc.GetExternalRoutes() - externalResponses := make([]remote.ExternalRoute, 0, len(externalRoutes)) - for domainName := range externalRoutes { - externalResponses = append(externalResponses, remote.ExternalRoute{Domain: domainName}) - } - sort.Slice(externalResponses, func(i, j int) bool { - return externalResponses[i].Domain < externalResponses[j].Domain - }) - cfg := &remote.Config{ - Routes: l.configSvc.GetRoutes(ctx), - ExternalRoutes: externalResponses, - } - cfg.Server.Port = l.configSvc.GetServerPort() - cfg.Server.RegistryPort = l.configSvc.GetRegistryPort() - cfg.Server.RegistryDomain = l.configSvc.GetRegistryDomain() - cfg.AutoRoute.Enabled = l.configSvc.IsAutoRouteEnabled() - cfg.NetworkIsolation.Enabled = l.configSvc.IsNetworkIsolationEnabled() - cfg.NetworkIsolation.Prefix = l.configSvc.GetNetworkPrefix() - if volumeCfg, ok := any(l.configSvc).(interface{ GetVolumeConfig() (bool, string, bool) }); ok { - cfg.Volumes.AutoCreate, cfg.Volumes.Prefix, cfg.Volumes.Preserve = volumeCfg.GetVolumeConfig() - } - return cfg, nil -} - -func (l *localControlPlane) DeployIntent(_ context.Context, imageName string) error { - if l.deployCoord != nil { - l.deployCoord.SuppressDeployEvent(imageName) - } - return nil -} - -func (l *localControlPlane) Deploy(ctx context.Context, deployDomain string) (*remote.DeployResult, error) { - if l.containerSvc != nil && l.configSvc != nil { - route, err := l.configSvc.GetRoute(ctx, deployDomain) - if err != nil { - return nil, err - } - container, err := l.containerSvc.Deploy(domain.WithInternalDeploy(ctx), *route) - if err != nil { - return nil, err - } - result := &remote.DeployResult{Status: "deployed", Domain: deployDomain} - if container != nil { - result.ContainerID = container.ID - } - return result, nil - } - - domainName, err := app.SendDeploySignal(deployDomain) - if err != nil { - return nil, err - } - return &remote.DeployResult{Status: "queued", Domain: domainName}, nil -} - -func (l *localControlPlane) Restart(ctx context.Context, restartDomain string, withAttachments bool) (*remote.RestartResult, error) { - if l.containerSvc == nil { - if withAttachments { - return nil, fmt.Errorf("local restart with attachments requires active local container service") - } - domainName, err := app.SendDeploySignal(restartDomain) - if err != nil { - return nil, err - } - return &remote.RestartResult{Status: "queued", Domain: domainName}, nil - } - - if err := l.containerSvc.SyncContainers(ctx); err != nil { - return nil, err - } - if err := l.containerSvc.Restart(ctx, restartDomain, withAttachments); err != nil { - return nil, err - } - - return &remote.RestartResult{Status: "restarted", Domain: restartDomain}, nil -} - -func (l *localControlPlane) ListTags(ctx context.Context, repository string) ([]string, error) { - if l.registrySvc == nil { - return nil, fmt.Errorf("local registry service unavailable") - } - return l.registrySvc.ListTags(ctx, repository) -} - -func (l *localControlPlane) ListBackups(ctx context.Context, backupDomain string) ([]dto.BackupJob, error) { - if l.backupSvc == nil { - return nil, fmt.Errorf("local backup service unavailable") - } - jobs, err := l.backupSvc.ListBackups(ctx, backupDomain) - if err != nil { - return nil, err - } - return toDTOBackupJobs(jobs), nil -} - -func (l *localControlPlane) BackupStatus(ctx context.Context) ([]dto.BackupJob, error) { - if l.backupSvc == nil { - return nil, fmt.Errorf("local backup service unavailable") - } - jobs, err := l.backupSvc.Status(ctx) - if err != nil { - return nil, err - } - return toDTOBackupJobs(jobs), nil -} - -func (l *localControlPlane) RunBackup(ctx context.Context, backupDomain, dbName string) (*dto.BackupRunResponse, error) { - if l.backupSvc == nil { - return nil, fmt.Errorf("local backup service unavailable") - } - result, err := l.backupSvc.RunBackup(ctx, backupDomain, dbName) - if err != nil { - return nil, err - } - if result == nil { - return &dto.BackupRunResponse{Status: "ok"}, nil - } - job := toDTOBackupJob(result.Job) - return &dto.BackupRunResponse{Status: "ok", Backup: &job}, nil -} - -func (l *localControlPlane) DetectDatabases(ctx context.Context, backupDomain string) ([]dto.DatabaseInfo, error) { - if l.backupSvc == nil { - return nil, fmt.Errorf("local backup service unavailable") - } - dbs, err := l.backupSvc.DetectDatabases(ctx, backupDomain) - if err != nil { - return nil, err - } - out := make([]dto.DatabaseInfo, 0, len(dbs)) - for _, db := range dbs { - out = append(out, dto.DatabaseInfo{ - Type: string(db.Type), - Name: db.Name, - Version: db.Version, - Host: db.Host, - Port: db.Port, - ContainerID: db.ContainerID, - ImageName: db.ImageName, - }) - } - return out, nil -} - -func (l *localControlPlane) ListVolumeBackups(ctx context.Context, backupDomain string) ([]dto.VolumeBackupJob, error) { - if l.volumeBackupSvc == nil { - return nil, localVolumeBackupServiceUnavailable() - } - jobs, err := l.volumeBackupSvc.ListVolumeBackups(ctx, backupDomain) - if err != nil { - return nil, fmt.Errorf("list volume backups: %w", err) - } - return toDTOVolumeBackupJobs(jobs), nil -} - -func (l *localControlPlane) VolumeBackupStatus(ctx context.Context) ([]dto.VolumeBackupJob, error) { - if l.volumeBackupSvc == nil { - return nil, localVolumeBackupServiceUnavailable() - } - jobs, err := l.volumeBackupSvc.VolumeBackupStatus(ctx) - if err != nil { - return nil, fmt.Errorf("get volume backup status: %w", err) - } - return toDTOVolumeBackupJobs(jobs), nil -} - -func (l *localControlPlane) RunVolumeBackups(ctx context.Context, backupDomain, volumeName string) (*dto.VolumeBackupRunResponse, error) { - if l.volumeBackupSvc == nil { - return nil, localVolumeBackupServiceUnavailable() - } - jobs, err := l.volumeBackupSvc.RunVolumeBackups(ctx, backupDomain, volumeName) - if err != nil { - if len(jobs) > 0 { - resp := &dto.VolumeBackupRunResponse{Status: "partial", Backups: toDTOVolumeBackupJobs(jobs), Error: err.Error()} - return resp, fmt.Errorf("run volume backups: %w", err) - } - return nil, fmt.Errorf("run volume backups: %w", err) - } - return &dto.VolumeBackupRunResponse{Status: "ok", Backups: toDTOVolumeBackupJobs(jobs)}, nil -} - -func localVolumeBackupServiceUnavailable() error { - return fmt.Errorf("local volume backup service unavailable: %w", domain.ErrVolumeBackupUnavailable) -} - -func (l *localControlPlane) GetProcessLogs(ctx context.Context, lines int) ([]string, error) { - if l.logSvc == nil { - return nil, fmt.Errorf("local log service unavailable") - } - return l.logSvc.GetProcessLogs(ctx, lines) -} - -func (l *localControlPlane) GetContainerLogs(ctx context.Context, logDomain string, lines int) ([]string, error) { - if l.logSvc == nil { - return nil, fmt.Errorf("local log service unavailable") - } - return l.logSvc.GetContainerLogs(ctx, logDomain, lines) -} - -func (l *localControlPlane) StreamProcessLogs(ctx context.Context, lines int) (<-chan string, error) { - if l.logSvc == nil { - return nil, fmt.Errorf("local log service unavailable") - } - return l.logSvc.FollowProcessLogs(ctx, lines) -} - -func (l *localControlPlane) StreamContainerLogs(ctx context.Context, logDomain string, lines int) (<-chan string, error) { - if l.logSvc == nil { - return nil, fmt.Errorf("local log service unavailable") - } - return l.logSvc.FollowContainerLogs(ctx, logDomain, lines) -} - -func (l *localControlPlane) ListVolumes(ctx context.Context) ([]dto.Volume, error) { - if l.volumeSvc == nil { - return nil, fmt.Errorf("volume service unavailable") - } - vols, err := l.volumeSvc.ListVolumes(ctx) - if err != nil { - return nil, err - } - - result := make([]dto.Volume, len(vols)) - for i, v := range vols { - result[i] = dto.Volume{ - Name: v.Name, - Driver: v.Driver, - MountPoint: v.MountPoint, - Size: v.Size, - CreatedAt: v.CreatedAt, - InUse: v.InUse, - Containers: v.Containers, - Labels: v.Labels, - } - } - return result, nil -} - -func (l *localControlPlane) PruneVolumes(ctx context.Context, req dto.VolumePruneRequest) (*dto.VolumePruneResponse, error) { - if l.volumeSvc == nil { - return nil, fmt.Errorf("volume service unavailable") - } - report, removed, err := l.volumeSvc.PruneVolumes(ctx, req.DryRun) - if err != nil { - return nil, err - } - - vols := make([]dto.Volume, len(removed)) - for i, v := range removed { - vols[i] = dto.Volume{ - Name: v.Name, - Size: v.Size, - } - } - - return &dto.VolumePruneResponse{ - VolumesRemoved: report.VolumesRemoved, - SpaceReclaimed: report.SpaceReclaimed, - Volumes: vols, - }, nil -} - -func mergeConfiguredRemoteRouteInfos(configured []domain.Route, detailed []domain.RouteInfo) []remote.RouteInfo { - detailedByDomain := make(map[string]domain.RouteInfo, len(detailed)) - for _, info := range detailed { - detailedByDomain[info.Domain] = info - } - - merged := make([]domain.RouteInfo, 0, len(configured)) - for _, route := range configured { - info, ok := detailedByDomain[route.Domain] - if !ok { - info = domain.RouteInfo{Domain: route.Domain, Image: route.Image} - } else if info.Image == "" { - info.Image = route.Image - } - merged = append(merged, info) - } - - return toRemoteRouteInfos(merged) -} - -func toRemoteRouteInfos(routes []domain.RouteInfo) []remote.RouteInfo { - out := make([]remote.RouteInfo, 0, len(routes)) - for _, route := range routes { - attachments := make([]dto.Attachment, 0, len(route.Attachments)) - for _, attachment := range route.Attachments { - attachments = append(attachments, dto.Attachment{ - Name: attachment.Name, - Image: attachment.Image, - ContainerID: attachment.ContainerID, - Status: attachment.Status, - Network: attachment.Network, - }) - } - - out = append(out, remote.RouteInfo{ - Domain: route.Domain, - Image: route.Image, - ContainerID: route.ContainerID, - ContainerStatus: route.ContainerStatus, - Network: route.Network, - Attachments: attachments, - }) - } - return out -} - -func toDTOBackupJobs(jobs []domain.BackupJob) []dto.BackupJob { - out := make([]dto.BackupJob, 0, len(jobs)) - for _, job := range jobs { - out = append(out, toDTOBackupJob(job)) - } - return out -} - -func toDTOBackupJob(job domain.BackupJob) dto.BackupJob { - var startedAt *time.Time - if !job.StartedAt.IsZero() { - t := job.StartedAt - startedAt = &t - } - var completedAt *time.Time - if !job.CompletedAt.IsZero() { - t := job.CompletedAt - completedAt = &t - } - - return dto.BackupJob{ - ID: job.ID, - Domain: job.Domain, - DBName: job.DBName, - Schedule: string(job.Schedule), - Type: string(job.Type), - Status: string(job.Status), - StartedAt: startedAt, - CompletedAt: completedAt, - SizeBytes: job.SizeBytes, - Error: job.Error, - } -} - -func toDTOVolumeBackupJobs(jobs []domain.VolumeBackupJob) []dto.VolumeBackupJob { - out := make([]dto.VolumeBackupJob, 0, len(jobs)) - for _, job := range jobs { - out = append(out, toDTOVolumeBackupJob(job)) - } - return out -} - -func toDTOVolumeBackupJob(job domain.VolumeBackupJob) dto.VolumeBackupJob { - var startedAt *time.Time - if !job.StartedAt.IsZero() { - t := job.StartedAt - startedAt = &t - } - var completedAt *time.Time - if !job.CompletedAt.IsZero() { - t := job.CompletedAt - completedAt = &t - } - - return dto.VolumeBackupJob{ - ID: job.ID, - Domain: job.Domain, - ContainerName: job.ContainerName, - ContainerID: job.ContainerID, - VolumeName: job.VolumeName, - MountPath: job.MountPath, - Compression: job.Metadata["compression"], - Type: string(job.Type), - Status: string(job.Status), - StartedAt: startedAt, - CompletedAt: completedAt, - SizeBytes: job.SizeBytes, - ArtifactRef: job.ArtifactRef, - Error: job.Error, - } -} diff --git a/internal/adapters/in/cli/controlplane_local_test.go b/internal/adapters/in/cli/controlplane_local_test.go deleted file mode 100644 index f23e0159d..000000000 --- a/internal/adapters/in/cli/controlplane_local_test.go +++ /dev/null @@ -1,308 +0,0 @@ -package cli - -import ( - "context" - "errors" - "testing" - "time" - - "github.com/bnema/gordon/internal/adapters/dto" - inmocks "github.com/bnema/gordon/internal/boundaries/in/mocks" - "github.com/bnema/gordon/internal/domain" - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/mock" - "github.com/stretchr/testify/require" -) - -type previewContainerService struct { - *inmocks.MockContainerService - preview func(context.Context, string) (*domain.CleanupReport, error) -} - -func (s *previewContainerService) PreviewRemovedRouteCleanup(ctx context.Context, routeDomain string) (*domain.CleanupReport, error) { - if s.preview == nil { - return nil, nil - } - return s.preview(ctx, routeDomain) -} - -func TestLocalControlPlane_GetStatus(t *testing.T) { - t.Parallel() - - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - - ctx := context.Background() - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{{Domain: "app.local", Image: "repo/app:latest"}}) - configSvc.EXPECT().GetRegistryDomain().Return("registry.local") - configSvc.EXPECT().GetRegistryPort().Return(5000) - configSvc.EXPECT().GetServerPort().Return(80) - configSvc.EXPECT().IsAutoRouteEnabled().Return(true) - configSvc.EXPECT().IsNetworkIsolationEnabled().Return(true) - containerSvc.EXPECT().List(mock.Anything).Return(map[string]*domain.Container{ - "app.local": {ID: "abc123", Status: "running"}, - }) - - cp := &localControlPlane{configSvc: configSvc, containerSvc: containerSvc} - status, err := cp.GetStatus(ctx) - require.NoError(t, err) - require.Equal(t, 1, status.Routes) - require.Equal(t, "registry.local", status.RegistryDomain) - require.Equal(t, "running", status.ContainerStatus["app.local"]) -} - -func TestLocalControlPlane_DeployUsesInternalDeployContext(t *testing.T) { - t.Parallel() - - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - - ctx := context.Background() - route := &domain.Route{Domain: "app.local", Image: "repo/app:latest"} - - require.False(t, domain.IsInternalDeploy(ctx)) - - configSvc.EXPECT().GetRoute(mock.Anything, "app.local").Return(route, nil) - containerSvc.EXPECT().Deploy(mock.Anything, *route).RunAndReturn(func(deployCtx context.Context, deployedRoute domain.Route) (*domain.Container, error) { - require.True(t, domain.IsInternalDeploy(deployCtx)) - require.Equal(t, *route, deployedRoute) - return &domain.Container{ID: "container-1"}, nil - }) - - cp := &localControlPlane{configSvc: configSvc, containerSvc: containerSvc} - result, err := cp.Deploy(ctx, "app.local") - require.NoError(t, err) - require.NotNil(t, result) - require.Equal(t, "deployed", result.Status) - require.Equal(t, "app.local", result.Domain) - require.Equal(t, "container-1", result.ContainerID) -} - -func TestLocalControlPlane_Backups(t *testing.T) { - t.Parallel() - - backupSvc := inmocks.NewMockBackupService(t) - ctx := context.Background() - - now := time.Now().UTC() - jobs := []domain.BackupJob{{ID: "b1", Domain: "app.local", DBName: "postgres", Status: domain.BackupStatusCompleted, StartedAt: now}} - backupSvc.EXPECT().ListBackups(mock.Anything, "app.local").Return(jobs, nil) - backupSvc.EXPECT().Status(mock.Anything).Return(jobs, nil) - backupSvc.EXPECT().RunBackup(mock.Anything, "app.local", "postgres").Return(&domain.BackupResult{Job: jobs[0]}, nil) - backupSvc.EXPECT().DetectDatabases(mock.Anything, "app.local").Return([]domain.DBInfo{{Type: domain.DBTypePostgreSQL, Name: "postgres", Host: "postgres", Port: 5432}}, nil) - - cp := &localControlPlane{backupSvc: backupSvc} - - list, err := cp.ListBackups(ctx, "app.local") - require.NoError(t, err) - require.Len(t, list, 1) - - status, err := cp.BackupStatus(ctx) - require.NoError(t, err) - require.Len(t, status, 1) - - run, err := cp.RunBackup(ctx, "app.local", "postgres") - require.NoError(t, err) - require.NotNil(t, run.Backup) - require.Equal(t, "b1", run.Backup.ID) - - dbs, err := cp.DetectDatabases(ctx, "app.local") - require.NoError(t, err) - require.Len(t, dbs, 1) - require.Equal(t, "postgres", dbs[0].Name) -} - -func TestLocalControlPlane_RunVolumeBackupsPreservesPartialJobs(t *testing.T) { - t.Parallel() - - volumeBackupSvc := inmocks.NewMockVolumeBackupService(t) - ctx := context.Background() - runErr := errors.New("one volume failed") - jobs := []domain.VolumeBackupJob{{ID: "v1", Domain: "app.local", VolumeName: "gordon-app-data", Status: domain.BackupStatusCompleted}} - volumeBackupSvc.EXPECT().RunVolumeBackups(mock.Anything, "app.local", "").Return(jobs, runErr) - - cp := &localControlPlane{volumeBackupSvc: volumeBackupSvc} - result, err := cp.RunVolumeBackups(ctx, "app.local", "") - - require.ErrorIs(t, err, runErr) - require.NotNil(t, result) - assert.Equal(t, "partial", result.Status) - assert.Equal(t, runErr.Error(), result.Error) - require.Len(t, result.Backups, 1) - assert.Equal(t, "v1", result.Backups[0].ID) -} - -func TestLocalControlPlane_RunVolumeBackupsMissingServiceUsesSentinel(t *testing.T) { - t.Parallel() - - cp := &localControlPlane{} - result, err := cp.RunVolumeBackups(context.Background(), "app.local", "") - - require.Error(t, err) - assert.Nil(t, result) - assert.ErrorIs(t, err, domain.ErrVolumeBackupUnavailable) -} - -func TestLocalControlPlane_ListVolumeBackupsWrapsServiceError(t *testing.T) { - t.Parallel() - - volumeBackupSvc := inmocks.NewMockVolumeBackupService(t) - wantErr := errors.New("store unavailable") - volumeBackupSvc.EXPECT().ListVolumeBackups(mock.Anything, "app.local").Return(nil, wantErr) - - cp := &localControlPlane{volumeBackupSvc: volumeBackupSvc} - jobs, err := cp.ListVolumeBackups(context.Background(), "app.local") - - require.ErrorIs(t, err, wantErr) - assert.Contains(t, err.Error(), "list volume backups") - assert.Nil(t, jobs) -} - -func TestLocalControlPlane_ListTags(t *testing.T) { - t.Parallel() - - registrySvc := inmocks.NewMockRegistryService(t) - registrySvc.EXPECT().ListTags(mock.Anything, "repo/app").Return([]string{"v1.0.0", "latest"}, nil) - - cp := &localControlPlane{registrySvc: registrySvc} - tags, err := cp.ListTags(context.Background(), "repo/app") - require.NoError(t, err) - require.Equal(t, []string{"v1.0.0", "latest"}, tags) -} - -func TestLocalControlPlane_RemoveRoutePersistsConfigThenReconcilesRuntime(t *testing.T) { - t.Parallel() - - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - report := &domain.CleanupReport{ - Domain: "app.local", - RemovedContainers: []domain.CleanupContainer{{ - ID: "container-1", - Name: "gordon-app.local", - }}, - } - - removeCall := configSvc.EXPECT().RemoveRoute(mock.Anything, "app.local").Return(nil).Once() - cleanupCall := containerSvc.EXPECT().ReconcileRemovedRoute(mock.Anything, "app.local").Return(report, nil).Once() - mock.InOrder(removeCall, cleanupCall) - - cp := &localControlPlane{configSvc: configSvc, containerSvc: containerSvc} - err := cp.RemoveRoute(context.Background(), "app.local") - require.NoError(t, err) -} - -func TestLocalControlPlane_RestartWithAttachments(t *testing.T) { - t.Parallel() - - containerSvc := inmocks.NewMockContainerService(t) - containerSvc.EXPECT().SyncContainers(mock.Anything).Return(nil) - containerSvc.EXPECT().Restart(mock.Anything, "app.local", true).Return(nil) - - cp := &localControlPlane{containerSvc: containerSvc} - result, err := cp.Restart(context.Background(), "app.local", true) - require.NoError(t, err) - require.Equal(t, "app.local", result.Domain) -} - -func TestLocalControlPlane_ListRoutesWithDetailsSyncsBeforeListing(t *testing.T) { - t.Parallel() - - containerSvc := inmocks.NewMockContainerService(t) - syncCall := containerSvc.EXPECT().SyncContainers(mock.Anything).Return(nil).Once() - listCall := containerSvc.EXPECT().ListRoutesWithDetails(mock.Anything).Return([]domain.RouteInfo{{ - Domain: "app.local", - Image: "repo/app:latest", - ContainerID: "container-1", - ContainerStatus: "running", - }}).Once() - mock.InOrder(syncCall, listCall) - - cp := &localControlPlane{containerSvc: containerSvc} - routes, err := cp.ListRoutesWithDetails(context.Background()) - require.NoError(t, err) - require.Len(t, routes, 1) - require.Equal(t, "app.local", routes[0].Domain) - require.Equal(t, "repo/app:latest", routes[0].Image) - require.Equal(t, "container-1", routes[0].ContainerID) - require.Equal(t, "running", routes[0].ContainerStatus) -} - -func TestLocalControlPlane_ListRoutesWithDetails_IncludesConfiguredRouteWithoutRuntimeDetail(t *testing.T) { - t.Parallel() - - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{{ - Domain: "app.local", - Image: "repo/app:latest", - }}).Once() - containerSvc.EXPECT().SyncContainers(mock.Anything).Return(nil).Once() - containerSvc.EXPECT().ListRoutesWithDetails(mock.Anything).Return(nil).Once() - - cp := &localControlPlane{configSvc: configSvc, containerSvc: containerSvc} - routes, err := cp.ListRoutesWithDetails(context.Background()) - require.NoError(t, err) - require.Len(t, routes, 1) - require.Equal(t, "app.local", routes[0].Domain) - require.Equal(t, "repo/app:latest", routes[0].Image) - require.Empty(t, routes[0].ContainerID) - require.Empty(t, routes[0].ContainerStatus) -} - -func TestLocalControlPlane_GetTLSStatusWithoutService(t *testing.T) { - t.Parallel() - - cp := &localControlPlane{} - status, err := cp.GetTLSStatus(context.Background()) - require.NoError(t, err) - require.Equal(t, &dto.TLSStatusResponse{ - ACMEEnabled: false, - SelectionReason: "public TLS service not configured", - }, status) -} - -func TestLocalControlPlane_GetContainerLogs(t *testing.T) { - t.Parallel() - - logSvc := inmocks.NewMockLogService(t) - logSvc.EXPECT().GetContainerLogs(mock.Anything, "app.local", 50).Return([]string{"line1", "line2"}, nil) - - cp := &localControlPlane{logSvc: logSvc} - lines, err := cp.GetContainerLogs(context.Background(), "app.local", 50) - require.NoError(t, err) - require.Equal(t, []string{"line1", "line2"}, lines) -} - -func TestLocalControlPlane_GetRouteCleanupPreview_ReturnsNilReportWithoutPanic(t *testing.T) { - ctx := context.Background() - volumeSvc := inmocks.NewMockVolumeService(t) - containerSvc := &previewContainerService{ - MockContainerService: inmocks.NewMockContainerService(t), - preview: func(context.Context, string) (*domain.CleanupReport, error) { - return nil, nil - }, - } - cp := &localControlPlane{containerSvc: containerSvc, volumeSvc: volumeSvc} - - report, err := cp.GetRouteCleanupPreview(ctx, "app.local") - require.NoError(t, err) - assert.Nil(t, report) -} - -func TestLocalControlPlane_RemoveRouteReconcilesRuntimeWhenRouteAlreadyMissing(t *testing.T) { - ctx := context.Background() - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - report := &domain.CleanupReport{Domain: "app.local"} - - configSvc.EXPECT().RemoveRoute(mock.Anything, "app.local").Return(domain.ErrRouteNotFound).Once() - containerSvc.EXPECT().ReconcileRemovedRoute(mock.Anything, "app.local").Return(report, nil).Once() - - cp := &localControlPlane{configSvc: configSvc, containerSvc: containerSvc} - resp, err := cp.RemoveRouteWithCleanup(ctx, "app.local") - require.NoError(t, err) - require.NotNil(t, resp) - assert.Equal(t, dto.CleanupReportFromDomain(report), resp.Cleanup) -} diff --git a/internal/adapters/in/cli/controlplane_remote.go b/internal/adapters/in/cli/controlplane_remote.go deleted file mode 100644 index 205ab0751..000000000 --- a/internal/adapters/in/cli/controlplane_remote.go +++ /dev/null @@ -1,223 +0,0 @@ -package cli - -import ( - "context" - "fmt" - - "github.com/bnema/gordon/internal/adapters/dto" - "github.com/bnema/gordon/internal/adapters/in/cli/remote" - "github.com/bnema/gordon/internal/domain" -) - -type remoteControlPlane struct { - client *remote.Client -} - -func NewRemoteControlPlane(client *remote.Client) ControlPlane { - return &remoteControlPlane{client: client} -} - -func (r *remoteControlPlane) ListRoutesWithDetails(ctx context.Context) ([]remote.RouteInfo, error) { - return r.client.ListRoutesWithDetails(ctx) -} - -func (r *remoteControlPlane) GetHealth(ctx context.Context) (map[string]*remote.RouteHealth, error) { - return r.client.GetHealth(ctx) -} - -func (r *remoteControlPlane) GetRoute(ctx context.Context, routeDomain string) (*domain.Route, error) { - return r.client.GetRoute(ctx, routeDomain) -} - -func (r *remoteControlPlane) FindRoutesByImage(ctx context.Context, imageName string) ([]domain.Route, error) { - return r.client.FindRoutesByImage(ctx, imageName) -} - -func (r *remoteControlPlane) AddRoute(ctx context.Context, route domain.Route) error { - return r.client.AddRoute(ctx, route) -} - -func (r *remoteControlPlane) UpdateRoute(ctx context.Context, route domain.Route) error { - return r.client.UpdateRoute(ctx, route) -} - -func (r *remoteControlPlane) RemoveRoute(ctx context.Context, routeDomain string) error { - _, err := r.RemoveRouteWithCleanup(ctx, routeDomain) - return err -} - -func (r *remoteControlPlane) RemoveRouteWithCleanup(ctx context.Context, routeDomain string) (*dto.RouteDeleteResponse, error) { - return r.client.RemoveRouteWithCleanup(ctx, routeDomain) -} - -func (r *remoteControlPlane) GetRouteCleanupPreview(ctx context.Context, routeDomain string) (*domain.CleanupReport, error) { - return r.client.GetRouteCleanupPreview(ctx, routeDomain) -} - -func (r *remoteControlPlane) Bootstrap(ctx context.Context, req dto.BootstrapRequest) (*dto.BootstrapResponse, error) { - return r.client.Bootstrap(ctx, req) -} - -func (r *remoteControlPlane) ListSecretsWithAttachments(ctx context.Context, secretDomain string) (*remote.SecretsListResult, error) { - return r.client.ListSecretsWithAttachments(ctx, secretDomain) -} - -func (r *remoteControlPlane) SetSecrets(ctx context.Context, secretDomain string, secrets map[string]string) error { - return r.client.SetSecrets(ctx, secretDomain, secrets) -} - -func (r *remoteControlPlane) DeleteSecret(ctx context.Context, secretDomain, key string) error { - return r.client.DeleteSecret(ctx, secretDomain, key) -} - -func (r *remoteControlPlane) SetAttachmentSecrets(ctx context.Context, domainName, service string, secrets map[string]string) error { - return r.client.SetAttachmentSecrets(ctx, domainName, service, secrets) -} - -func (r *remoteControlPlane) DeleteAttachmentSecret(ctx context.Context, domainName, service, key string) error { - return r.client.DeleteAttachmentSecret(ctx, domainName, service, key) -} - -func (r *remoteControlPlane) ListOrphanedAttachments(ctx context.Context) ([]domain.CleanupAttachment, error) { - return r.client.ListOrphanedAttachments(ctx) -} - -func (r *remoteControlPlane) CleanupOrphanedAttachments(ctx context.Context, owner string, stop bool) (*domain.CleanupReport, error) { - return r.client.CleanupOrphanedAttachments(ctx, owner, stop) -} - -func (r *remoteControlPlane) GetAllAttachmentsConfig(ctx context.Context) (map[string][]string, error) { - return r.client.GetAllAttachmentsConfig(ctx) -} - -func (r *remoteControlPlane) GetAttachmentsConfig(ctx context.Context, domainOrGroup string) ([]string, error) { - return r.client.GetAttachmentsConfig(ctx, domainOrGroup) -} - -func (r *remoteControlPlane) FindAttachmentTargetsByImage(ctx context.Context, imageName string) ([]string, error) { - return r.client.FindAttachmentTargetsByImage(ctx, imageName) -} - -func (r *remoteControlPlane) AddAttachment(ctx context.Context, domainOrGroup, image string) error { - return r.client.AddAttachment(ctx, domainOrGroup, image) -} - -func (r *remoteControlPlane) RemoveAttachment(ctx context.Context, domainOrGroup, image string) error { - return r.client.RemoveAttachment(ctx, domainOrGroup, image) -} - -func (r *remoteControlPlane) GetAutoRouteAllowedDomains(ctx context.Context) ([]string, error) { - return r.client.GetAutoRouteAllowedDomains(ctx) -} - -func (r *remoteControlPlane) AddAutoRouteAllowedDomain(ctx context.Context, pattern string) error { - return r.client.AddAutoRouteAllowedDomain(ctx, pattern) -} - -func (r *remoteControlPlane) RemoveAutoRouteAllowedDomain(ctx context.Context, pattern string) error { - return r.client.RemoveAutoRouteAllowedDomain(ctx, pattern) -} - -func (r *remoteControlPlane) GetStatus(ctx context.Context) (*remote.Status, error) { - return r.client.GetStatus(ctx) -} - -func (r *remoteControlPlane) GetTLSStatus(ctx context.Context) (*dto.TLSStatusResponse, error) { - return r.client.GetTLSStatus(ctx) -} - -func (r *remoteControlPlane) GetTrafficStatus(ctx context.Context) (*dto.TrafficStatusResponse, error) { - return r.client.GetTrafficStatus(ctx) -} - -func (r *remoteControlPlane) Reload(ctx context.Context) error { - return r.client.Reload(ctx) -} - -func (r *remoteControlPlane) ListNetworks(ctx context.Context) ([]*domain.NetworkInfo, error) { - return r.client.ListNetworks(ctx) -} - -func (r *remoteControlPlane) GetConfig(ctx context.Context) (*remote.Config, error) { - return r.client.GetConfig(ctx) -} - -func (r *remoteControlPlane) DeployIntent(ctx context.Context, imageName string) error { - return r.client.DeployIntent(ctx, imageName) -} - -func (r *remoteControlPlane) Deploy(ctx context.Context, deployDomain string) (*remote.DeployResult, error) { - return r.client.Deploy(ctx, deployDomain) -} - -func (r *remoteControlPlane) Restart(ctx context.Context, restartDomain string, withAttachments bool) (*remote.RestartResult, error) { - return r.client.Restart(ctx, restartDomain, withAttachments) -} - -func (r *remoteControlPlane) ListTags(ctx context.Context, repository string) ([]string, error) { - return r.client.ListTags(ctx, repository) -} - -func (r *remoteControlPlane) ListBackups(ctx context.Context, backupDomain string) ([]dto.BackupJob, error) { - return r.client.ListBackups(ctx, backupDomain) -} - -func (r *remoteControlPlane) BackupStatus(ctx context.Context) ([]dto.BackupJob, error) { - return r.client.BackupStatus(ctx) -} - -func (r *remoteControlPlane) RunBackup(ctx context.Context, backupDomain, dbName string) (*dto.BackupRunResponse, error) { - return r.client.RunBackup(ctx, backupDomain, dbName) -} - -func (r *remoteControlPlane) DetectDatabases(ctx context.Context, backupDomain string) ([]dto.DatabaseInfo, error) { - return r.client.DetectDatabases(ctx, backupDomain) -} - -func (r *remoteControlPlane) ListVolumeBackups(ctx context.Context, backupDomain string) ([]dto.VolumeBackupJob, error) { - jobs, err := r.client.ListVolumeBackups(ctx, backupDomain) - if err != nil { - return nil, fmt.Errorf("list volume backups: %w", err) - } - return jobs, nil -} - -func (r *remoteControlPlane) VolumeBackupStatus(ctx context.Context) ([]dto.VolumeBackupJob, error) { - jobs, err := r.client.VolumeBackupStatus(ctx) - if err != nil { - return nil, fmt.Errorf("get volume backup status: %w", err) - } - return jobs, nil -} - -func (r *remoteControlPlane) RunVolumeBackups(ctx context.Context, backupDomain, volumeName string) (*dto.VolumeBackupRunResponse, error) { - result, err := r.client.RunVolumeBackups(ctx, backupDomain, volumeName) - if err != nil { - return result, fmt.Errorf("run volume backups: %w", err) - } - return result, nil -} - -func (r *remoteControlPlane) GetProcessLogs(ctx context.Context, lines int) ([]string, error) { - return r.client.GetProcessLogs(ctx, lines) -} - -func (r *remoteControlPlane) GetContainerLogs(ctx context.Context, logDomain string, lines int) ([]string, error) { - return r.client.GetContainerLogs(ctx, logDomain, lines) -} - -func (r *remoteControlPlane) StreamProcessLogs(ctx context.Context, lines int) (<-chan string, error) { - return r.client.StreamProcessLogs(ctx, lines) -} - -func (r *remoteControlPlane) StreamContainerLogs(ctx context.Context, logDomain string, lines int) (<-chan string, error) { - return r.client.StreamContainerLogs(ctx, logDomain, lines) -} - -func (r *remoteControlPlane) ListVolumes(ctx context.Context) ([]dto.Volume, error) { - return r.client.ListVolumes(ctx) -} - -func (r *remoteControlPlane) PruneVolumes(ctx context.Context, req dto.VolumePruneRequest) (*dto.VolumePruneResponse, error) { - return r.client.PruneVolumes(ctx, req) -} diff --git a/internal/adapters/in/cli/controlplane_resolver.go b/internal/adapters/in/cli/controlplane_resolver.go index 32f1fc882..8c1429e07 100644 --- a/internal/adapters/in/cli/controlplane_resolver.go +++ b/internal/adapters/in/cli/controlplane_resolver.go @@ -5,7 +5,6 @@ import ( "fmt" "github.com/bnema/gordon/internal/adapters/in/cli/remote" - "github.com/bnema/gordon/internal/app" ) type controlPlaneHandle struct { @@ -24,39 +23,32 @@ func (h *controlPlaneHandle) close() { } } -func resolveControlPlane(configPath string) (*controlPlaneHandle, error) { +var newLocalControlPlaneClient = remote.NewLocalClient + +func resolveDaemonClient() (*remote.Client, bool, error) { client, isRemote, err := GetRemoteClient() - if err != nil { - return nil, err + if err != nil || isRemote { + return client, isRemote, err } - if isRemote { - return &controlPlaneHandle{plane: NewRemoteControlPlane(client), isRemote: true}, nil + + client, err = newLocalControlPlaneClient() + if err != nil { + return nil, false, fmt.Errorf("local control plane unavailable: %w", err) } + return client, false, nil +} - return resolveLocalControlPlane(configPath) +func resolveControlPlane(_ string) (*controlPlaneHandle, error) { + client, isRemote, err := resolveDaemonClient() + if err != nil { + return nil, err + } + return &controlPlaneHandle{plane: client, isRemote: isRemote}, nil } func newRemoteControlPlaneHandle(target *remote.ResolvedRemote) *controlPlaneHandle { client := remote.NewClient(target.URL, remoteClientOptions(target.Token, target.InsecureTLS)...) - return &controlPlaneHandle{plane: NewRemoteControlPlane(client), isRemote: true} -} - -func resolveControlPlaneForRouteDomain(ctx context.Context, routeDomain string) (*controlPlaneHandle, error) { - return resolveControlPlaneWithInference(ctx, func(ctx context.Context) (*remote.ResolvedRemote, error) { - return inferRemoteForRouteDomain(ctx, routeDomain) - }) -} - -func resolveControlPlaneForRouteCleanupDomain(ctx context.Context, routeDomain string) (*controlPlaneHandle, error) { - return resolveControlPlaneWithInference(ctx, func(ctx context.Context) (*remote.ResolvedRemote, error) { - return inferRemoteForRouteCleanupDomain(ctx, routeDomain) - }) -} - -func resolveControlPlaneForAttachmentTarget(ctx context.Context, target string) (*controlPlaneHandle, error) { - return resolveControlPlaneWithInference(ctx, func(ctx context.Context) (*remote.ResolvedRemote, error) { - return inferRemoteForAttachmentTarget(ctx, target) - }) + return &controlPlaneHandle{plane: client, isRemote: true} } func resolveControlPlaneForRepository(ctx context.Context, repository string) (*controlPlaneHandle, error) { @@ -73,17 +65,5 @@ func resolveControlPlaneWithInference(ctx context.Context, infer func(context.Co if resolved != nil { return newRemoteControlPlaneHandle(resolved), nil } - return resolveControlPlane(configPath) -} - -func resolveLocalControlPlane(configPath string) (*controlPlaneHandle, error) { - kernel, err := app.NewKernelQuiet(configPath) - if err != nil { - return nil, fmt.Errorf("failed to initialize local control plane: %w", err) - } - - return &controlPlaneHandle{ - plane: NewLocalControlPlane(kernel), - closeFn: kernel.Close, - }, nil + return resolveControlPlane(cliConfigPath) } diff --git a/internal/adapters/in/cli/controlplane_resolver_test.go b/internal/adapters/in/cli/controlplane_resolver_test.go index 9f3ccf4d1..5251c5cf5 100644 --- a/internal/adapters/in/cli/controlplane_resolver_test.go +++ b/internal/adapters/in/cli/controlplane_resolver_test.go @@ -1,110 +1,134 @@ package cli import ( + "context" + "encoding/json" + "errors" + "net/http" "os" "path/filepath" "testing" + "time" + "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/in/cli/remote" + "github.com/bnema/gordon/internal/adapters/localadmin" ) -func TestResolveControlPlane_LocalAllowedWhenAuthEnabled(t *testing.T) { - tmpDir := t.TempDir() - t.Setenv("XDG_CONFIG_HOME", filepath.Join(tmpDir, "xdg")) +func isolateControlPlaneResolver(t *testing.T) { + t.Helper() + t.Setenv("XDG_CONFIG_HOME", filepath.Join(t.TempDir(), "xdg")) t.Setenv("GORDON_REMOTE", "") t.Setenv("GORDON_TOKEN", "") originalRemoteFlag, originalTokenFlag, originalInsecureTLSFlag := remoteFlag, tokenFlag, insecureTLSFlag + originalFactory := newLocalControlPlaneClient t.Cleanup(func() { remoteFlag = originalRemoteFlag tokenFlag = originalTokenFlag insecureTLSFlag = originalInsecureTLSFlag + newLocalControlPlaneClient = originalFactory }) - remoteFlag = "" tokenFlag = "" insecureTLSFlag = false +} - configPath := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configPath, []byte(`[server] -gordon_domain = "gordon.local" -data_dir = "`+filepath.Join(tmpDir, "data")+`" - -[auth] -enabled = true -secrets_backend = "unsafe" -`), 0o600) +func localControlPlaneTestClient(t *testing.T, handler http.Handler) *remote.Client { + t.Helper() + dir := t.TempDir() + listener, _, err := localadmin.Listen(dir) require.NoError(t, err) + server := &http.Server{Handler: handler, ReadHeaderTimeout: 5 * time.Second} + go func() { _ = server.Serve(listener) }() + t.Cleanup(func() { _ = server.Close() }) + return remote.NewLocalClientForSocket(localadmin.SocketPath(dir)) +} - handle, err := resolveControlPlane(configPath) +func TestResolveControlPlane_LocalUsesSocketBackedRemotePlane(t *testing.T) { + isolateControlPlaneResolver(t) + requests := make(chan string, 1) + client := localControlPlaneTestClient(t, http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + requests <- r.URL.Path + w.Header().Set("Content-Type", "application/json") + _ = json.NewEncoder(w).Encode(remote.Status{RegistryDomain: "socket.local"}) + })) + factoryCalls := 0 + newLocalControlPlaneClient = func() (*remote.Client, error) { + factoryCalls++ + return client, nil + } + + // A deliberately unusable config path proves local resolution does not open a kernel/store. + handle, err := resolveControlPlane(filepath.Join(t.TempDir(), "missing", "gordon.toml")) require.NoError(t, err) require.NotNil(t, handle) - require.NotNil(t, handle.plane) - defer handle.close() -} + assert.False(t, handle.isRemote) + assert.IsType(t, &remote.Client{}, handle.plane) + assert.Equal(t, 1, factoryCalls) -func TestResolveControlPlane_LocalAllowedWhenAuthDisabled(t *testing.T) { - tmpDir := t.TempDir() - t.Setenv("XDG_CONFIG_HOME", filepath.Join(tmpDir, "xdg")) - t.Setenv("GORDON_REMOTE", "") - t.Setenv("GORDON_TOKEN", "") - originalRemoteFlag, originalTokenFlag, originalInsecureTLSFlag := remoteFlag, tokenFlag, insecureTLSFlag - t.Cleanup(func() { - remoteFlag = originalRemoteFlag - tokenFlag = originalTokenFlag - insecureTLSFlag = originalInsecureTLSFlag - }) + status, err := handle.plane.GetStatus(context.Background()) + require.NoError(t, err) + assert.Equal(t, "socket.local", status.RegistryDomain) + select { + case path := <-requests: + assert.Equal(t, "/admin/status", path) + case <-time.After(2 * time.Second): + t.Fatal("socket server received no request") + } +} - remoteFlag = "" - tokenFlag = "" - insecureTLSFlag = false +func TestResolveControlPlane_LocalFailsClosedWhenDaemonUnavailable(t *testing.T) { + isolateControlPlaneResolver(t) + newLocalControlPlaneClient = func() (*remote.Client, error) { + return nil, remote.ErrDaemonUnavailable + } - configPath := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configPath, []byte(`[server] -gordon_domain = "gordon.local" -data_dir = "`+filepath.Join(tmpDir, "data")+`" + handle, err := resolveControlPlane(filepath.Join(t.TempDir(), "would-create-store.toml")) + require.Error(t, err) + assert.ErrorIs(t, err, remote.ErrDaemonUnavailable) + assert.Nil(t, handle) +} -[auth] -enabled = false -secrets_backend = "unsafe" -`), 0o600) - require.NoError(t, err) +func TestResolveControlPlane_ExplicitRemoteDoesNotUseLocalFactory(t *testing.T) { + isolateControlPlaneResolver(t) + remoteFlag = "https://example.invalid" + tokenFlag = "token" + newLocalControlPlaneClient = func() (*remote.Client, error) { + t.Fatal("local factory invoked for explicit remote") + return nil, errors.New("unreachable") + } - handle, err := resolveControlPlane(configPath) + handle, err := resolveControlPlane("") require.NoError(t, err) - require.NotNil(t, handle) - require.NotNil(t, handle.plane) - defer handle.close() + assert.True(t, handle.isRemote) + assert.IsType(t, &remote.Client{}, handle.plane) } func TestResolveControlPlane_ExplicitUnknownRemoteReturnsError(t *testing.T) { - tmpDir := t.TempDir() - t.Setenv("XDG_CONFIG_HOME", filepath.Join(tmpDir, "xdg")) - t.Setenv("GORDON_REMOTE", "") - t.Setenv("GORDON_TOKEN", "") - originalRemoteFlag, originalTokenFlag, originalInsecureTLSFlag := remoteFlag, tokenFlag, insecureTLSFlag - t.Cleanup(func() { - remoteFlag = originalRemoteFlag - tokenFlag = originalTokenFlag - insecureTLSFlag = originalInsecureTLSFlag - }) - + isolateControlPlaneResolver(t) + t.Setenv("XDG_CONFIG_HOME", filepath.Join(t.TempDir(), "xdg")) remoteFlag = "does-not-exist" - tokenFlag = "" - insecureTLSFlag = false - configPath := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configPath, []byte(`[server] -gordon_domain = "gordon.local" -data_dir = "`+filepath.Join(tmpDir, "data")+`" + handle, err := resolveControlPlane("") + require.Error(t, err) + assert.Nil(t, handle) + assert.Contains(t, err.Error(), "does-not-exist") +} -[auth] -enabled = false -secrets_backend = "unsafe" -`), 0o600) +func TestResolveControlPlane_LocalIgnoresConfigAndStoreState(t *testing.T) { + isolateControlPlaneResolver(t) + client := localControlPlaneTestClient(t, http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.Header().Set("Content-Type", "application/json") + _ = json.NewEncoder(w).Encode(remote.Status{}) + })) + newLocalControlPlaneClient = func() (*remote.Client, error) { return client, nil } + + lockedStore := filepath.Join(t.TempDir(), "state.db") + require.NoError(t, os.WriteFile(lockedStore, []byte("active daemon store"), 0o600)) + handle, err := resolveControlPlane(lockedStore) + require.NoError(t, err) + _, err = handle.plane.GetStatus(context.Background()) require.NoError(t, err) - - handle, err := resolveControlPlane(configPath) - require.Error(t, err) - require.Nil(t, handle) - require.Contains(t, err.Error(), "does-not-exist") } diff --git a/internal/adapters/in/cli/daemon.go b/internal/adapters/in/cli/daemon.go new file mode 100644 index 000000000..aec8c90b6 --- /dev/null +++ b/internal/adapters/in/cli/daemon.go @@ -0,0 +1,29 @@ +package cli + +import "github.com/spf13/cobra" + +// newDaemonCmd groups commands that inspect or operate the Gordon daemon +// itself rather than the apps it runs. They target the local daemon or the +// one selected with --remote. +func newDaemonCmd() *cobra.Command { + cmd := &cobra.Command{ + Use: "daemon", + Short: "Inspect and operate the Gordon daemon", + Long: `Inspect and operate the Gordon daemon: status, process logs, +configuration, TLS, traffic, and networks. + +Targets the local daemon, or the one selected with --remote.`, + } + + cmd.AddCommand( + newStatusCmd(), + newLogsCmd(), + newReloadCmd(), + newConfigCmd(), + newTLSCmd(), + newTrafficCmd(), + newNetworksCmd(), + ) + + return cmd +} diff --git a/internal/adapters/in/cli/deploy.go b/internal/adapters/in/cli/deploy.go deleted file mode 100644 index 26cddd984..000000000 --- a/internal/adapters/in/cli/deploy.go +++ /dev/null @@ -1,114 +0,0 @@ -package cli - -import ( - "context" - "fmt" - "io" - - "github.com/spf13/cobra" - - "github.com/bnema/gordon/internal/adapters/in/cli/remote" - "github.com/bnema/gordon/internal/app" -) - -type deployer interface { - Deploy(ctx context.Context, deployDomain string) (*remote.DeployResult, error) -} - -var sendDeploySignal = app.SendDeploySignal - -// newDeployCmd creates the deploy command. -func newDeployCmd() *cobra.Command { - var jsonOut bool - cmd := &cobra.Command{ - Use: "deploy ", - Short: "Manually deploy or redeploy a route", - Long: `Triggers a deployment for the specified route domain. -The route must be configured in config.toml. - -This will pull the latest image and deploy/redeploy the container, -even if a container is already running. - -Examples: - gordon deploy myapp.example.com - gordon deploy api.example.com - gordon deploy myapp.example.com --remote https://gordon.mydomain.com --token $TOKEN`, - Args: cobra.ExactArgs(1), - RunE: func(cmd *cobra.Command, args []string) error { - handle, err := resolveControlPlaneForRouteDomain(cmd.Context(), args[0]) - if err != nil { - return err - } - defer handle.close() - - return runDeploy(cmd.Context(), handle.plane, handle.isRemote, args[0], cmd.OutOrStdout(), jsonOut) - }, - } - cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") - return cmd -} - -func runDeploy(ctx context.Context, deployer deployer, isRemote bool, deployDomain string, out io.Writer, jsonOut bool) error { - result, err := deployer.Deploy(ctx, deployDomain) - if err != nil { - if formatted, ok := structuredDeployFailure(err); ok { - return formatted - } - - if isRemote && shouldFallbackToLocal(err) { - return handleLocalDeployFallback(err, deployDomain, out, jsonOut) - } - - return formatDeployFailure(err) - } - - if jsonOut { - return writeJSON(out, result) - } - - messageDomain := deployDomain - if result.Domain != "" { - messageDomain = result.Domain - } - - containerID := shortContainerID(result.ContainerID) - - switch result.Status { - case "deployed": - msg := fmt.Sprintf("Deployed %s", messageDomain) - if containerID != "" { - msg += fmt.Sprintf(" (container: %s)", containerID) - } - return cliWriteLine(out, cliRenderSuccess(msg)) - case "queued": - return cliWriteLine(out, cliRenderSuccess(fmt.Sprintf("Deploy queued for domain: %s", messageDomain))) - default: - msg := fmt.Sprintf("Deploy status for %s: %s", messageDomain, result.Status) - if containerID != "" { - msg += fmt.Sprintf(" (container: %s)", containerID) - } - return cliWriteLine(out, cliRenderInfo(msg)) - } -} - -func handleLocalDeployFallback(err error, deployDomain string, out io.Writer, jsonOut bool) error { - domain, localErr := sendDeploySignal(deployDomain) - if localErr != nil { - return fmt.Errorf("failed to deploy: remote error: %w; local fallback also failed: %w", err, localErr) - } - - warning := sanitizeDeployLogLine(fmt.Sprintf("Remote deploy failed (%v), used local signal fallback", err)) - if jsonOut { - return writeJSON(out, map[string]string{ - "warning": warning, - "domain": domain, - "status": "success", - }) - } - - if writeErr := cliWriteLine(out, cliRenderWarning(warning)); writeErr != nil { - return writeErr - } - - return cliWriteLine(out, cliRenderSuccess(fmt.Sprintf("Deploy signal sent for domain: %s", domain))) -} diff --git a/internal/adapters/in/cli/deploy_test.go b/internal/adapters/in/cli/deploy_test.go deleted file mode 100644 index 94a0966d7..000000000 --- a/internal/adapters/in/cli/deploy_test.go +++ /dev/null @@ -1,119 +0,0 @@ -package cli - -import ( - "bytes" - "context" - "errors" - "testing" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" - - "github.com/bnema/gordon/internal/adapters/in/cli/remote" - "github.com/bnema/gordon/internal/domain" -) - -type deployTestDeployer struct { - result *remote.DeployResult - err error -} - -func (d *deployTestDeployer) Deploy(context.Context, string) (*remote.DeployResult, error) { - return d.result, d.err -} - -func TestRunDeploy_LocalTypedErrorUsesDeployFormatter(t *testing.T) { - root := &domain.DeployFailureError{ - Summary: "failed to deploy", - Cause: "health check failed", - Hint: "check DATABASE_URL", - Logs: []string{"booting app", "connection refused"}, - } - err := runDeploy(context.Background(), &deployTestDeployer{err: root}, false, "app.example.com", &bytes.Buffer{}, false) - - require.Error(t, err) - assert.Contains(t, err.Error(), "failed to deploy\nCause: health check failed\nHint: check DATABASE_URL\n\nRecent container logs:\n booting app\n connection refused") - assert.ErrorIs(t, err, root) -} - -func TestRunDeploy_StructuredRemoteNoFallbackReturnsFormattedError(t *testing.T) { - originalSendDeploySignal := sendDeploySignal - sendDeploySignal = func(string) (string, error) { - t.Fatalf("sendDeploySignal should not be called for structured deploy failures") - return "", nil - } - defer func() { sendDeploySignal = originalSendDeploySignal }() - - root := &remote.HTTPError{ - StatusCode: 503, - Status: "503 Service Unavailable", - Body: "failed to deploy", - Cause: "health check failed", - Hint: "check DATABASE_URL", - Logs: []string{"booting app", "connection refused"}, - Structured: true, - } - err := runDeploy(context.Background(), &deployTestDeployer{err: root}, true, "app.example.com", &bytes.Buffer{}, false) - - require.Error(t, err) - assert.Equal(t, "failed to deploy\nCause: health check failed\nHint: check DATABASE_URL\n\nRecent container logs:\n booting app\n connection refused", err.Error()) - assert.ErrorIs(t, err, root) -} - -func TestRunDeploy_UsesResultDomainWhenPresent(t *testing.T) { - var out bytes.Buffer - err := runDeploy(context.Background(), &deployTestDeployer{result: &remote.DeployResult{ - Status: "deployed", - Domain: "actual.example.com", - ContainerID: "1234567890abcdef", - }}, false, "requested.example.com", &out, false) - - require.NoError(t, err) - assert.Contains(t, out.String(), "Deployed actual.example.com (container: 1234567890ab)") - assert.NotContains(t, out.String(), "requested.example.com") -} - -func TestRunDeploy_RemoteFallbackUsesJSONWhenRequested(t *testing.T) { - originalSendDeploySignal := sendDeploySignal - sendDeploySignal = func(string) (string, error) { - return "app.example.com", nil - } - defer func() { sendDeploySignal = originalSendDeploySignal }() - - var out bytes.Buffer - err := runDeploy(context.Background(), &deployTestDeployer{err: errors.New("503 Service Unavailable: Registry Unavailable")}, true, "app.example.com", &out, true) - - require.NoError(t, err) - assert.JSONEq(t, `{"domain":"app.example.com","status":"success","warning":"Remote deploy failed (503 Service Unavailable: Registry Unavailable), used local signal fallback"}`, out.String()) -} - -func TestHandleLocalDeployFallback_WrapsRemoteAndLocalErrors(t *testing.T) { - originalSendDeploySignal := sendDeploySignal - localErr := errors.New("local signal failed") - sendDeploySignal = func(string) (string, error) { - return "", localErr - } - defer func() { sendDeploySignal = originalSendDeploySignal }() - - remoteErr := errors.New("remote deploy failed") - err := handleLocalDeployFallback(remoteErr, "app.example.com", &bytes.Buffer{}, false) - - require.Error(t, err) - assert.Equal(t, "failed to deploy: remote error: remote deploy failed; local fallback also failed: local signal failed", err.Error()) - assert.ErrorIs(t, err, remoteErr) - assert.ErrorIs(t, err, localErr) -} - -func TestRunDeploy_RemoteFallbackSanitizesWarning(t *testing.T) { - originalSendDeploySignal := sendDeploySignal - sendDeploySignal = func(string) (string, error) { - return "app.example.com", nil - } - defer func() { sendDeploySignal = originalSendDeploySignal }() - - var out bytes.Buffer - err := runDeploy(context.Background(), &deployTestDeployer{err: errors.New("503 Service Unavailable:\n\x1b[31mRegistry Unavailable\x1b[0m")}, true, "app.example.com", &out, true) - - require.NoError(t, err) - assert.JSONEq(t, `{"domain":"app.example.com","status":"success","warning":"Remote deploy failed (503 Service Unavailable:Registry Unavailable), used local signal fallback"}`, out.String()) -} diff --git a/internal/adapters/in/cli/images.go b/internal/adapters/in/cli/images.go index dc514ad53..665514ed4 100644 --- a/internal/adapters/in/cli/images.go +++ b/internal/adapters/in/cli/images.go @@ -69,12 +69,11 @@ var imagesListTableColumns = []components.TableColumn{ func newImagesCmd() *cobra.Command { cmd := &cobra.Command{ Use: "images", - Short: "List and prune images", - Long: `Inspect and clean up runtime and registry images. - -These commands currently require remote mode with a configured target.`, + Short: "Push, list, and prune images", + Long: `Push, inspect, and clean up runtime and registry images through the selected daemon.`, } + cmd.AddCommand(newPushCmd()) cmd.AddCommand(newImagesListCmd()) cmd.AddCommand(newImagesPruneCmd()) cmd.AddCommand(newImagesTagsCmd()) @@ -89,14 +88,10 @@ func newImagesListCmd() *cobra.Command { Use: "list", Short: "List runtime images and registry tags", RunE: func(cmd *cobra.Command, _ []string) error { - client, isRemote, err := GetRemoteClient() + client, _, err := resolveDaemonClient() if err != nil { return err } - if !isRemote { - return fmt.Errorf("images commands require a configured remote target") - } - return runImagesList(cmd.Context(), client, cmd.OutOrStdout(), jsonOut) }, } @@ -115,21 +110,21 @@ func newImagesPruneCmd() *cobra.Command { Long: `Remove dangling runtime images and/or old registry tags. By default both runtime and registry cleanup run, keeping latest + 3 previous -release tags per repository. Use --dangling or --registry to restrict scope.`, +release tags per repository. Use --dangling or --registry to restrict scope. + +Only resources proven safe are deleted: every candidate gets an eligible, +protected, or unknown verdict, and protected or unknown candidates are +reported and left in place. A prune that deletes nothing succeeds.`, RunE: func(cmd *cobra.Command, _ []string) error { - client, isRemote, err := GetRemoteClient() + client, _, err := resolveDaemonClient() if err != nil { return err } - if !isRemote { - return fmt.Errorf("images commands require a configured remote target") - } - return runImagesPrune(cmd.Context(), client, opts, cmd.OutOrStdout()) }, } - cmd.Flags().BoolVar(&opts.DryRun, "dry-run", false, "Show what would be pruned without applying changes") + cmd.Flags().BoolVar(&opts.DryRun, "dry-run", false, "Run the full inventory and planning, then report without deleting") cmd.Flags().IntVar(&opts.KeepReleases, "keep-releases", domain.DefaultImagePruneKeepLast, "Number of previous non-latest tags to keep per repository (latest is always kept)") cmd.Flags().BoolVar(&opts.Dangling, "dangling", false, "Include dangling runtime images (default: both scopes)") @@ -150,7 +145,7 @@ func newImagesTagsCmd() *cobra.Command { Examples: gordon images tags myapp gordon images tags myapp --json - gordon images tags myapp --remote https://gordon.mydomain.com --token $TOKEN`, + gordon images tags myapp --remote prod`, Args: cobra.ExactArgs(1), RunE: func(cmd *cobra.Command, args []string) error { ctx := cmd.Context() @@ -357,7 +352,7 @@ func runImagesPrune(ctx context.Context, client imagesClient, opts imagesPruneOp } } - req := buildPruneRequest(opts.KeepReleases, pruneDangling, pruneRegistry) + req := buildPruneRequest(opts.KeepReleases, pruneDangling, pruneRegistry, false) resp, err := pruneWithSpinner(ctx, client, req, pruneDangling, pruneRegistry) if err != nil { @@ -378,12 +373,14 @@ func runImagesPrune(ctx context.Context, client imagesClient, opts imagesPruneOp } if pruneRegistry { - err = cliWritef(out, "Registry: tags_removed=%d blobs_removed=%d space_reclaimed=%s\n", resp.Registry.TagsRemoved, resp.Registry.BlobsRemoved, bytesize.Format(resp.Registry.SpaceReclaimed)) + if err := cliWritef(out, "Registry: tags_removed=%d blobs_removed=%d space_reclaimed=%s\n", resp.Registry.TagsRemoved, resp.Registry.BlobsRemoved, bytesize.Format(resp.Registry.SpaceReclaimed)); err != nil { + return err + } + } else if err := cliWriteLine(out, cliRenderMuted("Registry cleanup skipped (--dangling)")); err != nil { return err } - err = cliWriteLine(out, cliRenderMuted("Registry cleanup skipped (--dangling)")) - return err + return cliRenderPruneSummary(out, resp.Plan) } func runImagesPruneDryRun(ctx context.Context, client imagesClient, opts imagesPruneOptions, pruneDangling, pruneRegistry bool, out io.Writer) error { @@ -391,33 +388,34 @@ func runImagesPruneDryRun(ctx context.Context, client imagesClient, opts imagesP return err } - if pruneDangling { - images, err := client.ListImages(ctx) - if err != nil { - return fmt.Errorf("failed to list images: %w", err) - } - - danglingCount := 0 - for _, img := range images { - if img.Dangling { - danglingCount++ - } - } + // A dry run executes the full inventory and planning path: only the + // deletion step is skipped, so the report is the executed plan. + req := buildPruneRequest(opts.KeepReleases, pruneDangling, pruneRegistry, true) + resp, err := client.PruneImages(ctx, req) + if err != nil { + return fmt.Errorf("failed to plan image prune: %w", err) + } + if resp == nil { + return fmt.Errorf("failed to plan image prune: empty response") + } - if err := cliWritef(out, "Runtime: would prune %d dangling runtime images\n", danglingCount); err != nil { - return err - } - } else { - if err := cliWriteLine(out, cliRenderMuted("Runtime cleanup skipped (--registry)")); err != nil { + if pruneDangling { + if err := cliWritef(out, "Runtime: would delete %d images\n", resp.Plan.CountByKind(domain.PruneResourceRuntimeImage)); err != nil { return err } + } else if err := cliWriteLine(out, cliRenderMuted("Runtime cleanup skipped (--registry)")); err != nil { + return err } if pruneRegistry { - return cliWritef(out, "Registry: would keep latest + %d previous tags per repository\n", opts.KeepReleases) + if err := cliWritef(out, "Registry: latest + %d previous tags per repository\n", opts.KeepReleases); err != nil { + return err + } + } else if err := cliWriteLine(out, cliRenderMuted("Registry cleanup skipped (--dangling)")); err != nil { + return err } - return cliWriteLine(out, cliRenderMuted("Registry cleanup skipped (--dangling)")) + return cliRenderPruneSummary(out, resp.Plan) } func loadPrunePreview(ctx context.Context, client imagesClient, keepReleases int, pruneDangling, pruneRegistry bool) (prunePreview, error) { @@ -561,11 +559,12 @@ func selectKeptTags(tagInfos []previewRegistryTag, keepReleases int) map[string] return kept } -func buildPruneRequest(keepReleases int, pruneDangling, pruneRegistry bool) dto.ImagePruneRequest { +func buildPruneRequest(keepReleases int, pruneDangling, pruneRegistry, dryRun bool) dto.ImagePruneRequest { return dto.ImagePruneRequest{ KeepLast: &keepReleases, PruneDangling: &pruneDangling, PruneRegistry: &pruneRegistry, + DryRun: &dryRun, } } diff --git a/internal/adapters/in/cli/images_test.go b/internal/adapters/in/cli/images_test.go index a7f8a1367..1cf109e5d 100644 --- a/internal/adapters/in/cli/images_test.go +++ b/internal/adapters/in/cli/images_test.go @@ -297,7 +297,18 @@ func TestRunImagesPrune_BothScopeFlagsExplicit(t *testing.T) { func TestRunImagesPrune_DryRunBothScopes(t *testing.T) { client := &imagesClientMock{ - listImagesResp: []dto.Image{{Dangling: true}, {Dangling: false}, {Dangling: true}}, + pruneResp: &dto.ImagePruneResponse{ + Plan: dto.PruneSummary{ + Applied: false, + Eligible: 2, + Protected: 1, + Candidates: []dto.PruneCandidate{ + {Kind: "runtime-image", Ref: "sha256:a", Verdict: "eligible"}, + {Kind: "runtime-image", Ref: "sha256:b", Verdict: "eligible"}, + {Kind: "registry-tag", Ref: "app:v1", Verdict: "protected", Reasons: []string{"protected-active-service"}}, + }, + }, + }, } var out bytes.Buffer @@ -307,15 +318,22 @@ func TestRunImagesPrune_DryRunBothScopes(t *testing.T) { text := out.String() assert.Contains(t, text, "Dry run") - assert.Contains(t, text, "would prune 2 dangling runtime images") - assert.Contains(t, text, "would keep latest + 3 previous tags") - assert.Equal(t, 1, client.listImagesCalls) - assert.Equal(t, 0, client.pruneCalls) + assert.Contains(t, text, "would delete 2 images") + assert.Contains(t, text, "latest + 3 previous tags") + assert.Contains(t, text, "Would delete: 2 (eligible=2 protected=1 unknown=0)") + assert.Contains(t, text, "protected-active-service") + assert.Equal(t, 1, client.pruneCalls) + assert.NotNil(t, client.lastPruneOpts.DryRun) + assert.True(t, *client.lastPruneOpts.DryRun) } func TestRunImagesPrune_DryRunDanglingOnlyScope(t *testing.T) { client := &imagesClientMock{ - listImagesResp: []dto.Image{{Dangling: true}}, + pruneResp: &dto.ImagePruneResponse{ + Plan: dto.PruneSummary{ + Candidates: []dto.PruneCandidate{{Kind: "runtime-image", Ref: "sha256:a", Verdict: "eligible"}}, + }, + }, } var out bytes.Buffer @@ -324,14 +342,18 @@ func TestRunImagesPrune_DryRunDanglingOnlyScope(t *testing.T) { require.NoError(t, err) text := out.String() - assert.Contains(t, text, "would prune 1 dangling runtime images") + assert.Contains(t, text, "would delete 1 images") assert.Contains(t, text, "Registry cleanup skipped") - assert.Equal(t, 0, client.pruneCalls) + assert.Equal(t, 1, client.pruneCalls) } func TestRunImagesPrune_DryRunRegistryOnlyScope(t *testing.T) { client := &imagesClientMock{ - listImagesResp: []dto.Image{{Dangling: true}, {Dangling: false}}, + pruneResp: &dto.ImagePruneResponse{ + Plan: dto.PruneSummary{ + Candidates: []dto.PruneCandidate{{Kind: "registry-tag", Ref: "app:v1", Verdict: "eligible"}}, + }, + }, } var out bytes.Buffer @@ -341,8 +363,8 @@ func TestRunImagesPrune_DryRunRegistryOnlyScope(t *testing.T) { text := out.String() assert.Contains(t, text, "Runtime cleanup skipped") - assert.Contains(t, text, "would keep latest + 5 previous tags") - assert.Equal(t, 0, client.pruneCalls) + assert.Contains(t, text, "latest + 5 previous tags") + assert.Equal(t, 1, client.pruneCalls) } // --------------------------------------------------------------------------- @@ -484,7 +506,7 @@ func TestRunImagesPrune_DryRunNeverPrompts(t *testing.T) { t.Cleanup(func() { pruneConfirmFunc = origConfirm }) client := &imagesClientMock{ - listImagesResp: []dto.Image{{Dangling: true}}, + pruneResp: &dto.ImagePruneResponse{Plan: dto.PruneSummary{}}, } var out bytes.Buffer @@ -493,6 +515,7 @@ func TestRunImagesPrune_DryRunNeverPrompts(t *testing.T) { require.NoError(t, err) assert.False(t, confirmCalled, "confirmation should not be called during dry-run") + assert.Equal(t, 1, client.pruneCalls) } // --------------------------------------------------------------------------- diff --git a/internal/adapters/in/cli/json_parity_test.go b/internal/adapters/in/cli/json_parity_test.go index f70ca0a33..b7dc81836 100644 --- a/internal/adapters/in/cli/json_parity_test.go +++ b/internal/adapters/in/cli/json_parity_test.go @@ -12,10 +12,6 @@ func TestAllListCommands_AcceptJSONFlag(t *testing.T) { name string builder func() *cobra.Command }{ - {"routes list", newRoutesListCmd}, - {"routes status", newRoutesStatusCmd}, - {"attachments list", newAttachmentsListCmd}, - {"secrets list", newSecretsListCmd}, {"images list", newImagesListCmd}, {"backup list", newBackupListCmd}, {"backup volumes list", newVolumeBackupListCmd}, @@ -23,12 +19,10 @@ func TestAllListCommands_AcceptJSONFlag(t *testing.T) { {"backup volumes status", newVolumeBackupStatusCmd}, {"auth token list", newTokenListCmd}, {"remotes list", newRemotesListCmd}, - {"pin list", newPinListCmd}, - {"config show", newConfigShowCmd}, - {"routes show", newRoutesShowCmd}, - {"networks list", newNetworksListCmd}, + {"daemon config show", newConfigShowCmd}, + {"daemon networks", newNetworksCmd}, {"images tags", newImagesTagsCmd}, - {"tls status", newTLSStatusCmd}, + {"daemon tls", newTLSCmd}, } for _, tc := range commands { diff --git a/internal/adapters/in/cli/json_test.go b/internal/adapters/in/cli/json_test.go index 2a6a6be88..dcaf9fe17 100644 --- a/internal/adapters/in/cli/json_test.go +++ b/internal/adapters/in/cli/json_test.go @@ -10,37 +10,9 @@ import ( "github.com/stretchr/testify/require" "github.com/bnema/gordon/internal/adapters/dto" - "github.com/bnema/gordon/internal/adapters/in/cli/remote" "github.com/bnema/gordon/internal/domain" ) -func TestRoutesList_JSONFlag_Accepted(t *testing.T) { - cmd := newRoutesListCmd() - f := cmd.Flags().Lookup("json") - assert.NotNil(t, f) - if f != nil { - assert.Equal(t, "false", f.DefValue) - } -} - -func TestAttachmentsList_JSONFlag_Accepted(t *testing.T) { - cmd := newAttachmentsListCmd() - f := cmd.Flags().Lookup("json") - assert.NotNil(t, f) - if f != nil { - assert.Equal(t, "false", f.DefValue) - } -} - -func TestSecretsList_JSONFlag_Accepted(t *testing.T) { - cmd := newSecretsListCmd() - f := cmd.Flags().Lookup("json") - assert.NotNil(t, f) - if f != nil { - assert.Equal(t, "false", f.DefValue) - } -} - func TestImagesList_JSONFlag_Accepted(t *testing.T) { cmd := newImagesListCmd() f := cmd.Flags().Lookup("json") @@ -77,15 +49,6 @@ func TestRemotesList_JSONFlag_Accepted(t *testing.T) { } } -func TestPinList_JSONFlag_Accepted(t *testing.T) { - cmd := newPinListCmd() - f := cmd.Flags().Lookup("json") - assert.NotNil(t, f) - if f != nil { - assert.Equal(t, "false", f.DefValue) - } -} - func TestImagesList_JSONShape_RoundTripsDTO(t *testing.T) { createdAt := time.Date(2026, 2, 8, 12, 0, 0, 0, time.UTC) images := []dto.Image{{Repository: "registry.example.com/app", Tag: "latest", Size: 12_000_000, Created: createdAt, ID: "sha256:1111", Dangling: false}} @@ -113,7 +76,7 @@ func TestPinList_JSONShape_RoundTripsTags(t *testing.T) { func TestBackupList_JSONShape_RoundTripsJobs(t *testing.T) { startedAt := time.Date(2026, 3, 1, 10, 0, 0, 0, time.UTC) - jobs := []domain.BackupJob{{ID: "b1", Domain: "app.example.com", DBName: "postgres", Status: domain.BackupStatusCompleted, StartedAt: startedAt}} + jobs := []domain.BackupJob{{ID: "b1", App: "shop", Service: "api", DBName: "orders", Status: domain.BackupStatusCompleted, StartedAt: startedAt}} payload, err := json.Marshal(jobs) require.NoError(t, err) @@ -122,7 +85,9 @@ func TestBackupList_JSONShape_RoundTripsJobs(t *testing.T) { require.NoError(t, json.Unmarshal(payload, &got)) require.Len(t, got, 1) assert.Equal(t, "b1", got[0].ID) - assert.Equal(t, "app.example.com", got[0].Domain) + assert.Equal(t, "shop", got[0].App) + assert.Equal(t, "api", got[0].Service) + assert.Equal(t, "orders", got[0].DBName) } func TestTokenList_JSONShape_RoundTripsTokens(t *testing.T) { @@ -172,159 +137,6 @@ func TestRemotesList_JSONShape_RoundTripsRemoteObjects(t *testing.T) { assert.True(t, got[0].InsecureTLS) } -func TestAttachmentsList_JSONShape_RoundTripsTargetPayload(t *testing.T) { - payload := struct { - Target string `json:"target"` - Images []string `json:"images"` - }{ - Target: "app.example.com", - Images: []string{"postgres:16"}, - } - - encoded, err := json.Marshal(payload) - require.NoError(t, err) - - var got struct { - Target string `json:"target"` - Images []string `json:"images"` - } - require.NoError(t, json.Unmarshal(encoded, &got)) - assert.Equal(t, payload.Target, got.Target) - assert.Equal(t, payload.Images, got.Images) -} - -func TestSecretsList_JSONShape_RoundTripsPayload(t *testing.T) { - payload := struct { - Domain string `json:"domain"` - Keys []string `json:"keys"` - Attachments []remote.AttachmentSecrets `json:"attachments"` - }{ - Domain: "app.example.com", - Keys: []string{"API_KEY"}, - Attachments: []remote.AttachmentSecrets{{ - Service: "postgres", - Keys: []string{"POSTGRES_PASSWORD"}, - }}, - } - - encoded, err := json.Marshal(payload) - require.NoError(t, err) - - var got struct { - Domain string `json:"domain"` - Keys []string `json:"keys"` - Attachments []remote.AttachmentSecrets `json:"attachments"` - } - require.NoError(t, json.Unmarshal(encoded, &got)) - assert.Equal(t, payload.Domain, got.Domain) - assert.Equal(t, payload.Keys, got.Keys) - assert.Equal(t, payload.Attachments, got.Attachments) -} - -func TestSecretsList_JSONShape_NormalizesNilAttachmentsToEmptySlice(t *testing.T) { - payload := struct { - Domain string `json:"domain"` - Keys []string `json:"keys"` - Attachments []remote.AttachmentSecrets `json:"attachments"` - }{ - Domain: "app.example.com", - Keys: []string{"API_KEY"}, - Attachments: []remote.AttachmentSecrets{}, - } - - encoded, err := json.Marshal(payload) - require.NoError(t, err) - assert.Contains(t, string(encoded), `"attachments":[]`) - - var got struct { - Domain string `json:"domain"` - Keys []string `json:"keys"` - Attachments []remote.AttachmentSecrets `json:"attachments"` - } - require.NoError(t, json.Unmarshal(encoded, &got)) - assert.NotNil(t, got.Attachments) - assert.Empty(t, got.Attachments) -} - -func TestRoutesList_JSONShape_RoundTripsLocalPayload(t *testing.T) { - payload := []struct { - Domain string `json:"domain"` - Image string `json:"image"` - }{ - {Domain: "app.example.com", Image: "app:latest"}, - } - - encoded, err := json.Marshal(payload) - require.NoError(t, err) - - var got []struct { - Domain string `json:"domain"` - Image string `json:"image"` - } - require.NoError(t, json.Unmarshal(encoded, &got)) - require.Len(t, got, 1) - assert.Equal(t, payload[0], got[0]) -} - -func TestRoutesStatus_JSONFlag_Accepted(t *testing.T) { - cmd := newRoutesStatusCmd() - f := cmd.Flags().Lookup("json") - assert.NotNil(t, f) - if f != nil { - assert.Equal(t, "false", f.DefValue) - } -} - -func TestRoutesList_JSONShape_RoundTripsSections(t *testing.T) { - payload := []routeListSection{{ - Kind: "local", - Name: "local", - Routes: []routeListItem{{ - Domain: "app.local", - Image: "myapp:latest", - }}, - }} - - encoded, err := json.Marshal(payload) - require.NoError(t, err) - - var got []routeListSection - require.NoError(t, json.Unmarshal(encoded, &got)) - require.Len(t, got, 1) - assert.Equal(t, "local", got[0].Kind) - assert.Equal(t, "app.local", got[0].Routes[0].Domain) -} - -func TestRoutesStatus_JSONShape_RoundTripsSections(t *testing.T) { - payload := []routeStatusSection{{ - Kind: "remote", - Name: "igor", - URL: "https://gordon.supri.xyz", - Routes: []routeStatusItem{{ - Domain: "grafana.supri.xyz", - Image: "grafana", - ContainerStatus: "running", - HTTPStatus: 200, - Network: "gordon-shared", - Attachments: []routeStatusAttachment{{ - Name: "prometheus", - Image: "prometheus:v5", - Status: "running", - }}, - }}, - }} - - encoded, err := json.Marshal(payload) - require.NoError(t, err) - - var got []routeStatusSection - require.NoError(t, json.Unmarshal(encoded, &got)) - require.Len(t, got, 1) - assert.Equal(t, "igor", got[0].Name) - assert.Equal(t, 200, got[0].Routes[0].HTTPStatus) - assert.Equal(t, "prometheus", got[0].Routes[0].Attachments[0].Name) -} - func TestWriteJSON_ProducesValidIndentedJSON(t *testing.T) { var out bytes.Buffer err := writeJSON(&out, map[string]any{"ok": true}) diff --git a/internal/adapters/in/cli/local.go b/internal/adapters/in/cli/local.go index 4c6693224..4a3e98a6b 100644 --- a/internal/adapters/in/cli/local.go +++ b/internal/adapters/in/cli/local.go @@ -4,25 +4,18 @@ package cli import ( "context" "fmt" - "path/filepath" "strings" - "github.com/bnema/zerowrap" "github.com/spf13/viper" - "github.com/bnema/gordon/internal/adapters/out/domainsecrets" "github.com/bnema/gordon/internal/app" "github.com/bnema/gordon/internal/boundaries/in" - "github.com/bnema/gordon/internal/boundaries/out" - "github.com/bnema/gordon/internal/domain" "github.com/bnema/gordon/internal/usecase/config" - secretsSvc "github.com/bnema/gordon/internal/usecase/secrets" ) // LocalServices provides direct access to local services for CLI operations. type LocalServices struct { configSvc in.ConfigService - secretSvc in.SecretService dataDir string tlsEnabled bool } @@ -32,11 +25,6 @@ func (l *LocalServices) GetConfigService() in.ConfigService { return l.configSvc } -// GetSecretService returns the secret service. -func (l *LocalServices) GetSecretService() in.SecretService { - return l.secretSvc -} - // GetDataDir returns the data directory. func (l *LocalServices) GetDataDir() string { return l.dataDir @@ -49,7 +37,7 @@ func (l *LocalServices) HasInternalTLS() bool { // GetLocalServices creates local services for CLI operations. // It loads the config and initializes services without starting the server. -func GetLocalServices(configPath string) (*LocalServices, error) { +func GetLocalServices(cliConfigPath string) (*LocalServices, error) { // Set up viper with defaults v := viper.New() v.SetDefault("server.port", 8088) @@ -57,7 +45,7 @@ func GetLocalServices(configPath string) (*LocalServices, error) { v.SetDefault("server.data_dir", app.DefaultDataDir()) // Configure viper with config file paths - app.ConfigureViper(v, configPath) + app.ConfigureViper(v, cliConfigPath) // Read config file if err := v.ReadInConfig(); err != nil { @@ -67,8 +55,6 @@ func GetLocalServices(configPath string) (*LocalServices, error) { // Config file not found is OK for some operations } - log := zerowrap.New(cliLogConfig) - // Create config service (without event bus for CLI operations) configSvc := config.NewService(v, nil) ctx := context.Background() @@ -76,27 +62,13 @@ func GetLocalServices(configPath string) (*LocalServices, error) { return nil, fmt.Errorf("failed to load configuration: %w", err) } - // Determine data directory and env directory dataDir := v.GetString("server.data_dir") if dataDir == "" { dataDir = app.DefaultDataDir() } - envDir := v.GetString("env.dir") - if envDir == "" { - envDir = filepath.Join(dataDir, "env") - } - - // Create domain secret store using configured backend (pass or file-based) - domainSecretStore, err := createLocalDomainSecretStore(v, envDir, log) - if err != nil { - return nil, fmt.Errorf("failed to create domain secret store: %w", err) - } - secretSvc := secretsSvc.NewService(domainSecretStore, log, nil) - return &LocalServices{ configSvc: configSvc, - secretSvc: secretSvc, dataDir: dataDir, tlsEnabled: hasLocalTLSCapableEntrypoint(v), }, nil @@ -115,36 +87,3 @@ func hasLocalTLSCapableEntrypoint(v *viper.Viper) bool { } return false } - -func createLocalDomainSecretStore(v *viper.Viper, envDir string, log zerowrap.Logger) (out.DomainSecretStore, error) { - backend := resolveLocalSecretsBackend(v) - switch backend { - case domain.SecretsBackendPass: - return domainsecrets.NewPassStore(log) - case domain.SecretsBackendSops: - return nil, fmt.Errorf("sops backend not yet supported for domain secrets") - default: - return domainsecrets.NewFileStore(envDir, log) - } -} - -func resolveLocalSecretsBackend(v *viper.Viper) domain.SecretsBackend { - backend := strings.TrimSpace(v.GetString("auth.secrets_backend")) - if backend == "" { - // Legacy key support for older configs. - backend = strings.TrimSpace(v.GetString("secrets.backend")) - } - - switch backend { - case "pass": - return domain.SecretsBackendPass - case "sops": - return domain.SecretsBackendSops - case "unsafe", "": - return domain.SecretsBackendUnsafe - default: - log := zerowrap.Default() - log.Warn().Str("backend", backend).Msg("unrecognized secrets backend, falling back to unsafe") - return domain.SecretsBackendUnsafe - } -} diff --git a/internal/adapters/in/cli/local_test.go b/internal/adapters/in/cli/local_test.go index 4091ed948..177cfa32a 100644 --- a/internal/adapters/in/cli/local_test.go +++ b/internal/adapters/in/cli/local_test.go @@ -3,41 +3,10 @@ package cli import ( "testing" - "github.com/bnema/zerowrap" "github.com/spf13/viper" "github.com/stretchr/testify/require" - - "github.com/bnema/gordon/internal/adapters/out/domainsecrets" - "github.com/bnema/gordon/internal/domain" ) -func TestResolveLocalSecretsBackend(t *testing.T) { - t.Run("auth backend takes precedence", func(t *testing.T) { - v := viper.New() - v.Set("auth.secrets_backend", "pass") - v.Set("secrets.backend", "unsafe") - - got := resolveLocalSecretsBackend(v) - require.Equal(t, domain.SecretsBackendPass, got) - }) - - t.Run("falls back to legacy secrets backend key", func(t *testing.T) { - v := viper.New() - v.Set("secrets.backend", "pass") - - got := resolveLocalSecretsBackend(v) - require.Equal(t, domain.SecretsBackendPass, got) - }) - - t.Run("defaults to unsafe for unknown backend", func(t *testing.T) { - v := viper.New() - v.Set("auth.secrets_backend", "wat") - - got := resolveLocalSecretsBackend(v) - require.Equal(t, domain.SecretsBackendUnsafe, got) - }) -} - func TestHasLocalTLSCapableEntrypoint(t *testing.T) { for _, tt := range []struct { name string @@ -56,38 +25,3 @@ func TestHasLocalTLSCapableEntrypoint(t *testing.T) { }) } } - -func TestCreateLocalDomainSecretStore_UsesPassStoreForPass(t *testing.T) { - // pass(1) must be available on the system for this test - if _, err := domainsecrets.NewPassStore(zerowrap.New(zerowrap.Config{Level: "error", Format: "console"})); err != nil { - t.Skipf("pass store not available: %v", err) - } - - v := viper.New() - v.Set("auth.secrets_backend", "pass") - - log := zerowrap.New(zerowrap.Config{Level: "error", Format: "console"}) - store, err := createLocalDomainSecretStore(v, t.TempDir(), log) - require.NoError(t, err) - require.IsType(t, &domainsecrets.PassStore{}, store) -} - -func TestCreateLocalDomainSecretStore_RejectsSopsBackend(t *testing.T) { - v := viper.New() - v.Set("auth.secrets_backend", "sops") - - log := zerowrap.New(zerowrap.Config{Level: "error", Format: "console"}) - _, err := createLocalDomainSecretStore(v, t.TempDir(), log) - require.Error(t, err) - require.Contains(t, err.Error(), "sops backend not yet supported") -} - -func TestCreateLocalDomainSecretStore_UsesFileStoreForUnsafe(t *testing.T) { - v := viper.New() - v.Set("auth.secrets_backend", "unsafe") - - log := zerowrap.New(zerowrap.Config{Level: "error", Format: "console"}) - store, err := createLocalDomainSecretStore(v, t.TempDir(), log) - require.NoError(t, err) - require.IsType(t, &domainsecrets.FileStore{}, store) -} diff --git a/internal/adapters/in/cli/mocks/mock_control_plane.go b/internal/adapters/in/cli/mocks/mock_control_plane.go index 180462135..855551af1 100644 --- a/internal/adapters/in/cli/mocks/mock_control_plane.go +++ b/internal/adapters/in/cli/mocks/mock_control_plane.go @@ -19,10 +19,19 @@ func NewMockControlPlane(t interface { mock.TestingT Cleanup(func()) }) *MockControlPlane { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockControlPlane{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } @@ -40,164 +49,55 @@ func (_m *MockControlPlane) EXPECT() *MockControlPlane_Expecter { return &MockControlPlane_Expecter{mock: &_m.Mock} } -// AddAttachment provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) AddAttachment(ctx context.Context, domainOrGroup string, image string) error { - ret := _mock.Called(ctx, domainOrGroup, image) +// ApplyApp provides a mock function for the type MockControlPlane +func (_mock *MockControlPlane) ApplyApp(ctx context.Context, req dto.AppApplyRequest) (*dto.AppApplyResponse, error) { + ret := _mock.Called(ctx, req) if len(ret) == 0 { - panic("no return value specified for AddAttachment") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) error); ok { - r0 = returnFunc(ctx, domainOrGroup, image) - } else { - r0 = ret.Error(0) + panic("no return value specified for ApplyApp") } - return r0 -} - -// MockControlPlane_AddAttachment_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'AddAttachment' -type MockControlPlane_AddAttachment_Call struct { - *mock.Call -} - -// AddAttachment is a helper method to define mock.On call -// - ctx context.Context -// - domainOrGroup string -// - image string -func (_e *MockControlPlane_Expecter) AddAttachment(ctx any, domainOrGroup any, image any) *MockControlPlane_AddAttachment_Call { - return &MockControlPlane_AddAttachment_Call{Call: _e.mock.On("AddAttachment", ctx, domainOrGroup, image)} -} - -func (_c *MockControlPlane_AddAttachment_Call) Run(run func(ctx context.Context, domainOrGroup string, image string)) *MockControlPlane_AddAttachment_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - var arg2 string - if args[2] != nil { - arg2 = args[2].(string) - } - run( - arg0, - arg1, - arg2, - ) - }) - return _c -} - -func (_c *MockControlPlane_AddAttachment_Call) Return(err error) *MockControlPlane_AddAttachment_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockControlPlane_AddAttachment_Call) RunAndReturn(run func(ctx context.Context, domainOrGroup string, image string) error) *MockControlPlane_AddAttachment_Call { - _c.Call.Return(run) - return _c -} -// AddAutoRouteAllowedDomain provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) AddAutoRouteAllowedDomain(ctx context.Context, pattern string) error { - ret := _mock.Called(ctx, pattern) - - if len(ret) == 0 { - panic("no return value specified for AddAutoRouteAllowedDomain") + var r0 *dto.AppApplyResponse + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, dto.AppApplyRequest) (*dto.AppApplyResponse, error)); ok { + return returnFunc(ctx, req) } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) error); ok { - r0 = returnFunc(ctx, pattern) + if returnFunc, ok := ret.Get(0).(func(context.Context, dto.AppApplyRequest) *dto.AppApplyResponse); ok { + r0 = returnFunc(ctx, req) } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockControlPlane_AddAutoRouteAllowedDomain_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'AddAutoRouteAllowedDomain' -type MockControlPlane_AddAutoRouteAllowedDomain_Call struct { - *mock.Call -} - -// AddAutoRouteAllowedDomain is a helper method to define mock.On call -// - ctx context.Context -// - pattern string -func (_e *MockControlPlane_Expecter) AddAutoRouteAllowedDomain(ctx any, pattern any) *MockControlPlane_AddAutoRouteAllowedDomain_Call { - return &MockControlPlane_AddAutoRouteAllowedDomain_Call{Call: _e.mock.On("AddAutoRouteAllowedDomain", ctx, pattern)} -} - -func (_c *MockControlPlane_AddAutoRouteAllowedDomain_Call) Run(run func(ctx context.Context, pattern string)) *MockControlPlane_AddAutoRouteAllowedDomain_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) + if ret.Get(0) != nil { + r0 = ret.Get(0).(*dto.AppApplyResponse) } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockControlPlane_AddAutoRouteAllowedDomain_Call) Return(err error) *MockControlPlane_AddAutoRouteAllowedDomain_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockControlPlane_AddAutoRouteAllowedDomain_Call) RunAndReturn(run func(ctx context.Context, pattern string) error) *MockControlPlane_AddAutoRouteAllowedDomain_Call { - _c.Call.Return(run) - return _c -} - -// AddRoute provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) AddRoute(ctx context.Context, route domain.Route) error { - ret := _mock.Called(ctx, route) - - if len(ret) == 0 { - panic("no return value specified for AddRoute") } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, domain.Route) error); ok { - r0 = returnFunc(ctx, route) + if returnFunc, ok := ret.Get(1).(func(context.Context, dto.AppApplyRequest) error); ok { + r1 = returnFunc(ctx, req) } else { - r0 = ret.Error(0) + r1 = ret.Error(1) } - return r0 + return r0, r1 } -// MockControlPlane_AddRoute_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'AddRoute' -type MockControlPlane_AddRoute_Call struct { +// MockControlPlane_ApplyApp_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ApplyApp' +type MockControlPlane_ApplyApp_Call struct { *mock.Call } -// AddRoute is a helper method to define mock.On call +// ApplyApp is a helper method to define mock.On call // - ctx context.Context -// - route domain.Route -func (_e *MockControlPlane_Expecter) AddRoute(ctx any, route any) *MockControlPlane_AddRoute_Call { - return &MockControlPlane_AddRoute_Call{Call: _e.mock.On("AddRoute", ctx, route)} +// - req dto.AppApplyRequest +func (_e *MockControlPlane_Expecter) ApplyApp(ctx any, req any) *MockControlPlane_ApplyApp_Call { + return &MockControlPlane_ApplyApp_Call{Call: _e.mock.On("ApplyApp", ctx, req)} } -func (_c *MockControlPlane_AddRoute_Call) Run(run func(ctx context.Context, route domain.Route)) *MockControlPlane_AddRoute_Call { +func (_c *MockControlPlane_ApplyApp_Call) Run(run func(ctx context.Context, req dto.AppApplyRequest)) *MockControlPlane_ApplyApp_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { arg0 = args[0].(context.Context) } - var arg1 domain.Route + var arg1 dto.AppApplyRequest if args[1] != nil { - arg1 = args[1].(domain.Route) + arg1 = args[1].(dto.AppApplyRequest) } run( arg0, @@ -207,12 +107,12 @@ func (_c *MockControlPlane_AddRoute_Call) Run(run func(ctx context.Context, rout return _c } -func (_c *MockControlPlane_AddRoute_Call) Return(err error) *MockControlPlane_AddRoute_Call { - _c.Call.Return(err) +func (_c *MockControlPlane_ApplyApp_Call) Return(appApplyResponse *dto.AppApplyResponse, err error) *MockControlPlane_ApplyApp_Call { + _c.Call.Return(appApplyResponse, err) return _c } -func (_c *MockControlPlane_AddRoute_Call) RunAndReturn(run func(ctx context.Context, route domain.Route) error) *MockControlPlane_AddRoute_Call { +func (_c *MockControlPlane_ApplyApp_Call) RunAndReturn(run func(ctx context.Context, req dto.AppApplyRequest) (*dto.AppApplyResponse, error)) *MockControlPlane_ApplyApp_Call { _c.Call.Return(run) return _c } @@ -279,248 +179,37 @@ func (_c *MockControlPlane_BackupStatus_Call) RunAndReturn(run func(ctx context. return _c } -// Bootstrap provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) Bootstrap(ctx context.Context, req dto.BootstrapRequest) (*dto.BootstrapResponse, error) { - ret := _mock.Called(ctx, req) - - if len(ret) == 0 { - panic("no return value specified for Bootstrap") - } - - var r0 *dto.BootstrapResponse - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, dto.BootstrapRequest) (*dto.BootstrapResponse, error)); ok { - return returnFunc(ctx, req) - } - if returnFunc, ok := ret.Get(0).(func(context.Context, dto.BootstrapRequest) *dto.BootstrapResponse); ok { - r0 = returnFunc(ctx, req) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).(*dto.BootstrapResponse) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context, dto.BootstrapRequest) error); ok { - r1 = returnFunc(ctx, req) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockControlPlane_Bootstrap_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Bootstrap' -type MockControlPlane_Bootstrap_Call struct { - *mock.Call -} - -// Bootstrap is a helper method to define mock.On call -// - ctx context.Context -// - req dto.BootstrapRequest -func (_e *MockControlPlane_Expecter) Bootstrap(ctx any, req any) *MockControlPlane_Bootstrap_Call { - return &MockControlPlane_Bootstrap_Call{Call: _e.mock.On("Bootstrap", ctx, req)} -} - -func (_c *MockControlPlane_Bootstrap_Call) Run(run func(ctx context.Context, req dto.BootstrapRequest)) *MockControlPlane_Bootstrap_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 dto.BootstrapRequest - if args[1] != nil { - arg1 = args[1].(dto.BootstrapRequest) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockControlPlane_Bootstrap_Call) Return(bootstrapResponse *dto.BootstrapResponse, err error) *MockControlPlane_Bootstrap_Call { - _c.Call.Return(bootstrapResponse, err) - return _c -} - -func (_c *MockControlPlane_Bootstrap_Call) RunAndReturn(run func(ctx context.Context, req dto.BootstrapRequest) (*dto.BootstrapResponse, error)) *MockControlPlane_Bootstrap_Call { - _c.Call.Return(run) - return _c -} - -// CleanupOrphanedAttachments provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) CleanupOrphanedAttachments(ctx context.Context, owner string, stop bool) (*domain.CleanupReport, error) { - ret := _mock.Called(ctx, owner, stop) - - if len(ret) == 0 { - panic("no return value specified for CleanupOrphanedAttachments") - } - - var r0 *domain.CleanupReport - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, bool) (*domain.CleanupReport, error)); ok { - return returnFunc(ctx, owner, stop) - } - if returnFunc, ok := ret.Get(0).(func(context.Context, string, bool) *domain.CleanupReport); ok { - r0 = returnFunc(ctx, owner, stop) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).(*domain.CleanupReport) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context, string, bool) error); ok { - r1 = returnFunc(ctx, owner, stop) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockControlPlane_CleanupOrphanedAttachments_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'CleanupOrphanedAttachments' -type MockControlPlane_CleanupOrphanedAttachments_Call struct { - *mock.Call -} - -// CleanupOrphanedAttachments is a helper method to define mock.On call -// - ctx context.Context -// - owner string -// - stop bool -func (_e *MockControlPlane_Expecter) CleanupOrphanedAttachments(ctx any, owner any, stop any) *MockControlPlane_CleanupOrphanedAttachments_Call { - return &MockControlPlane_CleanupOrphanedAttachments_Call{Call: _e.mock.On("CleanupOrphanedAttachments", ctx, owner, stop)} -} - -func (_c *MockControlPlane_CleanupOrphanedAttachments_Call) Run(run func(ctx context.Context, owner string, stop bool)) *MockControlPlane_CleanupOrphanedAttachments_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - var arg2 bool - if args[2] != nil { - arg2 = args[2].(bool) - } - run( - arg0, - arg1, - arg2, - ) - }) - return _c -} - -func (_c *MockControlPlane_CleanupOrphanedAttachments_Call) Return(cleanupReport *domain.CleanupReport, err error) *MockControlPlane_CleanupOrphanedAttachments_Call { - _c.Call.Return(cleanupReport, err) - return _c -} - -func (_c *MockControlPlane_CleanupOrphanedAttachments_Call) RunAndReturn(run func(ctx context.Context, owner string, stop bool) (*domain.CleanupReport, error)) *MockControlPlane_CleanupOrphanedAttachments_Call { - _c.Call.Return(run) - return _c -} - -// DeleteAttachmentSecret provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) DeleteAttachmentSecret(ctx context.Context, domain1 string, service string, key string) error { - ret := _mock.Called(ctx, domain1, service, key) - - if len(ret) == 0 { - panic("no return value specified for DeleteAttachmentSecret") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, string) error); ok { - r0 = returnFunc(ctx, domain1, service, key) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockControlPlane_DeleteAttachmentSecret_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'DeleteAttachmentSecret' -type MockControlPlane_DeleteAttachmentSecret_Call struct { - *mock.Call -} - -// DeleteAttachmentSecret is a helper method to define mock.On call -// - ctx context.Context -// - domain1 string -// - service string -// - key string -func (_e *MockControlPlane_Expecter) DeleteAttachmentSecret(ctx any, domain1 any, service any, key any) *MockControlPlane_DeleteAttachmentSecret_Call { - return &MockControlPlane_DeleteAttachmentSecret_Call{Call: _e.mock.On("DeleteAttachmentSecret", ctx, domain1, service, key)} -} - -func (_c *MockControlPlane_DeleteAttachmentSecret_Call) Run(run func(ctx context.Context, domain1 string, service string, key string)) *MockControlPlane_DeleteAttachmentSecret_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - var arg2 string - if args[2] != nil { - arg2 = args[2].(string) - } - var arg3 string - if args[3] != nil { - arg3 = args[3].(string) - } - run( - arg0, - arg1, - arg2, - arg3, - ) - }) - return _c -} - -func (_c *MockControlPlane_DeleteAttachmentSecret_Call) Return(err error) *MockControlPlane_DeleteAttachmentSecret_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockControlPlane_DeleteAttachmentSecret_Call) RunAndReturn(run func(ctx context.Context, domain1 string, service string, key string) error) *MockControlPlane_DeleteAttachmentSecret_Call { - _c.Call.Return(run) - return _c -} - -// DeleteSecret provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) DeleteSecret(ctx context.Context, secretDomain string, key string) error { - ret := _mock.Called(ctx, secretDomain, key) +// DeleteAppSecret provides a mock function for the type MockControlPlane +func (_mock *MockControlPlane) DeleteAppSecret(ctx context.Context, app string, req dto.AppSecretDeleteRequest) error { + ret := _mock.Called(ctx, app, req) if len(ret) == 0 { - panic("no return value specified for DeleteSecret") + panic("no return value specified for DeleteAppSecret") } var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) error); ok { - r0 = returnFunc(ctx, secretDomain, key) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, dto.AppSecretDeleteRequest) error); ok { + r0 = returnFunc(ctx, app, req) } else { r0 = ret.Error(0) } return r0 } -// MockControlPlane_DeleteSecret_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'DeleteSecret' -type MockControlPlane_DeleteSecret_Call struct { +// MockControlPlane_DeleteAppSecret_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'DeleteAppSecret' +type MockControlPlane_DeleteAppSecret_Call struct { *mock.Call } -// DeleteSecret is a helper method to define mock.On call +// DeleteAppSecret is a helper method to define mock.On call // - ctx context.Context -// - secretDomain string -// - key string -func (_e *MockControlPlane_Expecter) DeleteSecret(ctx any, secretDomain any, key any) *MockControlPlane_DeleteSecret_Call { - return &MockControlPlane_DeleteSecret_Call{Call: _e.mock.On("DeleteSecret", ctx, secretDomain, key)} +// - app string +// - req dto.AppSecretDeleteRequest +func (_e *MockControlPlane_Expecter) DeleteAppSecret(ctx any, app any, req any) *MockControlPlane_DeleteAppSecret_Call { + return &MockControlPlane_DeleteAppSecret_Call{Call: _e.mock.On("DeleteAppSecret", ctx, app, req)} } -func (_c *MockControlPlane_DeleteSecret_Call) Run(run func(ctx context.Context, secretDomain string, key string)) *MockControlPlane_DeleteSecret_Call { +func (_c *MockControlPlane_DeleteAppSecret_Call) Run(run func(ctx context.Context, app string, req dto.AppSecretDeleteRequest)) *MockControlPlane_DeleteAppSecret_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -530,9 +219,9 @@ func (_c *MockControlPlane_DeleteSecret_Call) Run(run func(ctx context.Context, if args[1] != nil { arg1 = args[1].(string) } - var arg2 string + var arg2 dto.AppSecretDeleteRequest if args[2] != nil { - arg2 = args[2].(string) + arg2 = args[2].(dto.AppSecretDeleteRequest) } run( arg0, @@ -543,641 +232,137 @@ func (_c *MockControlPlane_DeleteSecret_Call) Run(run func(ctx context.Context, return _c } -func (_c *MockControlPlane_DeleteSecret_Call) Return(err error) *MockControlPlane_DeleteSecret_Call { +func (_c *MockControlPlane_DeleteAppSecret_Call) Return(err error) *MockControlPlane_DeleteAppSecret_Call { _c.Call.Return(err) return _c } -func (_c *MockControlPlane_DeleteSecret_Call) RunAndReturn(run func(ctx context.Context, secretDomain string, key string) error) *MockControlPlane_DeleteSecret_Call { +func (_c *MockControlPlane_DeleteAppSecret_Call) RunAndReturn(run func(ctx context.Context, app string, req dto.AppSecretDeleteRequest) error) *MockControlPlane_DeleteAppSecret_Call { _c.Call.Return(run) return _c } -// Deploy provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) Deploy(ctx context.Context, deployDomain string) (*remote.DeployResult, error) { - ret := _mock.Called(ctx, deployDomain) +// DeployApp provides a mock function for the type MockControlPlane +func (_mock *MockControlPlane) DeployApp(ctx context.Context, app string, req dto.AppDeployRequest) (*dto.AppDeployResponse, string, error) { + ret := _mock.Called(ctx, app, req) if len(ret) == 0 { - panic("no return value specified for Deploy") + panic("no return value specified for DeployApp") } - var r0 *remote.DeployResult - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) (*remote.DeployResult, error)); ok { - return returnFunc(ctx, deployDomain) - } - if returnFunc, ok := ret.Get(0).(func(context.Context, string) *remote.DeployResult); ok { - r0 = returnFunc(ctx, deployDomain) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).(*remote.DeployResult) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { - r1 = returnFunc(ctx, deployDomain) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockControlPlane_Deploy_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Deploy' -type MockControlPlane_Deploy_Call struct { - *mock.Call -} - -// Deploy is a helper method to define mock.On call -// - ctx context.Context -// - deployDomain string -func (_e *MockControlPlane_Expecter) Deploy(ctx any, deployDomain any) *MockControlPlane_Deploy_Call { - return &MockControlPlane_Deploy_Call{Call: _e.mock.On("Deploy", ctx, deployDomain)} -} - -func (_c *MockControlPlane_Deploy_Call) Run(run func(ctx context.Context, deployDomain string)) *MockControlPlane_Deploy_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockControlPlane_Deploy_Call) Return(deployResult *remote.DeployResult, err error) *MockControlPlane_Deploy_Call { - _c.Call.Return(deployResult, err) - return _c -} - -func (_c *MockControlPlane_Deploy_Call) RunAndReturn(run func(ctx context.Context, deployDomain string) (*remote.DeployResult, error)) *MockControlPlane_Deploy_Call { - _c.Call.Return(run) - return _c -} - -// DeployIntent provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) DeployIntent(ctx context.Context, imageName string) error { - ret := _mock.Called(ctx, imageName) - - if len(ret) == 0 { - panic("no return value specified for DeployIntent") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) error); ok { - r0 = returnFunc(ctx, imageName) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockControlPlane_DeployIntent_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'DeployIntent' -type MockControlPlane_DeployIntent_Call struct { - *mock.Call -} - -// DeployIntent is a helper method to define mock.On call -// - ctx context.Context -// - imageName string -func (_e *MockControlPlane_Expecter) DeployIntent(ctx any, imageName any) *MockControlPlane_DeployIntent_Call { - return &MockControlPlane_DeployIntent_Call{Call: _e.mock.On("DeployIntent", ctx, imageName)} -} - -func (_c *MockControlPlane_DeployIntent_Call) Run(run func(ctx context.Context, imageName string)) *MockControlPlane_DeployIntent_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockControlPlane_DeployIntent_Call) Return(err error) *MockControlPlane_DeployIntent_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockControlPlane_DeployIntent_Call) RunAndReturn(run func(ctx context.Context, imageName string) error) *MockControlPlane_DeployIntent_Call { - _c.Call.Return(run) - return _c -} - -// DetectDatabases provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) DetectDatabases(ctx context.Context, backupDomain string) ([]dto.DatabaseInfo, error) { - ret := _mock.Called(ctx, backupDomain) - - if len(ret) == 0 { - panic("no return value specified for DetectDatabases") - } - - var r0 []dto.DatabaseInfo - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) ([]dto.DatabaseInfo, error)); ok { - return returnFunc(ctx, backupDomain) - } - if returnFunc, ok := ret.Get(0).(func(context.Context, string) []dto.DatabaseInfo); ok { - r0 = returnFunc(ctx, backupDomain) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]dto.DatabaseInfo) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { - r1 = returnFunc(ctx, backupDomain) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockControlPlane_DetectDatabases_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'DetectDatabases' -type MockControlPlane_DetectDatabases_Call struct { - *mock.Call -} - -// DetectDatabases is a helper method to define mock.On call -// - ctx context.Context -// - backupDomain string -func (_e *MockControlPlane_Expecter) DetectDatabases(ctx any, backupDomain any) *MockControlPlane_DetectDatabases_Call { - return &MockControlPlane_DetectDatabases_Call{Call: _e.mock.On("DetectDatabases", ctx, backupDomain)} -} - -func (_c *MockControlPlane_DetectDatabases_Call) Run(run func(ctx context.Context, backupDomain string)) *MockControlPlane_DetectDatabases_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockControlPlane_DetectDatabases_Call) Return(databaseInfos []dto.DatabaseInfo, err error) *MockControlPlane_DetectDatabases_Call { - _c.Call.Return(databaseInfos, err) - return _c -} - -func (_c *MockControlPlane_DetectDatabases_Call) RunAndReturn(run func(ctx context.Context, backupDomain string) ([]dto.DatabaseInfo, error)) *MockControlPlane_DetectDatabases_Call { - _c.Call.Return(run) - return _c -} - -// FindAttachmentTargetsByImage provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) FindAttachmentTargetsByImage(ctx context.Context, imageName string) ([]string, error) { - ret := _mock.Called(ctx, imageName) - - if len(ret) == 0 { - panic("no return value specified for FindAttachmentTargetsByImage") - } - - var r0 []string - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) ([]string, error)); ok { - return returnFunc(ctx, imageName) - } - if returnFunc, ok := ret.Get(0).(func(context.Context, string) []string); ok { - r0 = returnFunc(ctx, imageName) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]string) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { - r1 = returnFunc(ctx, imageName) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockControlPlane_FindAttachmentTargetsByImage_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'FindAttachmentTargetsByImage' -type MockControlPlane_FindAttachmentTargetsByImage_Call struct { - *mock.Call -} - -// FindAttachmentTargetsByImage is a helper method to define mock.On call -// - ctx context.Context -// - imageName string -func (_e *MockControlPlane_Expecter) FindAttachmentTargetsByImage(ctx any, imageName any) *MockControlPlane_FindAttachmentTargetsByImage_Call { - return &MockControlPlane_FindAttachmentTargetsByImage_Call{Call: _e.mock.On("FindAttachmentTargetsByImage", ctx, imageName)} -} - -func (_c *MockControlPlane_FindAttachmentTargetsByImage_Call) Run(run func(ctx context.Context, imageName string)) *MockControlPlane_FindAttachmentTargetsByImage_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockControlPlane_FindAttachmentTargetsByImage_Call) Return(strings []string, err error) *MockControlPlane_FindAttachmentTargetsByImage_Call { - _c.Call.Return(strings, err) - return _c -} - -func (_c *MockControlPlane_FindAttachmentTargetsByImage_Call) RunAndReturn(run func(ctx context.Context, imageName string) ([]string, error)) *MockControlPlane_FindAttachmentTargetsByImage_Call { - _c.Call.Return(run) - return _c -} - -// FindRoutesByImage provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) FindRoutesByImage(ctx context.Context, imageName string) ([]domain.Route, error) { - ret := _mock.Called(ctx, imageName) - - if len(ret) == 0 { - panic("no return value specified for FindRoutesByImage") - } - - var r0 []domain.Route - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) ([]domain.Route, error)); ok { - return returnFunc(ctx, imageName) - } - if returnFunc, ok := ret.Get(0).(func(context.Context, string) []domain.Route); ok { - r0 = returnFunc(ctx, imageName) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]domain.Route) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { - r1 = returnFunc(ctx, imageName) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockControlPlane_FindRoutesByImage_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'FindRoutesByImage' -type MockControlPlane_FindRoutesByImage_Call struct { - *mock.Call -} - -// FindRoutesByImage is a helper method to define mock.On call -// - ctx context.Context -// - imageName string -func (_e *MockControlPlane_Expecter) FindRoutesByImage(ctx any, imageName any) *MockControlPlane_FindRoutesByImage_Call { - return &MockControlPlane_FindRoutesByImage_Call{Call: _e.mock.On("FindRoutesByImage", ctx, imageName)} -} - -func (_c *MockControlPlane_FindRoutesByImage_Call) Run(run func(ctx context.Context, imageName string)) *MockControlPlane_FindRoutesByImage_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockControlPlane_FindRoutesByImage_Call) Return(routes []domain.Route, err error) *MockControlPlane_FindRoutesByImage_Call { - _c.Call.Return(routes, err) - return _c -} - -func (_c *MockControlPlane_FindRoutesByImage_Call) RunAndReturn(run func(ctx context.Context, imageName string) ([]domain.Route, error)) *MockControlPlane_FindRoutesByImage_Call { - _c.Call.Return(run) - return _c -} - -// GetAllAttachmentsConfig provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) GetAllAttachmentsConfig(ctx context.Context) (map[string][]string, error) { - ret := _mock.Called(ctx) - - if len(ret) == 0 { - panic("no return value specified for GetAllAttachmentsConfig") - } - - var r0 map[string][]string - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context) (map[string][]string, error)); ok { - return returnFunc(ctx) - } - if returnFunc, ok := ret.Get(0).(func(context.Context) map[string][]string); ok { - r0 = returnFunc(ctx) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).(map[string][]string) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { - r1 = returnFunc(ctx) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockControlPlane_GetAllAttachmentsConfig_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetAllAttachmentsConfig' -type MockControlPlane_GetAllAttachmentsConfig_Call struct { - *mock.Call -} - -// GetAllAttachmentsConfig is a helper method to define mock.On call -// - ctx context.Context -func (_e *MockControlPlane_Expecter) GetAllAttachmentsConfig(ctx any) *MockControlPlane_GetAllAttachmentsConfig_Call { - return &MockControlPlane_GetAllAttachmentsConfig_Call{Call: _e.mock.On("GetAllAttachmentsConfig", ctx)} -} - -func (_c *MockControlPlane_GetAllAttachmentsConfig_Call) Run(run func(ctx context.Context)) *MockControlPlane_GetAllAttachmentsConfig_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - run( - arg0, - ) - }) - return _c -} - -func (_c *MockControlPlane_GetAllAttachmentsConfig_Call) Return(stringToStrings map[string][]string, err error) *MockControlPlane_GetAllAttachmentsConfig_Call { - _c.Call.Return(stringToStrings, err) - return _c -} - -func (_c *MockControlPlane_GetAllAttachmentsConfig_Call) RunAndReturn(run func(ctx context.Context) (map[string][]string, error)) *MockControlPlane_GetAllAttachmentsConfig_Call { - _c.Call.Return(run) - return _c -} - -// GetAttachmentsConfig provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) GetAttachmentsConfig(ctx context.Context, domainOrGroup string) ([]string, error) { - ret := _mock.Called(ctx, domainOrGroup) - - if len(ret) == 0 { - panic("no return value specified for GetAttachmentsConfig") - } - - var r0 []string - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) ([]string, error)); ok { - return returnFunc(ctx, domainOrGroup) - } - if returnFunc, ok := ret.Get(0).(func(context.Context, string) []string); ok { - r0 = returnFunc(ctx, domainOrGroup) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]string) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { - r1 = returnFunc(ctx, domainOrGroup) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockControlPlane_GetAttachmentsConfig_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetAttachmentsConfig' -type MockControlPlane_GetAttachmentsConfig_Call struct { - *mock.Call -} - -// GetAttachmentsConfig is a helper method to define mock.On call -// - ctx context.Context -// - domainOrGroup string -func (_e *MockControlPlane_Expecter) GetAttachmentsConfig(ctx any, domainOrGroup any) *MockControlPlane_GetAttachmentsConfig_Call { - return &MockControlPlane_GetAttachmentsConfig_Call{Call: _e.mock.On("GetAttachmentsConfig", ctx, domainOrGroup)} -} - -func (_c *MockControlPlane_GetAttachmentsConfig_Call) Run(run func(ctx context.Context, domainOrGroup string)) *MockControlPlane_GetAttachmentsConfig_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockControlPlane_GetAttachmentsConfig_Call) Return(strings []string, err error) *MockControlPlane_GetAttachmentsConfig_Call { - _c.Call.Return(strings, err) - return _c -} - -func (_c *MockControlPlane_GetAttachmentsConfig_Call) RunAndReturn(run func(ctx context.Context, domainOrGroup string) ([]string, error)) *MockControlPlane_GetAttachmentsConfig_Call { - _c.Call.Return(run) - return _c -} - -// GetAutoRouteAllowedDomains provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) GetAutoRouteAllowedDomains(ctx context.Context) ([]string, error) { - ret := _mock.Called(ctx) - - if len(ret) == 0 { - panic("no return value specified for GetAutoRouteAllowedDomains") - } - - var r0 []string - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context) ([]string, error)); ok { - return returnFunc(ctx) - } - if returnFunc, ok := ret.Get(0).(func(context.Context) []string); ok { - r0 = returnFunc(ctx) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]string) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { - r1 = returnFunc(ctx) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockControlPlane_GetAutoRouteAllowedDomains_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetAutoRouteAllowedDomains' -type MockControlPlane_GetAutoRouteAllowedDomains_Call struct { - *mock.Call -} - -// GetAutoRouteAllowedDomains is a helper method to define mock.On call -// - ctx context.Context -func (_e *MockControlPlane_Expecter) GetAutoRouteAllowedDomains(ctx any) *MockControlPlane_GetAutoRouteAllowedDomains_Call { - return &MockControlPlane_GetAutoRouteAllowedDomains_Call{Call: _e.mock.On("GetAutoRouteAllowedDomains", ctx)} -} - -func (_c *MockControlPlane_GetAutoRouteAllowedDomains_Call) Run(run func(ctx context.Context)) *MockControlPlane_GetAutoRouteAllowedDomains_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - run( - arg0, - ) - }) - return _c -} - -func (_c *MockControlPlane_GetAutoRouteAllowedDomains_Call) Return(strings []string, err error) *MockControlPlane_GetAutoRouteAllowedDomains_Call { - _c.Call.Return(strings, err) - return _c -} - -func (_c *MockControlPlane_GetAutoRouteAllowedDomains_Call) RunAndReturn(run func(ctx context.Context) ([]string, error)) *MockControlPlane_GetAutoRouteAllowedDomains_Call { - _c.Call.Return(run) - return _c -} - -// GetConfig provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) GetConfig(ctx context.Context) (*remote.Config, error) { - ret := _mock.Called(ctx) - - if len(ret) == 0 { - panic("no return value specified for GetConfig") - } - - var r0 *remote.Config - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context) (*remote.Config, error)); ok { - return returnFunc(ctx) + var r0 *dto.AppDeployResponse + var r1 string + var r2 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, dto.AppDeployRequest) (*dto.AppDeployResponse, string, error)); ok { + return returnFunc(ctx, app, req) } - if returnFunc, ok := ret.Get(0).(func(context.Context) *remote.Config); ok { - r0 = returnFunc(ctx) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, dto.AppDeployRequest) *dto.AppDeployResponse); ok { + r0 = returnFunc(ctx, app, req) } else { if ret.Get(0) != nil { - r0 = ret.Get(0).(*remote.Config) + r0 = ret.Get(0).(*dto.AppDeployResponse) } } - if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { - r1 = returnFunc(ctx) + if returnFunc, ok := ret.Get(1).(func(context.Context, string, dto.AppDeployRequest) string); ok { + r1 = returnFunc(ctx, app, req) } else { - r1 = ret.Error(1) + r1 = ret.Get(1).(string) } - return r0, r1 + if returnFunc, ok := ret.Get(2).(func(context.Context, string, dto.AppDeployRequest) error); ok { + r2 = returnFunc(ctx, app, req) + } else { + r2 = ret.Error(2) + } + return r0, r1, r2 } -// MockControlPlane_GetConfig_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetConfig' -type MockControlPlane_GetConfig_Call struct { +// MockControlPlane_DeployApp_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'DeployApp' +type MockControlPlane_DeployApp_Call struct { *mock.Call } -// GetConfig is a helper method to define mock.On call +// DeployApp is a helper method to define mock.On call // - ctx context.Context -func (_e *MockControlPlane_Expecter) GetConfig(ctx any) *MockControlPlane_GetConfig_Call { - return &MockControlPlane_GetConfig_Call{Call: _e.mock.On("GetConfig", ctx)} +// - app string +// - req dto.AppDeployRequest +func (_e *MockControlPlane_Expecter) DeployApp(ctx any, app any, req any) *MockControlPlane_DeployApp_Call { + return &MockControlPlane_DeployApp_Call{Call: _e.mock.On("DeployApp", ctx, app, req)} } -func (_c *MockControlPlane_GetConfig_Call) Run(run func(ctx context.Context)) *MockControlPlane_GetConfig_Call { +func (_c *MockControlPlane_DeployApp_Call) Run(run func(ctx context.Context, app string, req dto.AppDeployRequest)) *MockControlPlane_DeployApp_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { arg0 = args[0].(context.Context) } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 dto.AppDeployRequest + if args[2] != nil { + arg2 = args[2].(dto.AppDeployRequest) + } run( arg0, + arg1, + arg2, ) }) return _c } -func (_c *MockControlPlane_GetConfig_Call) Return(config *remote.Config, err error) *MockControlPlane_GetConfig_Call { - _c.Call.Return(config, err) +func (_c *MockControlPlane_DeployApp_Call) Return(appDeployResponse *dto.AppDeployResponse, s string, err error) *MockControlPlane_DeployApp_Call { + _c.Call.Return(appDeployResponse, s, err) return _c } -func (_c *MockControlPlane_GetConfig_Call) RunAndReturn(run func(ctx context.Context) (*remote.Config, error)) *MockControlPlane_GetConfig_Call { +func (_c *MockControlPlane_DeployApp_Call) RunAndReturn(run func(ctx context.Context, app string, req dto.AppDeployRequest) (*dto.AppDeployResponse, string, error)) *MockControlPlane_DeployApp_Call { _c.Call.Return(run) return _c } -// GetContainerLogs provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) GetContainerLogs(ctx context.Context, logDomain string, lines int) ([]string, error) { - ret := _mock.Called(ctx, logDomain, lines) +// DiffApp provides a mock function for the type MockControlPlane +func (_mock *MockControlPlane) DiffApp(ctx context.Context, app string) (*dto.AppDiffResponse, error) { + ret := _mock.Called(ctx, app) if len(ret) == 0 { - panic("no return value specified for GetContainerLogs") + panic("no return value specified for DiffApp") } - var r0 []string + var r0 *dto.AppDiffResponse var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, int) ([]string, error)); ok { - return returnFunc(ctx, logDomain, lines) + if returnFunc, ok := ret.Get(0).(func(context.Context, string) (*dto.AppDiffResponse, error)); ok { + return returnFunc(ctx, app) } - if returnFunc, ok := ret.Get(0).(func(context.Context, string, int) []string); ok { - r0 = returnFunc(ctx, logDomain, lines) + if returnFunc, ok := ret.Get(0).(func(context.Context, string) *dto.AppDiffResponse); ok { + r0 = returnFunc(ctx, app) } else { if ret.Get(0) != nil { - r0 = ret.Get(0).([]string) + r0 = ret.Get(0).(*dto.AppDiffResponse) } } - if returnFunc, ok := ret.Get(1).(func(context.Context, string, int) error); ok { - r1 = returnFunc(ctx, logDomain, lines) + if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { + r1 = returnFunc(ctx, app) } else { r1 = ret.Error(1) } return r0, r1 } -// MockControlPlane_GetContainerLogs_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetContainerLogs' -type MockControlPlane_GetContainerLogs_Call struct { +// MockControlPlane_DiffApp_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'DiffApp' +type MockControlPlane_DiffApp_Call struct { *mock.Call } -// GetContainerLogs is a helper method to define mock.On call +// DiffApp is a helper method to define mock.On call // - ctx context.Context -// - logDomain string -// - lines int -func (_e *MockControlPlane_Expecter) GetContainerLogs(ctx any, logDomain any, lines any) *MockControlPlane_GetContainerLogs_Call { - return &MockControlPlane_GetContainerLogs_Call{Call: _e.mock.On("GetContainerLogs", ctx, logDomain, lines)} +// - app string +func (_e *MockControlPlane_Expecter) DiffApp(ctx any, app any) *MockControlPlane_DiffApp_Call { + return &MockControlPlane_DiffApp_Call{Call: _e.mock.On("DiffApp", ctx, app)} } -func (_c *MockControlPlane_GetContainerLogs_Call) Run(run func(ctx context.Context, logDomain string, lines int)) *MockControlPlane_GetContainerLogs_Call { +func (_c *MockControlPlane_DiffApp_Call) Run(run func(ctx context.Context, app string)) *MockControlPlane_DiffApp_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -1187,47 +372,42 @@ func (_c *MockControlPlane_GetContainerLogs_Call) Run(run func(ctx context.Conte if args[1] != nil { arg1 = args[1].(string) } - var arg2 int - if args[2] != nil { - arg2 = args[2].(int) - } run( arg0, arg1, - arg2, ) }) return _c } -func (_c *MockControlPlane_GetContainerLogs_Call) Return(strings []string, err error) *MockControlPlane_GetContainerLogs_Call { - _c.Call.Return(strings, err) +func (_c *MockControlPlane_DiffApp_Call) Return(appDiffResponse *dto.AppDiffResponse, err error) *MockControlPlane_DiffApp_Call { + _c.Call.Return(appDiffResponse, err) return _c } -func (_c *MockControlPlane_GetContainerLogs_Call) RunAndReturn(run func(ctx context.Context, logDomain string, lines int) ([]string, error)) *MockControlPlane_GetContainerLogs_Call { +func (_c *MockControlPlane_DiffApp_Call) RunAndReturn(run func(ctx context.Context, app string) (*dto.AppDiffResponse, error)) *MockControlPlane_DiffApp_Call { _c.Call.Return(run) return _c } -// GetHealth provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) GetHealth(ctx context.Context) (map[string]*remote.RouteHealth, error) { +// GetConfig provides a mock function for the type MockControlPlane +func (_mock *MockControlPlane) GetConfig(ctx context.Context) (*remote.Config, error) { ret := _mock.Called(ctx) if len(ret) == 0 { - panic("no return value specified for GetHealth") + panic("no return value specified for GetConfig") } - var r0 map[string]*remote.RouteHealth + var r0 *remote.Config var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context) (map[string]*remote.RouteHealth, error)); ok { + if returnFunc, ok := ret.Get(0).(func(context.Context) (*remote.Config, error)); ok { return returnFunc(ctx) } - if returnFunc, ok := ret.Get(0).(func(context.Context) map[string]*remote.RouteHealth); ok { + if returnFunc, ok := ret.Get(0).(func(context.Context) *remote.Config); ok { r0 = returnFunc(ctx) } else { if ret.Get(0) != nil { - r0 = ret.Get(0).(map[string]*remote.RouteHealth) + r0 = ret.Get(0).(*remote.Config) } } if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { @@ -1238,18 +418,18 @@ func (_mock *MockControlPlane) GetHealth(ctx context.Context) (map[string]*remot return r0, r1 } -// MockControlPlane_GetHealth_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetHealth' -type MockControlPlane_GetHealth_Call struct { +// MockControlPlane_GetConfig_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetConfig' +type MockControlPlane_GetConfig_Call struct { *mock.Call } -// GetHealth is a helper method to define mock.On call +// GetConfig is a helper method to define mock.On call // - ctx context.Context -func (_e *MockControlPlane_Expecter) GetHealth(ctx any) *MockControlPlane_GetHealth_Call { - return &MockControlPlane_GetHealth_Call{Call: _e.mock.On("GetHealth", ctx)} +func (_e *MockControlPlane_Expecter) GetConfig(ctx any) *MockControlPlane_GetConfig_Call { + return &MockControlPlane_GetConfig_Call{Call: _e.mock.On("GetConfig", ctx)} } -func (_c *MockControlPlane_GetHealth_Call) Run(run func(ctx context.Context)) *MockControlPlane_GetHealth_Call { +func (_c *MockControlPlane_GetConfig_Call) Run(run func(ctx context.Context)) *MockControlPlane_GetConfig_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -1262,12 +442,12 @@ func (_c *MockControlPlane_GetHealth_Call) Run(run func(ctx context.Context)) *M return _c } -func (_c *MockControlPlane_GetHealth_Call) Return(stringToRouteHealth map[string]*remote.RouteHealth, err error) *MockControlPlane_GetHealth_Call { - _c.Call.Return(stringToRouteHealth, err) +func (_c *MockControlPlane_GetConfig_Call) Return(config *remote.Config, err error) *MockControlPlane_GetConfig_Call { + _c.Call.Return(config, err) return _c } -func (_c *MockControlPlane_GetHealth_Call) RunAndReturn(run func(ctx context.Context) (map[string]*remote.RouteHealth, error)) *MockControlPlane_GetHealth_Call { +func (_c *MockControlPlane_GetConfig_Call) RunAndReturn(run func(ctx context.Context) (*remote.Config, error)) *MockControlPlane_GetConfig_Call { _c.Call.Return(run) return _c } @@ -1340,74 +520,6 @@ func (_c *MockControlPlane_GetProcessLogs_Call) RunAndReturn(run func(ctx contex return _c } -// GetRoute provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) GetRoute(ctx context.Context, routeDomain string) (*domain.Route, error) { - ret := _mock.Called(ctx, routeDomain) - - if len(ret) == 0 { - panic("no return value specified for GetRoute") - } - - var r0 *domain.Route - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) (*domain.Route, error)); ok { - return returnFunc(ctx, routeDomain) - } - if returnFunc, ok := ret.Get(0).(func(context.Context, string) *domain.Route); ok { - r0 = returnFunc(ctx, routeDomain) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).(*domain.Route) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { - r1 = returnFunc(ctx, routeDomain) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockControlPlane_GetRoute_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetRoute' -type MockControlPlane_GetRoute_Call struct { - *mock.Call -} - -// GetRoute is a helper method to define mock.On call -// - ctx context.Context -// - routeDomain string -func (_e *MockControlPlane_Expecter) GetRoute(ctx any, routeDomain any) *MockControlPlane_GetRoute_Call { - return &MockControlPlane_GetRoute_Call{Call: _e.mock.On("GetRoute", ctx, routeDomain)} -} - -func (_c *MockControlPlane_GetRoute_Call) Run(run func(ctx context.Context, routeDomain string)) *MockControlPlane_GetRoute_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockControlPlane_GetRoute_Call) Return(route *domain.Route, err error) *MockControlPlane_GetRoute_Call { - _c.Call.Return(route, err) - return _c -} - -func (_c *MockControlPlane_GetRoute_Call) RunAndReturn(run func(ctx context.Context, routeDomain string) (*domain.Route, error)) *MockControlPlane_GetRoute_Call { - _c.Call.Return(run) - return _c -} - // GetStatus provides a mock function for the type MockControlPlane func (_mock *MockControlPlane) GetStatus(ctx context.Context) (*remote.Status, error) { ret := _mock.Called(ctx) @@ -1594,47 +706,48 @@ func (_c *MockControlPlane_GetTrafficStatus_Call) RunAndReturn(run func(ctx cont return _c } -// ListBackups provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) ListBackups(ctx context.Context, backupDomain string) ([]dto.BackupJob, error) { - ret := _mock.Called(ctx, backupDomain) +// ListAppSecrets provides a mock function for the type MockControlPlane +func (_mock *MockControlPlane) ListAppSecrets(ctx context.Context, app string, service string) ([]dto.AppSecretMetadataDTO, error) { + ret := _mock.Called(ctx, app, service) if len(ret) == 0 { - panic("no return value specified for ListBackups") + panic("no return value specified for ListAppSecrets") } - var r0 []dto.BackupJob + var r0 []dto.AppSecretMetadataDTO var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) ([]dto.BackupJob, error)); ok { - return returnFunc(ctx, backupDomain) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) ([]dto.AppSecretMetadataDTO, error)); ok { + return returnFunc(ctx, app, service) } - if returnFunc, ok := ret.Get(0).(func(context.Context, string) []dto.BackupJob); ok { - r0 = returnFunc(ctx, backupDomain) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) []dto.AppSecretMetadataDTO); ok { + r0 = returnFunc(ctx, app, service) } else { if ret.Get(0) != nil { - r0 = ret.Get(0).([]dto.BackupJob) + r0 = ret.Get(0).([]dto.AppSecretMetadataDTO) } } - if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { - r1 = returnFunc(ctx, backupDomain) + if returnFunc, ok := ret.Get(1).(func(context.Context, string, string) error); ok { + r1 = returnFunc(ctx, app, service) } else { r1 = ret.Error(1) } return r0, r1 } -// MockControlPlane_ListBackups_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListBackups' -type MockControlPlane_ListBackups_Call struct { +// MockControlPlane_ListAppSecrets_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListAppSecrets' +type MockControlPlane_ListAppSecrets_Call struct { *mock.Call } -// ListBackups is a helper method to define mock.On call +// ListAppSecrets is a helper method to define mock.On call // - ctx context.Context -// - backupDomain string -func (_e *MockControlPlane_Expecter) ListBackups(ctx any, backupDomain any) *MockControlPlane_ListBackups_Call { - return &MockControlPlane_ListBackups_Call{Call: _e.mock.On("ListBackups", ctx, backupDomain)} +// - app string +// - service string +func (_e *MockControlPlane_Expecter) ListAppSecrets(ctx any, app any, service any) *MockControlPlane_ListAppSecrets_Call { + return &MockControlPlane_ListAppSecrets_Call{Call: _e.mock.On("ListAppSecrets", ctx, app, service)} } -func (_c *MockControlPlane_ListBackups_Call) Run(run func(ctx context.Context, backupDomain string)) *MockControlPlane_ListBackups_Call { +func (_c *MockControlPlane_ListAppSecrets_Call) Run(run func(ctx context.Context, app string, service string)) *MockControlPlane_ListAppSecrets_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -1644,42 +757,47 @@ func (_c *MockControlPlane_ListBackups_Call) Run(run func(ctx context.Context, b if args[1] != nil { arg1 = args[1].(string) } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } run( arg0, arg1, + arg2, ) }) return _c } -func (_c *MockControlPlane_ListBackups_Call) Return(backupJobs []dto.BackupJob, err error) *MockControlPlane_ListBackups_Call { - _c.Call.Return(backupJobs, err) +func (_c *MockControlPlane_ListAppSecrets_Call) Return(appSecretMetadataDTOs []dto.AppSecretMetadataDTO, err error) *MockControlPlane_ListAppSecrets_Call { + _c.Call.Return(appSecretMetadataDTOs, err) return _c } -func (_c *MockControlPlane_ListBackups_Call) RunAndReturn(run func(ctx context.Context, backupDomain string) ([]dto.BackupJob, error)) *MockControlPlane_ListBackups_Call { +func (_c *MockControlPlane_ListAppSecrets_Call) RunAndReturn(run func(ctx context.Context, app string, service string) ([]dto.AppSecretMetadataDTO, error)) *MockControlPlane_ListAppSecrets_Call { _c.Call.Return(run) return _c } -// ListNetworks provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) ListNetworks(ctx context.Context) ([]*domain.NetworkInfo, error) { +// ListApps provides a mock function for the type MockControlPlane +func (_mock *MockControlPlane) ListApps(ctx context.Context) ([]dto.AppSummaryDTO, error) { ret := _mock.Called(ctx) if len(ret) == 0 { - panic("no return value specified for ListNetworks") + panic("no return value specified for ListApps") } - var r0 []*domain.NetworkInfo + var r0 []dto.AppSummaryDTO var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context) ([]*domain.NetworkInfo, error)); ok { + if returnFunc, ok := ret.Get(0).(func(context.Context) ([]dto.AppSummaryDTO, error)); ok { return returnFunc(ctx) } - if returnFunc, ok := ret.Get(0).(func(context.Context) []*domain.NetworkInfo); ok { + if returnFunc, ok := ret.Get(0).(func(context.Context) []dto.AppSummaryDTO); ok { r0 = returnFunc(ctx) } else { if ret.Get(0) != nil { - r0 = ret.Get(0).([]*domain.NetworkInfo) + r0 = ret.Get(0).([]dto.AppSummaryDTO) } } if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { @@ -1690,18 +808,18 @@ func (_mock *MockControlPlane) ListNetworks(ctx context.Context) ([]*domain.Netw return r0, r1 } -// MockControlPlane_ListNetworks_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListNetworks' -type MockControlPlane_ListNetworks_Call struct { +// MockControlPlane_ListApps_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListApps' +type MockControlPlane_ListApps_Call struct { *mock.Call } -// ListNetworks is a helper method to define mock.On call +// ListApps is a helper method to define mock.On call // - ctx context.Context -func (_e *MockControlPlane_Expecter) ListNetworks(ctx any) *MockControlPlane_ListNetworks_Call { - return &MockControlPlane_ListNetworks_Call{Call: _e.mock.On("ListNetworks", ctx)} +func (_e *MockControlPlane_Expecter) ListApps(ctx any) *MockControlPlane_ListApps_Call { + return &MockControlPlane_ListApps_Call{Call: _e.mock.On("ListApps", ctx)} } -func (_c *MockControlPlane_ListNetworks_Call) Run(run func(ctx context.Context)) *MockControlPlane_ListNetworks_Call { +func (_c *MockControlPlane_ListApps_Call) Run(run func(ctx context.Context)) *MockControlPlane_ListApps_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -1714,96 +832,102 @@ func (_c *MockControlPlane_ListNetworks_Call) Run(run func(ctx context.Context)) return _c } -func (_c *MockControlPlane_ListNetworks_Call) Return(networkInfos []*domain.NetworkInfo, err error) *MockControlPlane_ListNetworks_Call { - _c.Call.Return(networkInfos, err) +func (_c *MockControlPlane_ListApps_Call) Return(appSummaryDTOs []dto.AppSummaryDTO, err error) *MockControlPlane_ListApps_Call { + _c.Call.Return(appSummaryDTOs, err) return _c } -func (_c *MockControlPlane_ListNetworks_Call) RunAndReturn(run func(ctx context.Context) ([]*domain.NetworkInfo, error)) *MockControlPlane_ListNetworks_Call { +func (_c *MockControlPlane_ListApps_Call) RunAndReturn(run func(ctx context.Context) ([]dto.AppSummaryDTO, error)) *MockControlPlane_ListApps_Call { _c.Call.Return(run) return _c } -// ListOrphanedAttachments provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) ListOrphanedAttachments(ctx context.Context) ([]domain.CleanupAttachment, error) { - ret := _mock.Called(ctx) +// ListBackups provides a mock function for the type MockControlPlane +func (_mock *MockControlPlane) ListBackups(ctx context.Context, app string) ([]dto.BackupJob, error) { + ret := _mock.Called(ctx, app) if len(ret) == 0 { - panic("no return value specified for ListOrphanedAttachments") + panic("no return value specified for ListBackups") } - var r0 []domain.CleanupAttachment + var r0 []dto.BackupJob var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context) ([]domain.CleanupAttachment, error)); ok { - return returnFunc(ctx) + if returnFunc, ok := ret.Get(0).(func(context.Context, string) ([]dto.BackupJob, error)); ok { + return returnFunc(ctx, app) } - if returnFunc, ok := ret.Get(0).(func(context.Context) []domain.CleanupAttachment); ok { - r0 = returnFunc(ctx) + if returnFunc, ok := ret.Get(0).(func(context.Context, string) []dto.BackupJob); ok { + r0 = returnFunc(ctx, app) } else { if ret.Get(0) != nil { - r0 = ret.Get(0).([]domain.CleanupAttachment) + r0 = ret.Get(0).([]dto.BackupJob) } } - if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { - r1 = returnFunc(ctx) + if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { + r1 = returnFunc(ctx, app) } else { r1 = ret.Error(1) } return r0, r1 } -// MockControlPlane_ListOrphanedAttachments_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListOrphanedAttachments' -type MockControlPlane_ListOrphanedAttachments_Call struct { +// MockControlPlane_ListBackups_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListBackups' +type MockControlPlane_ListBackups_Call struct { *mock.Call } -// ListOrphanedAttachments is a helper method to define mock.On call +// ListBackups is a helper method to define mock.On call // - ctx context.Context -func (_e *MockControlPlane_Expecter) ListOrphanedAttachments(ctx any) *MockControlPlane_ListOrphanedAttachments_Call { - return &MockControlPlane_ListOrphanedAttachments_Call{Call: _e.mock.On("ListOrphanedAttachments", ctx)} +// - app string +func (_e *MockControlPlane_Expecter) ListBackups(ctx any, app any) *MockControlPlane_ListBackups_Call { + return &MockControlPlane_ListBackups_Call{Call: _e.mock.On("ListBackups", ctx, app)} } -func (_c *MockControlPlane_ListOrphanedAttachments_Call) Run(run func(ctx context.Context)) *MockControlPlane_ListOrphanedAttachments_Call { +func (_c *MockControlPlane_ListBackups_Call) Run(run func(ctx context.Context, app string)) *MockControlPlane_ListBackups_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { arg0 = args[0].(context.Context) } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } run( arg0, + arg1, ) }) return _c } -func (_c *MockControlPlane_ListOrphanedAttachments_Call) Return(cleanupAttachments []domain.CleanupAttachment, err error) *MockControlPlane_ListOrphanedAttachments_Call { - _c.Call.Return(cleanupAttachments, err) +func (_c *MockControlPlane_ListBackups_Call) Return(backupJobs []dto.BackupJob, err error) *MockControlPlane_ListBackups_Call { + _c.Call.Return(backupJobs, err) return _c } -func (_c *MockControlPlane_ListOrphanedAttachments_Call) RunAndReturn(run func(ctx context.Context) ([]domain.CleanupAttachment, error)) *MockControlPlane_ListOrphanedAttachments_Call { +func (_c *MockControlPlane_ListBackups_Call) RunAndReturn(run func(ctx context.Context, app string) ([]dto.BackupJob, error)) *MockControlPlane_ListBackups_Call { _c.Call.Return(run) return _c } -// ListRoutesWithDetails provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) ListRoutesWithDetails(ctx context.Context) ([]remote.RouteInfo, error) { +// ListNetworks provides a mock function for the type MockControlPlane +func (_mock *MockControlPlane) ListNetworks(ctx context.Context) ([]*domain.NetworkInfo, error) { ret := _mock.Called(ctx) if len(ret) == 0 { - panic("no return value specified for ListRoutesWithDetails") + panic("no return value specified for ListNetworks") } - var r0 []remote.RouteInfo + var r0 []*domain.NetworkInfo var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context) ([]remote.RouteInfo, error)); ok { + if returnFunc, ok := ret.Get(0).(func(context.Context) ([]*domain.NetworkInfo, error)); ok { return returnFunc(ctx) } - if returnFunc, ok := ret.Get(0).(func(context.Context) []remote.RouteInfo); ok { + if returnFunc, ok := ret.Get(0).(func(context.Context) []*domain.NetworkInfo); ok { r0 = returnFunc(ctx) } else { if ret.Get(0) != nil { - r0 = ret.Get(0).([]remote.RouteInfo) + r0 = ret.Get(0).([]*domain.NetworkInfo) } } if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { @@ -1814,18 +938,18 @@ func (_mock *MockControlPlane) ListRoutesWithDetails(ctx context.Context) ([]rem return r0, r1 } -// MockControlPlane_ListRoutesWithDetails_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListRoutesWithDetails' -type MockControlPlane_ListRoutesWithDetails_Call struct { +// MockControlPlane_ListNetworks_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListNetworks' +type MockControlPlane_ListNetworks_Call struct { *mock.Call } -// ListRoutesWithDetails is a helper method to define mock.On call +// ListNetworks is a helper method to define mock.On call // - ctx context.Context -func (_e *MockControlPlane_Expecter) ListRoutesWithDetails(ctx any) *MockControlPlane_ListRoutesWithDetails_Call { - return &MockControlPlane_ListRoutesWithDetails_Call{Call: _e.mock.On("ListRoutesWithDetails", ctx)} +func (_e *MockControlPlane_Expecter) ListNetworks(ctx any) *MockControlPlane_ListNetworks_Call { + return &MockControlPlane_ListNetworks_Call{Call: _e.mock.On("ListNetworks", ctx)} } -func (_c *MockControlPlane_ListRoutesWithDetails_Call) Run(run func(ctx context.Context)) *MockControlPlane_ListRoutesWithDetails_Call { +func (_c *MockControlPlane_ListNetworks_Call) Run(run func(ctx context.Context)) *MockControlPlane_ListNetworks_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -1838,57 +962,57 @@ func (_c *MockControlPlane_ListRoutesWithDetails_Call) Run(run func(ctx context. return _c } -func (_c *MockControlPlane_ListRoutesWithDetails_Call) Return(vs []remote.RouteInfo, err error) *MockControlPlane_ListRoutesWithDetails_Call { - _c.Call.Return(vs, err) +func (_c *MockControlPlane_ListNetworks_Call) Return(networkInfos []*domain.NetworkInfo, err error) *MockControlPlane_ListNetworks_Call { + _c.Call.Return(networkInfos, err) return _c } -func (_c *MockControlPlane_ListRoutesWithDetails_Call) RunAndReturn(run func(ctx context.Context) ([]remote.RouteInfo, error)) *MockControlPlane_ListRoutesWithDetails_Call { +func (_c *MockControlPlane_ListNetworks_Call) RunAndReturn(run func(ctx context.Context) ([]*domain.NetworkInfo, error)) *MockControlPlane_ListNetworks_Call { _c.Call.Return(run) return _c } -// ListSecretsWithAttachments provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) ListSecretsWithAttachments(ctx context.Context, secretDomain string) (*remote.SecretsListResult, error) { - ret := _mock.Called(ctx, secretDomain) +// ListTags provides a mock function for the type MockControlPlane +func (_mock *MockControlPlane) ListTags(ctx context.Context, repository string) ([]string, error) { + ret := _mock.Called(ctx, repository) if len(ret) == 0 { - panic("no return value specified for ListSecretsWithAttachments") + panic("no return value specified for ListTags") } - var r0 *remote.SecretsListResult + var r0 []string var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) (*remote.SecretsListResult, error)); ok { - return returnFunc(ctx, secretDomain) + if returnFunc, ok := ret.Get(0).(func(context.Context, string) ([]string, error)); ok { + return returnFunc(ctx, repository) } - if returnFunc, ok := ret.Get(0).(func(context.Context, string) *remote.SecretsListResult); ok { - r0 = returnFunc(ctx, secretDomain) + if returnFunc, ok := ret.Get(0).(func(context.Context, string) []string); ok { + r0 = returnFunc(ctx, repository) } else { if ret.Get(0) != nil { - r0 = ret.Get(0).(*remote.SecretsListResult) + r0 = ret.Get(0).([]string) } } if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { - r1 = returnFunc(ctx, secretDomain) + r1 = returnFunc(ctx, repository) } else { r1 = ret.Error(1) } return r0, r1 } -// MockControlPlane_ListSecretsWithAttachments_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListSecretsWithAttachments' -type MockControlPlane_ListSecretsWithAttachments_Call struct { +// MockControlPlane_ListTags_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListTags' +type MockControlPlane_ListTags_Call struct { *mock.Call } -// ListSecretsWithAttachments is a helper method to define mock.On call +// ListTags is a helper method to define mock.On call // - ctx context.Context -// - secretDomain string -func (_e *MockControlPlane_Expecter) ListSecretsWithAttachments(ctx any, secretDomain any) *MockControlPlane_ListSecretsWithAttachments_Call { - return &MockControlPlane_ListSecretsWithAttachments_Call{Call: _e.mock.On("ListSecretsWithAttachments", ctx, secretDomain)} +// - repository string +func (_e *MockControlPlane_Expecter) ListTags(ctx any, repository any) *MockControlPlane_ListTags_Call { + return &MockControlPlane_ListTags_Call{Call: _e.mock.On("ListTags", ctx, repository)} } -func (_c *MockControlPlane_ListSecretsWithAttachments_Call) Run(run func(ctx context.Context, secretDomain string)) *MockControlPlane_ListSecretsWithAttachments_Call { +func (_c *MockControlPlane_ListTags_Call) Run(run func(ctx context.Context, repository string)) *MockControlPlane_ListTags_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -1906,57 +1030,57 @@ func (_c *MockControlPlane_ListSecretsWithAttachments_Call) Run(run func(ctx con return _c } -func (_c *MockControlPlane_ListSecretsWithAttachments_Call) Return(secretsListResult *remote.SecretsListResult, err error) *MockControlPlane_ListSecretsWithAttachments_Call { - _c.Call.Return(secretsListResult, err) +func (_c *MockControlPlane_ListTags_Call) Return(strings []string, err error) *MockControlPlane_ListTags_Call { + _c.Call.Return(strings, err) return _c } -func (_c *MockControlPlane_ListSecretsWithAttachments_Call) RunAndReturn(run func(ctx context.Context, secretDomain string) (*remote.SecretsListResult, error)) *MockControlPlane_ListSecretsWithAttachments_Call { +func (_c *MockControlPlane_ListTags_Call) RunAndReturn(run func(ctx context.Context, repository string) ([]string, error)) *MockControlPlane_ListTags_Call { _c.Call.Return(run) return _c } -// ListTags provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) ListTags(ctx context.Context, repository string) ([]string, error) { - ret := _mock.Called(ctx, repository) +// ListVolumeBackups provides a mock function for the type MockControlPlane +func (_mock *MockControlPlane) ListVolumeBackups(ctx context.Context, app string) ([]dto.VolumeBackupJob, error) { + ret := _mock.Called(ctx, app) if len(ret) == 0 { - panic("no return value specified for ListTags") + panic("no return value specified for ListVolumeBackups") } - var r0 []string + var r0 []dto.VolumeBackupJob var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) ([]string, error)); ok { - return returnFunc(ctx, repository) + if returnFunc, ok := ret.Get(0).(func(context.Context, string) ([]dto.VolumeBackupJob, error)); ok { + return returnFunc(ctx, app) } - if returnFunc, ok := ret.Get(0).(func(context.Context, string) []string); ok { - r0 = returnFunc(ctx, repository) + if returnFunc, ok := ret.Get(0).(func(context.Context, string) []dto.VolumeBackupJob); ok { + r0 = returnFunc(ctx, app) } else { if ret.Get(0) != nil { - r0 = ret.Get(0).([]string) + r0 = ret.Get(0).([]dto.VolumeBackupJob) } } if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { - r1 = returnFunc(ctx, repository) + r1 = returnFunc(ctx, app) } else { r1 = ret.Error(1) } return r0, r1 } -// MockControlPlane_ListTags_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListTags' -type MockControlPlane_ListTags_Call struct { +// MockControlPlane_ListVolumeBackups_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListVolumeBackups' +type MockControlPlane_ListVolumeBackups_Call struct { *mock.Call } -// ListTags is a helper method to define mock.On call +// ListVolumeBackups is a helper method to define mock.On call // - ctx context.Context -// - repository string -func (_e *MockControlPlane_Expecter) ListTags(ctx any, repository any) *MockControlPlane_ListTags_Call { - return &MockControlPlane_ListTags_Call{Call: _e.mock.On("ListTags", ctx, repository)} +// - app string +func (_e *MockControlPlane_Expecter) ListVolumeBackups(ctx any, app any) *MockControlPlane_ListVolumeBackups_Call { + return &MockControlPlane_ListVolumeBackups_Call{Call: _e.mock.On("ListVolumeBackups", ctx, app)} } -func (_c *MockControlPlane_ListTags_Call) Run(run func(ctx context.Context, repository string)) *MockControlPlane_ListTags_Call { +func (_c *MockControlPlane_ListVolumeBackups_Call) Run(run func(ctx context.Context, app string)) *MockControlPlane_ListVolumeBackups_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -1974,142 +1098,148 @@ func (_c *MockControlPlane_ListTags_Call) Run(run func(ctx context.Context, repo return _c } -func (_c *MockControlPlane_ListTags_Call) Return(strings []string, err error) *MockControlPlane_ListTags_Call { - _c.Call.Return(strings, err) +func (_c *MockControlPlane_ListVolumeBackups_Call) Return(volumeBackupJobs []dto.VolumeBackupJob, err error) *MockControlPlane_ListVolumeBackups_Call { + _c.Call.Return(volumeBackupJobs, err) return _c } -func (_c *MockControlPlane_ListTags_Call) RunAndReturn(run func(ctx context.Context, repository string) ([]string, error)) *MockControlPlane_ListTags_Call { +func (_c *MockControlPlane_ListVolumeBackups_Call) RunAndReturn(run func(ctx context.Context, app string) ([]dto.VolumeBackupJob, error)) *MockControlPlane_ListVolumeBackups_Call { _c.Call.Return(run) return _c } -// ListVolumeBackups provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) ListVolumeBackups(ctx context.Context, backupDomain string) ([]dto.VolumeBackupJob, error) { - ret := _mock.Called(ctx, backupDomain) +// ListVolumes provides a mock function for the type MockControlPlane +func (_mock *MockControlPlane) ListVolumes(ctx context.Context) ([]dto.Volume, error) { + ret := _mock.Called(ctx) if len(ret) == 0 { - panic("no return value specified for ListVolumeBackups") + panic("no return value specified for ListVolumes") } - var r0 []dto.VolumeBackupJob + var r0 []dto.Volume var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) ([]dto.VolumeBackupJob, error)); ok { - return returnFunc(ctx, backupDomain) + if returnFunc, ok := ret.Get(0).(func(context.Context) ([]dto.Volume, error)); ok { + return returnFunc(ctx) } - if returnFunc, ok := ret.Get(0).(func(context.Context, string) []dto.VolumeBackupJob); ok { - r0 = returnFunc(ctx, backupDomain) + if returnFunc, ok := ret.Get(0).(func(context.Context) []dto.Volume); ok { + r0 = returnFunc(ctx) } else { if ret.Get(0) != nil { - r0 = ret.Get(0).([]dto.VolumeBackupJob) + r0 = ret.Get(0).([]dto.Volume) } } - if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { - r1 = returnFunc(ctx, backupDomain) + if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { + r1 = returnFunc(ctx) } else { r1 = ret.Error(1) } return r0, r1 } -// MockControlPlane_ListVolumeBackups_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListVolumeBackups' -type MockControlPlane_ListVolumeBackups_Call struct { +// MockControlPlane_ListVolumes_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListVolumes' +type MockControlPlane_ListVolumes_Call struct { *mock.Call } -// ListVolumeBackups is a helper method to define mock.On call +// ListVolumes is a helper method to define mock.On call // - ctx context.Context -// - backupDomain string -func (_e *MockControlPlane_Expecter) ListVolumeBackups(ctx any, backupDomain any) *MockControlPlane_ListVolumeBackups_Call { - return &MockControlPlane_ListVolumeBackups_Call{Call: _e.mock.On("ListVolumeBackups", ctx, backupDomain)} +func (_e *MockControlPlane_Expecter) ListVolumes(ctx any) *MockControlPlane_ListVolumes_Call { + return &MockControlPlane_ListVolumes_Call{Call: _e.mock.On("ListVolumes", ctx)} } -func (_c *MockControlPlane_ListVolumeBackups_Call) Run(run func(ctx context.Context, backupDomain string)) *MockControlPlane_ListVolumeBackups_Call { +func (_c *MockControlPlane_ListVolumes_Call) Run(run func(ctx context.Context)) *MockControlPlane_ListVolumes_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) + arg0 = args[0].(context.Context) } run( arg0, - arg1, ) }) return _c } -func (_c *MockControlPlane_ListVolumeBackups_Call) Return(volumeBackupJobs []dto.VolumeBackupJob, err error) *MockControlPlane_ListVolumeBackups_Call { - _c.Call.Return(volumeBackupJobs, err) +func (_c *MockControlPlane_ListVolumes_Call) Return(volumes []dto.Volume, err error) *MockControlPlane_ListVolumes_Call { + _c.Call.Return(volumes, err) return _c } -func (_c *MockControlPlane_ListVolumeBackups_Call) RunAndReturn(run func(ctx context.Context, backupDomain string) ([]dto.VolumeBackupJob, error)) *MockControlPlane_ListVolumeBackups_Call { +func (_c *MockControlPlane_ListVolumes_Call) RunAndReturn(run func(ctx context.Context) ([]dto.Volume, error)) *MockControlPlane_ListVolumes_Call { _c.Call.Return(run) return _c } -// ListVolumes provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) ListVolumes(ctx context.Context) ([]dto.Volume, error) { - ret := _mock.Called(ctx) +// OperationByKey provides a mock function for the type MockControlPlane +func (_mock *MockControlPlane) OperationByKey(ctx context.Context, app string, key string) (*dto.AppDeployResponse, error) { + ret := _mock.Called(ctx, app, key) if len(ret) == 0 { - panic("no return value specified for ListVolumes") + panic("no return value specified for OperationByKey") } - var r0 []dto.Volume + var r0 *dto.AppDeployResponse var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context) ([]dto.Volume, error)); ok { - return returnFunc(ctx) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) (*dto.AppDeployResponse, error)); ok { + return returnFunc(ctx, app, key) } - if returnFunc, ok := ret.Get(0).(func(context.Context) []dto.Volume); ok { - r0 = returnFunc(ctx) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) *dto.AppDeployResponse); ok { + r0 = returnFunc(ctx, app, key) } else { if ret.Get(0) != nil { - r0 = ret.Get(0).([]dto.Volume) + r0 = ret.Get(0).(*dto.AppDeployResponse) } } - if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { - r1 = returnFunc(ctx) + if returnFunc, ok := ret.Get(1).(func(context.Context, string, string) error); ok { + r1 = returnFunc(ctx, app, key) } else { r1 = ret.Error(1) } return r0, r1 } -// MockControlPlane_ListVolumes_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListVolumes' -type MockControlPlane_ListVolumes_Call struct { +// MockControlPlane_OperationByKey_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'OperationByKey' +type MockControlPlane_OperationByKey_Call struct { *mock.Call } -// ListVolumes is a helper method to define mock.On call +// OperationByKey is a helper method to define mock.On call // - ctx context.Context -func (_e *MockControlPlane_Expecter) ListVolumes(ctx any) *MockControlPlane_ListVolumes_Call { - return &MockControlPlane_ListVolumes_Call{Call: _e.mock.On("ListVolumes", ctx)} +// - app string +// - key string +func (_e *MockControlPlane_Expecter) OperationByKey(ctx any, app any, key any) *MockControlPlane_OperationByKey_Call { + return &MockControlPlane_OperationByKey_Call{Call: _e.mock.On("OperationByKey", ctx, app, key)} } -func (_c *MockControlPlane_ListVolumes_Call) Run(run func(ctx context.Context)) *MockControlPlane_ListVolumes_Call { +func (_c *MockControlPlane_OperationByKey_Call) Run(run func(ctx context.Context, app string, key string)) *MockControlPlane_OperationByKey_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { arg0 = args[0].(context.Context) } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } run( arg0, + arg1, + arg2, ) }) return _c } -func (_c *MockControlPlane_ListVolumes_Call) Return(volumes []dto.Volume, err error) *MockControlPlane_ListVolumes_Call { - _c.Call.Return(volumes, err) +func (_c *MockControlPlane_OperationByKey_Call) Return(appDeployResponse *dto.AppDeployResponse, err error) *MockControlPlane_OperationByKey_Call { + _c.Call.Return(appDeployResponse, err) return _c } -func (_c *MockControlPlane_ListVolumes_Call) RunAndReturn(run func(ctx context.Context) ([]dto.Volume, error)) *MockControlPlane_ListVolumes_Call { +func (_c *MockControlPlane_OperationByKey_Call) RunAndReturn(run func(ctx context.Context, app string, key string) (*dto.AppDeployResponse, error)) *MockControlPlane_OperationByKey_Call { _c.Call.Return(run) return _c } @@ -2233,156 +1363,53 @@ func (_c *MockControlPlane_Reload_Call) RunAndReturn(run func(ctx context.Contex return _c } -// RemoveAttachment provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) RemoveAttachment(ctx context.Context, domainOrGroup string, image string) error { - ret := _mock.Called(ctx, domainOrGroup, image) +// RemoveApp provides a mock function for the type MockControlPlane +func (_mock *MockControlPlane) RemoveApp(ctx context.Context, app string) (*dto.AppDeployResponse, string, error) { + ret := _mock.Called(ctx, app) if len(ret) == 0 { - panic("no return value specified for RemoveAttachment") + panic("no return value specified for RemoveApp") } - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) error); ok { - r0 = returnFunc(ctx, domainOrGroup, image) - } else { - r0 = ret.Error(0) + var r0 *dto.AppDeployResponse + var r1 string + var r2 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string) (*dto.AppDeployResponse, string, error)); ok { + return returnFunc(ctx, app) } - return r0 -} - -// MockControlPlane_RemoveAttachment_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'RemoveAttachment' -type MockControlPlane_RemoveAttachment_Call struct { - *mock.Call -} - -// RemoveAttachment is a helper method to define mock.On call -// - ctx context.Context -// - domainOrGroup string -// - image string -func (_e *MockControlPlane_Expecter) RemoveAttachment(ctx any, domainOrGroup any, image any) *MockControlPlane_RemoveAttachment_Call { - return &MockControlPlane_RemoveAttachment_Call{Call: _e.mock.On("RemoveAttachment", ctx, domainOrGroup, image)} -} - -func (_c *MockControlPlane_RemoveAttachment_Call) Run(run func(ctx context.Context, domainOrGroup string, image string)) *MockControlPlane_RemoveAttachment_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - var arg2 string - if args[2] != nil { - arg2 = args[2].(string) + if returnFunc, ok := ret.Get(0).(func(context.Context, string) *dto.AppDeployResponse); ok { + r0 = returnFunc(ctx, app) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).(*dto.AppDeployResponse) } - run( - arg0, - arg1, - arg2, - ) - }) - return _c -} - -func (_c *MockControlPlane_RemoveAttachment_Call) Return(err error) *MockControlPlane_RemoveAttachment_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockControlPlane_RemoveAttachment_Call) RunAndReturn(run func(ctx context.Context, domainOrGroup string, image string) error) *MockControlPlane_RemoveAttachment_Call { - _c.Call.Return(run) - return _c -} - -// RemoveAutoRouteAllowedDomain provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) RemoveAutoRouteAllowedDomain(ctx context.Context, pattern string) error { - ret := _mock.Called(ctx, pattern) - - if len(ret) == 0 { - panic("no return value specified for RemoveAutoRouteAllowedDomain") } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) error); ok { - r0 = returnFunc(ctx, pattern) + if returnFunc, ok := ret.Get(1).(func(context.Context, string) string); ok { + r1 = returnFunc(ctx, app) } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockControlPlane_RemoveAutoRouteAllowedDomain_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'RemoveAutoRouteAllowedDomain' -type MockControlPlane_RemoveAutoRouteAllowedDomain_Call struct { - *mock.Call -} - -// RemoveAutoRouteAllowedDomain is a helper method to define mock.On call -// - ctx context.Context -// - pattern string -func (_e *MockControlPlane_Expecter) RemoveAutoRouteAllowedDomain(ctx any, pattern any) *MockControlPlane_RemoveAutoRouteAllowedDomain_Call { - return &MockControlPlane_RemoveAutoRouteAllowedDomain_Call{Call: _e.mock.On("RemoveAutoRouteAllowedDomain", ctx, pattern)} -} - -func (_c *MockControlPlane_RemoveAutoRouteAllowedDomain_Call) Run(run func(ctx context.Context, pattern string)) *MockControlPlane_RemoveAutoRouteAllowedDomain_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockControlPlane_RemoveAutoRouteAllowedDomain_Call) Return(err error) *MockControlPlane_RemoveAutoRouteAllowedDomain_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockControlPlane_RemoveAutoRouteAllowedDomain_Call) RunAndReturn(run func(ctx context.Context, pattern string) error) *MockControlPlane_RemoveAutoRouteAllowedDomain_Call { - _c.Call.Return(run) - return _c -} - -// RemoveRoute provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) RemoveRoute(ctx context.Context, routeDomain string) error { - ret := _mock.Called(ctx, routeDomain) - - if len(ret) == 0 { - panic("no return value specified for RemoveRoute") + r1 = ret.Get(1).(string) } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) error); ok { - r0 = returnFunc(ctx, routeDomain) + if returnFunc, ok := ret.Get(2).(func(context.Context, string) error); ok { + r2 = returnFunc(ctx, app) } else { - r0 = ret.Error(0) + r2 = ret.Error(2) } - return r0 + return r0, r1, r2 } -// MockControlPlane_RemoveRoute_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'RemoveRoute' -type MockControlPlane_RemoveRoute_Call struct { +// MockControlPlane_RemoveApp_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'RemoveApp' +type MockControlPlane_RemoveApp_Call struct { *mock.Call } -// RemoveRoute is a helper method to define mock.On call +// RemoveApp is a helper method to define mock.On call // - ctx context.Context -// - routeDomain string -func (_e *MockControlPlane_Expecter) RemoveRoute(ctx any, routeDomain any) *MockControlPlane_RemoveRoute_Call { - return &MockControlPlane_RemoveRoute_Call{Call: _e.mock.On("RemoveRoute", ctx, routeDomain)} +// - app string +func (_e *MockControlPlane_Expecter) RemoveApp(ctx any, app any) *MockControlPlane_RemoveApp_Call { + return &MockControlPlane_RemoveApp_Call{Call: _e.mock.On("RemoveApp", ctx, app)} } -func (_c *MockControlPlane_RemoveRoute_Call) Run(run func(ctx context.Context, routeDomain string)) *MockControlPlane_RemoveRoute_Call { +func (_c *MockControlPlane_RemoveApp_Call) Run(run func(ctx context.Context, app string)) *MockControlPlane_RemoveApp_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -2400,58 +1427,65 @@ func (_c *MockControlPlane_RemoveRoute_Call) Run(run func(ctx context.Context, r return _c } -func (_c *MockControlPlane_RemoveRoute_Call) Return(err error) *MockControlPlane_RemoveRoute_Call { - _c.Call.Return(err) +func (_c *MockControlPlane_RemoveApp_Call) Return(appDeployResponse *dto.AppDeployResponse, s string, err error) *MockControlPlane_RemoveApp_Call { + _c.Call.Return(appDeployResponse, s, err) return _c } -func (_c *MockControlPlane_RemoveRoute_Call) RunAndReturn(run func(ctx context.Context, routeDomain string) error) *MockControlPlane_RemoveRoute_Call { +func (_c *MockControlPlane_RemoveApp_Call) RunAndReturn(run func(ctx context.Context, app string) (*dto.AppDeployResponse, string, error)) *MockControlPlane_RemoveApp_Call { _c.Call.Return(run) return _c } -// Restart provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) Restart(ctx context.Context, restartDomain string, withAttachments bool) (*remote.RestartResult, error) { - ret := _mock.Called(ctx, restartDomain, withAttachments) +// RestartApp provides a mock function for the type MockControlPlane +func (_mock *MockControlPlane) RestartApp(ctx context.Context, app string, service string, all bool) (*dto.AppDeployResponse, string, error) { + ret := _mock.Called(ctx, app, service, all) if len(ret) == 0 { - panic("no return value specified for Restart") + panic("no return value specified for RestartApp") } - var r0 *remote.RestartResult - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, bool) (*remote.RestartResult, error)); ok { - return returnFunc(ctx, restartDomain, withAttachments) + var r0 *dto.AppDeployResponse + var r1 string + var r2 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, bool) (*dto.AppDeployResponse, string, error)); ok { + return returnFunc(ctx, app, service, all) } - if returnFunc, ok := ret.Get(0).(func(context.Context, string, bool) *remote.RestartResult); ok { - r0 = returnFunc(ctx, restartDomain, withAttachments) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, bool) *dto.AppDeployResponse); ok { + r0 = returnFunc(ctx, app, service, all) } else { if ret.Get(0) != nil { - r0 = ret.Get(0).(*remote.RestartResult) + r0 = ret.Get(0).(*dto.AppDeployResponse) } } - if returnFunc, ok := ret.Get(1).(func(context.Context, string, bool) error); ok { - r1 = returnFunc(ctx, restartDomain, withAttachments) + if returnFunc, ok := ret.Get(1).(func(context.Context, string, string, bool) string); ok { + r1 = returnFunc(ctx, app, service, all) } else { - r1 = ret.Error(1) + r1 = ret.Get(1).(string) } - return r0, r1 + if returnFunc, ok := ret.Get(2).(func(context.Context, string, string, bool) error); ok { + r2 = returnFunc(ctx, app, service, all) + } else { + r2 = ret.Error(2) + } + return r0, r1, r2 } -// MockControlPlane_Restart_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Restart' -type MockControlPlane_Restart_Call struct { +// MockControlPlane_RestartApp_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'RestartApp' +type MockControlPlane_RestartApp_Call struct { *mock.Call } -// Restart is a helper method to define mock.On call +// RestartApp is a helper method to define mock.On call // - ctx context.Context -// - restartDomain string -// - withAttachments bool -func (_e *MockControlPlane_Expecter) Restart(ctx any, restartDomain any, withAttachments any) *MockControlPlane_Restart_Call { - return &MockControlPlane_Restart_Call{Call: _e.mock.On("Restart", ctx, restartDomain, withAttachments)} +// - app string +// - service string +// - all bool +func (_e *MockControlPlane_Expecter) RestartApp(ctx any, app any, service any, all any) *MockControlPlane_RestartApp_Call { + return &MockControlPlane_RestartApp_Call{Call: _e.mock.On("RestartApp", ctx, app, service, all)} } -func (_c *MockControlPlane_Restart_Call) Run(run func(ctx context.Context, restartDomain string, withAttachments bool)) *MockControlPlane_Restart_Call { +func (_c *MockControlPlane_RestartApp_Call) Run(run func(ctx context.Context, app string, service string, all bool)) *MockControlPlane_RestartApp_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -2461,32 +1495,37 @@ func (_c *MockControlPlane_Restart_Call) Run(run func(ctx context.Context, resta if args[1] != nil { arg1 = args[1].(string) } - var arg2 bool + var arg2 string if args[2] != nil { - arg2 = args[2].(bool) + arg2 = args[2].(string) + } + var arg3 bool + if args[3] != nil { + arg3 = args[3].(bool) } run( arg0, arg1, arg2, + arg3, ) }) return _c } -func (_c *MockControlPlane_Restart_Call) Return(restartResult *remote.RestartResult, err error) *MockControlPlane_Restart_Call { - _c.Call.Return(restartResult, err) +func (_c *MockControlPlane_RestartApp_Call) Return(appDeployResponse *dto.AppDeployResponse, s string, err error) *MockControlPlane_RestartApp_Call { + _c.Call.Return(appDeployResponse, s, err) return _c } -func (_c *MockControlPlane_Restart_Call) RunAndReturn(run func(ctx context.Context, restartDomain string, withAttachments bool) (*remote.RestartResult, error)) *MockControlPlane_Restart_Call { +func (_c *MockControlPlane_RestartApp_Call) RunAndReturn(run func(ctx context.Context, app string, service string, all bool) (*dto.AppDeployResponse, string, error)) *MockControlPlane_RestartApp_Call { _c.Call.Return(run) return _c } // RunBackup provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) RunBackup(ctx context.Context, backupDomain string, dbName string) (*dto.BackupRunResponse, error) { - ret := _mock.Called(ctx, backupDomain, dbName) +func (_mock *MockControlPlane) RunBackup(ctx context.Context, app string, service string, database string) (*dto.BackupRunResponse, error) { + ret := _mock.Called(ctx, app, service, database) if len(ret) == 0 { panic("no return value specified for RunBackup") @@ -2494,18 +1533,18 @@ func (_mock *MockControlPlane) RunBackup(ctx context.Context, backupDomain strin var r0 *dto.BackupRunResponse var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) (*dto.BackupRunResponse, error)); ok { - return returnFunc(ctx, backupDomain, dbName) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, string) (*dto.BackupRunResponse, error)); ok { + return returnFunc(ctx, app, service, database) } - if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) *dto.BackupRunResponse); ok { - r0 = returnFunc(ctx, backupDomain, dbName) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, string) *dto.BackupRunResponse); ok { + r0 = returnFunc(ctx, app, service, database) } else { if ret.Get(0) != nil { r0 = ret.Get(0).(*dto.BackupRunResponse) } } - if returnFunc, ok := ret.Get(1).(func(context.Context, string, string) error); ok { - r1 = returnFunc(ctx, backupDomain, dbName) + if returnFunc, ok := ret.Get(1).(func(context.Context, string, string, string) error); ok { + r1 = returnFunc(ctx, app, service, database) } else { r1 = ret.Error(1) } @@ -2519,13 +1558,14 @@ type MockControlPlane_RunBackup_Call struct { // RunBackup is a helper method to define mock.On call // - ctx context.Context -// - backupDomain string -// - dbName string -func (_e *MockControlPlane_Expecter) RunBackup(ctx any, backupDomain any, dbName any) *MockControlPlane_RunBackup_Call { - return &MockControlPlane_RunBackup_Call{Call: _e.mock.On("RunBackup", ctx, backupDomain, dbName)} +// - app string +// - service string +// - database string +func (_e *MockControlPlane_Expecter) RunBackup(ctx any, app any, service any, database any) *MockControlPlane_RunBackup_Call { + return &MockControlPlane_RunBackup_Call{Call: _e.mock.On("RunBackup", ctx, app, service, database)} } -func (_c *MockControlPlane_RunBackup_Call) Run(run func(ctx context.Context, backupDomain string, dbName string)) *MockControlPlane_RunBackup_Call { +func (_c *MockControlPlane_RunBackup_Call) Run(run func(ctx context.Context, app string, service string, database string)) *MockControlPlane_RunBackup_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -2539,10 +1579,15 @@ func (_c *MockControlPlane_RunBackup_Call) Run(run func(ctx context.Context, bac if args[2] != nil { arg2 = args[2].(string) } + var arg3 string + if args[3] != nil { + arg3 = args[3].(string) + } run( arg0, arg1, arg2, + arg3, ) }) return _c @@ -2553,14 +1598,14 @@ func (_c *MockControlPlane_RunBackup_Call) Return(backupRunResponse *dto.BackupR return _c } -func (_c *MockControlPlane_RunBackup_Call) RunAndReturn(run func(ctx context.Context, backupDomain string, dbName string) (*dto.BackupRunResponse, error)) *MockControlPlane_RunBackup_Call { +func (_c *MockControlPlane_RunBackup_Call) RunAndReturn(run func(ctx context.Context, app string, service string, database string) (*dto.BackupRunResponse, error)) *MockControlPlane_RunBackup_Call { _c.Call.Return(run) return _c } // RunVolumeBackups provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) RunVolumeBackups(ctx context.Context, backupDomain string, volumeName string) (*dto.VolumeBackupRunResponse, error) { - ret := _mock.Called(ctx, backupDomain, volumeName) +func (_mock *MockControlPlane) RunVolumeBackups(ctx context.Context, app string, service string, volume string) (*dto.VolumeBackupRunResponse, error) { + ret := _mock.Called(ctx, app, service, volume) if len(ret) == 0 { panic("no return value specified for RunVolumeBackups") @@ -2568,18 +1613,18 @@ func (_mock *MockControlPlane) RunVolumeBackups(ctx context.Context, backupDomai var r0 *dto.VolumeBackupRunResponse var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) (*dto.VolumeBackupRunResponse, error)); ok { - return returnFunc(ctx, backupDomain, volumeName) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, string) (*dto.VolumeBackupRunResponse, error)); ok { + return returnFunc(ctx, app, service, volume) } - if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) *dto.VolumeBackupRunResponse); ok { - r0 = returnFunc(ctx, backupDomain, volumeName) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, string) *dto.VolumeBackupRunResponse); ok { + r0 = returnFunc(ctx, app, service, volume) } else { if ret.Get(0) != nil { r0 = ret.Get(0).(*dto.VolumeBackupRunResponse) } } - if returnFunc, ok := ret.Get(1).(func(context.Context, string, string) error); ok { - r1 = returnFunc(ctx, backupDomain, volumeName) + if returnFunc, ok := ret.Get(1).(func(context.Context, string, string, string) error); ok { + r1 = returnFunc(ctx, app, service, volume) } else { r1 = ret.Error(1) } @@ -2593,13 +1638,14 @@ type MockControlPlane_RunVolumeBackups_Call struct { // RunVolumeBackups is a helper method to define mock.On call // - ctx context.Context -// - backupDomain string -// - volumeName string -func (_e *MockControlPlane_Expecter) RunVolumeBackups(ctx any, backupDomain any, volumeName any) *MockControlPlane_RunVolumeBackups_Call { - return &MockControlPlane_RunVolumeBackups_Call{Call: _e.mock.On("RunVolumeBackups", ctx, backupDomain, volumeName)} +// - app string +// - service string +// - volume string +func (_e *MockControlPlane_Expecter) RunVolumeBackups(ctx any, app any, service any, volume any) *MockControlPlane_RunVolumeBackups_Call { + return &MockControlPlane_RunVolumeBackups_Call{Call: _e.mock.On("RunVolumeBackups", ctx, app, service, volume)} } -func (_c *MockControlPlane_RunVolumeBackups_Call) Run(run func(ctx context.Context, backupDomain string, volumeName string)) *MockControlPlane_RunVolumeBackups_Call { +func (_c *MockControlPlane_RunVolumeBackups_Call) Run(run func(ctx context.Context, app string, service string, volume string)) *MockControlPlane_RunVolumeBackups_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -2613,10 +1659,15 @@ func (_c *MockControlPlane_RunVolumeBackups_Call) Run(run func(ctx context.Conte if args[2] != nil { arg2 = args[2].(string) } + var arg3 string + if args[3] != nil { + arg3 = args[3].(string) + } run( arg0, arg1, arg2, + arg3, ) }) return _c @@ -2627,43 +1678,42 @@ func (_c *MockControlPlane_RunVolumeBackups_Call) Return(volumeBackupRunResponse return _c } -func (_c *MockControlPlane_RunVolumeBackups_Call) RunAndReturn(run func(ctx context.Context, backupDomain string, volumeName string) (*dto.VolumeBackupRunResponse, error)) *MockControlPlane_RunVolumeBackups_Call { +func (_c *MockControlPlane_RunVolumeBackups_Call) RunAndReturn(run func(ctx context.Context, app string, service string, volume string) (*dto.VolumeBackupRunResponse, error)) *MockControlPlane_RunVolumeBackups_Call { _c.Call.Return(run) return _c } -// SetAttachmentSecrets provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) SetAttachmentSecrets(ctx context.Context, domain1 string, service string, secrets map[string]string) error { - ret := _mock.Called(ctx, domain1, service, secrets) +// SetAppSecrets provides a mock function for the type MockControlPlane +func (_mock *MockControlPlane) SetAppSecrets(ctx context.Context, app string, req dto.AppSecretSetRequest) error { + ret := _mock.Called(ctx, app, req) if len(ret) == 0 { - panic("no return value specified for SetAttachmentSecrets") + panic("no return value specified for SetAppSecrets") } var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, map[string]string) error); ok { - r0 = returnFunc(ctx, domain1, service, secrets) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, dto.AppSecretSetRequest) error); ok { + r0 = returnFunc(ctx, app, req) } else { r0 = ret.Error(0) } return r0 } -// MockControlPlane_SetAttachmentSecrets_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'SetAttachmentSecrets' -type MockControlPlane_SetAttachmentSecrets_Call struct { +// MockControlPlane_SetAppSecrets_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'SetAppSecrets' +type MockControlPlane_SetAppSecrets_Call struct { *mock.Call } -// SetAttachmentSecrets is a helper method to define mock.On call +// SetAppSecrets is a helper method to define mock.On call // - ctx context.Context -// - domain1 string -// - service string -// - secrets map[string]string -func (_e *MockControlPlane_Expecter) SetAttachmentSecrets(ctx any, domain1 any, service any, secrets any) *MockControlPlane_SetAttachmentSecrets_Call { - return &MockControlPlane_SetAttachmentSecrets_Call{Call: _e.mock.On("SetAttachmentSecrets", ctx, domain1, service, secrets)} +// - app string +// - req dto.AppSecretSetRequest +func (_e *MockControlPlane_Expecter) SetAppSecrets(ctx any, app any, req any) *MockControlPlane_SetAppSecrets_Call { + return &MockControlPlane_SetAppSecrets_Call{Call: _e.mock.On("SetAppSecrets", ctx, app, req)} } -func (_c *MockControlPlane_SetAttachmentSecrets_Call) Run(run func(ctx context.Context, domain1 string, service string, secrets map[string]string)) *MockControlPlane_SetAttachmentSecrets_Call { +func (_c *MockControlPlane_SetAppSecrets_Call) Run(run func(ctx context.Context, app string, req dto.AppSecretSetRequest)) *MockControlPlane_SetAppSecrets_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -2673,65 +1723,70 @@ func (_c *MockControlPlane_SetAttachmentSecrets_Call) Run(run func(ctx context.C if args[1] != nil { arg1 = args[1].(string) } - var arg2 string + var arg2 dto.AppSecretSetRequest if args[2] != nil { - arg2 = args[2].(string) - } - var arg3 map[string]string - if args[3] != nil { - arg3 = args[3].(map[string]string) + arg2 = args[2].(dto.AppSecretSetRequest) } run( arg0, arg1, arg2, - arg3, ) }) return _c } -func (_c *MockControlPlane_SetAttachmentSecrets_Call) Return(err error) *MockControlPlane_SetAttachmentSecrets_Call { +func (_c *MockControlPlane_SetAppSecrets_Call) Return(err error) *MockControlPlane_SetAppSecrets_Call { _c.Call.Return(err) return _c } -func (_c *MockControlPlane_SetAttachmentSecrets_Call) RunAndReturn(run func(ctx context.Context, domain1 string, service string, secrets map[string]string) error) *MockControlPlane_SetAttachmentSecrets_Call { +func (_c *MockControlPlane_SetAppSecrets_Call) RunAndReturn(run func(ctx context.Context, app string, req dto.AppSecretSetRequest) error) *MockControlPlane_SetAppSecrets_Call { _c.Call.Return(run) return _c } -// SetSecrets provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) SetSecrets(ctx context.Context, secretDomain string, secrets map[string]string) error { - ret := _mock.Called(ctx, secretDomain, secrets) +// ShowApp provides a mock function for the type MockControlPlane +func (_mock *MockControlPlane) ShowApp(ctx context.Context, app string) (*dto.AppShowResponse, error) { + ret := _mock.Called(ctx, app) if len(ret) == 0 { - panic("no return value specified for SetSecrets") + panic("no return value specified for ShowApp") } - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, map[string]string) error); ok { - r0 = returnFunc(ctx, secretDomain, secrets) + var r0 *dto.AppShowResponse + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string) (*dto.AppShowResponse, error)); ok { + return returnFunc(ctx, app) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string) *dto.AppShowResponse); ok { + r0 = returnFunc(ctx, app) } else { - r0 = ret.Error(0) + if ret.Get(0) != nil { + r0 = ret.Get(0).(*dto.AppShowResponse) + } } - return r0 + if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { + r1 = returnFunc(ctx, app) + } else { + r1 = ret.Error(1) + } + return r0, r1 } -// MockControlPlane_SetSecrets_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'SetSecrets' -type MockControlPlane_SetSecrets_Call struct { +// MockControlPlane_ShowApp_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ShowApp' +type MockControlPlane_ShowApp_Call struct { *mock.Call } -// SetSecrets is a helper method to define mock.On call +// ShowApp is a helper method to define mock.On call // - ctx context.Context -// - secretDomain string -// - secrets map[string]string -func (_e *MockControlPlane_Expecter) SetSecrets(ctx any, secretDomain any, secrets any) *MockControlPlane_SetSecrets_Call { - return &MockControlPlane_SetSecrets_Call{Call: _e.mock.On("SetSecrets", ctx, secretDomain, secrets)} +// - app string +func (_e *MockControlPlane_Expecter) ShowApp(ctx any, app any) *MockControlPlane_ShowApp_Call { + return &MockControlPlane_ShowApp_Call{Call: _e.mock.On("ShowApp", ctx, app)} } -func (_c *MockControlPlane_SetSecrets_Call) Run(run func(ctx context.Context, secretDomain string, secrets map[string]string)) *MockControlPlane_SetSecrets_Call { +func (_c *MockControlPlane_ShowApp_Call) Run(run func(ctx context.Context, app string)) *MockControlPlane_ShowApp_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -2741,71 +1796,71 @@ func (_c *MockControlPlane_SetSecrets_Call) Run(run func(ctx context.Context, se if args[1] != nil { arg1 = args[1].(string) } - var arg2 map[string]string - if args[2] != nil { - arg2 = args[2].(map[string]string) - } run( arg0, arg1, - arg2, ) }) return _c } -func (_c *MockControlPlane_SetSecrets_Call) Return(err error) *MockControlPlane_SetSecrets_Call { - _c.Call.Return(err) +func (_c *MockControlPlane_ShowApp_Call) Return(appShowResponse *dto.AppShowResponse, err error) *MockControlPlane_ShowApp_Call { + _c.Call.Return(appShowResponse, err) return _c } -func (_c *MockControlPlane_SetSecrets_Call) RunAndReturn(run func(ctx context.Context, secretDomain string, secrets map[string]string) error) *MockControlPlane_SetSecrets_Call { +func (_c *MockControlPlane_ShowApp_Call) RunAndReturn(run func(ctx context.Context, app string) (*dto.AppShowResponse, error)) *MockControlPlane_ShowApp_Call { _c.Call.Return(run) return _c } -// StreamContainerLogs provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) StreamContainerLogs(ctx context.Context, logDomain string, lines int) (<-chan string, error) { - ret := _mock.Called(ctx, logDomain, lines) +// StartApp provides a mock function for the type MockControlPlane +func (_mock *MockControlPlane) StartApp(ctx context.Context, app string) (*dto.AppDeployResponse, string, error) { + ret := _mock.Called(ctx, app) if len(ret) == 0 { - panic("no return value specified for StreamContainerLogs") + panic("no return value specified for StartApp") } - var r0 <-chan string - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, int) (<-chan string, error)); ok { - return returnFunc(ctx, logDomain, lines) + var r0 *dto.AppDeployResponse + var r1 string + var r2 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string) (*dto.AppDeployResponse, string, error)); ok { + return returnFunc(ctx, app) } - if returnFunc, ok := ret.Get(0).(func(context.Context, string, int) <-chan string); ok { - r0 = returnFunc(ctx, logDomain, lines) + if returnFunc, ok := ret.Get(0).(func(context.Context, string) *dto.AppDeployResponse); ok { + r0 = returnFunc(ctx, app) } else { if ret.Get(0) != nil { - r0 = ret.Get(0).(<-chan string) + r0 = ret.Get(0).(*dto.AppDeployResponse) } } - if returnFunc, ok := ret.Get(1).(func(context.Context, string, int) error); ok { - r1 = returnFunc(ctx, logDomain, lines) + if returnFunc, ok := ret.Get(1).(func(context.Context, string) string); ok { + r1 = returnFunc(ctx, app) } else { - r1 = ret.Error(1) + r1 = ret.Get(1).(string) } - return r0, r1 + if returnFunc, ok := ret.Get(2).(func(context.Context, string) error); ok { + r2 = returnFunc(ctx, app) + } else { + r2 = ret.Error(2) + } + return r0, r1, r2 } -// MockControlPlane_StreamContainerLogs_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'StreamContainerLogs' -type MockControlPlane_StreamContainerLogs_Call struct { +// MockControlPlane_StartApp_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'StartApp' +type MockControlPlane_StartApp_Call struct { *mock.Call } -// StreamContainerLogs is a helper method to define mock.On call +// StartApp is a helper method to define mock.On call // - ctx context.Context -// - logDomain string -// - lines int -func (_e *MockControlPlane_Expecter) StreamContainerLogs(ctx any, logDomain any, lines any) *MockControlPlane_StreamContainerLogs_Call { - return &MockControlPlane_StreamContainerLogs_Call{Call: _e.mock.On("StreamContainerLogs", ctx, logDomain, lines)} +// - app string +func (_e *MockControlPlane_Expecter) StartApp(ctx any, app any) *MockControlPlane_StartApp_Call { + return &MockControlPlane_StartApp_Call{Call: _e.mock.On("StartApp", ctx, app)} } -func (_c *MockControlPlane_StreamContainerLogs_Call) Run(run func(ctx context.Context, logDomain string, lines int)) *MockControlPlane_StreamContainerLogs_Call { +func (_c *MockControlPlane_StartApp_Call) Run(run func(ctx context.Context, app string)) *MockControlPlane_StartApp_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -2815,78 +1870,79 @@ func (_c *MockControlPlane_StreamContainerLogs_Call) Run(run func(ctx context.Co if args[1] != nil { arg1 = args[1].(string) } - var arg2 int - if args[2] != nil { - arg2 = args[2].(int) - } run( arg0, arg1, - arg2, ) }) return _c } -func (_c *MockControlPlane_StreamContainerLogs_Call) Return(stringCh <-chan string, err error) *MockControlPlane_StreamContainerLogs_Call { - _c.Call.Return(stringCh, err) +func (_c *MockControlPlane_StartApp_Call) Return(appDeployResponse *dto.AppDeployResponse, s string, err error) *MockControlPlane_StartApp_Call { + _c.Call.Return(appDeployResponse, s, err) return _c } -func (_c *MockControlPlane_StreamContainerLogs_Call) RunAndReturn(run func(ctx context.Context, logDomain string, lines int) (<-chan string, error)) *MockControlPlane_StreamContainerLogs_Call { +func (_c *MockControlPlane_StartApp_Call) RunAndReturn(run func(ctx context.Context, app string) (*dto.AppDeployResponse, string, error)) *MockControlPlane_StartApp_Call { _c.Call.Return(run) return _c } -// StreamProcessLogs provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) StreamProcessLogs(ctx context.Context, lines int) (<-chan string, error) { - ret := _mock.Called(ctx, lines) +// StopApp provides a mock function for the type MockControlPlane +func (_mock *MockControlPlane) StopApp(ctx context.Context, app string) (*dto.AppDeployResponse, string, error) { + ret := _mock.Called(ctx, app) if len(ret) == 0 { - panic("no return value specified for StreamProcessLogs") + panic("no return value specified for StopApp") } - var r0 <-chan string - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, int) (<-chan string, error)); ok { - return returnFunc(ctx, lines) + var r0 *dto.AppDeployResponse + var r1 string + var r2 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string) (*dto.AppDeployResponse, string, error)); ok { + return returnFunc(ctx, app) } - if returnFunc, ok := ret.Get(0).(func(context.Context, int) <-chan string); ok { - r0 = returnFunc(ctx, lines) + if returnFunc, ok := ret.Get(0).(func(context.Context, string) *dto.AppDeployResponse); ok { + r0 = returnFunc(ctx, app) } else { if ret.Get(0) != nil { - r0 = ret.Get(0).(<-chan string) + r0 = ret.Get(0).(*dto.AppDeployResponse) } } - if returnFunc, ok := ret.Get(1).(func(context.Context, int) error); ok { - r1 = returnFunc(ctx, lines) + if returnFunc, ok := ret.Get(1).(func(context.Context, string) string); ok { + r1 = returnFunc(ctx, app) } else { - r1 = ret.Error(1) + r1 = ret.Get(1).(string) } - return r0, r1 + if returnFunc, ok := ret.Get(2).(func(context.Context, string) error); ok { + r2 = returnFunc(ctx, app) + } else { + r2 = ret.Error(2) + } + return r0, r1, r2 } -// MockControlPlane_StreamProcessLogs_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'StreamProcessLogs' -type MockControlPlane_StreamProcessLogs_Call struct { +// MockControlPlane_StopApp_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'StopApp' +type MockControlPlane_StopApp_Call struct { *mock.Call } -// StreamProcessLogs is a helper method to define mock.On call +// StopApp is a helper method to define mock.On call // - ctx context.Context -// - lines int -func (_e *MockControlPlane_Expecter) StreamProcessLogs(ctx any, lines any) *MockControlPlane_StreamProcessLogs_Call { - return &MockControlPlane_StreamProcessLogs_Call{Call: _e.mock.On("StreamProcessLogs", ctx, lines)} +// - app string +func (_e *MockControlPlane_Expecter) StopApp(ctx any, app any) *MockControlPlane_StopApp_Call { + return &MockControlPlane_StopApp_Call{Call: _e.mock.On("StopApp", ctx, app)} } -func (_c *MockControlPlane_StreamProcessLogs_Call) Run(run func(ctx context.Context, lines int)) *MockControlPlane_StreamProcessLogs_Call { +func (_c *MockControlPlane_StopApp_Call) Run(run func(ctx context.Context, app string)) *MockControlPlane_StopApp_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { arg0 = args[0].(context.Context) } - var arg1 int + var arg1 string if args[1] != nil { - arg1 = args[1].(int) + arg1 = args[1].(string) } run( arg0, @@ -2896,54 +1952,65 @@ func (_c *MockControlPlane_StreamProcessLogs_Call) Run(run func(ctx context.Cont return _c } -func (_c *MockControlPlane_StreamProcessLogs_Call) Return(stringCh <-chan string, err error) *MockControlPlane_StreamProcessLogs_Call { - _c.Call.Return(stringCh, err) +func (_c *MockControlPlane_StopApp_Call) Return(appDeployResponse *dto.AppDeployResponse, s string, err error) *MockControlPlane_StopApp_Call { + _c.Call.Return(appDeployResponse, s, err) return _c } -func (_c *MockControlPlane_StreamProcessLogs_Call) RunAndReturn(run func(ctx context.Context, lines int) (<-chan string, error)) *MockControlPlane_StreamProcessLogs_Call { +func (_c *MockControlPlane_StopApp_Call) RunAndReturn(run func(ctx context.Context, app string) (*dto.AppDeployResponse, string, error)) *MockControlPlane_StopApp_Call { _c.Call.Return(run) return _c } -// UpdateRoute provides a mock function for the type MockControlPlane -func (_mock *MockControlPlane) UpdateRoute(ctx context.Context, route domain.Route) error { - ret := _mock.Called(ctx, route) +// StreamProcessLogs provides a mock function for the type MockControlPlane +func (_mock *MockControlPlane) StreamProcessLogs(ctx context.Context, lines int) (<-chan string, error) { + ret := _mock.Called(ctx, lines) if len(ret) == 0 { - panic("no return value specified for UpdateRoute") + panic("no return value specified for StreamProcessLogs") } - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, domain.Route) error); ok { - r0 = returnFunc(ctx, route) + var r0 <-chan string + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, int) (<-chan string, error)); ok { + return returnFunc(ctx, lines) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, int) <-chan string); ok { + r0 = returnFunc(ctx, lines) } else { - r0 = ret.Error(0) + if ret.Get(0) != nil { + r0 = ret.Get(0).(<-chan string) + } } - return r0 + if returnFunc, ok := ret.Get(1).(func(context.Context, int) error); ok { + r1 = returnFunc(ctx, lines) + } else { + r1 = ret.Error(1) + } + return r0, r1 } -// MockControlPlane_UpdateRoute_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'UpdateRoute' -type MockControlPlane_UpdateRoute_Call struct { +// MockControlPlane_StreamProcessLogs_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'StreamProcessLogs' +type MockControlPlane_StreamProcessLogs_Call struct { *mock.Call } -// UpdateRoute is a helper method to define mock.On call +// StreamProcessLogs is a helper method to define mock.On call // - ctx context.Context -// - route domain.Route -func (_e *MockControlPlane_Expecter) UpdateRoute(ctx any, route any) *MockControlPlane_UpdateRoute_Call { - return &MockControlPlane_UpdateRoute_Call{Call: _e.mock.On("UpdateRoute", ctx, route)} +// - lines int +func (_e *MockControlPlane_Expecter) StreamProcessLogs(ctx any, lines any) *MockControlPlane_StreamProcessLogs_Call { + return &MockControlPlane_StreamProcessLogs_Call{Call: _e.mock.On("StreamProcessLogs", ctx, lines)} } -func (_c *MockControlPlane_UpdateRoute_Call) Run(run func(ctx context.Context, route domain.Route)) *MockControlPlane_UpdateRoute_Call { +func (_c *MockControlPlane_StreamProcessLogs_Call) Run(run func(ctx context.Context, lines int)) *MockControlPlane_StreamProcessLogs_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { arg0 = args[0].(context.Context) } - var arg1 domain.Route + var arg1 int if args[1] != nil { - arg1 = args[1].(domain.Route) + arg1 = args[1].(int) } run( arg0, @@ -2953,12 +2020,12 @@ func (_c *MockControlPlane_UpdateRoute_Call) Run(run func(ctx context.Context, r return _c } -func (_c *MockControlPlane_UpdateRoute_Call) Return(err error) *MockControlPlane_UpdateRoute_Call { - _c.Call.Return(err) +func (_c *MockControlPlane_StreamProcessLogs_Call) Return(stringCh <-chan string, err error) *MockControlPlane_StreamProcessLogs_Call { + _c.Call.Return(stringCh, err) return _c } -func (_c *MockControlPlane_UpdateRoute_Call) RunAndReturn(run func(ctx context.Context, route domain.Route) error) *MockControlPlane_UpdateRoute_Call { +func (_c *MockControlPlane_StreamProcessLogs_Call) RunAndReturn(run func(ctx context.Context, lines int) (<-chan string, error)) *MockControlPlane_StreamProcessLogs_Call { _c.Call.Return(run) return _c } diff --git a/internal/adapters/in/cli/mocks/mock_images_client.go b/internal/adapters/in/cli/mocks/mock_images_client.go index 4300d7ac0..b5dd78b9d 100644 --- a/internal/adapters/in/cli/mocks/mock_images_client.go +++ b/internal/adapters/in/cli/mocks/mock_images_client.go @@ -17,10 +17,19 @@ func NewMockimagesClient(t interface { mock.TestingT Cleanup(func()) }) *MockimagesClient { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockimagesClient{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/adapters/in/cli/mocks/mock_push_image_ops.go b/internal/adapters/in/cli/mocks/mock_push_image_ops.go index 47557e21d..f69cb7f17 100644 --- a/internal/adapters/in/cli/mocks/mock_push_image_ops.go +++ b/internal/adapters/in/cli/mocks/mock_push_image_ops.go @@ -16,10 +16,19 @@ func NewMockpushImageOps(t interface { mock.TestingT Cleanup(func()) }) *MockpushImageOps { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockpushImageOps{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/adapters/in/cli/mocks/mock_volumes_client.go b/internal/adapters/in/cli/mocks/mock_volumes_client.go index 5b95f9f0d..b636f85e6 100644 --- a/internal/adapters/in/cli/mocks/mock_volumes_client.go +++ b/internal/adapters/in/cli/mocks/mock_volumes_client.go @@ -17,10 +17,19 @@ func NewMockvolumesClient(t interface { mock.TestingT Cleanup(func()) }) *MockvolumesClient { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockvolumesClient{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/adapters/in/cli/networks.go b/internal/adapters/in/cli/networks.go index 646defcdc..38683080d 100644 --- a/internal/adapters/in/cli/networks.go +++ b/internal/adapters/in/cli/networks.go @@ -12,32 +12,21 @@ import ( ) func newNetworksCmd() *cobra.Command { - cmd := &cobra.Command{ - Use: "networks", - Short: "Inspect Gordon-managed networks", - } - - cmd.AddCommand(newNetworksListCmd()) - - return cmd -} - -func newNetworksListCmd() *cobra.Command { var jsonOut bool cmd := &cobra.Command{ - Use: "list", - Short: "List Gordon-managed Docker networks", + Use: "networks", + Short: "List Gordon-managed networks", Long: `Display Docker networks managed by Gordon, including which containers are connected to each network. Examples: - gordon networks list - gordon networks list --json - gordon networks list --remote https://gordon.mydomain.com --token $TOKEN`, + gordon daemon networks + gordon daemon networks --json + gordon daemon networks --remote prod`, RunE: func(cmd *cobra.Command, _ []string) error { ctx := cmd.Context() - handle, err := resolveControlPlane(configPath) + handle, err := resolveControlPlane(cliConfigPath) if err != nil { return err } diff --git a/internal/adapters/in/cli/parity_matrix_test.go b/internal/adapters/in/cli/parity_matrix_test.go deleted file mode 100644 index 82b68ca66..000000000 --- a/internal/adapters/in/cli/parity_matrix_test.go +++ /dev/null @@ -1,91 +0,0 @@ -package cli - -import ( - "go/ast" - "go/parser" - "go/token" - "os" - "strings" - "testing" -) - -type uiAdoptionExpectation struct { - family string - file string - functions []string -} - -var uiAdoptionExpectations = []uiAdoptionExpectation{ - { - family: "root/server", - file: "root.go", - functions: []string{"newVersionCmd", "runReloadRemote", "runLogsRemote", "streamLogsRemote", "showContainerLogsLocal"}, - }, - { - family: "backups", - file: "backup.go", - functions: []string{"newBackupListCmd", "newBackupRunCmd", "newBackupDetectCmd", "newBackupStatusCmd"}, - }, - { - family: "images", - file: "images.go", - functions: []string{"runImagesList", "runImagesPrune"}, - }, -} - -func TestLocalParityMatrix(t *testing.T) { - checks := []struct { - file string - legacyText string - command string - }{ - {file: "push.go", legacyText: "push requires remote mode", command: "gordon push"}, - // Guards against reintroducing the old rollback-only remote-mode error text during the command rename. - {file: "pin.go", legacyText: "rollback requires remote mode", command: "gordon pin"}, - {file: "backup.go", legacyText: "backup commands require a configured remote target", command: "gordon backups"}, - {file: "routes.go", legacyText: "status command requires --remote flag or GORDON_REMOTE env var", command: "gordon status"}, - {file: "restart.go", legacyText: "local restart does not support --with-attachments; use --remote", command: "gordon restart --with-attachments"}, - } - - for _, check := range checks { - content, err := os.ReadFile(check.file) - if err != nil { - t.Fatalf("failed to read %s: %v", check.file, err) - } - if strings.Contains(string(content), check.legacyText) { - t.Fatalf("local parity gap: %s still blocks local mode (%q found in %s)", check.command, check.legacyText, check.file) - } - } -} - -func TestUIAdoptionMatrixCoverage(t *testing.T) { - for _, expect := range uiAdoptionExpectations { - t.Run(expect.family, func(t *testing.T) { - if len(expect.functions) == 0 { - t.Fatalf("ui adoption matrix for %s is empty", expect.family) - } - - fset := token.NewFileSet() - fileNode, err := parser.ParseFile(fset, expect.file, nil, parser.AllErrors) - if err != nil { - t.Fatalf("failed to parse %s: %v", expect.file, err) - } - - found := make(map[string]bool, len(expect.functions)) - ast.Inspect(fileNode, func(n ast.Node) bool { - fn, ok := n.(*ast.FuncDecl) - if !ok { - return true - } - found[fn.Name.Name] = true - return true - }) - - for _, fn := range expect.functions { - if !found[fn] { - t.Fatalf("ui adoption matrix references missing function %s in %s", fn, expect.file) - } - } - }) - } -} diff --git a/internal/adapters/in/cli/pin.go b/internal/adapters/in/cli/pin.go deleted file mode 100644 index d66c34a8f..000000000 --- a/internal/adapters/in/cli/pin.go +++ /dev/null @@ -1,249 +0,0 @@ -package cli - -import ( - "context" - "fmt" - "io" - "slices" - "sort" - "strings" - - "github.com/spf13/cobra" - "golang.org/x/mod/semver" - - "github.com/bnema/gordon/internal/adapters/in/cli/ui/components" - "github.com/bnema/gordon/internal/adapters/in/cli/ui/styles" - "github.com/bnema/gordon/internal/domain" -) - -func newPinCmd() *cobra.Command { - var targetTag string - - cmd := &cobra.Command{ - Use: "pin ", - Short: "Pin a route to a specific image tag", - Long: `Lists available image tags for a domain and deploys the selected version. -Tags are read from the Gordon registry. - -Examples: - gordon pin myapp.example.com --remote ... - gordon pin myapp.example.com --tag v1.0.0 --remote ... - gordon pin list myapp.example.com --remote ...`, - Args: cobra.ExactArgs(1), - RunE: func(cmd *cobra.Command, args []string) error { - return runPin(cmd.Context(), cmd.OutOrStdout(), cmd.ErrOrStderr(), args[0], targetTag) - }, - } - - cmd.Flags().StringVar(&targetTag, "tag", "", "Target tag (skips interactive selection)") - cmd.AddCommand(newPinListCmd()) - - return cmd -} - -func newPinListCmd() *cobra.Command { - var jsonOut bool - - cmd := &cobra.Command{ - Use: "list ", - Short: "List available tags", - Args: cobra.ExactArgs(1), - RunE: func(cmd *cobra.Command, args []string) error { - return runPinList(cmd.Context(), cmd.OutOrStdout(), args[0], jsonOut) - }, - } - - cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") - - return cmd -} - -func runPin(ctx context.Context, out, ew io.Writer, pinDomain, targetTag string) error { - handle, err := resolveControlPlaneForRouteDomain(ctx, pinDomain) - if err != nil { - return fmt.Errorf("failed to resolve control plane: %w", err) - } - defer handle.close() - - route, err := handle.plane.GetRoute(ctx, pinDomain) - if err != nil { - return fmt.Errorf("failed to get route: %w", err) - } - - _, imageName, currentTag := parseImageRef(route.Image) - if imageName == "" { - return fmt.Errorf("cannot parse image name from route: %s", route.Image) - } - - tags, err := fetchAndSortTags(ctx, handle.plane, imageName) - if err != nil { - return fmt.Errorf("failed to fetch tags: %w", err) - } - - selectedTag, err := selectTag(targetTag, tags, currentTag, pinDomain, out) - if err != nil { - return fmt.Errorf("failed to select tag: %w", err) - } - if selectedTag == "" { - return nil - } - - if selectedTag == currentTag { - return cliWriteLine(out, styles.RenderWarning(fmt.Sprintf("Already running %s", selectedTag))) - } - - return deploySelectedTag(ctx, handle.plane, route, out, ew, pinDomain, imageName, selectedTag) -} - -func runPinList(ctx context.Context, w io.Writer, pinDomain string, jsonOut bool) error { - handle, err := resolveControlPlaneForRouteDomain(ctx, pinDomain) - if err != nil { - return fmt.Errorf("failed to resolve control plane: %w", err) - } - defer handle.close() - - route, err := handle.plane.GetRoute(ctx, pinDomain) - if err != nil { - return fmt.Errorf("failed to get route: %w", err) - } - - _, imageName, currentTag := parseImageRef(route.Image) - if imageName == "" { - return fmt.Errorf("cannot parse image name from route: %s", route.Image) - } - - tags, err := fetchAndSortTags(ctx, handle.plane, imageName) - if err != nil { - return fmt.Errorf("failed to fetch tags: %w", err) - } - - return printPinTags(w, pinDomain, currentTag, tags, jsonOut) -} - -func printPinTags(w io.Writer, pinDomain, currentTag string, tags []string, jsonOut bool) error { - if jsonOut { - return writeJSON(w, map[string]any{ - "domain": pinDomain, - "current_tag": currentTag, - "tags": tags, - }) - } - - if _, err := fmt.Fprintf(w, "Available tags for %s:\n", styles.Theme.Bold.Render(pinDomain)); err != nil { - return err - } - for _, tag := range tags { - suffix := "" - if tag == currentTag { - suffix = " (current)" - } - if _, err := fmt.Fprintf(w, "- %s%s\n", tag, suffix); err != nil { - return err - } - } - return nil -} - -func fetchAndSortTags(ctx context.Context, cp ControlPlane, imageName string) ([]string, error) { - tags, err := cp.ListTags(ctx, imageName) - if err != nil { - return nil, fmt.Errorf("failed to list tags: %w", err) - } - if len(tags) == 0 { - return nil, fmt.Errorf("no tags found for %s", imageName) - } - - return sortSemverTags(tags), nil -} - -// sortSemverTags normalizes, classifies, and sorts tags. -// Semver tags (with or without 'v' prefix) are sorted in descending order first, -// followed by non-semver tags in descending lexicographic order. -func sortSemverTags(tags []string) []string { - var semverTags, otherTags []string - for _, tag := range tags { - sv := tag - if !strings.HasPrefix(sv, "v") { - sv = "v" + sv - } - if semver.IsValid(sv) { - semverTags = append(semverTags, tag) - } else { - otherTags = append(otherTags, tag) - } - } - - // Sort semver tags in descending order (latest first) - sort.Slice(semverTags, func(i, j int) bool { - si, sj := semverTags[i], semverTags[j] - if !strings.HasPrefix(si, "v") { - si = "v" + si - } - if !strings.HasPrefix(sj, "v") { - sj = "v" + sj - } - return semver.Compare(si, sj) > 0 - }) - - // Sort non-semver tags in descending lexicographic order - sort.Slice(otherTags, func(i, j int) bool { - return otherTags[i] > otherTags[j] - }) - - // Combine: semver tags first, then other tags - return append(semverTags, otherTags...) -} - -func selectTag(targetTag string, tags []string, currentTag, pinDomain string, out io.Writer) (string, error) { - if targetTag != "" { - if !validateTagExists(targetTag, tags) { - return "", fmt.Errorf("tag %s not found. Available: %s", targetTag, strings.Join(tags, ", ")) - } - return targetTag, nil - } - - selectedTag, err := components.RunSelector( - fmt.Sprintf("Select version for %s:", pinDomain), - tags, - currentTag, - ) - if err != nil { - return "", err - } - if selectedTag == "" { - _ = cliWriteLine(out, "Cancelled.") - } - return selectedTag, nil -} - -func validateTagExists(targetTag string, tags []string) bool { - return slices.Contains(tags, targetTag) -} - -func deploySelectedTag(ctx context.Context, cp ControlPlane, route *domain.Route, out, ew io.Writer, pinDomain, imageName, selectedTag string) error { - registry, _, _ := parseImageRef(route.Image) - oldImage := route.Image - route.Image = fmt.Sprintf("%s/%s:%s", registry, imageName, selectedTag) - - if err := cliWritef(out, "Pinning to %s...\n", styles.Theme.Bold.Render(selectedTag)); err != nil { - return err - } - - if err := cp.UpdateRoute(ctx, *route); err != nil { - return fmt.Errorf("failed to update route: %w", err) - } - - result, err := cp.Deploy(ctx, pinDomain) - if err != nil { - formattedErr := formatDeployFailure(err) - // Attempt to revert route to previous image - route.Image = oldImage - if revertErr := cp.UpdateRoute(ctx, *route); revertErr != nil { - _ = cliWritef(ew, "WARNING: deploy failed and could not revert route: %v\n", revertErr) - return fmt.Errorf("failed to deploy; revert failed: %w; deploy error: %v", revertErr, formattedErr) - } - return fmt.Errorf("deploy failed; route reverted to previous image: %w", formattedErr) - } - containerID := shortContainerID(result.ContainerID) - return cliWriteLine(out, styles.RenderSuccess(fmt.Sprintf("Pinned %s to %s (container: %s)", pinDomain, selectedTag, containerID))) -} diff --git a/internal/adapters/in/cli/pin_test.go b/internal/adapters/in/cli/pin_test.go deleted file mode 100644 index ad2ec30d5..000000000 --- a/internal/adapters/in/cli/pin_test.go +++ /dev/null @@ -1,104 +0,0 @@ -package cli - -import ( - "bytes" - "context" - "errors" - "testing" - - climocks "github.com/bnema/gordon/internal/adapters/in/cli/mocks" - "github.com/bnema/gordon/internal/domain" - "github.com/stretchr/testify/assert" -) - -type failingWriter struct{} - -func (failingWriter) Write(_ []byte) (int, error) { - return 0, errors.New("write failed") -} - -func TestSortSemverTags(t *testing.T) { - tests := []struct { - name string - tags []string - wantTags []string - }{ - { - name: "v-prefixed tags sorted descending", - tags: []string{"v1.0.0", "v2.0.0", "v1.5.0"}, - wantTags: []string{"v2.0.0", "v1.5.0", "v1.0.0"}, - }, - { - name: "non-v-prefixed tags sorted as semver", - tags: []string{"1.0.0", "2.0.0", "1.5.0"}, - wantTags: []string{"2.0.0", "1.5.0", "1.0.0"}, - }, - { - name: "mixed v and non-v prefixed", - tags: []string{"v1.0.0", "2.0.0", "1.5.0"}, - wantTags: []string{"2.0.0", "1.5.0", "v1.0.0"}, // all recognized as semver, sorted desc - }, - { - name: "pre-release tags", - tags: []string{"v1.0.0-rc1", "v1.0.0", "1.0.0-beta"}, - wantTags: []string{"v1.0.0", "v1.0.0-rc1", "1.0.0-beta"}, // release > pre-release - }, - { - name: "non-semver tags appended after", - tags: []string{"latest", "v1.0.0", "dev"}, - wantTags: []string{"v1.0.0", "latest", "dev"}, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - gotTags := sortSemverTags(tt.tags) - assert.Equal(t, tt.wantTags, gotTags) - }) - } -} - -func TestValidateTagExists(t *testing.T) { - assert.True(t, validateTagExists("v1.0.0", []string{"v1.0.0", "v2.0.0"})) - assert.False(t, validateTagExists("v3.0.0", []string{"v1.0.0", "v2.0.0"})) -} - -func TestPrintPinTags_WritesToProvidedWriter(t *testing.T) { - var buf bytes.Buffer - - err := printPinTags(&buf, "myapp.example.com", "v1.2.0", []string{"v1.2.0", "v1.1.0"}, false) - assert.NoError(t, err) - - output := buf.String() - assert.Contains(t, output, "Available tags for") - assert.Contains(t, output, "myapp.example.com") - assert.Contains(t, output, "- v1.2.0 (current)") - assert.Contains(t, output, "- v1.1.0") -} - -func TestPrintPinTags_ReturnsWriteError(t *testing.T) { - err := printPinTags(failingWriter{}, "myapp.example.com", "v1.2.0", []string{"v1.2.0"}, false) - assert.Error(t, err) - assert.Contains(t, err.Error(), "write failed") -} - -func TestDeploySelectedTag_FormatsDeployFailureAfterRevert(t *testing.T) { - cp := climocks.NewMockControlPlane(t) - route := &domain.Route{Domain: "app.example.com", Image: "registry.example.com/myapp:latest"} - deployErr := &domain.DeployFailureError{ - Summary: "failed to deploy", - Cause: "health check failed", - Hint: "check DATABASE_URL", - } - - cp.EXPECT().UpdateRoute(context.Background(), domain.Route{Domain: "app.example.com", Image: "registry.example.com/myapp:v2"}).Return(nil).Once() - cp.EXPECT().Deploy(context.Background(), "app.example.com").Return(nil, deployErr).Once() - cp.EXPECT().UpdateRoute(context.Background(), domain.Route{Domain: "app.example.com", Image: "registry.example.com/myapp:latest"}).Return(nil).Once() - - err := deploySelectedTag(context.Background(), cp, route, &bytes.Buffer{}, &bytes.Buffer{}, "app.example.com", "myapp", "v2") - assert.Error(t, err) - assert.Contains(t, err.Error(), "failed to deploy") - assert.Contains(t, err.Error(), "Cause: health check failed") - assert.Contains(t, err.Error(), "Hint: check DATABASE_URL") - assert.Contains(t, err.Error(), "route reverted to previous image") -} diff --git a/internal/adapters/in/cli/presentation.go b/internal/adapters/in/cli/presentation.go index 2b35f2529..6c5490d5b 100644 --- a/internal/adapters/in/cli/presentation.go +++ b/internal/adapters/in/cli/presentation.go @@ -63,11 +63,3 @@ func cliRenderWarning(msg string) string { func cliRenderInfo(msg string) string { return styles.RenderInfo(msg) } - -func shortContainerID(id string) string { - if len(id) > 12 { - return id[:12] - } - - return id -} diff --git a/internal/adapters/in/cli/preview.go b/internal/adapters/in/cli/preview.go deleted file mode 100644 index 9937ac022..000000000 --- a/internal/adapters/in/cli/preview.go +++ /dev/null @@ -1,469 +0,0 @@ -package cli - -import ( - "context" - "fmt" - "io" - "os" - "os/exec" - "strings" - "text/tabwriter" - "time" - - tea "github.com/charmbracelet/bubbletea" - "github.com/spf13/cobra" - - "github.com/bnema/gordon/internal/adapters/in/cli/remote" - "github.com/bnema/gordon/internal/adapters/in/cli/ui/components" - "github.com/bnema/gordon/internal/domain" - "github.com/bnema/gordon/internal/usecase/auto/preview" -) - -func newPreviewCmd() *cobra.Command { - cmd := &cobra.Command{ - Use: "preview", - Short: "Create or manage preview environments", - Long: "Build, push, and deploy an ephemeral preview environment. Use a subcommand (create, list, delete, extend).", - } - - createCmd := newPreviewCreateCmd() - - cmd.AddCommand( - createCmd, - newPreviewListCmd(), - newPreviewDeleteCmd(), - newPreviewExtendCmd(), - ) - - return cmd -} - -func newPreviewCreateCmd() *cobra.Command { - var ( - ttl string - noBuild bool - noData bool - platform string - ) - - cmd := &cobra.Command{ - Use: "create [name]", - Short: "Create a preview environment", - Args: cobra.MaximumNArgs(1), - RunE: func(cmd *cobra.Command, args []string) error { - ctx := cmd.Context() - out := cmd.OutOrStdout() - - // Resolve preview name from arg or git branch. - name, err := resolvePreviewName(ctx, args) - if err != nil { - return err - } - - previewTag := "preview-" + name - - if err := cliWriteLine(out, cliRenderTitle("Preview: "+name)); err != nil { - return err - } - - // Resolve control plane (remote or local) — same as push. - handle, err := resolveControlPlane(configPath) - if err != nil { - return err - } - defer handle.close() - - previewRef, err := resolvePreviewImageRef(ctx, handle.plane, out, previewTag) - if err != nil { - return err - } - - if err := buildAndPushPreview(ctx, out, previewRef, previewTag, platform, noBuild); err != nil { - return err - } - - return waitForPreview(ctx, out, name, ttl, noData) - }, - } - - cmd.Flags().StringVar(&ttl, "ttl", "", "Override TTL (e.g., 72h)") - cmd.Flags().BoolVar(&noBuild, "no-build", false, "Skip image build") - cmd.Flags().BoolVar(&noData, "no-data", false, "Skip volume cloning") - cmd.Flags().StringVar(&platform, "platform", "linux/amd64", "Target platform for build") - return cmd -} - -func newPreviewListCmd() *cobra.Command { - var jsonOutput bool - cmd := &cobra.Command{ - Use: "list", - Short: "List active preview environments", - RunE: func(cmd *cobra.Command, args []string) error { - ctx := cmd.Context() - out := cmd.OutOrStdout() - - client, isRemote, err := GetRemoteClient() - if err != nil { - return err - } - if !isRemote { - return fmt.Errorf("preview list requires --remote (local preview listing is not yet supported)") - } - - previews, err := client.ListPreviews(ctx) - if err != nil { - return fmt.Errorf("failed to list previews: %w", err) - } - - if jsonOutput { - return writeJSON(out, previews) - } - - if len(previews) == 0 { - return cliWriteLine(out, cliRenderEmptyState("No active previews")) - } - - return printPreviewTable(out, previews) - }, - } - cmd.Flags().BoolVar(&jsonOutput, "json", false, "Output as JSON") - return cmd -} - -func newPreviewDeleteCmd() *cobra.Command { - return &cobra.Command{ - Use: "delete ", - Short: "Delete a preview environment", - Args: cobra.ExactArgs(1), - RunE: func(cmd *cobra.Command, args []string) error { - ctx := cmd.Context() - out := cmd.OutOrStdout() - name := args[0] - - client, isRemote, err := GetRemoteClient() - if err != nil { - return err - } - if !isRemote { - return fmt.Errorf("preview delete requires --remote (local preview deletion is not yet supported)") - } - - if err := client.DeletePreview(ctx, name); err != nil { - return fmt.Errorf("failed to delete preview %q: %w", name, err) - } - - return cliWriteLine(out, cliRenderSuccess(fmt.Sprintf("Preview %q deleted", name))) - }, - } -} - -func newPreviewExtendCmd() *cobra.Command { - var ttl string - cmd := &cobra.Command{ - Use: "extend ", - Short: "Extend a preview environment's TTL", - Args: cobra.ExactArgs(1), - RunE: func(cmd *cobra.Command, args []string) error { - ctx := cmd.Context() - out := cmd.OutOrStdout() - name := args[0] - - client, isRemote, err := GetRemoteClient() - if err != nil { - return err - } - if !isRemote { - return fmt.Errorf("preview extend requires --remote (local preview extend is not yet supported)") - } - - if err := client.ExtendPreview(ctx, name, ttl); err != nil { - return fmt.Errorf("failed to extend preview %q: %w", name, err) - } - - return cliWriteLine(out, cliRenderSuccess(fmt.Sprintf("Preview %q extended by %s", name, ttl))) - }, - } - cmd.Flags().StringVar(&ttl, "ttl", "24h", "Additional TTL duration") - return cmd -} - -func printPreviewTable(out io.Writer, previews []domain.PreviewRoute) error { - w := tabwriter.NewWriter(out, 0, 0, 2, ' ', 0) - if _, err := fmt.Fprintln(w, "NAME\tDOMAIN\tIMAGE\tCREATED\tEXPIRES"); err != nil { - return err - } - now := time.Now() - for _, p := range previews { - remaining := time.Until(p.ExpiresAt).Truncate(time.Minute) - expiresStr := remaining.String() - if p.IsExpired(now) { - expiresStr = "expired" - } - if _, err := fmt.Fprintf(w, "%s\t%s\t%s\t%s\t%s\n", - p.Name, - p.Domain, - truncateImage(p.Image, 40), - p.CreatedAt.Format(time.DateTime), - expiresStr, - ); err != nil { - return err - } - } - return w.Flush() -} - -func resolvePreviewName(ctx context.Context, args []string) (string, error) { - if len(args) > 0 { - name := preview.SanitizeBranchName(args[0]) - if name == "" { - return "", fmt.Errorf("invalid preview name after sanitization: %q", args[0]) - } - return name, nil - } - branch, err := detectGitBranch(ctx) - if err != nil { - return "", fmt.Errorf("no preview name provided and could not detect git branch: %w", err) - } - name := preview.SanitizeBranchName(branch) - if name == "" { - return "", fmt.Errorf("git branch %q produced empty preview name after sanitization", branch) - } - return name, nil -} - -func resolvePreviewImageRef(ctx context.Context, cp ControlPlane, out io.Writer, previewTag string) (string, error) { - dockerfile := "Dockerfile" - imageName, err := detectImageName(dockerfile) - if err != nil { - return "", err - } - if err := cliWriteLine(out, cliRenderMeta("Image:", imageName)); err != nil { - return "", err - } - - // Resolve registry from existing routes without showing the interactive - // selector — preview only needs the registry URL, not a target domain. - registry, err := resolveRegistryForImage(ctx, cp, imageName) - if err != nil { - return "", err - } - - previewRef := fmt.Sprintf("%s/%s:%s", registry, imageName, previewTag) - if err := cliWriteLine(out, cliRenderMeta("Tag:", previewRef)); err != nil { - return "", err - } - return previewRef, nil -} - -// resolveRegistryForImage finds the registry URL from any existing non-preview -// route for the given image, without prompting. -func resolveRegistryForImage(ctx context.Context, cp ControlPlane, imageName string) (string, error) { - routes, err := cp.FindRoutesByImage(ctx, imageName) - if err != nil { - return "", fmt.Errorf("failed to find routes for image %q: %w", imageName, err) - } - - for _, r := range routes { - if strings.Contains(r.Domain, domain.DefaultPreviewSeparator) { - continue - } - reg, _, _ := parseImageRef(r.Image) - if reg != "" { - return reg, nil - } - } - - return "", fmt.Errorf("no route configured for image %q", imageName) -} - -func buildAndPushPreview(ctx context.Context, out io.Writer, previewRef, previewTag, platform string, noBuild bool) error { - imageOps, err := newImageOpsFn() - if err != nil { - return err - } - - if !noBuild { - dockerfile := "Dockerfile" - if _, statErr := os.Stat(dockerfile); os.IsNotExist(statErr) { - return fmt.Errorf("dockerfile not found: %s", dockerfile) - } - buildArgs := buildImageArgs(ctx, previewTag, platform, dockerfile, nil, previewRef, previewRef) - if err := cliWriteLine(out, "\nBuilding image..."); err != nil { - return err - } - if err := imageOps.Build(ctx, buildArgs); err != nil { - return err - } - } - - if err := cliWriteLine(out, "Pushing..."); err != nil { - return err - } - if err := imageOps.Push(ctx, previewRef); err != nil { - return fmt.Errorf("failed to push %s: %w", previewRef, err) - } - return nil -} - -func printPreviewFlags(out io.Writer, ttl string, noData bool) error { - if ttl != "" { - if err := cliWritef(out, "Requested TTL: %s (server config controls actual TTL)\n", ttl); err != nil { - return err - } - } - if noData { - if err := cliWriteLine(out, cliRenderInfo("Data copy: skipped (--no-data)")); err != nil { - return err - } - } - return nil -} - -func waitForPreview(ctx context.Context, out io.Writer, name, ttl string, noData bool) error { - if err := printPreviewFlags(out, ttl, noData); err != nil { - return err - } - - client, isRemote, err := GetRemoteClient() - if err != nil { - return err - } - if !isRemote { - return cliWriteLine(out, cliRenderSuccess(fmt.Sprintf("Push complete. Preview %q will be created by the server (use --remote to poll status).", name))) - } - - result, err := pollPreviewWithSpinner(ctx, out, client, name) - if err != nil { - return err - } - - switch result.Status { - case domain.PreviewStatusRunning: - scheme := "http" - if result.HTTPS { - scheme = "https" - } - previewURL := fmt.Sprintf("%s://%s", scheme, result.Domain) - if err := cliWriteLine(out, cliRenderSuccess("Preview deployed")); err != nil { - return err - } - return cliWriteLine(out, cliRenderMeta("URL:", previewURL)) - case domain.PreviewStatusFailed: - return fmt.Errorf("preview deployment failed") - default: - return cliWriteLine(out, cliRenderInfo("Preview is still deploying. Check status with: gordon preview list --remote")) - } -} - -type previewPollResult struct { - preview *domain.PreviewRoute - err error -} - -type previewPollDoneMsg previewPollResult - -type previewSpinnerModel struct { - spinner components.SpinnerModel - done <-chan previewPollResult - outcome previewPollResult - finished bool -} - -func newPreviewSpinnerModel(name string, done <-chan previewPollResult) previewSpinnerModel { - return previewSpinnerModel{ - spinner: components.NewSpinner( - components.WithMessage(fmt.Sprintf("Deploying preview %s...", name)), - components.WithSpinnerType(components.SpinnerMiniDot), - ), - done: done, - } -} - -func (m previewSpinnerModel) Init() tea.Cmd { - return tea.Batch(m.spinner.Init(), waitForPreviewPollDone(m.done)) -} - -func (m previewSpinnerModel) Update(msg tea.Msg) (tea.Model, tea.Cmd) { - switch msg := msg.(type) { - case previewPollDoneMsg: - m.outcome = previewPollResult(msg) - m.finished = true - return m, tea.Quit - default: - updated, cmd := m.spinner.Update(msg) - if sm, ok := updated.(components.SpinnerModel); ok { - m.spinner = sm - } - return m, cmd - } -} - -func (m previewSpinnerModel) View() string { - return m.spinner.View() -} - -func waitForPreviewPollDone(done <-chan previewPollResult) tea.Cmd { - return func() tea.Msg { - return previewPollDoneMsg(<-done) - } -} - -func pollPreviewWithSpinner(ctx context.Context, out io.Writer, client *remote.Client, name string) (*domain.PreviewRoute, error) { - done := make(chan previewPollResult, 1) - go func() { - result, err := pollPreviewStatus(ctx, client, name) - done <- previewPollResult{preview: result, err: err} - }() - - if !isInteractiveTerminal() { - if err := cliWritef(out, "Waiting for preview %s...\n", name); err != nil { - return nil, err - } - r := <-done - return r.preview, r.err - } - - model := newPreviewSpinnerModel(name, done) - final, err := tea.NewProgram(model, tea.WithContext(ctx)).Run() - fmt.Print("\r\033[K") - if err != nil { - return nil, err - } - - m, ok := final.(previewSpinnerModel) - if !ok || !m.finished { - return nil, fmt.Errorf("preview spinner exited unexpectedly") - } - return m.outcome.preview, m.outcome.err -} - -func pollPreviewStatus(ctx context.Context, client *remote.Client, name string) (*domain.PreviewRoute, error) { - ticker := time.NewTicker(2 * time.Second) - defer ticker.Stop() - timeout := time.After(3 * time.Minute) - - for { - select { - case <-ctx.Done(): - return nil, ctx.Err() - case <-timeout: - return &domain.PreviewRoute{Status: domain.PreviewStatusTimeout}, nil - case <-ticker.C: - p, err := client.GetPreview(ctx, name) - if err != nil { - continue - } - if p.Status == domain.PreviewStatusRunning || p.Status == domain.PreviewStatusFailed { - return p, nil - } - } - } -} - -func detectGitBranch(ctx context.Context) (string, error) { - out, err := exec.CommandContext(ctx, "git", "rev-parse", "--abbrev-ref", "HEAD").Output() // #nosec G204 - if err != nil { - return "", fmt.Errorf("detect git branch: %w", err) - } - return strings.TrimSpace(string(out)), nil -} diff --git a/internal/adapters/in/cli/prune.go b/internal/adapters/in/cli/prune.go new file mode 100644 index 000000000..4dce2b243 --- /dev/null +++ b/internal/adapters/in/cli/prune.go @@ -0,0 +1,76 @@ +package cli + +import ( + "fmt" + "io" + + "github.com/bnema/gordon/internal/adapters/dto" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/pkg/bytesize" +) + +// cliRenderPruneSummary writes the shared prune verdict summary: counts +// per verdict, what was removed, and the reasons protected or unknown +// candidates were skipped. Dry-run and execution use the same renderer. +func cliRenderPruneSummary(out io.Writer, summary dto.PruneSummary) error { + action := "Deleted" + if !summary.Applied { + action = "Would delete" + } + // The wire summary carries the authoritative counts, so the human + // output can never disagree with the JSON body. + actionCount := len(summary.Deleted) + if !summary.Applied { + actionCount = summary.Eligible + } + if err := cliWritef(out, "%s: %d (eligible=%d protected=%d unknown=%d)\n", + action, actionCount, summary.Eligible, summary.Protected, summary.Unknown); err != nil { + return err + } + if summary.ReclaimedKnown { + if err := cliWritef(out, "Space reclaimed: %s\n", bytesize.Format(summary.ReclaimedBytes)); err != nil { + return err + } + } + for _, failure := range summary.Failures { + line := fmt.Sprintf("failed to delete %s %s: %s", failure.Kind, failure.Ref, failure.Error) + if err := cliWriteLine(out, cliRenderWarning(line)); err != nil { + return err + } + } + for _, gap := range summary.Gaps { + line := fmt.Sprintf("inventory incomplete (%s: %s); dependent candidates were left untouched", gap.Source, gap.Reason) + if err := cliWriteLine(out, cliRenderWarning(line)); err != nil { + return err + } + } + return cliRenderSkippedCandidates(out, summary.Candidates) +} + +// cliRenderSkippedCandidates lists the not-deleted candidates with their +// first reason, so an operator can see why each resource survived. +func cliRenderSkippedCandidates(out io.Writer, candidates []dto.PruneCandidate) error { + skipped := make([]dto.PruneCandidate, 0, len(candidates)) + for _, candidate := range candidates { + if candidate.Verdict == string(domain.PruneVerdictEligible) { + continue + } + skipped = append(skipped, candidate) + } + if len(skipped) == 0 { + return nil + } + if err := cliWriteLine(out, cliRenderMuted("Skipped:")); err != nil { + return err + } + for _, candidate := range skipped { + reason := "" + if len(candidate.Reasons) > 0 { + reason = candidate.Reasons[0] + } + if err := cliWritef(out, " %s %s (%s) %s\n", candidate.Kind, candidate.Ref, candidate.Verdict, reason); err != nil { + return err + } + } + return nil +} diff --git a/internal/adapters/in/cli/push.go b/internal/adapters/in/cli/push.go index 3ae005a0f..81f62bcf7 100644 --- a/internal/adapters/in/cli/push.go +++ b/internal/adapters/in/cli/push.go @@ -6,7 +6,6 @@ import ( "errors" "fmt" "io" - "net/http" "os" "os/exec" "path/filepath" @@ -14,11 +13,9 @@ import ( "strings" "time" - tea "github.com/charmbracelet/bubbletea" "github.com/spf13/cobra" "github.com/bnema/gordon/internal/adapters/in/cli/remote" - "github.com/bnema/gordon/internal/adapters/in/cli/ui/components" "github.com/bnema/gordon/internal/adapters/in/cli/ui/styles" "github.com/bnema/gordon/internal/domain" "github.com/bnema/gordon/pkg/validation" @@ -38,6 +35,7 @@ type buildConfig struct { type imagePush struct { Registry string ImageName string + SourceRef string Version string VersionRef string LatestRef string @@ -45,48 +43,33 @@ type imagePush struct { // pushRequest holds all inputs for the push command. type pushRequest struct { - ImageArg string - Domain string - Tag string - Build buildConfig - NoDeploy bool - NoConfirm bool + ImageArg string + Tag string + Build buildConfig } func newPushCmd() *cobra.Command { var ( - noDeploy bool - noConfirm bool build bool platform string tag string dockerfile string buildArgs []string - domainFlag string ) cmd := &cobra.Command{ Use: "push [image]", - Short: "Tag, push, and optionally deploy an image", + Short: "Tag and push an image to the Gordon registry", Long: `Tags a local image for the Gordon registry and pushes it. -Uses git tags for versioning. Optionally triggers deployment after push. - -Image-first syntax is primary: - - Positional args resolve routes by image name first - - Tagged positional refs still resolve routes by image name - - The pushed version comes from --tag, CI tag refs, or git describe - - Dotted positional refs fall back to legacy domain lookup if no image route exists - - --domain is a deploy target override for legacy workflows - - No-arg mode still auto-detects from Dockerfile labels or current directory name - -With --build, builds the image first using docker buildx. +Uses git tags for versioning. Push transfers OCI content only: it never +deploys. Deploy separately with ` + "`gordon apps deploy`" + ` after applying +the manifest that references the pushed tag. Examples: - gordon push myapp --build --remote ... - gordon push myapp:v1.2.3 --tag v1.2.3 --no-deploy --remote ... - gordon push registry.example.com/myapp:v1.2.3 --build --remote ... - gordon push --domain myapp.example.com --build --remote ... - gordon push --build --build-arg CGO_ENABLED=0 --remote ...`, + gordon images push myapp --build --remote ... + gordon images push myapp:v1.2.3 --tag v1.2.3 --remote ... + gordon images push registry.example.com/myapp:v1.2.3 --build --remote ... + gordon images push --build --build-arg CGO_ENABLED=0 --remote ...`, Args: cobra.MaximumNArgs(1), RunE: func(cmd *cobra.Command, args []string) error { var imageArg string @@ -94,24 +77,18 @@ Examples: imageArg = args[0] } return runPush(cmd.Context(), cmd.OutOrStdout(), pushRequest{ - ImageArg: imageArg, - Domain: domainFlag, - Tag: tag, - Build: buildConfig{Enabled: build, Platform: platform, Dockerfile: dockerfile, BuildArgs: buildArgs}, - NoDeploy: noDeploy, - NoConfirm: noConfirm, + ImageArg: imageArg, + Tag: tag, + Build: buildConfig{Enabled: build, Platform: platform, Dockerfile: dockerfile, BuildArgs: buildArgs}, }) }, } - cmd.Flags().BoolVar(&noDeploy, "no-deploy", false, "Push only, don't trigger deploy") - cmd.Flags().BoolVar(&noConfirm, "no-confirm", false, "Skip deploy confirmation prompt") cmd.Flags().BoolVar(&build, "build", false, "Build the image first using docker buildx") cmd.Flags().StringVar(&platform, "platform", "linux/amd64", "Target platform (used with --build)") cmd.Flags().StringVarP(&dockerfile, "file", "f", "", "Path to Dockerfile (default: ./Dockerfile, used with --build)") cmd.Flags().StringVar(&tag, "tag", "", "Override pushed version tag (default: CI tag ref or git describe)") cmd.Flags().StringArrayVar(&buildArgs, "build-arg", nil, "Additional build args (used with --build)") - cmd.Flags().StringVar(&domainFlag, "domain", "", "Explicit domain override (legacy mode)") return cmd } @@ -141,7 +118,8 @@ const ( type classifiedPushArg struct { kind pushArgKind - lookupImage string + sourceRef string + repository string legacyDomain string } @@ -150,11 +128,11 @@ func classifyPushArgument(arg string) classifiedPushArg { return classifiedPushArg{kind: pushArgKindLegacyDomain, legacyDomain: arg} } - classified := classifiedPushArg{kind: pushArgKindImage, lookupImage: arg} - if parsedImage, ref := validation.ParseImageReference(arg); parsedImage != arg && !strings.HasPrefix(ref, "sha256:") { - classified.lookupImage = parsedImage + _, repository, _ := parseImageRef(arg) + if repository == "" { + repository, _ = validation.ParseImageReference(arg) } - return classified + return classifiedPushArg{kind: pushArgKindImage, sourceRef: arg, repository: repository} } func looksLikeLegacyDomain(arg string) bool { @@ -168,51 +146,29 @@ func looksLikeLegacyDomain(arg string) bool { return strings.Contains(host, ".") } -// resolveRoute determines the registry, image name, and domain from the input mode. -func resolveRoute(ctx context.Context, cp ControlPlane, imageArg, domainFlag, dockerfile string) (registry, imageName, pushDomain string, err error) { - if domainFlag != "" { - return resolveFromDomain(ctx, cp, domainFlag) - } - +// resolveImageTarget determines the registry and image name from the input. +// Push transfers OCI content only and never triggers deploys or route +// lookups. The deploy target comes from the app manifest applied +// separately via `gordon apps apply`. +func resolveImageTarget(out io.Writer, imageArg, dockerfile string) (registry, imageName, sourceRef string, err error) { if imageArg == "" { imageArg, err = detectImageName(dockerfile) if err != nil { return "", "", "", err } - fmt.Printf("Detected image: %s\n", styles.Theme.Bold.Render(imageArg)) - } - - if looksLikeLegacyDomain(imageArg) { - lookupImage := imageArg - domainLookupArg := imageArg - if parsedImage, ref := validation.ParseImageReference(imageArg); parsedImage != imageArg { - domainLookupArg = parsedImage - if !strings.HasPrefix(ref, "sha256:") { - lookupImage = parsedImage - } - } - - registry, imageName, pushDomain, err = resolveFromImage(ctx, cp, lookupImage, dockerfile) - if err == nil { - return registry, imageName, pushDomain, nil - } - if !isImageBootstrapError(err) { + if err := cliWritef(out, "Detected image: %s\n", styles.Theme.Bold.Render(imageArg)); err != nil { return "", "", "", err } - - registry, imageName, pushDomain, err = resolveFromDomain(ctx, cp, domainLookupArg) - if err == nil { - return registry, imageName, pushDomain, nil - } - if errors.Is(err, domain.ErrRouteNotFound) { - return "", "", "", noRouteForImageError(domainLookupArg) - } - return "", "", "", err } - + if looksLikeLegacyDomain(imageArg) { + return "", "", "", fmt.Errorf("domain-style push targets are retired: push an image name, then `gordon apps apply` (got %q)", imageArg) + } classified := classifyPushArgument(imageArg) - - return resolveFromImage(ctx, cp, classified.lookupImage, dockerfile) + registry, imageName, _ = parseImageRef(classified.sourceRef) + if registry == "" || imageName == "" { + return "", "", "", fmt.Errorf("cannot parse registry/image from %q", imageArg) + } + return registry, imageName, classified.sourceRef, nil } // resolveVersion determines and validates the version tag. @@ -233,7 +189,7 @@ func runPush(ctx context.Context, out io.Writer, req pushRequest) error { } defer handle.close() - registry, imageName, pushDomain, err := resolveRoute(ctx, handle.plane, req.ImageArg, req.Domain, dockerfile) + registry, imageName, sourceRef, err := resolveImageTarget(out, req.ImageArg, dockerfile) if err != nil { return err } @@ -246,7 +202,7 @@ func runPush(ctx context.Context, out io.Writer, req pushRequest) error { return err } - img := imagePush{Registry: registry, ImageName: imageName, Version: version} + img := imagePush{Registry: registry, ImageName: imageName, SourceRef: sourceRef, Version: version} img.VersionRef, img.LatestRef = resolveImageRefs(registry, imageName, version) imageOps, err := newPushImageOps(inferredRemote) @@ -262,22 +218,14 @@ func runPush(ctx context.Context, out io.Writer, req pushRequest) error { return err } } - if err := cliWritef(out, "Domain: %s\n", styles.Theme.Bold.Render(pushDomain)); err != nil { - return err - } - - skipExplicitDeploy := shouldSkipDeploy(ctx, handle.plane, imageName, req.NoDeploy) build := buildConfig{Enabled: req.Build.Enabled, Platform: req.Build.Platform, Dockerfile: dockerfile, BuildArgs: req.Build.BuildArgs} - if err := pushResolvedImage(ctx, imageOps, build, img); err != nil { + if err := pushResolvedImage(ctx, out, imageOps, build, img); err != nil { return err } if err := cliWriteLine(out, styles.RenderSuccess("Push complete")); err != nil { return err } - if !skipExplicitDeploy { - return deployAfterPush(ctx, handle.plane, pushDomain, req.NoConfirm) - } return nil } @@ -286,7 +234,7 @@ func resolvePushTarget(ctx context.Context, out io.Writer, req pushRequest) (doc if err != nil { return "", nil, nil, err } - inferredRemote, err = inferPushRemote(ctx, req.ImageArg, req.Domain, dockerfile) + inferredRemote, err = inferPushRemote(ctx, req.ImageArg, "", dockerfile) if err != nil { return "", nil, nil, err } @@ -297,7 +245,7 @@ func resolvePushTarget(ctx context.Context, out io.Writer, req pushRequest) (doc } return dockerfile, inferredRemote, handle, nil } - handle, err = resolveControlPlane(configPath) + handle, err = resolveControlPlane(cliConfigPath) if err != nil { return "", nil, nil, err } @@ -320,179 +268,15 @@ func newPushImageOps(inferredRemote *remote.ResolvedRemote) (pushImageOps, error return newImageOpsFn() } -func pushResolvedImage(ctx context.Context, imageOps pushImageOps, build buildConfig, img imagePush) error { +func pushResolvedImage(ctx context.Context, out io.Writer, imageOps pushImageOps, build buildConfig, img imagePush) error { if build.Enabled { - return buildAndPush(ctx, imageOps, build, img) - } - return tagAndPush(ctx, imageOps, img) -} - -// shouldSkipDeploy signals deploy intent and returns whether the explicit deploy -// should be skipped (either because --no-deploy was set, or the token lacks scope). -func shouldSkipDeploy(ctx context.Context, plane ControlPlane, imageName string, noDeploy bool) bool { - // Always call DeployIntent to suppress server-side auto-deploy, even with --no-deploy. - // Without this, the server's event listener would still auto-deploy the pushed image. - if err := plane.DeployIntent(ctx, imageName); err != nil { - if isInsufficientScope(err) { - fmt.Fprintln(os.Stderr, "info: deploy intent skipped (insufficient scope), server will auto-deploy on image receive") - return true - } - // Non-fatal: worst case we get a redundant deploy via event - fmt.Fprintf(os.Stderr, "warning: failed to register deploy intent: %v\n", err) - } - return noDeploy -} - -// isInsufficientScope returns true if the error is an HTTP 403 Forbidden response, -// meaning the token lacks the required admin scope. -func isInsufficientScope(err error) bool { - var httpErr *remote.HTTPError - return errors.As(err, &httpErr) && httpErr.StatusCode == http.StatusForbidden -} - -// resolveFromDomain resolves image info from a domain name (legacy mode). -func resolveFromDomain(ctx context.Context, cp ControlPlane, pushDomain string) (registry, imageName, resolvedDomain string, err error) { - route, err := cp.GetRoute(ctx, pushDomain) - if err != nil { - return "", "", "", fmt.Errorf("failed to get route for domain %q: %w", pushDomain, err) - } - - registry, imageName, _ = parseImageRef(route.Image) - if registry == "" || imageName == "" { - return "", "", "", fmt.Errorf("cannot parse registry/image from route image: %s", route.Image) - } - - return registry, imageName, pushDomain, nil -} - -// resolveFromImage resolves domain(s) from an image name using the backend. -func resolveFromImage(ctx context.Context, cp ControlPlane, imageArg, dockerfile string) (registry, imageName, resolvedDomain string, err error) { - // First, query the backend to find routes for this image - routes, err := cp.FindRoutesByImage(ctx, imageArg) - if err != nil { - return "", "", "", fmt.Errorf("failed to find routes for image %q: %w", imageArg, err) - } - - routes = filterPreviewRoutes(routes) - - if len(routes) == 0 { - return "", "", "", noRouteForImageError(imageArg) - } - - // Pick the target domain - var targetRoute domain.Route - if len(routes) == 1 { - targetRoute = routes[0] - } else { - // Multiple domains: check Dockerfile labels first, then prompt - targetRoute, err = selectDomain(routes, dockerfile) - if err != nil { - return "", "", "", err - } - } - - registry, imageName, _ = parseImageRef(targetRoute.Image) - if registry == "" || imageName == "" { - return "", "", "", fmt.Errorf("cannot parse registry/image from route image: %s", targetRoute.Image) - } - - return registry, imageName, targetRoute.Domain, nil -} - -func isImageBootstrapError(err error) bool { - return errors.Is(err, domain.ErrNoRouteForImage) -} - -func noRouteForImageError(imageArg string) error { - // bootstrap command is registered in bootstrap.go (see issue #98) - return noRouteForImageBootstrapError{imageArg: imageArg} -} - -type noRouteForImageBootstrapError struct { - imageArg string -} - -func (e noRouteForImageBootstrapError) Error() string { - return fmt.Sprintf( - "no route configured for image %q\n\nFor a first deploy, use 'gordon bootstrap %s'\nOr configure the route directly with 'gordon routes add %s'\nIf this is an attachment image, use 'gordon attachments push %s'", - e.imageArg, - e.imageArg, - e.imageArg, - e.imageArg, - ) -} - -func (e noRouteForImageBootstrapError) Unwrap() error { - return domain.ErrNoRouteForImage -} - -// selectDomain picks the target domain from multiple routes. -// Checks Dockerfile labels first, then falls back to interactive selection. -func selectDomain(routes []domain.Route, dockerfile string) (domain.Route, error) { - // Try to resolve from Dockerfile labels - labels := parseDockerfileLabels(dockerfile) - labelDomain := labels[domain.LabelDomain] - labelDomains := labels[domain.LabelDomains] - - // Collect domains from labels - var labelDomainList []string - if labelDomain != "" { - labelDomainList = append(labelDomainList, labelDomain) - } - if labelDomains != "" { - for d := range strings.SplitSeq(labelDomains, ",") { - d = strings.TrimSpace(d) - if d != "" { - labelDomainList = append(labelDomainList, d) - } - } - } - - // Try to find a matching route from labels - if len(labelDomainList) > 0 { - routeMap := make(map[string]domain.Route, len(routes)) - for _, r := range routes { - routeMap[r.Domain] = r - } - for _, ld := range labelDomainList { - if r, ok := routeMap[ld]; ok { - return r, nil - } - } - } - - // Fall back to interactive selection - items := make([]string, 0, len(routes)) - for _, r := range routes { - items = append(items, fmt.Sprintf("%s %s", r.Domain, styles.Theme.Muted.Render(r.Image))) - } - - selected, err := components.RunSelector( - "Multiple domains found for this image. Select target:", - items, - "", - ) - if err != nil { - return domain.Route{}, fmt.Errorf("selection error: %w", err) + return buildAndPush(ctx, out, imageOps, build, img) } - if selected == "" { - return domain.Route{}, fmt.Errorf("no domain selected") - } - - // Extract domain from the selected display string - for i, item := range items { - if item == selected { - return routes[i], nil - } - } - - return domain.Route{}, fmt.Errorf("selected domain not found") + return tagAndPush(ctx, out, imageOps, img) } -// detectImageName auto-detects the image name from context. -// Resolution order: -// 1. Dockerfile label gordon.domain (if Dockerfile exists) -// 2. Current directory name +// detectImageName resolves the image name from Dockerfile labels or the +// current directory name. func detectImageName(dockerfile string) (string, error) { // Try Dockerfile labels labels := parseDockerfileLabels(dockerfile) @@ -513,7 +297,7 @@ func detectImageName(dockerfile string) (string, error) { dirName := filepath.Base(cwd) if dirName == "." || dirName == "/" { - return "", fmt.Errorf("cannot detect image name from current directory; provide an image name or use --domain") + return "", fmt.Errorf("cannot detect image name from current directory; provide an image name") } return dirName, nil @@ -655,7 +439,7 @@ func parseTagRef(ref string) string { return tag } -func buildAndPush(ctx context.Context, ops pushImageOps, build buildConfig, img imagePush) error { +func buildAndPush(ctx context.Context, out io.Writer, ops pushImageOps, build buildConfig, img imagePush) error { if _, err := os.Stat(build.Dockerfile); os.IsNotExist(err) { return fmt.Errorf("dockerfile not found: %s", build.Dockerfile) } @@ -663,12 +447,16 @@ func buildAndPush(ctx context.Context, ops pushImageOps, build buildConfig, img // Build and load into local daemon (NOT --push). // The native registry push client handles chunked uploads to stay // within Cloudflare's 100MB per-request limit. - fmt.Println("\nBuilding image...") + if err := cliWriteLine(out, "\nBuilding image..."); err != nil { + return err + } if err := ops.Build(ctx, buildImageArgs(ctx, img.Version, build.Platform, build.Dockerfile, build.BuildArgs, img.VersionRef, img.LatestRef)); err != nil { return err } - fmt.Println("Pushing...") + if err := cliWriteLine(out, "Pushing..."); err != nil { + return err + } if err := ops.Push(ctx, img.LatestRef); err != nil { return fmt.Errorf("failed to push %s: %w", img.LatestRef, err) } @@ -731,10 +519,15 @@ func buildImageArgs(ctx context.Context, version, platform, dockerfile string, b return args } -func tagAndPush(ctx context.Context, ops pushImageOps, img imagePush) error { - localImage := fmt.Sprintf("%s/%s", img.Registry, img.ImageName) +func tagAndPush(ctx context.Context, out io.Writer, ops pushImageOps, img imagePush) error { + localImage := img.SourceRef + if localImage == "" { + return errors.New("push source image reference cannot be empty") + } - fmt.Println("\nChecking local image...") + if err := cliWriteLine(out, "\nChecking local image..."); err != nil { + return err + } exists, err := ops.Exists(ctx, localImage) if err != nil { return fmt.Errorf("failed to inspect local image %s: %w", localImage, err) @@ -743,7 +536,9 @@ func tagAndPush(ctx context.Context, ops pushImageOps, img imagePush) error { return fmt.Errorf("local image %s not found; build and tag it before pushing", localImage) } - fmt.Println("Tagging...") + if err := cliWriteLine(out, "Tagging..."); err != nil { + return err + } if err := ops.Tag(ctx, localImage, img.VersionRef); err != nil { return fmt.Errorf("failed to tag %s: %w", img.VersionRef, err) } @@ -753,7 +548,9 @@ func tagAndPush(ctx context.Context, ops pushImageOps, img imagePush) error { } } - fmt.Println("Pushing...") + if err := cliWriteLine(out, "Pushing..."); err != nil { + return err + } if err := ops.Push(ctx, img.VersionRef); err != nil { return fmt.Errorf("failed to push %s: %w", img.VersionRef, err) } @@ -765,118 +562,6 @@ func tagAndPush(ctx context.Context, ops pushImageOps, img imagePush) error { return nil } -func deployAfterPush(ctx context.Context, cp ControlPlane, pushDomain string, noConfirm bool) error { - if !noConfirm { - confirmed, err := components.RunConfirm("Deploy now?", components.WithDefaultYes()) - if err != nil { - return err - } - if !confirmed { - return nil - } - } - - var ( - result *remote.DeployResult - err error - ) - if remoteCP, ok := cp.(*remoteControlPlane); ok { - result, err = deployWithSpinner(ctx, remoteCP.client, pushDomain) - } else { - result, err = cp.Deploy(ctx, pushDomain) - } - if err != nil { - return formatDeployFailure(err) - } - containerID := shortContainerID(result.ContainerID) - fmt.Println(styles.RenderSuccess(fmt.Sprintf("Deployed %s (container: %s)", pushDomain, containerID))) - return nil -} - -func deployWithSpinner(ctx context.Context, client *remote.Client, pushDomain string) (*remote.DeployResult, error) { - if !isInteractiveTerminal() { - fmt.Printf("Deploying %s...\n", pushDomain) - return client.Deploy(ctx, pushDomain) - } - - done := make(chan deployOutcome, 1) - go func() { - result, err := client.Deploy(ctx, pushDomain) - done <- deployOutcome{result: result, err: err} - }() - - model := newDeploySpinnerModel(pushDomain, done) - final, err := tea.NewProgram(model, tea.WithContext(ctx)).Run() - fmt.Print("\r\033[K") - if err != nil { - return nil, err - } - - deployModel, ok := final.(deploySpinnerModel) - if !ok { - return nil, fmt.Errorf("spinner exited with unexpected model type %T", final) - } - if !deployModel.finished { - return nil, fmt.Errorf("deploy spinner exited before deploy result was received") - } - - return deployModel.outcome.result, deployModel.outcome.err -} - -type deployOutcome struct { - result *remote.DeployResult - err error -} - -type deployDoneMsg deployOutcome - -type deploySpinnerModel struct { - spinner components.SpinnerModel - done <-chan deployOutcome - outcome deployOutcome - finished bool -} - -func newDeploySpinnerModel(pushDomain string, done <-chan deployOutcome) deploySpinnerModel { - return deploySpinnerModel{ - spinner: components.NewSpinner( - components.WithMessage(fmt.Sprintf("Deploying %s...", pushDomain)), - components.WithSpinnerType(components.SpinnerMiniDot), - ), - done: done, - } -} - -func (m deploySpinnerModel) Init() tea.Cmd { - return tea.Batch(m.spinner.Init(), waitForDeployDone(m.done)) -} - -func (m deploySpinnerModel) Update(msg tea.Msg) (tea.Model, tea.Cmd) { - switch msg := msg.(type) { - case deployDoneMsg: - m.outcome = deployOutcome(msg) - m.finished = true - return m, tea.Quit - default: - updated, cmd := m.spinner.Update(msg) - spinnerModel, ok := updated.(components.SpinnerModel) - if ok { - m.spinner = spinnerModel - } - return m, cmd - } -} - -func (m deploySpinnerModel) View() string { - return m.spinner.View() -} - -func waitForDeployDone(done <-chan deployOutcome) tea.Cmd { - return func() tea.Msg { - return deployDoneMsg(<-done) - } -} - func isInteractiveTerminal() bool { term := os.Getenv("TERM") if term == "" || term == "dumb" { @@ -898,6 +583,9 @@ func parseImageRef(image string) (registry, name, tag string) { } registry = parts[0] nameTag := parts[1] + if idx := strings.LastIndex(nameTag, "@"); idx != -1 { + return registry, nameTag[:idx], nameTag[idx+1:] + } if idx := strings.LastIndex(nameTag, ":"); idx != -1 { name = nameTag[:idx] tag = nameTag[idx+1:] diff --git a/internal/adapters/in/cli/push_behavior_test.go b/internal/adapters/in/cli/push_behavior_test.go new file mode 100644 index 000000000..de30d2bee --- /dev/null +++ b/internal/adapters/in/cli/push_behavior_test.go @@ -0,0 +1,61 @@ +package cli + +import ( + "context" + "io" + "testing" + + "github.com/stretchr/testify/require" +) + +type recordingPushOps struct { + existsRef string + tags [][2]string + pushes []string + exists bool +} + +func (o *recordingPushOps) Exists(_ context.Context, ref string) (bool, error) { + o.existsRef = ref + return o.exists, nil +} + +func (o *recordingPushOps) Tag(_ context.Context, source, target string) error { + o.tags = append(o.tags, [2]string{source, target}) + return nil +} + +func (o *recordingPushOps) Push(_ context.Context, ref string) error { + o.pushes = append(o.pushes, ref) + return nil +} + +func (*recordingPushOps) Build(context.Context, []string) error { return nil } + +func TestTagAndPushPreservesExactSourceReference(t *testing.T) { + ops := &recordingPushOps{exists: true} + img := imagePush{ + SourceRef: "registry.example.com/team/app:v3", + Version: "v4", VersionRef: "registry.example.com/team/app:v4", + LatestRef: "registry.example.com/team/app:latest", + } + + require.NoError(t, tagAndPush(context.Background(), io.Discard, ops, img)) + require.Equal(t, img.SourceRef, ops.existsRef) + require.Equal(t, [][2]string{ + {img.SourceRef, img.VersionRef}, + {img.SourceRef, img.LatestRef}, + }, ops.tags) + require.Equal(t, []string{img.VersionRef, img.LatestRef}, ops.pushes) +} + +func TestTagAndPushRejectsMissingSourceReference(t *testing.T) { + ops := &recordingPushOps{exists: true} + err := tagAndPush(context.Background(), io.Discard, ops, imagePush{ + Registry: "registry.example.com", ImageName: "team/app", Version: "v4", + }) + require.EqualError(t, err, "push source image reference cannot be empty") + require.Empty(t, ops.existsRef) + require.Empty(t, ops.tags) + require.Empty(t, ops.pushes) +} diff --git a/internal/adapters/in/cli/push_image_ops.go b/internal/adapters/in/cli/push_image_ops.go index 12e6c118c..d96f95f00 100644 --- a/internal/adapters/in/cli/push_image_ops.go +++ b/internal/adapters/in/cli/push_image_ops.go @@ -35,6 +35,7 @@ type pushImageOps interface { type dockerImageOps struct { cli client.APIClient insecureTLS bool + plainHTTP bool progress io.Writer remoteClient *remote.Client // nil when local } @@ -70,8 +71,11 @@ func newImageOpsForResolvedRemote(resolved *remote.ResolvedRemote) (pushImageOps if err != nil { return nil, err } - if resolved != nil && resolved.Token != "" { - ops.remoteClient = remote.NewClient(resolved.URL, remoteClientOptions(resolved.Token, resolved.InsecureTLS)...) + if resolved != nil { + ops.plainHTTP = strings.HasPrefix(strings.ToLower(resolved.URL), "http://") + if resolved.Token != "" { + ops.remoteClient = remote.NewClient(resolved.URL, remoteClientOptions(resolved.Token, resolved.InsecureTLS)...) + } } return ops, nil } @@ -87,6 +91,7 @@ func (d *dockerImageOps) Push(ctx context.Context, ref string) error { opts := []registrypush.Option{ registrypush.WithProgress(d.progress), registrypush.WithInsecureTLS(d.insecureTLS), + registrypush.WithPlainHTTP(d.plainHTTP), } if d.remoteClient != nil { @@ -107,6 +112,7 @@ func (d *dockerImageOps) Push(ctx context.Context, ref string) error { retryOpts := []registrypush.Option{ registrypush.WithProgress(d.progress), registrypush.WithInsecureTLS(d.insecureTLS), + registrypush.WithPlainHTTP(d.plainHTTP), registrypush.WithAuth(authHeader), } pusher = registrypush.New(retryOpts...) @@ -175,7 +181,7 @@ func (d *dockerImageOps) Build(ctx context.Context, args []string) error { "Podman and other runtimes are not supported for --build.\n" + "Build the image manually and push with:\n" + " podman build -t .\n" + - " gordon push ", + " gordon images push ", ) } cmd := exec.CommandContext(ctx, "docker", args...) // #nosec G204 diff --git a/internal/adapters/in/cli/push_integration_test.go b/internal/adapters/in/cli/push_integration_test.go index 60e2920ff..ec8e86858 100644 --- a/internal/adapters/in/cli/push_integration_test.go +++ b/internal/adapters/in/cli/push_integration_test.go @@ -3,6 +3,7 @@ package cli import ( "context" "fmt" + "io" "net/http" "net/http/httptest" "strings" @@ -33,9 +34,13 @@ func TestIntegration_TagAndPush_NativeRegistry(t *testing.T) { img, err := random.Image(512, 1) require.NoError(t, err) + registry := strings.TrimPrefix(harness.server.URL, "http://") + versionRef := registry + "/testapp:v1.0.0" + latestRef := registry + "/testapp:latest" + ops := climocks.NewMockpushImageOps(t) ops.On("Tag", mock.Anything, mock.Anything, mock.Anything).Return(nil) - ops.On("Exists", mock.Anything, mock.Anything).Return(true, nil) + ops.On("Exists", mock.Anything, registry+"/testapp:v1.0.0").Return(true, nil) ops.EXPECT().Push(mock.Anything, mock.Anything).RunAndReturn(func(ctx context.Context, ref string) error { pusher := registrypush.New( registrypush.WithChunkSize(256), @@ -47,13 +52,10 @@ func TestIntegration_TagAndPush_NativeRegistry(t *testing.T) { return pusher.Push(ctx, ref) }) - registry := strings.TrimPrefix(harness.server.URL, "http://") - versionRef := registry + "/testapp:v1.0.0" - latestRef := registry + "/testapp:latest" - - err = tagAndPush(ctx, ops, imagePush{ + err = tagAndPush(ctx, io.Discard, ops, imagePush{ Registry: registry, ImageName: "testapp", + SourceRef: versionRef, Version: "v1.0.0", VersionRef: versionRef, LatestRef: latestRef, diff --git a/internal/adapters/in/cli/push_remote_inference_test.go b/internal/adapters/in/cli/push_remote_inference_test.go index 1858abe87..b32a795f2 100644 --- a/internal/adapters/in/cli/push_remote_inference_test.go +++ b/internal/adapters/in/cli/push_remote_inference_test.go @@ -13,16 +13,14 @@ import ( "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" - - "github.com/bnema/gordon/internal/domain" ) func TestInferPushRemote_UsesUniqueMatchingSavedRemote(t *testing.T) { - prod := newFindRoutesByImageTestServer(t, func(image string) ([]domain.Route, int) { + prod := newTagsTestServer(t, func(image string) ([]string, int) { assert.Equal(t, "myapp", image) - return []domain.Route{{Domain: "app.example.com", Image: "registry.example.com/myapp:latest"}}, http.StatusOK + return []string{"latest"}, http.StatusOK }) - staging := newFindRoutesByImageTestServer(t, func(image string) ([]domain.Route, int) { + staging := newTagsTestServer(t, func(image string) ([]string, int) { assert.Equal(t, "myapp", image) return nil, http.StatusOK }) @@ -43,11 +41,11 @@ url = "`+staging.URL+`" } func TestInferPushRemote_ReturnsAmbiguousErrorWhenMultipleRemotesMatch(t *testing.T) { - prod := newFindRoutesByImageTestServer(t, func(image string) ([]domain.Route, int) { - return []domain.Route{{Domain: "app.example.com", Image: "registry.example.com/myapp:latest"}}, http.StatusOK + prod := newTagsTestServer(t, func(image string) ([]string, int) { + return []string{"latest"}, http.StatusOK }) - staging := newFindRoutesByImageTestServer(t, func(image string) ([]domain.Route, int) { - return []domain.Route{{Domain: "staging.example.com", Image: "registry.example.com/myapp:latest"}}, http.StatusOK + staging := newTagsTestServer(t, func(image string) ([]string, int) { + return []string{"latest"}, http.StatusOK }) configurePushRemoteInferenceTestEnv(t, ` @@ -73,9 +71,9 @@ func TestInferPushRemote_SkipsGuessWhenExplicitRemoteSelected(t *testing.T) { }) called := false - prod := newFindRoutesByImageTestServer(t, func(image string) ([]domain.Route, int) { + prod := newTagsTestServer(t, func(image string) ([]string, int) { called = true - return []domain.Route{{Domain: "app.example.com", Image: "registry.example.com/myapp:latest"}}, http.StatusOK + return []string{"latest"}, http.StatusOK }) configurePushRemoteInferenceTestEnv(t, ` @@ -93,9 +91,9 @@ url = "`+prod.URL+`" func TestInferPushRemote_SkipsGuessWhenActiveRemoteConfigured(t *testing.T) { called := false - prod := newFindRoutesByImageTestServer(t, func(image string) ([]domain.Route, int) { + prod := newTagsTestServer(t, func(image string) ([]string, int) { called = true - return []domain.Route{{Domain: "app.example.com", Image: "registry.example.com/myapp:latest"}}, http.StatusOK + return []string{"latest"}, http.StatusOK }) configurePushRemoteInferenceTestEnv(t, ` @@ -111,26 +109,11 @@ url = "`+prod.URL+`" assert.False(t, called) } -func TestInferPushRemote_IgnoresPreviewOnlyMatches(t *testing.T) { - prod := newFindRoutesByImageTestServer(t, func(image string) ([]domain.Route, int) { - return []domain.Route{{Domain: "preview--pr-42.example.com", Image: "registry.example.com/myapp:latest"}}, http.StatusOK - }) - - configurePushRemoteInferenceTestEnv(t, ` -[remotes.prod] -url = "`+prod.URL+`" -`) - - resolved, err := inferPushRemote(context.Background(), "myapp", "", "Dockerfile") - require.NoError(t, err) - assert.Nil(t, resolved) -} - func TestInferPushRemote_FailsSafeWhenSomeRemotesCannotBeProbed(t *testing.T) { - prod := newFindRoutesByImageTestServer(t, func(image string) ([]domain.Route, int) { - return []domain.Route{{Domain: "app.example.com", Image: "registry.example.com/myapp:latest"}}, http.StatusOK + prod := newTagsTestServer(t, func(image string) ([]string, int) { + return []string{"latest"}, http.StatusOK }) - broken := newFindRoutesByImageTestServer(t, func(image string) ([]domain.Route, int) { + broken := newTagsTestServer(t, func(image string) ([]string, int) { return nil, http.StatusInternalServerError }) @@ -149,199 +132,6 @@ url = "`+broken.URL+`" assert.Contains(t, err.Error(), "broken") } -func TestInferRemoteForRouteDomain_UsesUniqueMatchingSavedRemote(t *testing.T) { - prod := newGetRouteTestServer(t, func(domainName string) (*domain.Route, int) { - return &domain.Route{Domain: domainName, Image: "registry.example.com/myapp:latest"}, http.StatusOK - }) - staging := newGetRouteTestServer(t, func(domainName string) (*domain.Route, int) { - return nil, http.StatusNotFound - }) - - configurePushRemoteInferenceTestEnv(t, ` -[remotes.prod] -url = "`+prod.URL+`" - -[remotes.staging] -url = "`+staging.URL+`" -`) - - resolved, err := inferRemoteForRouteDomain(context.Background(), "app.example.com") - require.NoError(t, err) - require.NotNil(t, resolved) - assert.Equal(t, "prod", resolved.Name) -} - -func TestInferRemoteForRouteDomain_ReturnsAmbiguousErrorWhenMultipleRemotesMatch(t *testing.T) { - prod := newGetRouteTestServer(t, func(domainName string) (*domain.Route, int) { - return &domain.Route{Domain: domainName, Image: "registry.example.com/myapp:latest"}, http.StatusOK - }) - staging := newGetRouteTestServer(t, func(domainName string) (*domain.Route, int) { - return &domain.Route{Domain: domainName, Image: "registry.example.com/myapp:latest"}, http.StatusOK - }) - - configurePushRemoteInferenceTestEnv(t, ` -[remotes.prod] -url = "`+prod.URL+`" - -[remotes.staging] -url = "`+staging.URL+`" -`) - - resolved, err := inferRemoteForRouteDomain(context.Background(), "app.example.com") - require.Error(t, err) - assert.Nil(t, resolved) - assert.Contains(t, err.Error(), `multiple saved remotes match route "app.example.com"`) -} - -func TestInferRemoteForRouteCleanupDomain_UsesCleanupPreviewWhenRouteConfigMissing(t *testing.T) { - prod := newMultiAdminProbeTestServer(t, multiAdminProbeHandlers{ - getRoute: func(domainName string) (*domain.Route, int) { - return nil, http.StatusNotFound - }, - getRouteCleanup: func(domainName string) (*domain.CleanupReport, int) { - return &domain.CleanupReport{ - Domain: domainName, - OrphanedEntities: []domain.CleanupOrphanedEntity{{ - Kind: "route_container", - ID: "abc123", - Status: "running", - }}, - }, http.StatusOK - }, - }) - staging := newMultiAdminProbeTestServer(t, multiAdminProbeHandlers{ - getRoute: func(domainName string) (*domain.Route, int) { - return nil, http.StatusNotFound - }, - getRouteCleanup: func(domainName string) (*domain.CleanupReport, int) { - return &domain.CleanupReport{Domain: domainName}, http.StatusOK - }, - }) - - configurePushRemoteInferenceTestEnv(t, ` -[remotes.prod] -url = "`+prod.URL+`" - -[remotes.staging] -url = "`+staging.URL+`" -`) - - resolved, err := inferRemoteForRouteCleanupDomain(context.Background(), "app.example.com") - require.NoError(t, err) - require.NotNil(t, resolved) - assert.Equal(t, "prod", resolved.Name) -} - -func TestInferRemoteForRouteCleanupDomain_UsesCleanupPreviewVolumesWhenRouteConfigMissing(t *testing.T) { - prod := newMultiAdminProbeTestServer(t, multiAdminProbeHandlers{ - getRoute: func(domainName string) (*domain.Route, int) { - return nil, http.StatusNotFound - }, - getRouteCleanup: func(domainName string) (*domain.CleanupReport, int) { - return &domain.CleanupReport{ - Domain: domainName, - PreservedVolumes: []domain.CleanupVolume{{Name: "gordon-app-example-com-data"}}, - }, http.StatusOK - }, - }) - staging := newMultiAdminProbeTestServer(t, multiAdminProbeHandlers{ - getRoute: func(domainName string) (*domain.Route, int) { - return nil, http.StatusNotFound - }, - getRouteCleanup: func(domainName string) (*domain.CleanupReport, int) { - return &domain.CleanupReport{Domain: domainName}, http.StatusOK - }, - }) - - configurePushRemoteInferenceTestEnv(t, ` -[remotes.prod] -url = "`+prod.URL+`" - -[remotes.staging] -url = "`+staging.URL+`" -`) - - resolved, err := inferRemoteForRouteCleanupDomain(context.Background(), "app.example.com") - require.NoError(t, err) - require.NotNil(t, resolved) - assert.Equal(t, "prod", resolved.Name) -} - -func TestInferRemoteForRouteCleanupDomain_FailsSafeWhenCleanupPreviewProbeFails(t *testing.T) { - prod := newMultiAdminProbeTestServer(t, multiAdminProbeHandlers{ - getRoute: func(domainName string) (*domain.Route, int) { - return nil, http.StatusNotFound - }, - getRouteCleanup: func(domainName string) (*domain.CleanupReport, int) { - return nil, http.StatusForbidden - }, - }) - - configurePushRemoteInferenceTestEnv(t, ` -[remotes.prod] -url = "`+prod.URL+`" -`) - - resolved, err := inferRemoteForRouteCleanupDomain(context.Background(), "app.example.com") - require.Error(t, err) - assert.Nil(t, resolved) - assert.Contains(t, err.Error(), `could not safely infer remote for route cleanup "app.example.com"`) -} - -func TestInferRemoteForAttachmentImage_UsesUniqueMatchingSavedRemote(t *testing.T) { - prod := newAttachmentTargetsByImageTestServer(t, func(image string) ([]string, int) { - return []string{"app.example.com"}, http.StatusOK - }) - staging := newAttachmentTargetsByImageTestServer(t, func(image string) ([]string, int) { - return nil, http.StatusOK - }) - - configurePushRemoteInferenceTestEnv(t, ` -[remotes.prod] -url = "`+prod.URL+`" - -[remotes.staging] -url = "`+staging.URL+`" -`) - - resolved, err := inferRemoteForAttachmentImage(context.Background(), "postgres:18") - require.NoError(t, err) - require.NotNil(t, resolved) - assert.Equal(t, "prod", resolved.Name) -} - -func TestInferRemoteForAttachmentTarget_FallsBackToRouteLookup(t *testing.T) { - prod := newMultiAdminProbeTestServer(t, multiAdminProbeHandlers{ - getRoute: func(domainName string) (*domain.Route, int) { - return &domain.Route{Domain: domainName, Image: "registry.example.com/myapp:latest"}, http.StatusOK - }, - getAttachmentsConfig: func(target string) ([]string, int) { - return nil, http.StatusNotFound - }, - }) - staging := newMultiAdminProbeTestServer(t, multiAdminProbeHandlers{ - getRoute: func(domainName string) (*domain.Route, int) { - return nil, http.StatusNotFound - }, - getAttachmentsConfig: func(target string) ([]string, int) { - return nil, http.StatusNotFound - }, - }) - - configurePushRemoteInferenceTestEnv(t, ` -[remotes.prod] -url = "`+prod.URL+`" - -[remotes.staging] -url = "`+staging.URL+`" -`) - - resolved, err := inferRemoteForAttachmentTarget(context.Background(), "app.example.com") - require.NoError(t, err) - require.NotNil(t, resolved) - assert.Equal(t, "prod", resolved.Name) -} - func TestInferRemoteForRepository_ReturnsAmbiguousErrorWhenMultipleRemotesMatch(t *testing.T) { prod := newTagsTestServer(t, func(repository string) ([]string, int) { return []string{"latest", "v1.0.0"}, http.StatusOK @@ -386,205 +176,20 @@ func configurePushRemoteInferenceTestEnv(t *testing.T, remotesTOML string) { configHome := t.TempDir() t.Setenv("XDG_CONFIG_HOME", configHome) - configPath := filepath.Join(configHome, "gordon", "remotes.toml") - require.NoError(t, os.MkdirAll(filepath.Dir(configPath), 0o755)) - require.NoError(t, os.WriteFile(configPath, []byte(strings.TrimSpace(remotesTOML)), 0o600)) + cliConfigPath := filepath.Join(configHome, "gordon", "remotes.toml") + require.NoError(t, os.MkdirAll(filepath.Dir(cliConfigPath), 0o755)) + require.NoError(t, os.WriteFile(cliConfigPath, []byte(strings.TrimSpace(remotesTOML)), 0o600)) } -func newFindRoutesByImageTestServer(t *testing.T, handler func(image string) ([]domain.Route, int)) *httptest.Server { - t.Helper() - - server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - if !strings.HasPrefix(r.URL.Path, "/admin/routes/by-image/") { - http.NotFound(w, r) - return - } - - image, err := url.PathUnescape(strings.TrimPrefix(r.URL.Path, "/admin/routes/by-image/")) - if !assert.NoError(t, err) { - return - } - - routes, status := handler(image) - if status >= http.StatusBadRequest { - w.Header().Set("Content-Type", "application/json") - w.WriteHeader(status) - if !assert.NoError(t, json.NewEncoder(w).Encode(map[string]string{"error": "boom"})) { - return - } - return - } - - w.Header().Set("Content-Type", "application/json") - if !assert.NoError(t, json.NewEncoder(w).Encode(map[string]any{ - "image": image, - "routes": routes, - })) { - return - } - })) - t.Cleanup(server.Close) - return server -} - -func newGetRouteTestServer(t *testing.T, handler func(domainName string) (*domain.Route, int)) *httptest.Server { - t.Helper() - return newMultiAdminProbeTestServer(t, multiAdminProbeHandlers{getRoute: handler}) -} - -type multiAdminProbeHandlers struct { - getRoute func(domainName string) (*domain.Route, int) - getRouteCleanup func(domainName string) (*domain.CleanupReport, int) - getAttachmentsConfig func(target string) ([]string, int) - listTags func(repository string) ([]string, int) - findAttachmentTarget func(image string) ([]string, int) -} - -func newMultiAdminProbeTestServer(t *testing.T, handlers multiAdminProbeHandlers) *httptest.Server { +func newTagsTestServer(t *testing.T, handler func(repository string) ([]string, int)) *httptest.Server { t.Helper() - server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - switch { - case strings.HasSuffix(r.URL.Path, "/cleanup") && strings.HasPrefix(r.URL.Path, "/admin/routes/"): - handleGetRouteCleanupProbe(t, w, r, handlers.getRouteCleanup) - case strings.HasPrefix(r.URL.Path, "/admin/routes/"): - handleGetRouteProbe(t, w, r, handlers.getRoute) - case strings.HasPrefix(r.URL.Path, "/admin/attachments/by-image/"): - handleAttachmentTargetsByImageProbe(t, w, r, handlers.findAttachmentTarget) - case strings.HasPrefix(r.URL.Path, "/admin/attachments/"): - handleGetAttachmentsConfigProbe(t, w, r, handlers.getAttachmentsConfig) - case strings.HasPrefix(r.URL.Path, "/admin/tags/"): - handleListTagsProbe(t, w, r, handlers.listTags) - default: - http.NotFound(w, r) - } + handleListTagsProbe(t, w, r, handler) })) t.Cleanup(server.Close) return server } -func newAttachmentTargetsByImageTestServer(t *testing.T, handler func(image string) ([]string, int)) *httptest.Server { - t.Helper() - return newMultiAdminProbeTestServer(t, multiAdminProbeHandlers{findAttachmentTarget: handler}) -} - -func newTagsTestServer(t *testing.T, handler func(repository string) ([]string, int)) *httptest.Server { - t.Helper() - return newMultiAdminProbeTestServer(t, multiAdminProbeHandlers{listTags: handler}) -} - -func handleGetRouteProbe(t *testing.T, w http.ResponseWriter, r *http.Request, handler func(domainName string) (*domain.Route, int)) { - t.Helper() - if handler == nil { - http.NotFound(w, r) - return - } - domainName, err := url.PathUnescape(strings.TrimPrefix(r.URL.Path, "/admin/routes/")) - if !assert.NoError(t, err) { - return - } - route, status := handler(domainName) - if status >= http.StatusBadRequest { - w.Header().Set("Content-Type", "application/json") - w.WriteHeader(status) - if status == http.StatusNotFound { - if !assert.NoError(t, json.NewEncoder(w).Encode(map[string]string{"error": domain.ErrRouteNotFound.Error()})) { - return - } - return - } - if !assert.NoError(t, json.NewEncoder(w).Encode(map[string]string{"error": "boom"})) { - return - } - return - } - w.Header().Set("Content-Type", "application/json") - if !assert.NoError(t, json.NewEncoder(w).Encode(route)) { - return - } -} - -func handleGetRouteCleanupProbe(t *testing.T, w http.ResponseWriter, r *http.Request, handler func(domainName string) (*domain.CleanupReport, int)) { - t.Helper() - if handler == nil { - http.NotFound(w, r) - return - } - domainName, err := url.PathUnescape(strings.TrimSuffix(strings.TrimPrefix(r.URL.Path, "/admin/routes/"), "/cleanup")) - if !assert.NoError(t, err) { - return - } - report, status := handler(domainName) - if status >= http.StatusBadRequest { - w.Header().Set("Content-Type", "application/json") - w.WriteHeader(status) - if !assert.NoError(t, json.NewEncoder(w).Encode(map[string]string{"error": "boom"})) { - return - } - return - } - w.Header().Set("Content-Type", "application/json") - if !assert.NoError(t, json.NewEncoder(w).Encode(report)) { - return - } -} - -func handleAttachmentTargetsByImageProbe(t *testing.T, w http.ResponseWriter, r *http.Request, handler func(image string) ([]string, int)) { - t.Helper() - if handler == nil { - http.NotFound(w, r) - return - } - image, err := url.PathUnescape(strings.TrimPrefix(r.URL.Path, "/admin/attachments/by-image/")) - if !assert.NoError(t, err) { - return - } - targets, status := handler(image) - if status >= http.StatusBadRequest { - w.Header().Set("Content-Type", "application/json") - w.WriteHeader(status) - if !assert.NoError(t, json.NewEncoder(w).Encode(map[string]string{"error": "boom"})) { - return - } - return - } - w.Header().Set("Content-Type", "application/json") - if !assert.NoError(t, json.NewEncoder(w).Encode(map[string]any{"image": image, "targets": targets})) { - return - } -} - -func handleGetAttachmentsConfigProbe(t *testing.T, w http.ResponseWriter, r *http.Request, handler func(target string) ([]string, int)) { - t.Helper() - if handler == nil { - http.NotFound(w, r) - return - } - target, err := url.PathUnescape(strings.TrimPrefix(r.URL.Path, "/admin/attachments/")) - if !assert.NoError(t, err) { - return - } - images, status := handler(target) - if status >= http.StatusBadRequest { - w.Header().Set("Content-Type", "application/json") - w.WriteHeader(status) - if status == http.StatusNotFound { - if !assert.NoError(t, json.NewEncoder(w).Encode(map[string]string{"error": "no attachments found for target"})) { - return - } - return - } - if !assert.NoError(t, json.NewEncoder(w).Encode(map[string]string{"error": "boom"})) { - return - } - return - } - w.Header().Set("Content-Type", "application/json") - if !assert.NoError(t, json.NewEncoder(w).Encode(map[string]any{"target": target, "images": images})) { - return - } -} - func handleListTagsProbe(t *testing.T, w http.ResponseWriter, r *http.Request, handler func(repository string) ([]string, int)) { t.Helper() if handler == nil { diff --git a/internal/adapters/in/cli/push_test.go b/internal/adapters/in/cli/push_test.go index 05c20028a..4b5e2160a 100644 --- a/internal/adapters/in/cli/push_test.go +++ b/internal/adapters/in/cli/push_test.go @@ -2,8 +2,6 @@ package cli import ( "context" - "errors" - "fmt" "io" "os" "os/exec" @@ -11,11 +9,7 @@ import ( "strings" "testing" - climocks "github.com/bnema/gordon/internal/adapters/in/cli/mocks" - "github.com/bnema/gordon/internal/adapters/in/cli/remote" - "github.com/bnema/gordon/internal/domain" "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/mock" ) func TestValidateBuildArg(t *testing.T) { @@ -79,12 +73,12 @@ func TestClassifyPushArgument(t *testing.T) { arg string want classifiedPushArg }{ - {name: "tagged image latest", arg: "myapp:latest", want: classifiedPushArg{kind: pushArgKindImage, lookupImage: "myapp"}}, - {name: "tagged image semver", arg: "myapp:v1.2.3", want: classifiedPushArg{kind: pushArgKindImage, lookupImage: "myapp"}}, - {name: "registry qualified image", arg: "registry.example.com/myapp:v1.2.3", want: classifiedPushArg{kind: pushArgKindImage, lookupImage: "registry.example.com/myapp"}}, - {name: "registry qualified digest", arg: "registry.example.com/myapp@sha256:deadbeef", want: classifiedPushArg{kind: pushArgKindImage, lookupImage: "registry.example.com/myapp@sha256:deadbeef"}}, + {name: "tagged image latest", arg: "myapp:latest", want: classifiedPushArg{kind: pushArgKindImage, sourceRef: "myapp:latest", repository: "myapp"}}, + {name: "tagged image semver", arg: "myapp:v1.2.3", want: classifiedPushArg{kind: pushArgKindImage, sourceRef: "myapp:v1.2.3", repository: "myapp"}}, + {name: "registry qualified image", arg: "registry.example.com/myapp:v1.2.3", want: classifiedPushArg{kind: pushArgKindImage, sourceRef: "registry.example.com/myapp:v1.2.3", repository: "myapp"}}, + {name: "registry qualified digest", arg: "registry.example.com/myapp@sha256:deadbeef", want: classifiedPushArg{kind: pushArgKindImage, sourceRef: "registry.example.com/myapp@sha256:deadbeef", repository: "myapp"}}, {name: "legacy domain", arg: "app.example.com", want: classifiedPushArg{kind: pushArgKindLegacyDomain, legacyDomain: "app.example.com"}}, - {name: "bare image name", arg: "myapp", want: classifiedPushArg{kind: pushArgKindImage, lookupImage: "myapp"}}, + {name: "bare image name", arg: "myapp", want: classifiedPushArg{kind: pushArgKindImage, sourceRef: "myapp", repository: "myapp"}}, } for _, tt := range tests { @@ -449,102 +443,3 @@ func TestParseLabelPair(t *testing.T) { }) } } - -func TestResolveFromImage_NoRouteSuggestsBootstrap(t *testing.T) { - cpMock := climocks.NewMockControlPlane(t) - cpMock.EXPECT().FindRoutesByImage(context.Background(), "myapp").Return(nil, nil).Once() - - _, _, _, err := resolveFromImage(context.Background(), cpMock, "myapp", "Dockerfile") - - assert.Error(t, err) - assert.Contains(t, err.Error(), `no route configured for image "myapp"`) - assert.Contains(t, err.Error(), "gordon bootstrap") -} - -func TestResolveRoute_DottedBareImageUsesImageLookup(t *testing.T) { - cpMock := climocks.NewMockControlPlane(t) - cpMock.EXPECT().FindRoutesByImage(context.Background(), "my.app").Return([]domain.Route{{Domain: "app.example.com", Image: "registry.example.com/my.app:latest"}}, nil).Once() - - registry, imageName, pushDomain, err := resolveRoute(context.Background(), cpMock, "my.app", "", "Dockerfile") - - assert.NoError(t, err) - assert.Equal(t, "registry.example.com", registry) - assert.Equal(t, "my.app", imageName) - assert.Equal(t, "app.example.com", pushDomain) -} - -func TestResolveRoute_DottedBareImageNoRoutesKeepsBootstrapError(t *testing.T) { - cpMock := climocks.NewMockControlPlane(t) - imageLookup := cpMock.EXPECT().FindRoutesByImage(context.Background(), "my.app").Return(nil, nil).Once() - routeLookup := cpMock.EXPECT().GetRoute(context.Background(), "my.app").Return(nil, domain.ErrRouteNotFound).Once() - mock.InOrder(imageLookup, routeLookup) - - _, _, _, err := resolveRoute(context.Background(), cpMock, "my.app", "", "Dockerfile") - - assert.Error(t, err) - assert.True(t, errors.Is(err, domain.ErrNoRouteForImage)) - assert.Contains(t, err.Error(), `no route configured for image "my.app"`) - assert.Contains(t, err.Error(), "gordon bootstrap") - assert.NotContains(t, err.Error(), "failed to get route for domain") -} - -func TestNoRouteForImageErrorWrapsSentinel(t *testing.T) { - err := noRouteForImageError("myapp") - - assert.True(t, errors.Is(err, domain.ErrNoRouteForImage)) - assert.Contains(t, err.Error(), `no route configured for image "myapp"`) -} - -func TestResolveRoute_DottedBareDomainFallsBackToLegacyLookup(t *testing.T) { - cpMock := climocks.NewMockControlPlane(t) - imageLookup := cpMock.EXPECT().FindRoutesByImage(context.Background(), "app.example.com").Return(nil, nil).Once() - routeLookup := cpMock.EXPECT().GetRoute(context.Background(), "app.example.com").Return(&domain.Route{Domain: "app.example.com", Image: "registry.example.com/myapp:latest"}, nil).Once() - mock.InOrder(imageLookup, routeLookup) - - registry, imageName, pushDomain, err := resolveRoute(context.Background(), cpMock, "app.example.com", "", "Dockerfile") - - assert.NoError(t, err) - assert.Equal(t, "registry.example.com", registry) - assert.Equal(t, "myapp", imageName) - assert.Equal(t, "app.example.com", pushDomain) -} - -func TestResolveRoute_TaggedImageStripsTagBeforeLookup(t *testing.T) { - cpMock := climocks.NewMockControlPlane(t) - cpMock.EXPECT().FindRoutesByImage(context.Background(), "myapp").Return([]domain.Route{{Domain: "app.example.com", Image: "registry.example.com/myapp:latest"}}, nil).Once() - - registry, imageName, pushDomain, err := resolveRoute(context.Background(), cpMock, "myapp:v1.2.3", "", "Dockerfile") - - assert.NoError(t, err) - assert.Equal(t, "registry.example.com", registry) - assert.Equal(t, "myapp", imageName) - assert.Equal(t, "app.example.com", pushDomain) -} - -func TestResolveRoute_RegistryQualifiedTaggedImageUsesImageLookup(t *testing.T) { - cpMock := climocks.NewMockControlPlane(t) - cpMock.EXPECT().FindRoutesByImage(context.Background(), "registry.example.com/myapp").Return([]domain.Route{{Domain: "app.example.com", Image: "registry.example.com/myapp:latest"}}, nil).Once() - - _, _, _, err := resolveRoute(context.Background(), cpMock, "registry.example.com/myapp:v1.2.3", "", "Dockerfile") - - assert.NoError(t, err) -} - -func TestIsInsufficientScope(t *testing.T) { - tests := []struct { - name string - err error - want bool - }{ - {"nil error", nil, false}, - {"HTTPError 403", &remote.HTTPError{StatusCode: 403, Status: "403 Forbidden", Body: "insufficient scope"}, true}, - {"wrapped HTTPError 403", fmt.Errorf("deploy intent: %w", &remote.HTTPError{StatusCode: 403, Status: "403 Forbidden", Body: "scope"}), true}, - {"HTTPError 500", &remote.HTTPError{StatusCode: 500, Status: "500 Internal Server Error", Body: "broke"}, false}, - {"plain error", fmt.Errorf("connection refused"), false}, - } - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - assert.Equal(t, tt.want, isInsufficientScope(tt.err)) - }) - } -} diff --git a/internal/adapters/in/cli/remote/apps.go b/internal/adapters/in/cli/remote/apps.go new file mode 100644 index 000000000..e55b64b72 --- /dev/null +++ b/internal/adapters/in/cli/remote/apps.go @@ -0,0 +1,306 @@ +package remote + +import ( + "bytes" + "context" + "encoding/hex" + "encoding/json" + "fmt" + "io" + "net/http" + "net/url" + + "github.com/google/uuid" + + "github.com/bnema/gordon/internal/adapters/dto" +) + +// Apps API: daemon-owned app mutations through the existing V2 +// administration transport. Every mutation carries an Idempotency-Key +// header (client-generated ULID); the daemon persists the claim before +// effects and replays recorded results. On ambiguous transport outcome +// the caller gets OutcomeUnknownError and must re-query by key — never +// blindly retry. + +// newIdempotencyKey allocates a time-ordered key (uuid v7). +func newIdempotencyKey() string { + id, err := uuid.NewV7() + if err != nil { + id = uuid.New() + } + var buf [32]byte + hex.Encode(buf[:], id[:]) + return string(buf[:]) +} + +// mutationPost sends a POST mutation with an idempotency key. +func (c *Client) mutationPost(ctx context.Context, path string, key string, body any) (*http.Response, error) { + var jsonBody []byte + if body != nil { + var err error + jsonBody, err = json.Marshal(body) + if err != nil { + return nil, fmt.Errorf("failed to marshal request body: %w", err) + } + } + return c.doMutation(ctx, http.MethodPost, path, key, jsonBody) +} + +// doMutation executes one mutation request with the idempotency header. +// Transport ambiguity surfaces as OutcomeUnknownError: the caller must +// re-query GET …/operations/by-key/{key} before any retry, and retries +// reuse the SAME key. +func (c *Client) doMutation(ctx context.Context, method, path, key string, jsonBody []byte) (*http.Response, error) { + reqURL := c.baseURL + "/admin" + path + + var bodyReader io.Reader + if jsonBody != nil { + bodyReader = bytes.NewReader(jsonBody) + } + req, err := http.NewRequestWithContext(ctx, method, reqURL, bodyReader) + if err != nil { + return nil, fmt.Errorf("failed to create request: %w", err) + } + req.Header.Set("Content-Type", "application/json") + req.Header.Set("Accept", "application/json") + req.Header.Set("Idempotency-Key", key) + + bearer, err := c.bearerToken(ctx) + if err != nil { + return nil, err + } + if bearer != "" { + req.Header.Set("Authorization", "Bearer "+bearer) + } + + resp, err := c.httpClient.Do(req) + if err != nil { + return nil, &OutcomeUnknownError{Method: method, Path: path, Err: err} + } + if isRetryableStatus(resp.StatusCode) { + body, _ := io.ReadAll(io.LimitReader(resp.Body, maxErrorBodySize)) + _ = resp.Body.Close() + return nil, &OutcomeUnknownError{ + Method: method, + Path: path, + Err: parseErrorResponse(resp, body), + } + } + return resp, nil +} + +// ApplyApp validates and persists desired state (or dry-runs). +func (c *Client) ApplyApp(ctx context.Context, req dto.AppApplyRequest) (*dto.AppApplyResponse, error) { + resp, err := c.mutationPost(ctx, "/apps/apply", newIdempotencyKey(), req) + if err != nil { + return nil, err + } + var result dto.AppApplyResponse + if err := parseResponse(resp, &result); err != nil { + return nil, fmt.Errorf("apply app: %w", err) + } + return &result, nil +} + +// ListApps returns desired+active summaries. +func (c *Client) ListApps(ctx context.Context) ([]dto.AppSummaryDTO, error) { + resp, err := c.request(ctx, http.MethodGet, "/apps", nil) + if err != nil { + return nil, err + } + var envelope struct { + Apps []dto.AppSummaryDTO `json:"apps"` + } + if err := parseResponse(resp, &envelope); err != nil { + return nil, fmt.Errorf("list apps: %w", err) + } + return envelope.Apps, nil +} + +// ShowApp inspects desired + active + intent + op ref. +func (c *Client) ShowApp(ctx context.Context, app string) (*dto.AppShowResponse, error) { + resp, err := c.request(ctx, http.MethodGet, "/apps/"+url.PathEscape(app), nil) + if err != nil { + return nil, err + } + var result dto.AppShowResponse + if err := parseResponse(resp, &result); err != nil { + return nil, fmt.Errorf("show app %s: %w", app, err) + } + return &result, nil +} + +// DiffApp returns the normalized desired-vs-active diff. +func (c *Client) DiffApp(ctx context.Context, app string) (*dto.AppDiffResponse, error) { + resp, err := c.request(ctx, http.MethodGet, "/apps/"+url.PathEscape(app)+"/diff", nil) + if err != nil { + return nil, err + } + var result dto.AppDiffResponse + if err := parseResponse(resp, &result); err != nil { + return nil, fmt.Errorf("diff app %s: %w", app, err) + } + return &result, nil +} + +// DeployApp activates a captured revision. It returns the key used so +// ambiguous outcomes can be recovered via OperationByKey. +func (c *Client) DeployApp(ctx context.Context, app string, req dto.AppDeployRequest) (*dto.AppDeployResponse, string, error) { + key := newIdempotencyKey() + resp, err := c.mutationPost(ctx, "/apps/"+url.PathEscape(app)+"/deploy", key, req) + if err != nil { + return nil, key, err + } + result, err := parseDeployResponse(resp, "deploy", app) + return result, key, err +} + +// StopApp persists durable stopped intent and stops exact containers. +func (c *Client) StopApp(ctx context.Context, app string) (*dto.AppDeployResponse, string, error) { + return c.lifecyclePost(ctx, app, "stop") +} + +// StartApp clears stopped intent and ensures running from active. +func (c *Client) StartApp(ctx context.Context, app string) (*dto.AppDeployResponse, string, error) { + return c.lifecyclePost(ctx, app, "start") +} + +// RestartApp restarts from pinned digests (no re-resolution). +// all confirms an app-wide restart of a multi-service app. +func (c *Client) RestartApp(ctx context.Context, app, service string, all bool) (*dto.AppDeployResponse, string, error) { + key := newIdempotencyKey() + path := "/apps/" + url.PathEscape(app) + "/restart" + query := url.Values{} + if service != "" { + query.Set("service", service) + } + if all { + query.Set("all", "true") + } + if len(query) > 0 { + path += "?" + query.Encode() + } + resp, err := c.doMutation(ctx, http.MethodPost, path, key, nil) + if err != nil { + return nil, key, err + } + result, err := parseDeployResponse(resp, "restart", app) + return result, key, err +} + +// RemoveApp withdraws workloads; volumes and secrets are retained. +func (c *Client) RemoveApp(ctx context.Context, app string) (*dto.AppDeployResponse, string, error) { + return c.lifecyclePost(ctx, app, "remove") +} + +// lifecyclePost posts a body-less lifecycle mutation. +func (c *Client) lifecyclePost(ctx context.Context, app, action string) (*dto.AppDeployResponse, string, error) { + key := newIdempotencyKey() + resp, err := c.doMutation(ctx, http.MethodPost, "/apps/"+url.PathEscape(app)+"/"+action, key, nil) + if err != nil { + return nil, key, err + } + result, err := parseDeployResponse(resp, action, app) + return result, key, err +} + +// maxDeployConflictBodySize bounds 409 journal reads: journals carry +// per-service results and steps and can exceed the 1 KB error cap. +const maxDeployConflictBodySize int64 = 1 << 20 + +// AppOpConflictError reports a 409 deploy/lifecycle outcome while carrying +// the decoded operation journal. Use errors.As to render the journal and +// retain failure semantics; the idempotency key travels alongside the +// returned error from DeployApp/RestartApp/lifecycle calls. +type AppOpConflictError struct { + StatusCode int + Status string + Response dto.AppDeployResponse +} + +func (e *AppOpConflictError) Error() string { + return fmt.Sprintf("%s: %s %s outcome %s", e.Status, e.Response.Op, e.Response.App, e.Response.Outcome) +} + +// parseDeployResponse decodes a deploy/lifecycle mutation response. A 202 +// Accepted is a success shape: it carries the same running AppDeployResponse +// journal as a 200 and is polled via OperationByKey, so it decodes through +// the generic 2xx path. A 409 carries the journaled AppDeployResponse body +// (per-service results, steps, effective/observed state) instead of an error +// envelope; it is decoded and surfaced as AppOpConflictError so the CLI can +// render the journal while retaining failure semantics. +func parseDeployResponse(resp *http.Response, action, app string) (*dto.AppDeployResponse, error) { + if resp.StatusCode == http.StatusConflict { + return decodeDeployConflict(resp, action, app) + } + var result dto.AppDeployResponse + if err := parseResponse(resp, &result); err != nil { + return nil, fmt.Errorf("%s app %s: %w", action, app, err) + } + return &result, nil +} + +// decodeDeployConflict decodes a 409 journal body into AppDeployResponse. +// Bodies that are not journals (preflight AppError envelopes) fall back to +// HTTPError so generic conflicts keep their existing shape. +func decodeDeployConflict(resp *http.Response, action, app string) (*dto.AppDeployResponse, error) { + defer resp.Body.Close() + body, _ := io.ReadAll(io.LimitReader(resp.Body, maxDeployConflictBodySize)) + var result dto.AppDeployResponse + if err := json.Unmarshal(body, &result); err != nil { + return nil, fmt.Errorf("%s app %s: %w", action, app, parseErrorResponse(resp, body)) + } + if result.Op == "" && result.Outcome == "" && len(result.Services) == 0 && len(result.Steps) == 0 { + return nil, fmt.Errorf("%s app %s: %w", action, app, parseErrorResponse(resp, body)) + } + conflict := &AppOpConflictError{StatusCode: resp.StatusCode, Status: resp.Status, Response: result} + return &result, fmt.Errorf("%s app %s: %w", action, app, conflict) +} + +// OperationByKey recovers an ambiguous mutation outcome by idempotency key. +func (c *Client) OperationByKey(ctx context.Context, app, key string) (*dto.AppDeployResponse, error) { + resp, err := c.request(ctx, http.MethodGet, "/apps/"+url.PathEscape(app)+"/operations/by-key/"+url.PathEscape(key), nil) + if err != nil { + return nil, err + } + var result dto.AppDeployResponse + if err := parseResponse(resp, &result); err != nil { + return nil, fmt.Errorf("operation by key: %w", err) + } + return &result, nil +} + +// ListAppSecrets returns metadata-only secret registrations. +func (c *Client) ListAppSecrets(ctx context.Context, app, service string) ([]dto.AppSecretMetadataDTO, error) { + path := "/apps/" + url.PathEscape(app) + "/secrets" + if service != "" { + path += "?service=" + url.QueryEscape(service) + } + resp, err := c.request(ctx, http.MethodGet, path, nil) + if err != nil { + return nil, fmt.Errorf("list app secrets %s: %w", app, err) + } + var result []dto.AppSecretMetadataDTO + if err := parseResponse(resp, &result); err != nil { + return nil, fmt.Errorf("list app secrets %s: %w", app, err) + } + return result, nil +} + +// SetAppSecrets writes secret values for pre-registered names. +func (c *Client) SetAppSecrets(ctx context.Context, app string, req dto.AppSecretSetRequest) error { + resp, err := c.mutationPost(ctx, "/apps/"+url.PathEscape(app)+"/secrets/set", newIdempotencyKey(), req) + if err != nil { + return err + } + return parseResponse(resp, nil) +} + +// DeleteAppSecret removes one secret value (refused when referenced). +func (c *Client) DeleteAppSecret(ctx context.Context, app string, req dto.AppSecretDeleteRequest) error { + resp, err := c.mutationPost(ctx, "/apps/"+url.PathEscape(app)+"/secrets/delete", newIdempotencyKey(), req) + if err != nil { + return err + } + return parseResponse(resp, nil) +} diff --git a/internal/adapters/in/cli/remote/apps_test.go b/internal/adapters/in/cli/remote/apps_test.go new file mode 100644 index 000000000..12f570bdf --- /dev/null +++ b/internal/adapters/in/cli/remote/apps_test.go @@ -0,0 +1,245 @@ +package remote + +// Tests for the v3 app mutation transport (apps.go): every mutation +// carries a client-generated Idempotency-Key, retryable gateway statuses +// surface OutcomeUnknownError without replaying the mutation, and +// ambiguous outcomes recover via the by-key endpoint. + +import ( + "context" + "encoding/json" + "net/http" + "net/http/httptest" + "sync/atomic" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/dto" +) + +func TestClientApplyApp_SendsIdempotencyKey(t *testing.T) { + var gotKey string + var gotBody dto.AppApplyRequest + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + require.Equal(t, "/admin/apps/apply", r.URL.Path) + require.Equal(t, http.MethodPost, r.Method) + gotKey = r.Header.Get("Idempotency-Key") + require.NoError(t, json.NewDecoder(r.Body).Decode(&gotBody)) + w.Header().Set("Content-Type", "application/json") + _ = json.NewEncoder(w).Encode(dto.AppApplyResponse{ + App: "blog", ResultingRevision: "rev-b", Pending: true, Intent: "apply-1", + }) + })) + defer srv.Close() + + resp, err := NewClient(srv.URL).ApplyApp(context.Background(), dto.AppApplyRequest{ManifestTOML: "x"}) + require.NoError(t, err) + assert.Equal(t, "blog", resp.App) + assert.NotEmpty(t, gotKey, "mutations must carry an Idempotency-Key") + assert.Equal(t, "x", gotBody.ManifestTOML) +} + +func TestClientDeployApp_GatewayErrorIsOutcomeUnknownWithoutReplay(t *testing.T) { + var mutations int32 + var gotKey string + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + require.Equal(t, "/admin/apps/blog/deploy", r.URL.Path) + atomic.AddInt32(&mutations, 1) + gotKey = r.Header.Get("Idempotency-Key") + w.WriteHeader(http.StatusServiceUnavailable) + _, _ = w.Write([]byte(`{"error":"upstream unavailable"}`)) + })) + defer srv.Close() + + resp, key, err := NewClient(srv.URL).DeployApp(context.Background(), "blog", dto.AppDeployRequest{}) + require.Nil(t, resp) + require.Error(t, err) + var unknown *OutcomeUnknownError + require.ErrorAs(t, err, &unknown) + assert.EqualValues(t, 1, atomic.LoadInt32(&mutations), "mutations must never replay on ambiguity") + assert.NotEmpty(t, key) + assert.Equal(t, gotKey, key, "recovery re-queries with the SAME key") +} + +func TestClientOperationByKey_RecoveryPath(t *testing.T) { + var gotPath string + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + gotPath = r.URL.EscapedPath() + require.Equal(t, http.MethodGet, r.Method) + w.Header().Set("Content-Type", "application/json") + _ = json.NewEncoder(w).Encode(dto.AppDeployResponse{ + Op: "op-1", App: "blog", Revision: "rev-b", Outcome: "success", + }) + })) + defer srv.Close() + + resp, err := NewClient(srv.URL).OperationByKey(context.Background(), "blog", "key-abc") + require.NoError(t, err) + assert.Equal(t, "op-1", resp.Op) + assert.Equal(t, "/admin/apps/blog/operations/by-key/key-abc", gotPath) +} + +func TestClientLifecycle_MutationsCarryKeys(t *testing.T) { + var keys []string + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + keys = append(keys, r.Header.Get("Idempotency-Key")) + w.Header().Set("Content-Type", "application/json") + _ = json.NewEncoder(w).Encode(dto.AppDeployResponse{Op: "op-x", Outcome: "success"}) + })) + defer srv.Close() + + client := NewClient(srv.URL) + ctx := context.Background() + _, _, err := client.StopApp(ctx, "blog") + require.NoError(t, err) + _, _, err = client.StartApp(ctx, "blog") + require.NoError(t, err) + _, _, err = client.RemoveApp(ctx, "blog") + require.NoError(t, err) + require.Len(t, keys, 3) + for _, key := range keys { + assert.NotEmpty(t, key) + } + assert.NotEqual(t, keys[0], keys[1], "each mutation gets a fresh key") +} + +func TestClientRestartApp_SendsServiceScope(t *testing.T) { + var queries []string + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + queries = append(queries, r.URL.RawQuery) + w.Header().Set("Content-Type", "application/json") + _ = json.NewEncoder(w).Encode(dto.AppDeployResponse{Op: "op-x", Outcome: "success"}) + })) + defer srv.Close() + + client := NewClient(srv.URL) + for _, tc := range []struct { + service string + all bool + }{{"", false}, {"web", false}, {"", true}} { + _, _, err := client.RestartApp(context.Background(), "blog", tc.service, tc.all) + require.NoError(t, err) + } + assert.Equal(t, []string{"", "service=web", "all=true"}, queries) +} + +// TestClientDeployApp_AcceptedRunningJournal proves the remote client accepts +// the 202 running response: it decodes the journal and returns the key so the +// caller can poll operations/by-key. +func TestClientDeployApp_AcceptedRunningJournal(t *testing.T) { + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + require.Equal(t, "/admin/apps/blog/deploy", r.URL.Path) + require.Equal(t, http.MethodPost, r.Method) + require.NotEmpty(t, r.Header.Get("Idempotency-Key")) + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(http.StatusAccepted) + _ = json.NewEncoder(w).Encode(dto.AppDeployResponse{ + Op: "op-run", App: "blog", Revision: "rev-b", Status: dto.AppStatusRunning, + }) + })) + defer srv.Close() + + resp, key, err := NewClient(srv.URL).DeployApp(context.Background(), "blog", dto.AppDeployRequest{}) + require.NoError(t, err) + require.NotNil(t, resp) + assert.Equal(t, "op-run", resp.Op) + assert.Equal(t, dto.AppStatusRunning, resp.Status) + assert.NotEmpty(t, key) +} + +func TestClientDeployApp_ConflictDecodesJournal(t *testing.T) { + journal := dto.AppDeployResponse{ + Op: "op-9", App: "blog", Revision: "rev-b", Outcome: "failed", + Services: map[string]dto.AppServiceResultDTO{ + "web": {Result: "failed", EffectiveRevision: "rev-b", Error: "nope"}, + }, + Steps: []dto.AppStepDTO{{ID: "service.web.start", State: "failed", Error: "nope"}}, + } + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + require.Equal(t, "/admin/apps/blog/deploy", r.URL.Path) + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(http.StatusConflict) + _ = json.NewEncoder(w).Encode(journal) + })) + defer srv.Close() + + resp, key, err := NewClient(srv.URL).DeployApp(context.Background(), "blog", dto.AppDeployRequest{}) + require.Error(t, err) + require.NotNil(t, resp, "conflict must preserve the decoded journal") + assert.Equal(t, "op-9", resp.Op) + assert.Equal(t, "failed", resp.Services["web"].Result) + assert.NotEmpty(t, key) + var conflict *AppOpConflictError + require.ErrorAs(t, err, &conflict) + assert.Equal(t, http.StatusConflict, conflict.StatusCode) + assert.Equal(t, "op-9", conflict.Response.Op) + assert.Len(t, conflict.Response.Steps, 1) +} + +func TestClientStopApp_ConflictDecodesJournal(t *testing.T) { + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + require.Equal(t, "/admin/apps/blog/stop", r.URL.Path) + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(http.StatusConflict) + _ = json.NewEncoder(w).Encode(dto.AppDeployResponse{ + Op: "op-7", App: "blog", Outcome: "partial", + Services: map[string]dto.AppServiceResultDTO{ + "web": {Result: "failed", EffectiveRevision: "rev-b", Error: "busy"}, + }, + }) + })) + defer srv.Close() + + resp, key, err := NewClient(srv.URL).StopApp(context.Background(), "blog") + require.Error(t, err) + require.NotNil(t, resp) + assert.Equal(t, "partial", resp.Outcome) + assert.NotEmpty(t, key) + var conflict *AppOpConflictError + require.ErrorAs(t, err, &conflict) +} + +func TestClientDeployApp_ConflictWithoutJournalFallsBackToHTTPError(t *testing.T) { + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.WriteHeader(http.StatusConflict) + _, _ = w.Write([]byte(`{"error":"state-conflict","message":"busy"}`)) + })) + defer srv.Close() + + resp, _, err := NewClient(srv.URL).DeployApp(context.Background(), "blog", dto.AppDeployRequest{}) + require.Error(t, err) + require.Nil(t, resp) + var httpErr *HTTPError + require.ErrorAs(t, err, &httpErr) + assert.Equal(t, http.StatusConflict, httpErr.StatusCode) +} + +func TestClientListAppSecrets_ServiceFilterAndErrorContext(t *testing.T) { + var gotService string + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + require.Equal(t, "/admin/apps/blog/secrets", r.URL.Path) + gotService = r.URL.Query().Get("service") + w.Header().Set("Content-Type", "application/json") + _ = json.NewEncoder(w).Encode([]dto.AppSecretMetadataDTO{ + {Service: "web", Key: "DATABASE_URL", Name: "database-url", Source: "desired", Presence: "unknown"}, + }) + })) + defer srv.Close() + + entries, err := NewClient(srv.URL).ListAppSecrets(context.Background(), "blog", "web") + require.NoError(t, err) + require.Len(t, entries, 1) + assert.Equal(t, "web", gotService) + + errSrv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(http.StatusInternalServerError) + _, _ = w.Write([]byte(`{"error":"internal"}`)) + })) + defer errSrv.Close() + + _, err = NewClient(errSrv.URL).ListAppSecrets(context.Background(), "blog", "") + require.Error(t, err) + assert.Contains(t, err.Error(), "list app secrets blog") +} diff --git a/internal/adapters/in/cli/remote/client.go b/internal/adapters/in/cli/remote/client.go index f3ba19e1e..37c5e961e 100644 --- a/internal/adapters/in/cli/remote/client.go +++ b/internal/adapters/in/cli/remote/client.go @@ -373,14 +373,17 @@ func parseErrorResponse(resp *http.Response, body []byte) error { msg := string(body) structured := false var errResp struct { - Error string `json:"error"` - Cause string `json:"cause"` - Hint string `json:"hint"` - Logs []string `json:"logs"` + Error string `json:"error"` + Message string `json:"message"` + Cause string `json:"cause"` + Hint string `json:"hint"` + Logs []string `json:"logs"` } if err := json.Unmarshal(body, &errResp); err == nil { structured = errResp.Cause != "" || errResp.Hint != "" || len(errResp.Logs) > 0 - if errResp.Error != "" { + if errResp.Message != "" { + msg = errResp.Message + } else if errResp.Error != "" { msg = errResp.Error } } @@ -478,276 +481,18 @@ func (c *Client) requestWithRetry(ctx context.Context, method, path string, body return nil, fmt.Errorf("request failed after retries") } -// Routes API - -// Type aliases for API types using shared DTO package. -type RouteInfo = dto.RouteInfo -type Attachment = dto.Attachment - -// ListRoutes returns all configured routes. -func (c *Client) ListRoutes(ctx context.Context) ([]domain.Route, error) { - resp, err := c.request(ctx, http.MethodGet, "/routes", nil) - if err != nil { - return nil, err - } - - var result struct { - Routes []domain.Route `json:"routes"` - } - if err := parseResponse(resp, &result); err != nil { - return nil, err - } - - return result.Routes, nil -} - -// ListRoutesWithDetails returns routes with network and attachment info. -func (c *Client) ListRoutesWithDetails(ctx context.Context) ([]RouteInfo, error) { - resp, err := c.request(ctx, http.MethodGet, "/routes?detailed=true", nil) - if err != nil { - return nil, err - } - - var result struct { - Routes []RouteInfo `json:"routes"` - } - if err := parseResponse(resp, &result); err != nil { - return nil, err - } - - return result.Routes, nil -} - -// ListNetworks returns Gordon-managed networks. -func (c *Client) ListNetworks(ctx context.Context) ([]*domain.NetworkInfo, error) { - resp, err := c.request(ctx, http.MethodGet, "/networks", nil) - if err != nil { - return nil, err - } - - var result struct { - Networks []*domain.NetworkInfo `json:"networks"` - } - if err := parseResponse(resp, &result); err != nil { - return nil, err - } - - return result.Networks, nil -} - -// ListAttachments returns attachments for a domain. -func (c *Client) ListAttachments(ctx context.Context, routeDomain string) ([]Attachment, error) { - if routeDomain == "" { - return nil, fmt.Errorf("domain is required") - } - path := "/routes/" + url.PathEscape(routeDomain) + "/attachments" - resp, err := c.request(ctx, http.MethodGet, path, nil) - if err != nil { - return nil, err - } - - var result struct { - Attachments []Attachment `json:"attachments"` - } - if err := parseResponse(resp, &result); err != nil { - return nil, err - } - - return result.Attachments, nil -} - -// GetRoute returns a specific route by domain. -func (c *Client) GetRoute(ctx context.Context, routeDomain string) (*domain.Route, error) { - resp, err := c.request(ctx, http.MethodGet, "/routes/"+url.PathEscape(routeDomain), nil) - if err != nil { - return nil, err - } - - var route domain.Route - if err := parseResponse(resp, &route); err != nil { - var httpErr *HTTPError - if errors.As(err, &httpErr) && httpErr.StatusCode == http.StatusNotFound { - return nil, domain.ErrRouteNotFound - } - return nil, err - } - - return &route, nil -} - -// GetRouteCleanupPreview returns retained cleanup state for a route that may no longer be configured. -func (c *Client) GetRouteCleanupPreview(ctx context.Context, routeDomain string) (*domain.CleanupReport, error) { - resp, err := c.request(ctx, http.MethodGet, "/routes/"+url.PathEscape(routeDomain)+"/cleanup", nil) - if err != nil { - return nil, err - } - - var report domain.CleanupReport - if err := parseResponse(resp, &report); err != nil { - return nil, err - } - - return &report, nil -} - -// FindRoutesByImage returns all routes associated with the given image name. -func (c *Client) FindRoutesByImage(ctx context.Context, imageName string) ([]domain.Route, error) { - resp, err := c.request(ctx, http.MethodGet, "/routes/by-image/"+url.PathEscape(imageName), nil) - if err != nil { - return nil, err - } - - var result struct { - Image string `json:"image"` - Routes []domain.Route `json:"routes"` - } - if err := parseResponse(resp, &result); err != nil { - return nil, err - } - - return result.Routes, nil -} - -// AddRoute adds a new route. -func (c *Client) AddRoute(ctx context.Context, route domain.Route) error { - resp, err := c.request(ctx, http.MethodPost, "/routes", route) - if err != nil { - return err - } - return parseResponse(resp, nil) -} - -// UpdateRoute updates an existing route. -func (c *Client) UpdateRoute(ctx context.Context, route domain.Route) error { - resp, err := c.request(ctx, http.MethodPut, "/routes/"+url.PathEscape(route.Domain), route) - if err != nil { - return err - } - return parseResponse(resp, nil) -} - -// RemoveRoute removes a route by domain. -func (c *Client) RemoveRoute(ctx context.Context, routeDomain string) error { - _, err := c.RemoveRouteWithCleanup(ctx, routeDomain) - return err -} - -// RemoveRouteWithCleanup removes a route and returns the server cleanup report. -func (c *Client) RemoveRouteWithCleanup(ctx context.Context, routeDomain string) (*dto.RouteDeleteResponse, error) { - resp, err := c.request(ctx, http.MethodDelete, "/routes/"+url.PathEscape(routeDomain), nil) - if err != nil { - return nil, fmt.Errorf("delete route %s failed: %w", routeDomain, err) - } - var result dto.RouteDeleteResponse - if err := parseResponse(resp, &result); err != nil { - return nil, fmt.Errorf("parse delete route %s response: %w", routeDomain, err) - } - return &result, nil -} - -func (c *Client) Bootstrap(ctx context.Context, req dto.BootstrapRequest) (*dto.BootstrapResponse, error) { - resp, err := c.request(ctx, http.MethodPost, "/bootstrap", req) - if err != nil { - return nil, err - } - var result dto.BootstrapResponse - if err := parseResponse(resp, &result); err != nil { - return nil, err - } - return &result, nil -} - -// Secrets API - -// AttachmentSecrets represents secrets for an attachment container. -type AttachmentSecrets struct { - Service string `json:"service"` - Keys []string `json:"keys"` -} - -// SecretsListResult contains domain secrets and any attachment secrets. -type SecretsListResult struct { - Domain string `json:"domain"` - Keys []string `json:"keys"` - Attachments []AttachmentSecrets `json:"attachments,omitempty"` -} - -// ListSecrets returns the list of secret keys for a domain. -func (c *Client) ListSecrets(ctx context.Context, secretDomain string) ([]string, error) { - result, err := c.ListSecretsWithAttachments(ctx, secretDomain) - if err != nil { - return nil, err - } - return result.Keys, nil -} - -// ListSecretsWithAttachments returns domain secrets and attachment secrets. -func (c *Client) ListSecretsWithAttachments(ctx context.Context, secretDomain string) (*SecretsListResult, error) { - resp, err := c.request(ctx, http.MethodGet, "/secrets/"+url.PathEscape(secretDomain), nil) - if err != nil { - return nil, err - } - - var result SecretsListResult - if err := parseResponse(resp, &result); err != nil { - return nil, err - } - - return &result, nil -} - -// SetSecrets sets secrets for a domain. -func (c *Client) SetSecrets(ctx context.Context, secretDomain string, secrets map[string]string) error { - resp, err := c.request(ctx, http.MethodPost, "/secrets/"+url.PathEscape(secretDomain), secrets) - if err != nil { - return err - } - return parseResponse(resp, nil) -} - -// DeleteSecret removes a secret from a domain. -func (c *Client) DeleteSecret(ctx context.Context, secretDomain, key string) error { - resp, err := c.request(ctx, http.MethodDelete, "/secrets/"+url.PathEscape(secretDomain)+"/"+url.PathEscape(key), nil) - if err != nil { - return err - } - return parseResponse(resp, nil) -} - -// SetAttachmentSecrets sets secrets for an attachment container. -func (c *Client) SetAttachmentSecrets(ctx context.Context, domain, service string, secrets map[string]string) error { - path := "/secrets/" + url.PathEscape(domain) + "/attachments/" + url.PathEscape(service) - resp, err := c.request(ctx, http.MethodPost, path, secrets) - if err != nil { - return err - } - return parseResponse(resp, nil) -} - -// DeleteAttachmentSecret removes a secret from an attachment container. -func (c *Client) DeleteAttachmentSecret(ctx context.Context, domain, service, key string) error { - path := "/secrets/" + url.PathEscape(domain) + "/attachments/" + url.PathEscape(service) + "/" + url.PathEscape(key) - resp, err := c.request(ctx, http.MethodDelete, path, nil) - if err != nil { - return err - } - return parseResponse(resp, nil) -} - // Status API // Status represents the Gordon server status. type Status struct { - Routes int `json:"routes"` + Apps int `json:"apps"` RegistryDomain string `json:"registry_domain"` RegistryPort int `json:"registry_port"` ServerPort int `json:"server_port"` - AutoRoute bool `json:"auto_route"` NetworkIsolation bool `json:"network_isolation"` ContainerStatus map[string]string `json:"container_status"` } -// GetTLSStatus returns the public TLS/ACME status. func (c *Client) GetTLSStatus(ctx context.Context) (*dto.TLSStatusResponse, error) { resp, err := c.request(ctx, http.MethodGet, "/tls/status", nil) if err != nil { @@ -840,10 +585,10 @@ func (c *Client) PruneImages(ctx context.Context, req dto.ImagePruneRequest) (*d } // ListBackups returns backups globally or for a domain. -func (c *Client) ListBackups(ctx context.Context, backupDomain string) ([]dto.BackupJob, error) { +func (c *Client) ListBackups(ctx context.Context, app string) ([]dto.BackupJob, error) { path := "/backups" - if backupDomain != "" { - path += "/" + url.PathEscape(backupDomain) + if app != "" { + path += "/" + url.PathEscape(app) } resp, err := c.request(ctx, http.MethodGet, path, nil) @@ -859,7 +604,8 @@ func (c *Client) ListBackups(ctx context.Context, backupDomain string) ([]dto.Ba return result.Backups, nil } -// BackupStatus returns aggregate backup status. +// BackupStatus returns stored backups plus declared targets that have no +// completed backup yet. func (c *Client) BackupStatus(ctx context.Context) ([]dto.BackupJob, error) { resp, err := c.request(ctx, http.MethodGet, "/backups/status", nil) if err != nil { @@ -874,13 +620,15 @@ func (c *Client) BackupStatus(ctx context.Context) ([]dto.BackupJob, error) { return result.Backups, nil } -// RunBackup triggers a backup for a domain. -func (c *Client) RunBackup(ctx context.Context, backupDomain, dbName string) (*dto.BackupRunResponse, error) { - if backupDomain == "" { - return nil, fmt.Errorf("domain cannot be empty") +// RunBackup runs one declared database backup of one app. service and +// database are explicit selectors. +func (c *Client) RunBackup(ctx context.Context, app, service, database string) (*dto.BackupRunResponse, error) { + if app == "" { + return nil, fmt.Errorf("app cannot be empty") } - resp, err := c.request(ctx, http.MethodPost, "/backups/"+url.PathEscape(backupDomain), dto.BackupRunRequest{DB: dbName}) + resp, err := c.request(ctx, http.MethodPost, "/backups/"+url.PathEscape(app), + dto.BackupRunRequest{Service: service, Database: database}) if err != nil { return nil, err } @@ -893,30 +641,11 @@ func (c *Client) RunBackup(ctx context.Context, backupDomain, dbName string) (*d return &result, nil } -// DetectDatabases detects supported databases for a domain. -func (c *Client) DetectDatabases(ctx context.Context, backupDomain string) ([]dto.DatabaseInfo, error) { - if backupDomain == "" { - return nil, fmt.Errorf("domain cannot be empty") - } - - resp, err := c.request(ctx, http.MethodGet, "/backups/"+url.PathEscape(backupDomain)+"/detect", nil) - if err != nil { - return nil, err - } - - var result dto.BackupDetectResponse - if err := parseResponse(resp, &result); err != nil { - return nil, err - } - - return result.Databases, nil -} - -// ListVolumeBackups returns volume backups globally or for a domain. -func (c *Client) ListVolumeBackups(ctx context.Context, backupDomain string) ([]dto.VolumeBackupJob, error) { +// ListVolumeBackups returns volume backups globally or for one app. +func (c *Client) ListVolumeBackups(ctx context.Context, app string) ([]dto.VolumeBackupJob, error) { path := "/backups/volumes" - if backupDomain != "" { - path += "/" + url.PathEscape(backupDomain) + if app != "" { + path += "/" + url.PathEscape(app) } resp, err := c.request(ctx, http.MethodGet, path, nil) @@ -945,14 +674,15 @@ func (c *Client) VolumeBackupStatus(ctx context.Context) ([]dto.VolumeBackupJob, return result.Backups, nil } -// RunVolumeBackups triggers volume backups. -func (c *Client) RunVolumeBackups(ctx context.Context, backupDomain, volumeName string) (*dto.VolumeBackupRunResponse, error) { - path := "/backups/volumes" - if backupDomain != "" { - path += "/" + url.PathEscape(backupDomain) +// RunVolumeBackups runs one declared volume backup of one app. service +// and volume are explicit selectors. +func (c *Client) RunVolumeBackups(ctx context.Context, app, service, volume string) (*dto.VolumeBackupRunResponse, error) { + if app == "" { + return nil, fmt.Errorf("app cannot be empty") } + path := "/backups/volumes/" + url.PathEscape(app) - resp, err := c.request(ctx, http.MethodPost, path, dto.VolumeBackupRunRequest{Volume: volumeName}) + resp, err := c.request(ctx, http.MethodPost, path, dto.VolumeBackupRunRequest{Service: service, Volume: volume}) if err != nil { return nil, err } @@ -1006,9 +736,6 @@ type Config struct { RegistryDomain string `json:"registry_domain"` DataDir string `json:"data_dir,omitempty"` } `json:"server"` - AutoRoute struct { - Enabled bool `json:"enabled"` - } `json:"auto_route"` NetworkIsolation struct { Enabled bool `json:"enabled"` Prefix string `json:"prefix"` @@ -1018,7 +745,6 @@ type Config struct { Prefix string `json:"prefix"` Preserve bool `json:"preserve"` } `json:"volumes"` - Routes []domain.Route `json:"routes"` ExternalRoutes []ExternalRoute `json:"external_routes"` } @@ -1029,97 +755,43 @@ type ExternalRoute struct { } // GetConfig returns the Gordon configuration. -func (c *Client) GetConfig(ctx context.Context) (*Config, error) { - resp, err := c.request(ctx, http.MethodGet, "/config", nil) +func (c *Client) ListNetworks(ctx context.Context) ([]*domain.NetworkInfo, error) { + resp, err := c.request(ctx, http.MethodGet, "/networks", nil) if err != nil { return nil, err } - var config Config - if err := parseResponse(resp, &config); err != nil { - return nil, err - } - - return &config, nil -} - -// Ping checks if the remote Gordon instance is reachable. -func (c *Client) Ping(ctx context.Context) error { - _, err := c.GetStatus(ctx) - return err -} - -// Deploy API - -// DeployResult contains the result of a deployment. -type DeployResult struct { - Status string `json:"status"` - ContainerID string `json:"container_id"` - Domain string `json:"domain"` -} - -// Deploy triggers a deployment for the specified domain. -func (c *Client) Deploy(ctx context.Context, deployDomain string) (*DeployResult, error) { - resp, err := c.requestWithRetry(ctx, http.MethodPost, "/deploy/"+url.PathEscape(deployDomain), nil) - if err != nil { - return nil, err + var result struct { + Networks []*domain.NetworkInfo `json:"networks"` } - - var result DeployResult if err := parseResponse(resp, &result); err != nil { return nil, err } - return &result, nil -} - -// DeployIntent tells the server that a CLI-managed push is about to happen, -// suppressing event-based deploys for this image. -func (c *Client) DeployIntent(ctx context.Context, imageName string) error { - imageName = strings.TrimSpace(imageName) - if imageName == "" { - return fmt.Errorf("image name cannot be empty") - } - resp, err := c.requestWithRetry(ctx, http.MethodPost, "/deploy-intent/"+url.PathEscape(imageName), nil) - if err != nil { - return err - } - return parseResponse(resp, nil) -} - -// Restart API - -// RestartResult contains the result of a restart. -type RestartResult struct { - Status string `json:"status"` - Domain string `json:"domain"` + return result.Networks, nil } -// Restart triggers a container restart for the specified domain. -func (c *Client) Restart(ctx context.Context, restartDomain string, withAttachments bool) (*RestartResult, error) { - if restartDomain == "" { - return nil, fmt.Errorf("domain cannot be empty") - } - path := "/restart/" + url.PathEscape(restartDomain) - if withAttachments { - path += "?attachments=true" - } - resp, err := c.requestWithRetry(ctx, http.MethodPost, path, nil) +// GetConfig returns the Gordon configuration. +func (c *Client) GetConfig(ctx context.Context) (*Config, error) { + resp, err := c.request(ctx, http.MethodGet, "/config", nil) if err != nil { return nil, err } - var result RestartResult - if err := parseResponse(resp, &result); err != nil { + var config Config + if err := parseResponse(resp, &config); err != nil { return nil, err } - return &result, nil + return &config, nil } -// Tags API +// Ping checks if the remote Gordon instance is reachable. +func (c *Client) Ping(ctx context.Context) error { + _, err := c.GetStatus(ctx) + return err +} -// ListTags returns available tags for a repository. func (c *Client) ListTags(ctx context.Context, repository string) ([]string, error) { if repository == "" { return nil, fmt.Errorf("repository cannot be empty") @@ -1194,169 +866,6 @@ func (c *Client) StreamContainerLogs(ctx context.Context, logDomain string, line // Attachments Config API // ListOrphanedAttachments returns running attachment containers no longer configured. -func (c *Client) ListOrphanedAttachments(ctx context.Context) ([]domain.CleanupAttachment, error) { - resp, err := c.request(ctx, http.MethodGet, "/attachments/orphans", nil) - if err != nil { - return nil, err - } - var result struct { - Attachments []domain.CleanupAttachment `json:"attachments"` - } - if err := parseResponse(resp, &result); err != nil { - return nil, err - } - return result.Attachments, nil -} - -// CleanupOrphanedAttachments optionally stops/removes orphaned attachment containers. -func (c *Client) CleanupOrphanedAttachments(ctx context.Context, owner string, stop bool) (*domain.CleanupReport, error) { - path := "/attachments/prune" - params := url.Values{} - if stop { - params.Set("stop", "true") - } - if owner != "" { - params.Set("owner", owner) - } - if encoded := params.Encode(); encoded != "" { - path += "?" + encoded - } - resp, err := c.request(ctx, http.MethodPost, path, nil) - if err != nil { - return nil, fmt.Errorf("cleanup orphaned attachments failed: %w", err) - } - var result domain.CleanupReport - if err := parseResponse(resp, &result); err != nil { - return nil, fmt.Errorf("parse orphaned attachment cleanup response: %w", err) - } - return &result, nil -} - -// GetAllAttachmentsConfig returns all configured attachments. -func (c *Client) GetAllAttachmentsConfig(ctx context.Context) (map[string][]string, error) { - resp, err := c.request(ctx, http.MethodGet, "/attachments", nil) - if err != nil { - return nil, err - } - - var result struct { - Attachments map[string][]string `json:"attachments"` - } - if err := parseResponse(resp, &result); err != nil { - return nil, err - } - - return result.Attachments, nil -} - -// GetAttachmentsConfig returns attachments for a specific domain/group from config. -func (c *Client) GetAttachmentsConfig(ctx context.Context, domainOrGroup string) ([]string, error) { - if domainOrGroup == "" { - return nil, fmt.Errorf("domain or group is required") - } - path := "/attachments/" + url.PathEscape(domainOrGroup) - resp, err := c.request(ctx, http.MethodGet, path, nil) - if err != nil { - return nil, err - } - - var result struct { - Target string `json:"target"` - Images []string `json:"images"` - } - if err := parseResponse(resp, &result); err != nil { - return nil, err - } - - return result.Images, nil -} - -// FindAttachmentTargetsByImage returns all attachment targets associated with the given image name. -func (c *Client) FindAttachmentTargetsByImage(ctx context.Context, imageName string) ([]string, error) { - resp, err := c.request(ctx, http.MethodGet, "/attachments/by-image/"+url.PathEscape(imageName), nil) - if err != nil { - return nil, err - } - - var result dto.AttachmentTargetsByImageResponse - if err := parseResponse(resp, &result); err != nil { - return nil, err - } - - return result.Targets, nil -} - -// AddAttachment adds an attachment to a domain/group. -func (c *Client) AddAttachment(ctx context.Context, domainOrGroup, image string) error { - if domainOrGroup == "" { - return fmt.Errorf("domain or group is required") - } - if image == "" { - return fmt.Errorf("image is required") - } - path := "/attachments/" + url.PathEscape(domainOrGroup) - resp, err := c.request(ctx, http.MethodPost, path, struct { - Image string `json:"image"` - }{Image: image}) - if err != nil { - return err - } - return parseResponse(resp, nil) -} - -// RemoveAttachment removes an attachment from a domain/group. -func (c *Client) RemoveAttachment(ctx context.Context, domainOrGroup, image string) error { - if domainOrGroup == "" { - return fmt.Errorf("domain or group is required") - } - if image == "" { - return fmt.Errorf("image is required") - } - path := "/attachments/" + url.PathEscape(domainOrGroup) + "/" + url.PathEscape(image) - resp, err := c.request(ctx, http.MethodDelete, path, nil) - if err != nil { - return err - } - return parseResponse(resp, nil) -} - -func (c *Client) GetAutoRouteAllowedDomains(ctx context.Context) ([]string, error) { - resp, err := c.request(ctx, http.MethodGet, "/autoroute/allowed-domains", nil) - if err != nil { - return nil, err - } - - var result dto.AutoRouteAllowedDomainsResponse - if err := parseResponse(resp, &result); err != nil { - return nil, err - } - - return result.Domains, nil -} - -func (c *Client) AddAutoRouteAllowedDomain(ctx context.Context, pattern string) error { - if strings.TrimSpace(pattern) == "" { - return fmt.Errorf("pattern must not be empty") - } - resp, err := c.request(ctx, http.MethodPost, "/autoroute/allowed-domains", dto.AutoRouteAllowedDomainRequest{Pattern: pattern}) - if err != nil { - return err - } - return parseResponse(resp, nil) -} - -func (c *Client) RemoveAutoRouteAllowedDomain(ctx context.Context, pattern string) error { - if strings.TrimSpace(pattern) == "" { - return fmt.Errorf("pattern must not be empty") - } - resp, err := c.request(ctx, http.MethodDelete, "/autoroute/allowed-domains/"+url.PathEscape(pattern), nil) - if err != nil { - return err - } - return parseResponse(resp, nil) -} - -// openSSEStream opens an SSE connection to the given admin path and returns the response. func (c *Client) openSSEStream(ctx context.Context, path string) (*http.Response, error) { streamURL := c.baseURL + "/admin" + path @@ -1377,7 +886,8 @@ func (c *Client) openSSEStream(ctx context.Context, path string) (*http.Response // Use the same transport as the main client (honoring TLS config and // custom transports) but without a timeout so streaming doesn't get cut off. streamClient := &http.Client{ - Transport: c.httpClient.Transport, + Transport: c.httpClient.Transport, + CheckRedirect: c.httpClient.CheckRedirect, } resp, err := streamClient.Do(req) if err != nil { diff --git a/internal/adapters/in/cli/remote/client_test.go b/internal/adapters/in/cli/remote/client_test.go index 33b72b86c..db00308f1 100644 --- a/internal/adapters/in/cli/remote/client_test.go +++ b/internal/adapters/in/cli/remote/client_test.go @@ -17,7 +17,6 @@ import ( "github.com/stretchr/testify/require" "github.com/bnema/gordon/internal/adapters/dto" - "github.com/bnema/gordon/internal/domain" ) func withFastRetry(t *testing.T) { @@ -32,50 +31,6 @@ func withFastRetry(t *testing.T) { }) } -func TestClientRestartDoesNotRetryGatewayResponseAfterSideEffect(t *testing.T) { - withFastRetry(t) - - var mutations int32 - srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - require.Equal(t, "/admin/restart/test.example.com", r.URL.Path) - require.Equal(t, http.MethodPost, r.Method) - atomic.AddInt32(&mutations, 1) - w.WriteHeader(http.StatusServiceUnavailable) - _, _ = w.Write([]byte(`{"error":"upstream response unavailable"}`)) - })) - defer srv.Close() - - client := NewClient(srv.URL) - result, err := client.Restart(context.Background(), "test.example.com", false) - - require.Nil(t, result) - var outcomeErr *OutcomeUnknownError - require.ErrorAs(t, err, &outcomeErr) - assert.Contains(t, err.Error(), "inspect current state before retrying") - assert.EqualValues(t, 1, atomic.LoadInt32(&mutations)) -} - -func TestClientRestartDoesNotRetryAmbiguousTransportFailure(t *testing.T) { - withFastRetry(t) - - var mutations int32 - srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { - atomic.AddInt32(&mutations, 1) - conn, _, err := w.(http.Hijacker).Hijack() - require.NoError(t, err) - require.NoError(t, conn.Close()) - })) - defer srv.Close() - - client := NewClient(srv.URL) - result, err := client.Restart(context.Background(), "test.example.com", true) - - require.Nil(t, result) - var outcomeErr *OutcomeUnknownError - require.ErrorAs(t, err, &outcomeErr) - assert.EqualValues(t, 1, atomic.LoadInt32(&mutations)) -} - func TestClientReloadDoesNotRetryGatewayResponse(t *testing.T) { withFastRetry(t) @@ -97,73 +52,6 @@ func TestClientReloadDoesNotRetryGatewayResponse(t *testing.T) { assert.EqualValues(t, 1, atomic.LoadInt32(&attempts)) } -func TestClientDeployDoesNotRetryGatewayResponse(t *testing.T) { - withFastRetry(t) - - var attempts int32 - srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - require.Equal(t, "/admin/deploy/test.example.com", r.URL.Path) - require.Equal(t, http.MethodPost, r.Method) - atomic.AddInt32(&attempts, 1) - w.WriteHeader(http.StatusBadGateway) - _, _ = w.Write([]byte(`{"error":"upstream unavailable"}`)) - })) - defer srv.Close() - - client := NewClient(srv.URL) - result, err := client.Deploy(context.Background(), "test.example.com") - - require.Nil(t, result) - var outcomeErr *OutcomeUnknownError - require.ErrorAs(t, err, &outcomeErr) - assert.ErrorContains(t, err, "502 Bad Gateway: upstream unavailable") - assert.EqualValues(t, 1, atomic.LoadInt32(&attempts)) -} - -func TestClientDeployConfirmedRejectionIsNotOutcomeUnknown(t *testing.T) { - var attempts int32 - srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { - atomic.AddInt32(&attempts, 1) - w.WriteHeader(http.StatusUnprocessableEntity) - _, _ = w.Write([]byte(`{"error":"invalid deployment"}`)) - })) - defer srv.Close() - - client := NewClient(srv.URL) - result, err := client.Deploy(context.Background(), "test.example.com") - - require.Nil(t, result) - var outcomeErr *OutcomeUnknownError - assert.NotErrorAs(t, err, &outcomeErr) - var httpErr *HTTPError - require.ErrorAs(t, err, &httpErr) - assert.Equal(t, http.StatusUnprocessableEntity, httpErr.StatusCode) - assert.EqualValues(t, 1, atomic.LoadInt32(&attempts)) -} - -func TestClientDeployAuthenticationRejectionIsNotOutcomeUnknown(t *testing.T) { - var deployAttempts int32 - srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - if r.URL.Path == "/auth/token" { - w.WriteHeader(http.StatusForbidden) - _, _ = w.Write([]byte(`{"error":"invalid credentials"}`)) - return - } - atomic.AddInt32(&deployAttempts, 1) - w.WriteHeader(http.StatusNoContent) - })) - defer srv.Close() - - client := NewClient(srv.URL, WithToken("invalid-token")) - result, err := client.Deploy(context.Background(), "test.example.com") - - require.Nil(t, result) - var outcomeErr *OutcomeUnknownError - assert.NotErrorAs(t, err, &outcomeErr) - assert.ErrorContains(t, err, "ephemeral token exchange: 403 Forbidden") - assert.Zero(t, atomic.LoadInt32(&deployAttempts)) -} - func TestClientGetStatusDoesNotRetryAuthenticationRejection(t *testing.T) { withFastRetry(t) @@ -197,7 +85,7 @@ func TestClientGetStatusRetriesTransientGatewayResponse(t *testing.T) { return } w.Header().Set("Content-Type", "application/json") - _, _ = w.Write([]byte(`{"routes":2,"container_status":{}}`)) + _, _ = w.Write([]byte(`{"apps":2,"container_status":{}}`)) })) defer srv.Close() @@ -206,7 +94,7 @@ func TestClientGetStatusRetriesTransientGatewayResponse(t *testing.T) { require.NoError(t, err) require.NotNil(t, status) - assert.Equal(t, 2, status.Routes) + assert.Equal(t, 2, status.Apps) assert.EqualValues(t, 3, atomic.LoadInt32(&attempts)) } @@ -222,7 +110,7 @@ func TestClientGetStatusRetriesTransientTransportFailure(t *testing.T) { return } w.Header().Set("Content-Type", "application/json") - _, _ = w.Write([]byte(`{"routes":2,"container_status":{}}`)) + _, _ = w.Write([]byte(`{"apps":2,"container_status":{}}`)) })) defer srv.Close() @@ -231,7 +119,7 @@ func TestClientGetStatusRetriesTransientTransportFailure(t *testing.T) { require.NoError(t, err) require.NotNil(t, status) - assert.Equal(t, 2, status.Routes) + assert.Equal(t, 2, status.Apps) assert.EqualValues(t, 3, atomic.LoadInt32(&attempts)) } @@ -259,7 +147,7 @@ func TestClientWithInsecureTLS_AllowsSelfSignedCertificate(t *testing.T) { srv := httptest.NewTLSServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { require.Equal(t, "/admin/status", r.URL.Path) w.Header().Set("Content-Type", "application/json") - _, _ = w.Write([]byte(`{"routes":0,"registry_domain":"","registry_port":0,"server_port":0,"auto_route":false,"network_isolation":false,"container_status":{}}`)) + _, _ = w.Write([]byte(`{"apps":0,"registry_domain":"","registry_port":0,"server_port":0,"auto_route":false,"network_isolation":false,"container_status":{}}`)) })) defer srv.Close() @@ -355,53 +243,6 @@ func TestClientPruneImages(t *testing.T) { assert.Equal(t, 3, resp.Registry.TagsRemoved) } -func TestClientFindAttachmentTargetsByImage(t *testing.T) { - srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - require.Equal(t, "/admin/attachments/by-image/postgres:16", r.URL.Path) - require.Equal(t, "", r.URL.RawQuery) - require.Equal(t, http.MethodGet, r.Method) - w.Header().Set("Content-Type", "application/json") - _, _ = w.Write([]byte(`{"image":"postgres:16","targets":["app.example.com","workers"]}`)) - })) - defer srv.Close() - - client := NewClient(srv.URL) - targets, err := client.FindAttachmentTargetsByImage(context.Background(), "postgres:16") - require.NoError(t, err) - assert.Equal(t, []string{"app.example.com", "workers"}, targets) -} - -func TestClientGetRoute_Maps404ToDomainErrRouteNotFound(t *testing.T) { - srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - require.Equal(t, "/admin/routes/missing.example.test", r.URL.Path) - require.Equal(t, http.MethodGet, r.Method) - w.WriteHeader(http.StatusNotFound) - _, _ = w.Write([]byte(`{"error":"route not found"}`)) - })) - defer srv.Close() - - client := NewClient(srv.URL) - route, err := client.GetRoute(context.Background(), "missing.example.test") - require.Nil(t, route) - require.Error(t, err) - assert.ErrorIs(t, err, domain.ErrRouteNotFound) -} - -func TestClientFindAttachmentTargetsByImage_WithSlashContainingImage(t *testing.T) { - srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - require.Equal(t, "/admin/attachments/by-image/registry/org/image:tag", r.URL.Path) - require.Equal(t, http.MethodGet, r.Method) - w.Header().Set("Content-Type", "application/json") - _, _ = w.Write([]byte(`{"image":"registry/org/image:tag","targets":["workers"]}`)) - })) - defer srv.Close() - - client := NewClient(srv.URL) - targets, err := client.FindAttachmentTargetsByImage(context.Background(), "registry/org/image:tag") - require.NoError(t, err) - assert.Equal(t, []string{"workers"}, targets) -} - func TestParseResponse_CapsErrorBodySize(t *testing.T) { largeBody := strings.Repeat("x", 10*1024*1024) resp := &http.Response{ @@ -538,6 +379,16 @@ func TestParseErrorResponse_ReturnsHTTPError(t *testing.T) { assert.Equal(t, "insufficient scope", httpErr.Body) } +func TestParseErrorResponse_AppErrorMessage(t *testing.T) { + resp := &http.Response{StatusCode: 400, Status: "400 Bad Request"} + err := parseErrorResponse(resp, []byte(`{"error":"invalid-manifest","message":"invalid app manifest: service \"metrics\" readiness.port 9090 matches no declared container port"}`)) + + var httpErr *HTTPError + require.True(t, errors.As(err, &httpErr)) + assert.Equal(t, `invalid app manifest: service "metrics" readiness.port 9090 matches no declared container port`, httpErr.Body) + assert.Contains(t, err.Error(), `service "metrics" readiness.port 9090`) +} + func TestParseErrorResponse_DeployFailureFields(t *testing.T) { resp := &http.Response{ StatusCode: 500, @@ -589,20 +440,20 @@ func TestParseErrorResponse_NonJSON(t *testing.T) { func TestRunVolumeBackups_PartialContentReturnsResultAndError(t *testing.T) { srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - require.Equal(t, "/admin/backups/volumes/app.example.com", r.URL.Path) + require.Equal(t, "/admin/backups/volumes/shop", r.URL.Path) require.Equal(t, http.MethodPost, r.Method) w.Header().Set("Content-Type", "application/json") w.WriteHeader(http.StatusPartialContent) require.NoError(t, json.NewEncoder(w).Encode(dto.VolumeBackupRunResponse{ Status: "partial", - Backups: []dto.VolumeBackupJob{{ID: "v1", Domain: "app.example.com", VolumeName: "gordon-app-data", Status: "completed"}}, + Backups: []dto.VolumeBackupJob{{ID: "v1", App: "shop", Service: "api", VolumeName: "data", Status: "completed"}}, Error: "one volume failed", })) })) defer srv.Close() client := NewClient(srv.URL) - result, err := client.RunVolumeBackups(context.Background(), "app.example.com", "") + result, err := client.RunVolumeBackups(context.Background(), "shop", "api", "data") require.Error(t, err) assert.Contains(t, err.Error(), "one volume failed") diff --git a/internal/adapters/in/cli/remote/local.go b/internal/adapters/in/cli/remote/local.go new file mode 100644 index 000000000..7f4d36e4a --- /dev/null +++ b/internal/adapters/in/cli/remote/local.go @@ -0,0 +1,106 @@ +package remote + +import ( + "context" + "errors" + "fmt" + "io/fs" + "net" + "net/http" + "os" + "path/filepath" + "syscall" + "time" + + "github.com/bnema/gordon/internal/adapters/localadmin" +) + +// LocalBaseURL is the synthetic origin of the Unix-socket admin client. The +// transport ignores the host and always dials the owner-only socket. +const LocalBaseURL = "http://gordon.local" + +// ErrDaemonUnavailable reports that no owner-only admin socket passed +// validation, so no local daemon can be reached. +var ErrDaemonUnavailable = errors.New("daemon-unavailable") + +// errLocalRedirect refuses redirects on the local socket: a redirect can never +// be a legitimate part of the owner-only admin surface. +var errLocalRedirect = errors.New("local admin client does not follow redirects") + +// localSocketDirs lists the discovery candidates in daemon priority order. +// It is a variable so tests can isolate from a real daemon. +var localSocketDirs = localadmin.ClientDirCandidates + +// DiscoverLocalSocket returns the first admin socket candidate, in daemon +// priority order, that is an owner-owned Unix socket without group/other +// permission bits. +func DiscoverLocalSocket() (string, error) { + return DiscoverLocalSocketIn(localSocketDirs()) +} + +// DiscoverLocalSocketIn is DiscoverLocalSocket over explicit candidate +// directories. Each candidate is validated, never followed through symlinks. +func DiscoverLocalSocketIn(dirs []string) (string, error) { + var lastErr error + for _, dir := range dirs { + path := filepath.Join(dir, localadmin.SocketName) + if _, err := os.Lstat(path); err != nil { + lastErr = err + continue + } + if err := localadmin.ValidateDir(dir); err != nil { + return "", fmt.Errorf("%w: unsafe local admin directory: %w", ErrDaemonUnavailable, err) + } + if err := localadmin.ValidateSocket(path); err != nil { + return "", fmt.Errorf("%w: unsafe local admin socket: %w", ErrDaemonUnavailable, err) + } + live, err := localadmin.ProbeSocket(path) + if err != nil { + return "", fmt.Errorf("%w: probe local admin socket: %w", ErrDaemonUnavailable, err) + } + if !live { + lastErr = syscall.ECONNREFUSED + continue + } + return path, nil + } + if lastErr == nil { + lastErr = fs.ErrNotExist + } + return "", fmt.Errorf("%w: no owner-only admin socket found: %w", ErrDaemonUnavailable, lastErr) +} + +// NewLocalClient discovers the daemon's owner-only admin socket and returns a +// client bound to it. It returns ErrDaemonUnavailable when no daemon is +// reachable; callers must not fall back to local writes. +func NewLocalClient() (*Client, error) { + socketPath, err := DiscoverLocalSocket() + if err != nil { + return nil, err + } + return NewLocalClientForSocket(socketPath), nil +} + +// NewLocalClientForSocket returns a client whose transport dials only the +// given Unix socket. It sends no Authorization header and never exchanges a +// token, so it works with auth.enabled=false. +func NewLocalClientForSocket(socketPath string) *Client { + transport := &http.Transport{ + // Never consult environment proxies for the local socket. DialContext + // authenticates SO_PEERCRED before the transport can write request bytes. + Proxy: nil, + DialContext: func(ctx context.Context, _, _ string) (net.Conn, error) { + return localadmin.DialContext(ctx, socketPath) + }, + } + + httpClient := &http.Client{ + Transport: transport, + Timeout: 2 * time.Minute, + CheckRedirect: func(*http.Request, []*http.Request) error { + return errLocalRedirect + }, + } + + return NewClient(LocalBaseURL, WithHTTPClient(httpClient)) +} diff --git a/internal/adapters/in/cli/remote/local_test.go b/internal/adapters/in/cli/remote/local_test.go new file mode 100644 index 000000000..dd0ed2927 --- /dev/null +++ b/internal/adapters/in/cli/remote/local_test.go @@ -0,0 +1,210 @@ +package remote + +import ( + "context" + "encoding/json" + "errors" + "net/http" + "os" + "path/filepath" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/localadmin" +) + +func mustListenUnix(t *testing.T, dir string, handler http.Handler) string { + t.Helper() + ln, _, err := localadmin.Listen(dir) + require.NoError(t, err) + + srv := &http.Server{Handler: handler, ReadHeaderTimeout: 5 * time.Second} + go func() { _ = srv.Serve(ln) }() + t.Cleanup(func() { _ = srv.Close() }) + + return localadmin.SocketPath(dir) +} + +func TestLocalAdminDiscoveryRejectsUnsafeSelectedCandidate(t *testing.T) { + unsafeDir := t.TempDir() + goodDir := t.TempDir() + + unsafeLn, _, err := localadmin.Listen(unsafeDir) + require.NoError(t, err) + t.Cleanup(func() { _ = unsafeLn.Close() }) + require.NoError(t, os.Chmod(localadmin.SocketPath(unsafeDir), 0o666)) + + goodLn, _, err := localadmin.Listen(goodDir) + require.NoError(t, err) + t.Cleanup(func() { _ = goodLn.Close() }) + + missingDir := t.TempDir() + + _, err = DiscoverLocalSocketIn([]string{missingDir, unsafeDir, goodDir}) + require.Error(t, err) + assert.ErrorIs(t, err, ErrDaemonUnavailable) + assert.ErrorIs(t, err, localadmin.ErrUnsafePath) +} + +func TestLocalAdminDiscoveryReportsUnavailable(t *testing.T) { + dir := t.TempDir() + _, err := DiscoverLocalSocketIn([]string{dir}) + require.Error(t, err) + assert.ErrorIs(t, err, ErrDaemonUnavailable) +} + +func TestLocalAdminDiscoveryUsesXDGCandidates(t *testing.T) { + xdg := t.TempDir() + t.Setenv("XDG_RUNTIME_DIR", xdg) + + ln, _, err := localadmin.Listen(filepath.Join(xdg, "gordon")) + require.NoError(t, err) + t.Cleanup(func() { _ = ln.Close() }) + + got, err := DiscoverLocalSocket() + require.NoError(t, err) + assert.Equal(t, localadmin.SocketPath(filepath.Join(xdg, "gordon")), got) +} + +func TestLocalAdminClientFailsWithoutSocket(t *testing.T) { + dirs := []string{t.TempDir(), t.TempDir()} + restore := localSocketDirs + localSocketDirs = func() []string { return dirs } + t.Cleanup(func() { localSocketDirs = restore }) + + _, err := NewLocalClient() + require.Error(t, err) + assert.ErrorIs(t, err, ErrDaemonUnavailable) +} + +// TestLocalAdminClientSendsNoTokenAndNoExchange proves tokenless local +// requests: no Authorization header and no /auth/token call. +func TestLocalAdminClientSendsNoTokenAndNoExchange(t *testing.T) { + dir := t.TempDir() + type captured struct { + path string + auth string + method string + } + requests := make(chan captured, 4) + + socketPath := mustListenUnix(t, dir, http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + requests <- captured{path: r.URL.Path, auth: r.Header.Get("Authorization"), method: r.Method} + w.Header().Set("Content-Type", "application/json") + _ = json.NewEncoder(w).Encode(map[string]any{"apps": []any{}}) + })) + + client := NewLocalClientForSocket(socketPath) + apps, err := client.ListApps(context.Background()) + require.NoError(t, err) + assert.Empty(t, apps) + + select { + case got := <-requests: + assert.Equal(t, "/admin/apps", got.path) + assert.Empty(t, got.auth) + case <-time.After(2 * time.Second): + t.Fatal("no request captured") + } + + select { + case extra := <-requests: + t.Fatalf("unexpected extra request: %+v", extra) + case <-time.After(100 * time.Millisecond): + } +} + +func TestLocalAdminClientIgnoresEnvironmentProxy(t *testing.T) { + dir := t.TempDir() + requests := make(chan string, 1) + + socketPath := mustListenUnix(t, dir, http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + requests <- r.URL.Path + w.Header().Set("Content-Type", "application/json") + _ = json.NewEncoder(w).Encode(map[string]any{"apps": []any{}}) + })) + + t.Setenv("HTTP_PROXY", "http://127.0.0.1:1") + t.Setenv("HTTPS_PROXY", "http://127.0.0.1:1") + + client := NewLocalClientForSocket(socketPath) + _, err := client.ListApps(context.Background()) + require.NoError(t, err) + + select { + case got := <-requests: + assert.Equal(t, "/admin/apps", got) + case <-time.After(2 * time.Second): + t.Fatal("no request captured") + } +} + +func TestLocalAdminClientRejectsRedirects(t *testing.T) { + dir := t.TempDir() + socketPath := mustListenUnix(t, dir, http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.URL.Path == "/admin/apps" { + http.Redirect(w, r, "http://evil.example/admin/apps", http.StatusFound) + return + } + w.WriteHeader(http.StatusTeapot) + })) + + client := NewLocalClientForSocket(socketPath) + _, err := client.ListApps(context.Background()) + require.Error(t, err) + assert.Contains(t, err.Error(), "redirect") +} + +func TestLocalAdminClientRespectsCancellation(t *testing.T) { + dir := t.TempDir() + release := make(chan struct{}) + socketPath := mustListenUnix(t, dir, http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + <-release + })) + + ctx, cancel := context.WithCancel(context.Background()) + done := make(chan error, 1) + go func() { + client := NewLocalClientForSocket(socketPath) + _, err := client.ListApps(ctx) + done <- err + }() + + cancel() + select { + case err := <-done: + require.Error(t, err) + assert.True(t, errors.Is(err, context.Canceled) || errors.Is(err, context.DeadlineExceeded), "got %v", err) + case <-time.After(3 * time.Second): + t.Fatal("request did not honor cancellation") + } + close(release) +} + +func TestLocalAdminClientFailsOnMissingSocket(t *testing.T) { + client := NewLocalClientForSocket(filepath.Join(t.TempDir(), "absent.sock")) + _, err := client.ListApps(context.Background()) + require.Error(t, err) + + var transportErr *requestTransportError + assert.True(t, errors.As(err, &transportErr), "expected transport error, got %T: %v", err, err) +} + +func TestLocalAdminTransportDialsUnixOnly(t *testing.T) { + dir := t.TempDir() + socketPath := mustListenUnix(t, dir, http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(http.StatusNoContent) + })) + + // The synthetic base URL host must never be resolved: only the socket is dialed. + client := NewLocalClientForSocket(socketPath) + req, err := http.NewRequestWithContext(context.Background(), http.MethodGet, LocalBaseURL+"/admin/apps", nil) + require.NoError(t, err) + resp, err := client.httpClient.Do(req) + require.NoError(t, err) + defer resp.Body.Close() + assert.Equal(t, http.StatusNoContent, resp.StatusCode) +} diff --git a/internal/adapters/in/cli/remote/preview.go b/internal/adapters/in/cli/remote/preview.go deleted file mode 100644 index 66633643e..000000000 --- a/internal/adapters/in/cli/remote/preview.go +++ /dev/null @@ -1,70 +0,0 @@ -package remote - -import ( - "context" - "fmt" - "net/http" - "net/url" - - "github.com/bnema/gordon/internal/domain" -) - -// ListPreviews returns all active preview environments. -func (c *Client) ListPreviews(ctx context.Context) ([]domain.PreviewRoute, error) { - resp, err := c.request(ctx, http.MethodGet, "/previews", nil) - if err != nil { - return nil, err - } - - var result struct { - Previews []domain.PreviewRoute `json:"previews"` - } - if err := parseResponse(resp, &result); err != nil { - return nil, err - } - - return result.Previews, nil -} - -// DeletePreview tears down a preview environment by name. -func (c *Client) DeletePreview(ctx context.Context, name string) error { - if name == "" { - return fmt.Errorf("preview name cannot be empty") - } - resp, err := c.request(ctx, http.MethodDelete, "/preview/"+url.PathEscape(name), nil) - if err != nil { - return err - } - return parseResponse(resp, nil) -} - -// ExtendPreview extends the TTL of a preview environment. -func (c *Client) ExtendPreview(ctx context.Context, name string, ttl string) error { - if name == "" { - return fmt.Errorf("preview name cannot be empty") - } - if ttl == "" { - return fmt.Errorf("ttl cannot be empty") - } - resp, err := c.request(ctx, http.MethodPatch, "/preview/"+url.PathEscape(name), struct { - TTL string `json:"ttl"` - }{TTL: ttl}) - if err != nil { - return err - } - return parseResponse(resp, nil) -} - -// GetPreview returns a single preview by name. -func (c *Client) GetPreview(ctx context.Context, name string) (*domain.PreviewRoute, error) { - previews, err := c.ListPreviews(ctx) - if err != nil { - return nil, fmt.Errorf("list previews: %w", err) - } - for _, p := range previews { - if p.Name == name { - return &p, nil - } - } - return nil, fmt.Errorf("preview %q: %w", name, domain.ErrPreviewNotFound) -} diff --git a/internal/adapters/in/cli/remote_inference.go b/internal/adapters/in/cli/remote_inference.go index a1f1ac592..3716553e4 100644 --- a/internal/adapters/in/cli/remote_inference.go +++ b/internal/adapters/in/cli/remote_inference.go @@ -5,20 +5,17 @@ import ( "errors" "fmt" "net/http" + "os" + "strconv" "strings" "github.com/bnema/gordon/internal/adapters/in/cli/remote" "github.com/bnema/gordon/internal/domain" - "github.com/bnema/gordon/pkg/validation" ) type remoteInferenceProbe func(context.Context, *remote.Client) (bool, error) -func inferPushRemote(ctx context.Context, imageArg, domainFlag, dockerfile string) (*remote.ResolvedRemote, error) { - if domainFlag != "" { - return inferRemoteForRouteDomain(ctx, domainFlag) - } - +func inferPushRemote(ctx context.Context, imageArg, _, dockerfile string) (*remote.ResolvedRemote, error) { if imageArg == "" { detected, err := detectImageName(dockerfile) if err != nil { @@ -26,105 +23,20 @@ func inferPushRemote(ctx context.Context, imageArg, domainFlag, dockerfile strin } imageArg = detected } - - if looksLikeLegacyDomain(imageArg) { - lookupImage := imageArg - domainLookupArg := imageArg - if parsedImage, ref := validation.ParseImageReference(imageArg); parsedImage != imageArg { - domainLookupArg = parsedImage - if !strings.HasPrefix(ref, "sha256:") { - lookupImage = parsedImage - } - } - - resolved, err := inferRemoteForImage(ctx, lookupImage) - if err != nil || resolved != nil { - return resolved, err - } - - return inferRemoteForRouteDomain(ctx, domainLookupArg) - } - classified := classifyPushArgument(imageArg) - return inferRemoteForImage(ctx, classified.lookupImage) + return inferRemoteForImage(ctx, classified.repository) } func inferRemoteForImage(ctx context.Context, imageName string) (*remote.ResolvedRemote, error) { return inferSavedRemote(ctx, "image", imageName, func(ctx context.Context, client *remote.Client) (bool, error) { - routes, err := client.FindRoutesByImage(ctx, imageName) - if err != nil { - return false, err - } - return len(filterPreviewRoutes(routes)) > 0, nil - }) -} - -func inferRemoteForRouteDomain(ctx context.Context, routeDomain string) (*remote.ResolvedRemote, error) { - return inferSavedRemote(ctx, "route", routeDomain, func(ctx context.Context, client *remote.Client) (bool, error) { - _, err := client.GetRoute(ctx, routeDomain) - if err != nil { - if isRemoteNotFoundError(err) { - return false, nil - } - return false, err - } - return true, nil - }) -} - -func inferRemoteForRouteCleanupDomain(ctx context.Context, routeDomain string) (*remote.ResolvedRemote, error) { - return inferSavedRemote(ctx, "route cleanup", routeDomain, func(ctx context.Context, client *remote.Client) (bool, error) { - _, err := client.GetRoute(ctx, routeDomain) - if err == nil { - return true, nil - } - if !isRemoteNotFoundError(err) { - return false, err - } - - preview, err := client.GetRouteCleanupPreview(ctx, routeDomain) - if err != nil { - return false, err - } - return routeCleanupReportHasEvidence(preview), nil - }) -} - -func routeCleanupReportHasEvidence(report *domain.CleanupReport) bool { - if report == nil { - return false - } - return len(report.PreservedVolumes) > 0 || len(report.PreservedAttachments) > 0 || len(report.OrphanedEntities) > 0 -} - -func inferRemoteForAttachmentImage(ctx context.Context, imageName string) (*remote.ResolvedRemote, error) { - return inferSavedRemote(ctx, "attachment image", imageName, func(ctx context.Context, client *remote.Client) (bool, error) { - targets, err := client.FindAttachmentTargetsByImage(ctx, imageName) - if err != nil { - return false, err - } - return len(targets) > 0, nil - }) -} - -func inferRemoteForAttachmentTarget(ctx context.Context, target string) (*remote.ResolvedRemote, error) { - return inferSavedRemote(ctx, "attachment target", target, func(ctx context.Context, client *remote.Client) (bool, error) { - images, err := client.GetAttachmentsConfig(ctx, target) - if err == nil { - return len(images) > 0, nil - } - if !isRemoteNotFoundError(err) { - return false, err - } - - _, err = client.GetRoute(ctx, target) + tags, err := client.ListTags(ctx, imageName) if err != nil { if isRemoteNotFoundError(err) { return false, nil } return false, err } - return true, nil + return len(tags) > 0, nil }) } @@ -153,7 +65,7 @@ func inferSavedRemote(ctx context.Context, targetKind, target string, probe remo for _, name := range names { entry := remotes[name] - matched, err := probe(ctx, newRoutesTargetClient(name, entry)) + matched, err := probe(ctx, newInferenceTargetClient(name, entry)) if err != nil { probeFailures = append(probeFailures, fmt.Sprintf("%s (%v)", name, err)) continue @@ -183,7 +95,7 @@ func inferSavedRemote(ctx context.Context, targetKind, target string, probe remo } func loadInferenceCandidateRemotes() (map[string]remote.RemoteEntry, bool) { - if _, ok := resolveRoutesExplicitTarget(); ok { + if _, ok := resolveExplicitTargetName(); ok { return nil, false } @@ -205,8 +117,8 @@ func resolvedRemoteFromEntry(name string, entry remote.RemoteEntry) *remote.Reso return &remote.ResolvedRemote{ Name: name, URL: entry.URL, - Token: resolveRoutesTokenForTarget(name, entry), - InsecureTLS: resolveRoutesInsecureForTarget(name, entry), + Token: resolveTokenForTarget(name, entry), + InsecureTLS: resolveInsecureForTarget(name, entry), } } @@ -217,12 +129,44 @@ func isRemoteNotFoundError(err error) bool { return errors.Is(err, domain.ErrRouteNotFound) } -func filterPreviewRoutes(routes []domain.Route) []domain.Route { - filtered := make([]domain.Route, 0, len(routes)) - for _, route := range routes { - if !strings.Contains(route.Domain, domain.DefaultPreviewSeparator) { - filtered = append(filtered, route) +func resolveExplicitTargetName() (string, bool) { + if target := strings.TrimSpace(remoteFlag); target != "" { + return target, true + } + if target := strings.TrimSpace(os.Getenv("GORDON_REMOTE")); target != "" { + return target, true + } + return "", false +} + +func resolveTokenForTarget(name string, entry remote.RemoteEntry) string { + if token := strings.TrimSpace(tokenFlag); token != "" { + return token + } + if token := strings.TrimSpace(os.Getenv("GORDON_TOKEN")); token != "" { + return token + } + if name != "" { + return remote.ResolveTokenForRemote(name, entry) + } + return "" +} + +func resolveInsecureForTarget(name string, entry remote.RemoteEntry) bool { + if insecureTLSFlag { + return true + } + if env := strings.TrimSpace(os.Getenv("GORDON_INSECURE")); env != "" { + if value, err := strconv.ParseBool(env); err == nil { + return value } } - return filtered + if name != "" { + return entry.InsecureTLS + } + return false +} + +func newInferenceTargetClient(name string, entry remote.RemoteEntry) *remote.Client { + return remote.NewClient(entry.URL, remoteClientOptions(resolveTokenForTarget(name, entry), resolveInsecureForTarget(name, entry))...) } diff --git a/internal/adapters/in/cli/remotes.go b/internal/adapters/in/cli/remotes.go index f47e0f6e6..450d49525 100644 --- a/internal/adapters/in/cli/remotes.go +++ b/internal/adapters/in/cli/remotes.go @@ -296,7 +296,7 @@ without needing to specify --remote. Examples: gordon remotes use prod - gordon routes list # Uses prod remote automatically`, + gordon daemon status # Uses prod remote automatically`, Args: cobra.ExactArgs(1), RunE: func(cmd *cobra.Command, args []string) error { name := args[0] diff --git a/internal/adapters/in/cli/restart.go b/internal/adapters/in/cli/restart.go deleted file mode 100644 index 0d8b59873..000000000 --- a/internal/adapters/in/cli/restart.go +++ /dev/null @@ -1,58 +0,0 @@ -package cli - -import ( - "fmt" - - "github.com/spf13/cobra" - - "github.com/bnema/gordon/internal/adapters/in/cli/ui/styles" - "github.com/bnema/gordon/internal/app" -) - -func newRestartCmd() *cobra.Command { - var withAttachments bool - - cmd := &cobra.Command{ - Use: "restart ", - Short: "Restart a running container", - Long: `Restarts the container for the specified route domain. -Useful after changing environment variables with 'gordon secrets set'. - -Use --with-attachments to also restart attached services (databases, caches). - -Examples: - gordon restart myapp.example.com --remote https://gordon.mydomain.com --token $TOKEN - gordon restart myapp.example.com --with-attachments --remote ...`, - Args: cobra.ExactArgs(1), - RunE: func(cmd *cobra.Command, args []string) error { - ctx := cmd.Context() - restartDomain := args[0] - - handle, err := resolveControlPlaneForRouteDomain(ctx, restartDomain) - if err != nil { - return err - } - defer handle.close() - - result, err := handle.plane.Restart(ctx, restartDomain, withAttachments) - if err != nil { - if handle.isRemote && !withAttachments && shouldFallbackToLocal(err) { - domain, localErr := app.SendDeploySignal(restartDomain) - if localErr == nil { - fmt.Println(styles.RenderWarning(fmt.Sprintf("Remote restart failed (%v), used local deploy-signal fallback", err))) - fmt.Println(styles.RenderSuccess(fmt.Sprintf("Restart signal sent for %s (local deploy path)", domain))) - return nil - } - return fmt.Errorf("remote restart failed: %w; local fallback failed: %v", err, localErr) - } - return fmt.Errorf("failed to restart: %w", err) - } - fmt.Println(styles.RenderSuccess(fmt.Sprintf("Restarted %s", result.Domain))) - return nil - }, - } - - cmd.Flags().BoolVar(&withAttachments, "with-attachments", false, "Also restart attached services (databases, caches)") - - return cmd -} diff --git a/internal/adapters/in/cli/root.go b/internal/adapters/in/cli/root.go index fd902d883..4e1a9a616 100644 --- a/internal/adapters/in/cli/root.go +++ b/internal/adapters/in/cli/root.go @@ -13,7 +13,6 @@ import ( "github.com/spf13/cobra" "github.com/bnema/gordon/internal/adapters/in/cli/remote" - "github.com/bnema/gordon/internal/app" ) var ( @@ -21,6 +20,7 @@ var ( Version = "dev" Commit = "unknown" BuildDate = "unknown" + Dirty = "unknown" // Global flags for remote targeting remoteFlag string @@ -50,16 +50,16 @@ func NewRootCmd() *cobra.Command { tokenFlag = token return nil }, - Short: "Gordon - A lightweight container deployment platform", - Long: `Gordon is a self-contained container deployment platform that combines -a Docker registry with automatic container deployment capabilities. + Short: "Gordon - A self-hosted container deployment platform", + Long: `Gordon is a self-hosted container deployment platform: a private +container registry, an app runtime, and a reverse proxy. -It listens for image pushes and automatically deploys containers based on -configuration rules, making it ideal for single-server deployments. +Applications are declared in standalone TOML files and managed with +gordon apps; gordon images push transfers OCI content only and never deploys. Commands are organized by where they run: - Server-only: Run on the machine hosting Gordon (serve, auth, reload) - Management: Work locally or remotely via --remote flag (routes, secrets, etc.) + Server-only: Run on the machine hosting Gordon (serve, auth, ca) + Management: Work locally or remotely via --remote flag (apps, daemon, images, etc.) Client-only: CLI utilities that don't require a running Gordon server`, } @@ -88,86 +88,16 @@ Commands are organized by where they run: caCmd.GroupID = groupServer rootCmd.AddCommand(caCmd) - // Management commands (work locally or via --remote) - routesCmd := newRoutesCmd() - routesCmd.GroupID = groupManage - rootCmd.AddCommand(routesCmd) - - attachmentsCmd := newAttachmentsCmd() - attachmentsCmd.GroupID = groupManage - rootCmd.AddCommand(attachmentsCmd) - - secretsCmd := newSecretsCmd() - secretsCmd.GroupID = groupManage - rootCmd.AddCommand(secretsCmd) - - deployCmd := newDeployCmd() - deployCmd.GroupID = groupManage - rootCmd.AddCommand(deployCmd) - - restartCmd := newRestartCmd() - restartCmd.GroupID = groupManage - rootCmd.AddCommand(restartCmd) - - pushCmd := newPushCmd() - pushCmd.GroupID = groupManage - rootCmd.AddCommand(pushCmd) - - pinCmd := newPinCmd() - pinCmd.GroupID = groupManage - rootCmd.AddCommand(pinCmd) - - reloadCmd := newReloadCmd() - reloadCmd.GroupID = groupManage - rootCmd.AddCommand(reloadCmd) - - logsCmd := newLogsCmd() - logsCmd.GroupID = groupManage - rootCmd.AddCommand(logsCmd) - - statusCmd := newStatusCmd() - statusCmd.GroupID = groupManage - rootCmd.AddCommand(statusCmd) - - backupCmd := newBackupCmd() - backupCmd.GroupID = groupManage - rootCmd.AddCommand(backupCmd) - - imagesCmd := newImagesCmd() - imagesCmd.GroupID = groupManage - rootCmd.AddCommand(imagesCmd) - - bootstrapCmd := newBootstrapCmd() - bootstrapCmd.GroupID = groupManage - rootCmd.AddCommand(bootstrapCmd) - - configCmd := newConfigCmd() - configCmd.GroupID = groupManage - rootCmd.AddCommand(configCmd) - - networksCmd := newNetworksCmd() - networksCmd.GroupID = groupManage - rootCmd.AddCommand(networksCmd) - - volumesCmd := newVolumesCmd() - volumesCmd.GroupID = groupManage - rootCmd.AddCommand(volumesCmd) - - autorouteCmd := newAutorouteCmd() - autorouteCmd.GroupID = groupManage - rootCmd.AddCommand(autorouteCmd) - - previewCmd := newPreviewCmd() - previewCmd.GroupID = groupManage - rootCmd.AddCommand(previewCmd) - - tlsCmd := newTLSCmd() - tlsCmd.GroupID = groupManage - rootCmd.AddCommand(tlsCmd) - - trafficCmd := newTrafficCmd() - trafficCmd.GroupID = groupManage - rootCmd.AddCommand(trafficCmd) + for _, cmd := range []*cobra.Command{ + newAppsCmd(), + newBackupCmd(), + newDaemonCmd(), + newImagesCmd(), + newVolumesCmd(), + } { + cmd.GroupID = groupManage + rootCmd.AddCommand(cmd) + } // Client-only commands (no server needed) remotesCmd := newRemotesCmd() @@ -222,21 +152,23 @@ func IsRemoteMode() bool { func newReloadCmd() *cobra.Command { return &cobra.Command{ Use: "reload", - Short: "Start containers for configured routes", - Long: `Starts containers for routes defined in config.toml that don't have -a running container. Running containers are never restarted to ensure 100% uptime. + Short: "Reload installation configuration (never activates app state)", + Long: `Reloads installation-only settings (entrypoints, TLS, limits, +external routes) after editing gordon.toml. -Use this command after editing config.toml to add new routes, or after pushing -images to the registry when the route was not yet configured.`, +Reload never activates pending app desired state, never re-resolves +image tags, and never starts app workloads. Apps are managed with +gordon apps (apply, deploy, lifecycle).`, RunE: func(cmd *cobra.Command, args []string) error { - client, isRemote, err := GetRemoteClient() + handle, err := resolveControlPlane(cliConfigPath) if err != nil { return err } - if isRemote { - return runReloadRemote(cmd.Context(), client) + defer handle.close() + if err := handle.plane.Reload(cmd.Context()); err != nil { + return fmt.Errorf("failed to reload: %w", err) } - return runReload() + return cliWriteLine(cmd.OutOrStdout(), cliRenderSuccess("Configuration reloaded successfully")) }, } } @@ -257,39 +189,17 @@ func newVersionCmd() *cobra.Command { if err := cliWriteLine(out, cliRenderMeta("Build Date:", BuildDate)); err != nil { return err } + if err := cliWriteLine(out, cliRenderMeta("Dirty:", Dirty)); err != nil { + return err + } return nil }, } } -// runReload sends SIGUSR1 to the running Gordon process. -func runReload() error { - return app.SendReloadSignal() -} - -// runReloadRemote triggers a reload on a remote Gordon instance. -func runReloadRemote(ctx context.Context, client *remote.Client) error { - if err := client.Reload(ctx); err != nil { - if shouldFallbackToLocal(err) { - localErr := runReload() - if localErr == nil { - if writeErr := cliWriteLine(os.Stdout, cliRenderWarning(fmt.Sprintf("Remote reload failed (%v), used local signal fallback", err))); writeErr != nil { - return writeErr - } - if writeErr := cliWriteLine(os.Stdout, cliRenderSuccess("Configuration reloaded successfully")); writeErr != nil { - return writeErr - } - return nil - } - return fmt.Errorf("remote reload failed: %w; local fallback failed: %v", err, localErr) - } - return fmt.Errorf("failed to reload: %w", err) - } - if err := cliWriteLine(os.Stdout, cliRenderSuccess("Configuration reloaded successfully")); err != nil { - return err - } - return nil -} +// cliConfigPath for local operations. If empty, config is auto-discovered +// from standard locations (/etc/gordon/gordon.toml, ~/.config/gordon/gordon.toml, ./gordon.toml). +var cliConfigPath string // newLogsCmd creates the logs command. func newLogsCmd() *cobra.Command { @@ -298,29 +208,23 @@ func newLogsCmd() *cobra.Command { var logsConfigPath string cmd := &cobra.Command{ - Use: "logs [domain]", - Short: "Show logs (Gordon process or container)", - Long: `Shows logs from the Gordon process or a specific container. + Use: "logs", + Short: "Show Gordon process logs", + Long: `Shows logs from the Gordon daemon process. -Without a domain argument, shows Gordon process logs. -With a domain argument, shows container logs for that domain. +Application workload output is read with gordon apps logs APP --service SVC, +which resolves the app's active container through the daemon. Examples: - gordon logs # Gordon process logs - gordon logs -f # Follow process logs - gordon logs myapp.local # Container logs for myapp.local - gordon logs myapp.local -f # Follow container logs + gordon daemon logs # Gordon process logs + gordon daemon logs -f # Follow process logs + gordon daemon logs -n 100 # Last 100 lines Remote mode: - gordon logs --remote https://gordon.mydomain.com --token $TOKEN - gordon logs myapp.local --remote https://gordon.mydomain.com --token $TOKEN`, - Args: cobra.MaximumNArgs(1), + gordon daemon logs --remote prod`, + Args: cobra.NoArgs, RunE: func(cmd *cobra.Command, args []string) error { - logDomain := "" - if len(args) > 0 { - logDomain = args[0] - } - return runLogs(cmd.Context(), logsConfigPath, logDomain, follow, lines, cmd.OutOrStdout()) + return runLogs(cmd.Context(), logsConfigPath, follow, lines, cmd.OutOrStdout()) }, } @@ -332,101 +236,23 @@ Remote mode: } // runLogs handles the logs command logic. -func runLogs(ctx context.Context, logsConfigPath, logDomain string, follow bool, lines int, out io.Writer) error { - if logDomain != "" { - handle, err := resolveControlPlaneForRouteDomain(ctx, logDomain) - if err != nil { - return err - } - defer handle.close() - return runContainerLogs(ctx, handle.plane, logDomain, follow, lines, out) - } - - client, isRemote, err := GetRemoteClient() +func runLogs(ctx context.Context, logsConfigPath string, follow bool, lines int, out io.Writer) error { + handle, err := resolveControlPlane(logsConfigPath) if err != nil { return err } - if isRemote { - return runLogsRemote(ctx, client, logDomain, follow, lines, out) - } - return runLogsLocal(logsConfigPath, logDomain, follow, lines, out) -} - -// runLogsRemote fetches logs from a remote Gordon instance. -func runLogsRemote(ctx context.Context, client *remote.Client, logDomain string, follow bool, lines int, out io.Writer) error { - ctx, cancel := signal.NotifyContext(ctx, os.Interrupt, syscall.SIGTERM) - defer cancel() - - if follow { - return streamLogsRemote(ctx, client, logDomain, lines, out) - } - - if logDomain == "" { - // Process logs - logLines, err := client.GetProcessLogs(ctx, lines) - if err != nil { - return fmt.Errorf("failed to get process logs: %w", err) - } - for _, line := range logLines { - if err := cliWriteLine(out, line); err != nil { - return err - } - } - } else { - // Container logs - logLines, err := client.GetContainerLogs(ctx, logDomain, lines) - if err != nil { - return fmt.Errorf("failed to get container logs: %w", err) - } - for _, line := range logLines { - if err := cliWriteLine(out, line); err != nil { - return err - } - } - } - return nil + defer handle.close() + return runProcessLogs(ctx, handle.plane, follow, lines, out) } -// streamLogsRemote streams logs from a remote Gordon instance. -func streamLogsRemote(ctx context.Context, client *remote.Client, logDomain string, lines int, out io.Writer) error { - var ch <-chan string - var err error - - if logDomain == "" { - ch, err = client.StreamProcessLogs(ctx, lines) - } else { - ch, err = client.StreamContainerLogs(ctx, logDomain, lines) - } - if err != nil { - return fmt.Errorf("failed to stream logs: %w", err) - } - - for line := range ch { - if err := cliWriteLine(out, line); err != nil { - return err - } - } - return nil -} - -// runLogsLocal shows logs from a local Gordon instance. -func runLogsLocal(logsConfigPath, logDomain string, follow bool, lines int, out io.Writer) error { - if logDomain == "" { - // Process logs - use existing app.ShowLogs - return app.ShowLogs(logsConfigPath, follow, lines) - } - - return showContainerLogsLocal(out, logsConfigPath, logDomain, follow, lines) -} - -func runContainerLogs(ctx context.Context, cp ControlPlane, logDomain string, follow bool, lines int, out io.Writer) error { +func runProcessLogs(ctx context.Context, cp ControlPlane, follow bool, lines int, out io.Writer) error { ctx, cancel := signal.NotifyContext(ctx, os.Interrupt, syscall.SIGTERM) defer cancel() if follow { - ch, err := cp.StreamContainerLogs(ctx, logDomain, lines) + ch, err := cp.StreamProcessLogs(ctx, lines) if err != nil { - return fmt.Errorf("failed to stream container logs: %w", err) + return fmt.Errorf("failed to stream process logs: %w", err) } for line := range ch { if err := cliWriteLine(out, line); err != nil { @@ -436,9 +262,9 @@ func runContainerLogs(ctx context.Context, cp ControlPlane, logDomain string, fo return nil } - logLines, err := cp.GetContainerLogs(ctx, logDomain, lines) + logLines, err := cp.GetProcessLogs(ctx, lines) if err != nil { - return fmt.Errorf("failed to get container logs: %w", err) + return fmt.Errorf("failed to get process logs: %w", err) } for _, line := range logLines { if err := cliWriteLine(out, line); err != nil { @@ -448,34 +274,10 @@ func runContainerLogs(ctx context.Context, cp ControlPlane, logDomain string, fo return nil } -// showContainerLogsLocal is retained for UI adoption coverage and fallback messaging. -func showContainerLogsLocal(out io.Writer, _ string, logDomain string, follow bool, lines int) error { - if err := cliWriteLine(out, cliRenderTitle(fmt.Sprintf("Container logs for %s", logDomain))); err != nil { - return err - } - if err := cliWriteLine(out, cliRenderInfo("To view container logs locally, use:")); err != nil { - return err - } - if err := cliWritef(out, " docker logs --tail %d %s\n", lines, logDomain); err != nil { - return err - } - if follow { - if err := cliWritef(out, " docker logs -f --tail %d %s\n", lines, logDomain); err != nil { - return err - } - } - if err := cliWriteLine(out, ""); err != nil { - return err - } - if err := cliWriteLine(out, cliRenderMuted("Or use remote mode to access logs via the admin API.")); err != nil { - return err - } - return nil -} - // SetVersionInfo sets the version information for the CLI. -func SetVersionInfo(version, commit, date string) { +func SetVersionInfo(version, commit, date, dirty string) { Version = version Commit = commit BuildDate = date + Dirty = dirty } diff --git a/internal/adapters/in/cli/routes.go b/internal/adapters/in/cli/routes.go deleted file mode 100644 index b95be363e..000000000 --- a/internal/adapters/in/cli/routes.go +++ /dev/null @@ -1,1837 +0,0 @@ -package cli - -import ( - "context" - "errors" - "fmt" - "io" - "os" - "sort" - "strconv" - "strings" - "sync" - - "github.com/spf13/cobra" - - "github.com/bnema/gordon/internal/adapters/dto" - "github.com/bnema/gordon/internal/adapters/in/cli/remote" - "github.com/bnema/gordon/internal/adapters/in/cli/ui/components" - "github.com/bnema/gordon/internal/adapters/in/cli/ui/styles" - "github.com/bnema/gordon/internal/domain" -) - -// configPath for local operations. If empty, config is auto-discovered -// from standard locations (/etc/gordon/gordon.toml, ~/.config/gordon/gordon.toml, ./gordon.toml). -var configPath string - -// truncateImage shortens long image references for display. -// For digests (image@sha256:...), shows first 12 chars of digest. -// For tags, truncates to maxLen with ellipsis if needed. -func truncateImage(image string, maxLen int) string { - // Handle digest references: image@sha256:abc123... - if maxLen <= 0 { - return "" - } - - if name, digest, ok := strings.Cut(image, "@sha256:"); ok { - if len(digest) > 12 { - digest = digest[:12] - } - short := fmt.Sprintf("%s@sha256:%s", name, digest) - if len(short) <= maxLen { - return short - } - // Truncate from the end to maintain valid reference format - if maxLen <= 3 { - return short[:maxLen] - } - return short[:maxLen-3] + "..." - } - - // Regular tag: truncate if needed - if len(image) <= maxLen { - return image - } - if maxLen <= 3 { - return image[:maxLen] - } - return image[:maxLen-3] + "..." -} - -const networkPrefix = "gordon-" - -// httpHealthToStatus maps an HTTP health probe result to a components.Status. -func httpHealthToStatus(health *remote.RouteHealth) components.Status { - if health == nil { - return components.StatusUnknown - } - if health.HTTPStatus == 0 { - if health.Error != "" { - return components.StatusError - } - return components.StatusUnknown - } - if health.HTTPStatus >= 200 && health.HTTPStatus < 400 { - return components.StatusSuccess - } - return components.StatusError -} - -// stripNetworkPrefix removes the "gordon-" prefix from a network name for display. -func stripNetworkPrefix(network string) string { - return strings.TrimPrefix(network, networkPrefix) -} - -// networkGroup holds routes that share a network. -type networkGroup struct { - name string - routes []remote.RouteInfo -} - -type routeListItem struct { - Domain string `json:"domain"` - Image string `json:"image"` -} - -type routeListSection struct { - Kind string `json:"kind"` - Name string `json:"name"` - URL string `json:"url,omitempty"` - Error string `json:"error,omitempty"` - Routes []routeListItem `json:"routes,omitempty"` -} - -type routeStatusAttachment struct { - Name string `json:"name"` - Image string `json:"image"` - Status string `json:"status"` -} - -type routeStatusItem struct { - Domain string `json:"domain"` - Image string `json:"image"` - ContainerID string `json:"container_id,omitempty"` - ContainerStatus string `json:"container_status"` - HTTPStatus int `json:"http_status"` - HealthError string `json:"health_error,omitempty"` - Network string `json:"network"` - Attachments []routeStatusAttachment `json:"attachments,omitempty"` -} - -type routeStatusSection struct { - Kind string `json:"kind"` - Name string `json:"name"` - URL string `json:"url,omitempty"` - Error string `json:"error,omitempty"` - Routes []routeStatusItem `json:"routes,omitempty"` -} - -type routeDiagnosis struct { - Domain string `json:"domain"` - Configured bool `json:"configured"` - Route *domain.Route `json:"route,omitempty"` - Runtime *remote.RouteInfo `json:"runtime,omitempty"` - Health *remote.RouteHealth `json:"health,omitempty"` - Volumes []dto.Volume `json:"volumes,omitempty"` - OrphanedAttachments []domain.CleanupAttachment `json:"orphaned_attachments,omitempty"` - OrphanedEntities []domain.CleanupOrphanedEntity `json:"orphaned_entities,omitempty"` - Warnings []string `json:"warnings,omitempty"` - Hints []string `json:"hints,omitempty"` -} - -type routesListDeps struct { - explicitRemote func() (*remote.ResolvedRemote, bool, error) - loadLocal func(context.Context, string) (routeListSection, error) - listRemotes func() (map[string]remote.RemoteEntry, string, error) - loadRemote func(context.Context, string, remote.RemoteEntry) (routeListSection, error) -} - -type routesStatusDeps struct { - explicitRemote func() (*remote.ResolvedRemote, bool, error) - loadLocal func(context.Context, string) (routeStatusSection, error) - listRemotes func() (map[string]remote.RemoteEntry, string, error) - loadRemote func(context.Context, string, remote.RemoteEntry) (routeStatusSection, error) -} - -// groupRoutesByNetwork separates routes into network groups (2+ routes sharing a network) -// and solo routes. Both are sorted alphabetically by domain. -func groupRoutesByNetwork(routes []remote.RouteInfo) ([]networkGroup, []remote.RouteInfo) { - byNetwork := make(map[string][]remote.RouteInfo) - var networkOrder []string - for _, route := range routes { - if _, seen := byNetwork[route.Network]; !seen { - networkOrder = append(networkOrder, route.Network) - } - byNetwork[route.Network] = append(byNetwork[route.Network], route) - } - - var groups []networkGroup - var solo []remote.RouteInfo - - for _, net := range networkOrder { - members := byNetwork[net] - if len(members) >= 2 { - sort.Slice(members, func(i, j int) bool { - return members[i].Domain < members[j].Domain - }) - groups = append(groups, networkGroup{ - name: stripNetworkPrefix(net), - routes: members, - }) - } else { - solo = append(solo, members[0]) - } - } - - sort.Slice(solo, func(i, j int) bool { - return solo[i].Domain < solo[j].Domain - }) - - return groups, solo -} - -// newRoutesCmd creates the routes command group. -func newRoutesCmd() *cobra.Command { - cmd := &cobra.Command{ - Use: "routes", - Short: "Manage routes", - Long: `Manage Gordon routes. Routes map domains to container images. - -When targeting a remote Gordon instance (via --remote flag or GORDON_REMOTE env var), -these commands operate on the remote server. Otherwise, they require access to -the local Gordon configuration.`, - } - - cmd.AddCommand(newRoutesListCmd()) - cmd.AddCommand(newRoutesStatusCmd()) - - cmd.AddCommand(newRoutesShowCmd()) - cmd.AddCommand(newRoutesDiagnoseCmd()) - cmd.AddCommand(newRoutesPurgeCmd()) - cmd.AddCommand(newRoutesAddCmd()) - cmd.AddCommand(newRoutesRemoveCmd()) - - return cmd -} - -// newRoutesListCmd creates the routes list command. -func newRoutesListCmd() *cobra.Command { - var jsonOut bool - - cmd := &cobra.Command{ - Use: "list", - Short: "List all routes", - RunE: func(cmd *cobra.Command, args []string) error { - sections, err := collectRoutesListSections(cmd.Context(), configPath, routesListDeps{}) - if err != nil { - return err - } - - if jsonOut { - return writeJSON(cmd.OutOrStdout(), sections) - } - - return renderRoutesListSections(cmd.OutOrStdout(), sections) - }, - } - - cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") - - return cmd -} - -func newRoutesStatusCmd() *cobra.Command { - var jsonOut bool - - cmd := &cobra.Command{ - Use: "status", - Short: "Show detailed route status", - RunE: func(cmd *cobra.Command, args []string) error { - sections, err := collectRoutesStatusSections(cmd.Context(), configPath, routesStatusDeps{}) - if err != nil { - return err - } - - if jsonOut { - return writeJSON(cmd.OutOrStdout(), sections) - } - - return renderRoutesStatusSections(cmd.OutOrStdout(), sections) - }, - } - - cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") - - return cmd -} - -func collectRoutesListSections(ctx context.Context, cfgPath string, deps routesListDeps) ([]routeListSection, error) { - if deps.explicitRemote == nil { - deps.explicitRemote = resolveRoutesExplicitRemote - } - if deps.loadLocal == nil { - deps.loadLocal = loadRoutesListLocalSection - } - if deps.listRemotes == nil { - deps.listRemotes = remote.ListRemotes - } - if deps.loadRemote == nil { - deps.loadRemote = loadRoutesListRemoteSection - } - - if resolved, ok, err := deps.explicitRemote(); err != nil { - return nil, err - } else if ok { - section := loadRoutesListExplicitRemoteSection(ctx, deps, resolved) - return []routeListSection{section}, nil - } - if target, ok := resolveRoutesExplicitTarget(); ok { - return []routeListSection{{ - Kind: "remote", - Name: target, - Error: unresolvedRoutesTargetError(target), - }}, nil - } - - return collectRoutesListAggregateSections(ctx, cfgPath, deps) -} - -func loadRoutesListExplicitRemoteSection(ctx context.Context, deps routesListDeps, resolved *remote.ResolvedRemote) routeListSection { - section, err := deps.loadRemote(ctx, resolved.DisplayName(), remote.RemoteEntry{URL: resolved.URL, Token: resolved.Token, InsecureTLS: resolved.InsecureTLS}) - if err != nil && section.Error == "" { - section.Error = err.Error() - } - return normalizeRouteListRemoteSection(section, resolved.DisplayName(), remote.RemoteEntry{URL: resolved.URL, Token: resolved.Token, InsecureTLS: resolved.InsecureTLS}) -} - -func collectRoutesListAggregateSections(ctx context.Context, cfgPath string, deps routesListDeps) ([]routeListSection, error) { - - type localResult struct { - section routeListSection - err error - } - type remotesResult struct { - entries map[string]remote.RemoteEntry - err error - } - - localCh := make(chan localResult, 1) - remotesCh := make(chan remotesResult, 1) - - go func() { - section, err := deps.loadLocal(ctx, cfgPath) - localCh <- localResult{section: section, err: err} - }() - go func() { - entries, _, err := deps.listRemotes() - remotesCh <- remotesResult{entries: entries, err: err} - }() - - localRes := <-localCh - remotes := <-remotesCh - local := localRes.section - if err := localRes.err; err != nil { - if local.Error == "" { - local.Error = err.Error() - } - } - - sections := make([]routeListSection, 0, 1) - sections = append(sections, normalizeRouteListLocalSection(local)) - - if remotes.err != nil { - sections = append(sections, routeListSection{Kind: "remote", Name: "remotes", Error: remotes.err.Error()}) - return sections, nil - } - - names := sortedRemoteNames(remotes.entries) - remoteSections := make([]routeListSection, len(names)) - var wg sync.WaitGroup - for i, name := range names { - entry := remotes.entries[name] - wg.Add(1) - go func(idx int, remoteName string, remoteEntry remote.RemoteEntry) { - defer wg.Done() - section, err := deps.loadRemote(ctx, remoteName, remoteEntry) - if err != nil && section.Error == "" { - section.Error = err.Error() - } - remoteSections[idx] = normalizeRouteListRemoteSection(section, remoteName, remoteEntry) - }(i, name, entry) - } - wg.Wait() - - sections = append(sections, remoteSections...) - return sections, nil -} - -func normalizeRouteListLocalSection(section routeListSection) routeListSection { - if section.Kind == "" { - section.Kind = "local" - } - if section.Name == "" { - section.Name = "local" - } - return section -} - -func normalizeRouteListRemoteSection(section routeListSection, name string, entry remote.RemoteEntry) routeListSection { - if section.Kind == "" { - section.Kind = "remote" - } - if section.Name == "" { - section.Name = name - } - if section.URL == "" { - section.URL = entry.URL - } - return section -} - -func collectRoutesStatusSections(ctx context.Context, cfgPath string, deps routesStatusDeps) ([]routeStatusSection, error) { - if deps.explicitRemote == nil { - deps.explicitRemote = resolveRoutesExplicitRemote - } - if deps.loadLocal == nil { - deps.loadLocal = loadRoutesStatusLocalSection - } - if deps.listRemotes == nil { - deps.listRemotes = remote.ListRemotes - } - if deps.loadRemote == nil { - deps.loadRemote = loadRoutesStatusRemoteSection - } - - if resolved, ok, err := deps.explicitRemote(); err != nil { - return nil, err - } else if ok { - section := loadRoutesStatusExplicitRemoteSection(ctx, deps, resolved) - return []routeStatusSection{section}, nil - } - if target, ok := resolveRoutesExplicitTarget(); ok { - return []routeStatusSection{{ - Kind: "remote", - Name: target, - Error: unresolvedRoutesTargetError(target), - }}, nil - } - - return collectRoutesStatusAggregateSections(ctx, cfgPath, deps) -} - -func loadRoutesStatusExplicitRemoteSection(ctx context.Context, deps routesStatusDeps, resolved *remote.ResolvedRemote) routeStatusSection { - section, err := deps.loadRemote(ctx, resolved.DisplayName(), remote.RemoteEntry{URL: resolved.URL, Token: resolved.Token, InsecureTLS: resolved.InsecureTLS}) - if err != nil && section.Error == "" { - section.Error = err.Error() - } - return normalizeRouteStatusRemoteSection(section, resolved.DisplayName(), remote.RemoteEntry{URL: resolved.URL, Token: resolved.Token, InsecureTLS: resolved.InsecureTLS}) -} - -func collectRoutesStatusAggregateSections(ctx context.Context, cfgPath string, deps routesStatusDeps) ([]routeStatusSection, error) { - - type localResult struct { - section routeStatusSection - err error - } - type remotesResult struct { - entries map[string]remote.RemoteEntry - err error - } - - localCh := make(chan localResult, 1) - remotesCh := make(chan remotesResult, 1) - - go func() { - section, err := deps.loadLocal(ctx, cfgPath) - localCh <- localResult{section: section, err: err} - }() - go func() { - entries, _, err := deps.listRemotes() - remotesCh <- remotesResult{entries: entries, err: err} - }() - - localRes := <-localCh - remotes := <-remotesCh - local := localRes.section - if err := localRes.err; err != nil { - if local.Error == "" { - local.Error = err.Error() - } - } - - sections := make([]routeStatusSection, 0, 1) - sections = append(sections, normalizeRouteStatusLocalSection(local)) - - if remotes.err != nil { - sections = append(sections, routeStatusSection{Kind: "remote", Name: "remotes", Error: remotes.err.Error()}) - return sections, nil - } - - names := sortedRemoteNames(remotes.entries) - remoteSections := make([]routeStatusSection, len(names)) - var wg sync.WaitGroup - for i, name := range names { - entry := remotes.entries[name] - wg.Add(1) - go func(idx int, remoteName string, remoteEntry remote.RemoteEntry) { - defer wg.Done() - section, err := deps.loadRemote(ctx, remoteName, remoteEntry) - if err != nil && section.Error == "" { - section.Error = err.Error() - } - remoteSections[idx] = normalizeRouteStatusRemoteSection(section, remoteName, remoteEntry) - }(i, name, entry) - } - wg.Wait() - - sections = append(sections, remoteSections...) - return sections, nil -} - -func normalizeRouteStatusLocalSection(section routeStatusSection) routeStatusSection { - if section.Kind == "" { - section.Kind = "local" - } - if section.Name == "" { - section.Name = "local" - } - return section -} - -func normalizeRouteStatusRemoteSection(section routeStatusSection, name string, entry remote.RemoteEntry) routeStatusSection { - if section.Kind == "" { - section.Kind = "remote" - } - if section.Name == "" { - section.Name = name - } - if section.URL == "" { - section.URL = entry.URL - } - return section -} - -func renderRoutesListSections(out io.Writer, sections []routeListSection) error { - sections = renderedRouteListSections(sections) - - if err := cliWriteLine(out, cliRenderTitle("Routes")); err != nil { - return err - } - if err := cliWriteLine(out, ""); err != nil { - return err - } - - for i, section := range sections { - if i > 0 { - if err := cliWriteLine(out, ""); err != nil { - return err - } - } - if err := renderRoutesListSection(out, section); err != nil { - return err - } - } - - return nil -} - -func renderRoutesListSection(out io.Writer, section routeListSection) error { - if err := cliWriteLine(out, routeSectionHeading(section.Kind, section.Name)); err != nil { - return err - } - - if len(section.Routes) > 0 { - const imageColWidth = 45 - rows := make([][]string, 0, len(section.Routes)) - for _, route := range section.Routes { - rows = append(rows, []string{route.Domain, truncateImage(route.Image, imageColWidth)}) - } - - table := components.NewTable( - components.WithColumns([]components.TableColumn{ - {Title: "Domain", Width: 30}, - {Title: "Image", Width: imageColWidth}, - }), - components.WithRows(rows), - ) - - if err := cliWriteLine(out, table.View()); err != nil { - return err - } - } - - if section.Error != "" { - return cliWriteLine(out, cliRenderWarning(section.Error)) - } - - if len(section.Routes) == 0 { - return cliWriteLine(out, cliRenderMuted("No routes configured")) - } - - return nil -} - -func renderRoutesStatusSections(out io.Writer, sections []routeStatusSection) error { - sections = renderedRouteStatusSections(sections) - - if err := cliWriteLine(out, cliRenderTitle("Route Status")); err != nil { - return err - } - if err := cliWriteLine(out, ""); err != nil { - return err - } - - for i, section := range sections { - if i > 0 { - if err := cliWriteLine(out, ""); err != nil { - return err - } - } - if err := renderRoutesStatusSection(out, section); err != nil { - return err - } - } - - return nil -} - -func renderedRouteListSections(sections []routeListSection) []routeListSection { - if !shouldHideLocalRouteListSection(sections) { - return sections - } - return sections[1:] -} - -func shouldHideLocalRouteListSection(sections []routeListSection) bool { - if len(sections) < 2 { - return false - } - local := sections[0] - if local.Kind != "local" || local.Error == "" { - return false - } - for _, section := range sections[1:] { - if section.Kind == "remote" && section.Error == "" { - return true - } - } - return false -} - -func renderedRouteStatusSections(sections []routeStatusSection) []routeStatusSection { - if !shouldHideLocalRouteStatusSection(sections) { - return sections - } - return sections[1:] -} - -func shouldHideLocalRouteStatusSection(sections []routeStatusSection) bool { - if len(sections) < 2 { - return false - } - local := sections[0] - if local.Kind != "local" || local.Error == "" { - return false - } - for _, section := range sections[1:] { - if section.Kind == "remote" && section.Error == "" { - return true - } - } - return false -} - -func renderRoutesStatusSection(out io.Writer, section routeStatusSection) error { - if err := cliWriteLine(out, routeSectionHeading(section.Kind, section.Name)); err != nil { - return err - } - - if len(section.Routes) > 0 { - tree := buildRouteStatusTree(section.Routes) - if err := cliWriteLine(out, strings.TrimSuffix(tree.Render(), "\n")); err != nil { - return err - } - } - - if section.Error != "" { - return cliWriteLine(out, cliRenderWarning(section.Error)) - } - - if len(section.Routes) == 0 { - return cliWriteLine(out, cliRenderMuted("No routes configured")) - } - - return nil -} - -func routeSectionHeading(kind, name string) string { - switch kind { - case "local": - return "Local" - case "remote": - if name != "" { - return "Remote: " + name - } - return "Remote" - default: - if name != "" { - return capitalizeKind(kind) + ": " + name - } - return capitalizeKind(kind) - } -} - -func capitalizeKind(kind string) string { - if kind == "" { - return "" - } - if len(kind) == 1 { - return strings.ToUpper(kind) - } - return strings.ToUpper(kind[:1]) + kind[1:] -} - -func unresolvedRoutesTargetError(target string) string { - return fmt.Sprintf("could not resolve remote target %q", target) -} - -func resolveRoutesExplicitRemote() (*remote.ResolvedRemote, bool, error) { - target, ok := resolveRoutesExplicitTarget() - if !ok { - return nil, false, nil - } - - if strings.HasPrefix(target, "http://") || strings.HasPrefix(target, "https://") { - return &remote.ResolvedRemote{ - URL: target, - Token: resolveRoutesTokenForTarget("", remote.RemoteEntry{}), - InsecureTLS: resolveRoutesInsecureForTarget("", remote.RemoteEntry{}), - }, true, nil - } - - remotes, err := remote.LoadRemotes("") - if err != nil { - return nil, false, err - } - - if remotes != nil { - if entry, found := remotes.Remotes[target]; found { - return &remote.ResolvedRemote{ - Name: target, - URL: entry.URL, - Token: resolveRoutesTokenForTarget(target, entry), - InsecureTLS: resolveRoutesInsecureForTarget(target, entry), - }, true, nil - } - } - - return nil, false, nil -} - -func resolveRoutesExplicitTarget() (string, bool) { - if target := strings.TrimSpace(remoteFlag); target != "" { - return target, true - } - if target := strings.TrimSpace(os.Getenv("GORDON_REMOTE")); target != "" { - return target, true - } - return "", false -} - -func resolveRoutesTokenForTarget(name string, entry remote.RemoteEntry) string { - if token := strings.TrimSpace(tokenFlag); token != "" { - return token - } - if token := strings.TrimSpace(os.Getenv("GORDON_TOKEN")); token != "" { - return token - } - if name != "" { - return remote.ResolveTokenForRemote(name, entry) - } - return "" -} - -func resolveRoutesInsecureForTarget(name string, entry remote.RemoteEntry) bool { - if insecureTLSFlag { - return true - } - if env := strings.TrimSpace(os.Getenv("GORDON_INSECURE")); env != "" { - if value, err := strconv.ParseBool(env); err == nil { - return value - } - } - if name != "" { - return entry.InsecureTLS - } - return false -} - -func newRoutesTargetClient(name string, entry remote.RemoteEntry) *remote.Client { - return remote.NewClient(entry.URL, remoteClientOptions(resolveRoutesTokenForTarget(name, entry), resolveRoutesInsecureForTarget(name, entry))...) -} - -func loadRoutesListLocalSection(ctx context.Context, cfgPath string) (routeListSection, error) { - local, err := GetLocalServices(cfgPath) - if err != nil { - return routeListSection{Kind: "local", Name: "local", Error: err.Error()}, nil - } - - routes := local.GetConfigService().GetRoutes(ctx) - section := routeListSection{Kind: "local", Name: "local", Routes: make([]routeListItem, 0, len(routes))} - for _, route := range routes { - section.Routes = append(section.Routes, routeListItem{Domain: route.Domain, Image: route.Image}) - } - - return section, nil -} - -func loadRoutesListRemoteSection(ctx context.Context, name string, entry remote.RemoteEntry) (routeListSection, error) { - section := routeListSection{Kind: "remote", Name: name, URL: entry.URL} - client := newRoutesTargetClient(name, entry) - routes, err := client.ListRoutes(ctx) - if err != nil { - section.Error = err.Error() - return section, nil - } - - section.Routes = routeListItemsFromRoutes(routes) - return section, nil -} - -func loadRoutesStatusLocalSection(ctx context.Context, cfgPath string) (routeStatusSection, error) { - section := routeStatusSection{Kind: "local", Name: "local"} - handle, err := resolveLocalControlPlane(cfgPath) - if err != nil { - section.Error = err.Error() - return section, nil - } - defer handle.close() - - routes, err := handle.plane.ListRoutesWithDetails(ctx) - if err != nil { - section.Error = err.Error() - return section, nil - } - - health, err := handle.plane.GetHealth(ctx) - if err != nil { - section.Error = err.Error() - health = nil - } - - section.Routes = routeStatusItemsFromInfos(routes, health) - return section, nil -} - -func loadRoutesStatusRemoteSection(ctx context.Context, name string, entry remote.RemoteEntry) (routeStatusSection, error) { - section := routeStatusSection{Kind: "remote", Name: name, URL: entry.URL} - client := newRoutesTargetClient(name, entry) - cp := NewRemoteControlPlane(client) - - routes, err := cp.ListRoutesWithDetails(ctx) - if err != nil { - section.Error = err.Error() - return section, nil - } - - health, err := cp.GetHealth(ctx) - if err != nil { - section.Error = err.Error() - health = nil - } - - section.Routes = routeStatusItemsFromInfos(routes, health) - return section, nil -} - -func routeStatusItemsFromInfos(routes []remote.RouteInfo, health map[string]*remote.RouteHealth) []routeStatusItem { - items := make([]routeStatusItem, 0, len(routes)) - for _, route := range routes { - items = append(items, routeStatusItemFromInfo(route, health[route.Domain])) - } - return items -} - -func routeStatusItemFromInfo(route remote.RouteInfo, routeHealth *remote.RouteHealth) routeStatusItem { - item := routeStatusItem{ - Domain: route.Domain, - Image: route.Image, - ContainerID: route.ContainerID, - ContainerStatus: route.ContainerStatus, - Network: route.Network, - } - if item.ContainerStatus == "" && routeHealth != nil { - item.ContainerStatus = routeHealth.ContainerStatus - item.HTTPStatus = routeHealth.HTTPStatus - item.HealthError = routeHealth.Error - } - if item.ContainerStatus == "" { - item.ContainerStatus = "unknown" - } - if item.HTTPStatus == 0 && routeHealth != nil { - item.HTTPStatus = routeHealth.HTTPStatus - item.HealthError = routeHealth.Error - } - if item.HealthError == "" && routeHealth != nil { - item.HealthError = routeHealth.Error - } - item.Attachments = make([]routeStatusAttachment, 0, len(route.Attachments)) - for _, attachment := range route.Attachments { - status := attachment.Status - if status == "" { - status = "unknown" - } - item.Attachments = append(item.Attachments, routeStatusAttachment{Name: attachment.Name, Image: attachment.Image, Status: status}) - } - return item -} - -func routeListItemsFromRoutes(routes []domain.Route) []routeListItem { - items := make([]routeListItem, 0, len(routes)) - for _, route := range routes { - items = append(items, routeListItem{Domain: route.Domain, Image: route.Image}) - } - return items -} - -func buildRouteStatusTree(routes []routeStatusItem) *components.Tree { - sortedRoutes := make([]routeStatusItem, len(routes)) - copy(sortedRoutes, routes) - sort.SliceStable(sortedRoutes, func(i, j int) bool { - if sortedRoutes[i].Network != sortedRoutes[j].Network { - return sortedRoutes[i].Network < sortedRoutes[j].Network - } - if sortedRoutes[i].Domain != sortedRoutes[j].Domain { - return sortedRoutes[i].Domain < sortedRoutes[j].Domain - } - if sortedRoutes[i].Image != sortedRoutes[j].Image { - return sortedRoutes[i].Image < sortedRoutes[j].Image - } - if sortedRoutes[i].ContainerID != sortedRoutes[j].ContainerID { - return sortedRoutes[i].ContainerID < sortedRoutes[j].ContainerID - } - if sortedRoutes[i].ContainerStatus != sortedRoutes[j].ContainerStatus { - return sortedRoutes[i].ContainerStatus < sortedRoutes[j].ContainerStatus - } - return sortedRoutes[i].HTTPStatus < sortedRoutes[j].HTTPStatus - }) - - infos := make([]remote.RouteInfo, 0, len(routes)) - itemsByDomain := make(map[string]routeStatusItem, len(routes)) - for _, route := range sortedRoutes { - infos = append(infos, remote.RouteInfo{ - Domain: route.Domain, - Image: route.Image, - ContainerID: route.ContainerID, - ContainerStatus: route.ContainerStatus, - Network: route.Network, - Attachments: routeStatusAttachmentsToRemote(route.Attachments), - }) - itemsByDomain[route.Domain] = route - } - - groups, solo := groupRoutesByNetwork(infos) - tree := components.NewTree() - - for _, group := range groups { - g := tree.AddGroup(group.name) - for _, route := range group.routes { - item := itemsByDomain[route.Domain] - node := g.AddNode(routeStatusTitle(item), item.Image) - addRouteStatusAttachmentChildren(node, item) - } - } - - for _, route := range solo { - item := itemsByDomain[route.Domain] - node := tree.AddNode(routeStatusTitle(item), item.Image) - addRouteStatusAttachmentChildren(node, item) - } - - return tree -} - -func routeStatusTitle(route routeStatusItem) string { - containerStatus := route.ContainerStatus - if containerStatus == "" { - containerStatus = "unknown" - } - - httpIcon := components.StatusIcon(styles.IconHTTPStatus, httpHealthToStatus(&remote.RouteHealth{HTTPStatus: route.HTTPStatus, Error: route.HealthError})) - containerIcon := components.StatusIcon(styles.IconContainerStatus, components.ParseStatus(containerStatus)) - - return httpIcon + " " + containerIcon + " " + route.Domain -} - -func addRouteStatusAttachmentChildren(node *components.Node, route routeStatusItem) { - for _, att := range route.Attachments { - status := att.Status - if status == "" { - status = "unknown" - } - attIcon := components.StatusIcon(styles.IconContainerStatus, components.ParseStatus(status)) - node.AddChild(attIcon+" "+att.Name, att.Image) - } -} - -func routeStatusAttachmentsToRemote(attachments []routeStatusAttachment) []remote.Attachment { - result := make([]remote.Attachment, 0, len(attachments)) - for _, att := range attachments { - result = append(result, remote.Attachment{Name: att.Name, Image: att.Image, Status: att.Status}) - } - return result -} - -func newRoutesShowCmd() *cobra.Command { - var jsonOut bool - - cmd := &cobra.Command{ - Use: "show ", - Short: "Show details for a single route", - Long: `Display detailed information about a specific route including its image, -container status, and health. - -Examples: - gordon routes show app.mydomain.com - gordon routes show app.mydomain.com --json - gordon routes show app.mydomain.com --remote https://gordon.mydomain.com`, - Args: cobra.ExactArgs(1), - RunE: func(cmd *cobra.Command, args []string) error { - ctx := cmd.Context() - handle, err := resolveControlPlaneForRouteDomain(ctx, args[0]) - if err != nil { - return err - } - defer handle.close() - return runRoutesShow(ctx, handle.plane, cmd.OutOrStdout(), args[0], jsonOut) - }, - } - - cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") - - return cmd -} - -func runRoutesShow(ctx context.Context, cp ControlPlane, out io.Writer, routeDomain string, jsonOut bool) error { - route, err := cp.GetRoute(ctx, routeDomain) - if err != nil { - if errors.Is(err, domain.ErrRouteNotFound) { - return fmt.Errorf("route %q not found", routeDomain) - } - return fmt.Errorf("failed to get route: %w", err) - } - - health, err := cp.GetHealth(ctx) - healthErr := "" - if err != nil { - healthErr = err.Error() - health = nil - } - var routeHealth *remote.RouteHealth - if health != nil { - routeHealth = health[routeDomain] - } - - containerStatus := "unknown" - httpStatus := 0 - if routeHealth != nil { - if routeHealth.ContainerStatus != "" { - containerStatus = routeHealth.ContainerStatus - } - httpStatus = routeHealth.HTTPStatus - if healthErr == "" && routeHealth.Error != "" { - healthErr = routeHealth.Error - } - } - - if jsonOut { - payload := map[string]any{ - "domain": route.Domain, - "image": route.Image, - "container_status": containerStatus, - "http_status": httpStatus, - } - if healthErr != "" { - payload["health_error"] = healthErr - } - return writeJSON(out, payload) - } - - return writeRouteShowText(out, route.Domain, route.Image, containerStatus, httpStatus, healthErr) -} - -func writeRouteShowText(out io.Writer, domainName, image, containerStatus string, httpStatus int, healthErr string) error { - if err := cliWriteLine(out, cliRenderTitle("Route: "+domainName)); err != nil { - return err - } - if err := cliWriteLine(out, ""); err != nil { - return err - } - if err := cliWriteLine(out, cliRenderMeta("Domain:", domainName)); err != nil { - return err - } - if err := cliWriteLine(out, cliRenderMeta("Image:", image)); err != nil { - return err - } - if err := cliWriteLine(out, cliRenderMeta("Container:", containerStatus)); err != nil { - return err - } - if httpStatus > 0 { - if err := cliWriteLine(out, cliRenderMeta("HTTP Status:", fmt.Sprintf("%d", httpStatus))); err != nil { - return err - } - } - if healthErr != "" { - if err := cliWriteLine(out, cliRenderWarning(healthErr)); err != nil { - return err - } - } - - return nil -} - -// newRoutesDiagnoseCmd creates the routes diagnose command. -func newRoutesDiagnoseCmd() *cobra.Command { - var jsonOut bool - cmd := &cobra.Command{ - Use: "diagnose ", - Short: "Diagnose route configuration, runtime state, and cleanup leftovers", - Args: cobra.ExactArgs(1), - RunE: func(cmd *cobra.Command, args []string) error { - return runRouteDiagnosis(cmd.Context(), args[0], cmd.OutOrStdout(), jsonOut) - }, - } - cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") - return cmd -} - -func runRouteDiagnosis(ctx context.Context, domainName string, out io.Writer, jsonOut bool) error { - handle, err := resolveControlPlaneForRouteCleanupDomain(ctx, domainName) - if err != nil { - return err - } - defer handle.close() - diag, err := buildRouteDiagnosis(ctx, handle.plane, domainName) - if err != nil { - return err - } - if jsonOut { - return writeJSON(out, diag) - } - return writeRouteDiagnosisText(out, diag) -} - -func buildRouteDiagnosis(ctx context.Context, cp ControlPlane, domainName string) (*routeDiagnosis, error) { - diag := &routeDiagnosis{Domain: domainName} - loadRouteDiagnosisConfig(ctx, cp, domainName, diag) - loadRouteDiagnosisCleanupPreview(ctx, cp, domainName, diag) - loadRouteDiagnosisRuntime(ctx, cp, domainName, diag) - loadRouteDiagnosisHealth(ctx, cp, domainName, diag) - loadRouteDiagnosisOrphanedAttachments(ctx, cp, domainName, diag) - loadRouteDiagnosisVolumes(ctx, cp, domainName, diag) - finalizeRouteDiagnosis(diag) - return diag, nil -} - -func loadRouteDiagnosisConfig(ctx context.Context, cp ControlPlane, domainName string, diag *routeDiagnosis) { - route, err := cp.GetRoute(ctx, domainName) - if err == nil && route != nil { - diag.Configured = true - diag.Route = route - return - } - if err != nil && !errors.Is(err, domain.ErrRouteNotFound) { - diag.Warnings = append(diag.Warnings, "failed to load route config: "+err.Error()) - } -} - -func loadRouteDiagnosisCleanupPreview(ctx context.Context, cp ControlPlane, domainName string, diag *routeDiagnosis) { - if diag.Configured { - return - } - previewer, ok := cp.(interface { - GetRouteCleanupPreview(context.Context, string) (*domain.CleanupReport, error) - }) - if !ok { - return - } - - report, err := previewer.GetRouteCleanupPreview(ctx, domainName) - if err != nil { - if isRemoteNotFoundError(err) { - return - } - diag.Warnings = append(diag.Warnings, "failed to load route cleanup preview: "+err.Error()) - return - } - if report == nil { - return - } - if len(diag.OrphanedAttachments) == 0 { - diag.OrphanedAttachments = append(diag.OrphanedAttachments, report.PreservedAttachments...) - } - if len(diag.Volumes) == 0 { - diag.Volumes = cleanupVolumesForDiagnosis(report.PreservedVolumes) - } - diag.OrphanedEntities = append(diag.OrphanedEntities, report.OrphanedEntities...) - diag.Warnings = append(diag.Warnings, report.Warnings...) - diag.Hints = append(diag.Hints, report.Hints...) -} - -func loadRouteDiagnosisRuntime(ctx context.Context, cp ControlPlane, domainName string, diag *routeDiagnosis) { - routes, err := cp.ListRoutesWithDetails(ctx) - if err != nil { - diag.Warnings = append(diag.Warnings, "failed to load route runtime details: "+err.Error()) - return - } - for _, info := range routes { - if strings.EqualFold(info.Domain, domainName) { - copy := info - diag.Runtime = © - return - } - } -} - -func loadRouteDiagnosisHealth(ctx context.Context, cp ControlPlane, domainName string, diag *routeDiagnosis) { - if health, err := cp.GetHealth(ctx); err == nil && health != nil { - diag.Health = health[domainName] - } -} - -func loadRouteDiagnosisVolumes(ctx context.Context, cp ControlPlane, domainName string, diag *routeDiagnosis) { - if len(diag.Volumes) > 0 { - return - } - if volumes, err := cp.ListVolumes(ctx); err == nil { - if len(volumes) == 0 { - return - } - scope := newRouteResourceScope(domainName, routeDiagnosisVolumePrefix(ctx, cp), diag.Runtime, diag.OrphanedAttachments) - diag.Volumes = volumesForRouteDiagnosis(volumes, scope) - } -} - -func routeDiagnosisVolumePrefix(ctx context.Context, cp ControlPlane) string { - const defaultPrefix = "gordon" - cfg, err := cp.GetConfig(ctx) - if err != nil || cfg == nil || cfg.Volumes.Prefix == "" { - return defaultPrefix - } - return cfg.Volumes.Prefix -} - -func loadRouteDiagnosisOrphanedAttachments(ctx context.Context, cp ControlPlane, domainName string, diag *routeDiagnosis) { - if len(diag.OrphanedAttachments) > 0 { - return - } - lister, ok := cp.(interface { - ListOrphanedAttachments(context.Context) ([]domain.CleanupAttachment, error) - }) - if !ok { - return - } - if attachments, err := lister.ListOrphanedAttachments(ctx); err == nil { - diag.OrphanedAttachments = attachmentsForRouteDiagnosis(attachments, domainName) - } -} - -func cleanupVolumesForDiagnosis(volumes []domain.CleanupVolume) []dto.Volume { - result := make([]dto.Volume, 0, len(volumes)) - for _, volume := range volumes { - result = append(result, dto.Volume{Name: volume.Name}) - } - return result -} - -func finalizeRouteDiagnosis(diag *routeDiagnosis) { - if !diag.Configured && (diag.Runtime != nil || hasRouteCleanupEntity(diag.OrphanedEntities, "route_container")) { - diag.Warnings = append(diag.Warnings, "route is not configured but runtime state still exists") - } - if len(diag.Volumes) > 0 { - diag.Hints = append(diag.Hints, "persistent volumes are preserved by default") - } - if len(diag.OrphanedAttachments) > 0 { - diag.Hints = append(diag.Hints, fmt.Sprintf("run 'gordon routes purge %s --attachments' to review orphaned attachment cleanup for this route", diag.Domain)) - } -} - -func hasRouteCleanupEntity(entities []domain.CleanupOrphanedEntity, kind string) bool { - for _, entity := range entities { - if entity.Kind == kind { - return true - } - } - return false -} - -type routeResourceScope struct { - containerNames map[string]struct{} - volumePrefixes []string -} - -func newRouteResourceScope(domainName, volumePrefix string, runtime *remote.RouteInfo, orphaned []domain.CleanupAttachment) routeResourceScope { - if volumePrefix == "" { - volumePrefix = "gordon" - } - scope := routeResourceScope{ - containerNames: make(map[string]struct{}), - volumePrefixes: []string{volumePrefix + "-" + strings.ReplaceAll(domainName, ".", "-") + "-"}, - } - for _, name := range []string{ - fmt.Sprintf("gordon-%s", domainName), - fmt.Sprintf("gordon-%s-new", domainName), - fmt.Sprintf("gordon-%s-next", domainName), - } { - scope.containerNames[name] = struct{}{} - } - if runtime != nil { - for _, attachment := range runtime.Attachments { - for _, name := range routeAttachmentContainerNameCandidates(domainName, attachment.Name) { - scope.containerNames[name] = struct{}{} - scope.volumePrefixes = append(scope.volumePrefixes, volumePrefix+"-"+name+"-") - } - } - } - for _, attachment := range orphaned { - if attachment.Name == "" { - continue - } - scope.containerNames[attachment.Name] = struct{}{} - scope.volumePrefixes = append(scope.volumePrefixes, volumePrefix+"-"+attachment.Name+"-") - } - return scope -} - -func routeAttachmentContainerNameCandidates(domainName, serviceName string) []string { - if serviceName == "" { - return nil - } - return []string{ - fmt.Sprintf("gordon-%s-%s", domain.SanitizeDomainForContainer(domainName), serviceName), - fmt.Sprintf("gordon-%s-%s", domain.SanitizeDomainForContainerLegacy(domainName), serviceName), - } -} - -func volumesForRouteDiagnosis(volumes []dto.Volume, scope routeResourceScope) []dto.Volume { - var out []dto.Volume - for _, volume := range volumes { - if scope.matchesVolume(volume) { - out = append(out, volume) - } - } - return out -} - -func (s routeResourceScope) matchesVolume(volume dto.Volume) bool { - for _, containerName := range volume.Containers { - if _, ok := s.containerNames[containerName]; ok { - return true - } - } - for _, prefix := range s.volumePrefixes { - if strings.HasPrefix(volume.Name, prefix) { - return true - } - } - return false -} - -func attachmentsForRouteDiagnosis(attachments []domain.CleanupAttachment, domainName string) []domain.CleanupAttachment { - var out []domain.CleanupAttachment - for _, attachment := range attachments { - if strings.EqualFold(attachment.Owner, domainName) { - out = append(out, attachment) - } - } - return out -} - -func writeRouteDiagnosisText(out io.Writer, diag *routeDiagnosis) error { - if err := cliWriteLine(out, cliRenderTitle("Route diagnosis: "+diag.Domain)); err != nil { - return err - } - if err := writeRouteDiagnosisSummary(out, diag); err != nil { - return err - } - if err := writeRouteDiagnosisLeftovers(out, diag); err != nil { - return err - } - return writeCleanupMessages(out, diag.Warnings, diag.Hints) -} - -func writeRouteDiagnosisSummary(out io.Writer, diag *routeDiagnosis) error { - if diag.Configured && diag.Route != nil { - if err := cliWriteLine(out, cliRenderMeta("Image:", diag.Route.Image)); err != nil { - return err - } - } else if err := cliWriteLine(out, cliRenderWarning("Route is not configured")); err != nil { - return err - } - if diag.Runtime == nil { - return nil - } - if err := cliWriteLine(out, cliRenderMeta("Container:", diag.Runtime.ContainerStatus+" "+shortContainerID(diag.Runtime.ContainerID))); err != nil { - return err - } - if len(diag.Runtime.Attachments) == 0 { - return nil - } - if err := cliWriteLine(out, cliRenderMeta("Active attachments:", "")); err != nil { - return err - } - for _, attachment := range diag.Runtime.Attachments { - label := attachment.Name - if label == "" { - label = attachment.ContainerID - } - if attachment.Status != "" { - label = fmt.Sprintf("%s (%s)", label, attachment.Status) - } - if err := cliWriteLine(out, " "+label); err != nil { - return err - } - } - return nil -} - -func writeRouteDiagnosisLeftovers(out io.Writer, diag *routeDiagnosis) error { - for _, entity := range diag.OrphanedEntities { - label := entity.Name - if label == "" { - label = entity.ID - } - if label == "" { - label = entity.Kind - } - if entity.Status != "" { - label = fmt.Sprintf("%s (%s)", label, entity.Status) - } - if err := cliWriteLine(out, cliRenderWarning("Orphaned "+strings.ReplaceAll(entity.Kind, "_", " ")+": "+label)); err != nil { - return err - } - } - volumeLabel := "Volume:" - if !diag.Configured { - volumeLabel = "Preserved volume:" - } - for _, volume := range diag.Volumes { - if err := cliWriteLine(out, cliRenderMeta(volumeLabel, volume.Name)); err != nil { - return err - } - } - for _, attachment := range diag.OrphanedAttachments { - if err := cliWriteLine(out, cliRenderWarning("Orphaned attachment: "+attachment.Name)); err != nil { - return err - } - } - return nil -} - -func newRoutesPurgeCmd() *cobra.Command { - var jsonOut bool - var force bool - var includeAttachments bool - var includeVolumes bool - cmd := &cobra.Command{ - Use: "purge ", - Short: "Review or explicitly purge retained route resources", - Args: cobra.ExactArgs(1), - RunE: func(cmd *cobra.Command, args []string) error { - return runRoutePurgeCommand(cmd.Context(), args[0], cmd.OutOrStdout(), routePurgeOptions{Force: force, Attachments: includeAttachments, Volumes: includeVolumes}, jsonOut) - }, - } - cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") - cmd.Flags().BoolVar(&force, "force", false, "Execute purge actions; default is dry-run") - cmd.Flags().BoolVar(&includeAttachments, "attachments", false, "Include orphaned attachment containers") - cmd.Flags().BoolVar(&includeVolumes, "volumes", false, "Include preserved volumes in the purge plan") - return cmd -} - -type routePurgeOptions struct { - Force bool - Attachments bool - Volumes bool -} - -func runRoutePurgeCommand(ctx context.Context, domainName string, out io.Writer, opts routePurgeOptions, jsonOut bool) error { - handle, err := resolveControlPlaneForRouteCleanupDomain(ctx, domainName) - if err != nil { - return err - } - defer handle.close() - report, err := runRoutePurge(ctx, handle.plane, domainName, opts) - if err != nil { - return err - } - if jsonOut { - return writeJSON(out, report) - } - return writeRoutePurgeText(out, report, opts.Force) -} - -func runRoutePurge(ctx context.Context, cp ControlPlane, domainName string, opts routePurgeOptions) (*domain.CleanupReport, error) { - diag, err := buildRouteDiagnosis(ctx, cp, domainName) - if err != nil { - return nil, err - } - report := &domain.CleanupReport{Domain: domainName} - if diag.Runtime != nil && diag.Runtime.ContainerID != "" { - report.OrphanedEntities = append(report.OrphanedEntities, domain.CleanupOrphanedEntity{Kind: "route_container", ID: diag.Runtime.ContainerID, Status: diag.Runtime.ContainerStatus, Reason: "route runtime state matched purge target"}) - } - report.OrphanedEntities = appendUniqueCleanupEntities(report.OrphanedEntities, diag.OrphanedEntities...) - if opts.Volumes { - for _, volume := range diag.Volumes { - report.PreservedVolumes = append(report.PreservedVolumes, domain.CleanupVolume{Name: volume.Name, Reason: "volume purge requires explicit --volumes and runtime support"}) - } - } - if !opts.Attachments || !opts.Force { - report.PreservedAttachments = append(report.PreservedAttachments, diag.OrphanedAttachments...) - } - if opts.Attachments && opts.Force { - cleaner, ok := cp.(interface { - CleanupOrphanedAttachments(context.Context, string, bool) (*domain.CleanupReport, error) - }) - if !ok { - return nil, fmt.Errorf("attachment purge is unavailable") - } - attachmentReport, err := cleaner.CleanupOrphanedAttachments(ctx, domainName, true) - if err != nil { - return nil, err - } - report.RemovedContainers = append(report.RemovedContainers, attachmentReport.RemovedContainers...) - report.PreservedVolumes = append(report.PreservedVolumes, attachmentReport.PreservedVolumes...) - report.PreservedAttachments = append(report.PreservedAttachments, attachmentReport.PreservedAttachments...) - report.OrphanedEntities = append(report.OrphanedEntities, attachmentReport.OrphanedEntities...) - report.Warnings = append(report.Warnings, attachmentReport.Warnings...) - report.Hints = append(report.Hints, attachmentReport.Hints...) - report.PartialFailures = append(report.PartialFailures, attachmentReport.PartialFailures...) - } - if opts.Volumes { - report.Warnings = append(report.Warnings, "volume purge is not executed yet; preserved volumes are reported for manual review") - } - if !opts.Force { - report.Hints = append(report.Hints, "dry-run only; add --force with explicit category flags to execute supported purge actions") - } - return report, nil -} - -func appendUniqueCleanupEntities(existing []domain.CleanupOrphanedEntity, candidates ...domain.CleanupOrphanedEntity) []domain.CleanupOrphanedEntity { - seen := make(map[string]struct{}, len(existing)) - for _, entity := range existing { - seen[cleanupEntityKey(entity)] = struct{}{} - } - for _, entity := range candidates { - key := cleanupEntityKey(entity) - if _, ok := seen[key]; ok { - continue - } - existing = append(existing, entity) - seen[key] = struct{}{} - } - return existing -} - -func cleanupEntityKey(entity domain.CleanupOrphanedEntity) string { - return entity.Kind + "|" + entity.ID + "|" + entity.Name -} - -func writeRoutePurgeText(out io.Writer, report *domain.CleanupReport, force bool) error { - mode := "Purge dry-run" - if force { - mode = "Purge completed" - } - if err := cliWriteLine(out, cliRenderTitle(mode+": "+report.Domain)); err != nil { - return err - } - if err := writePreservedAttachments(out, report.PreservedAttachments); err != nil { - return err - } - if err := writeRemovedContainers(out, report.RemovedContainers); err != nil { - return err - } - for _, entity := range report.OrphanedEntities { - if err := cliWriteLine(out, cliRenderWarning("Purge candidate: "+entity.Kind+" "+entity.ID)); err != nil { - return err - } - } - for _, volume := range report.PreservedVolumes { - if err := cliWriteLine(out, cliRenderMeta("Preserved volume:", volume.Name)); err != nil { - return err - } - } - return writeCleanupMessages(out, report.Warnings, report.Hints) -} - -func newRoutesAddCmd() *cobra.Command { - var image string - - cmd := &cobra.Command{ - Use: "add ", - Short: "Create or update a route", - Long: `Create or update a route mapping a domain to a container image. - -If the route already exists with the same image, this is a no-op. -If it exists with a different image, the image is updated. - -Examples: - gordon routes add app.mydomain.com myapp:latest - gordon --remote https://gordon.mydomain.com routes add api.mydomain.com api:v2`, - Args: cobra.RangeArgs(1, 2), - RunE: func(cmd *cobra.Command, args []string) error { - ctx := cmd.Context() - routeDomain := args[0] - - // Resolve image from positional argument or flag - hasPositionalImage := len(args) > 1 - hasFlagImage := image != "" - - // Check for conflicts: both positional and flag provided - if hasPositionalImage && hasFlagImage { - return fmt.Errorf("error: cannot use both positional image argument and --image flag") - } - - // Check that image is provided via either method - if !hasPositionalImage && !hasFlagImage { - return fmt.Errorf("error: image is required (use positional argument or --image flag)") - } - - if hasPositionalImage { - image = args[1] - } - - route := domain.Route{ - Domain: routeDomain, - Image: image, - } - - client, isRemote, err := GetRemoteClient() - if err != nil { - return err - } - if isRemote { - if err := client.AddRoute(ctx, route); err != nil { - return fmt.Errorf("failed to add route: %w", err) - } - } else { - local, err := GetLocalServices(configPath) - if err != nil { - return fmt.Errorf("failed to initialize local services: %w", err) - } - if err := local.GetConfigService().AddRoute(ctx, route); err != nil { - return fmt.Errorf("failed to add route: %w", err) - } - } - - return cliWriteLine(cmd.OutOrStdout(), styles.RenderSuccess(fmt.Sprintf("Route configured: %s -> %s", routeDomain, image))) - }, - } - - cmd.Flags().StringVarP(&image, "image", "i", "", "Container image") - - return cmd -} - -// newRoutesRemoveCmd creates the routes remove command. -func newRoutesRemoveCmd() *cobra.Command { - var force bool - - cmd := &cobra.Command{ - Use: "remove ", - Short: "Remove a route", - Long: `Remove a route by its domain name. - -Examples: - gordon routes remove app.mydomain.com - gordon routes remove app.mydomain.com --force`, - Args: cobra.ExactArgs(1), - RunE: func(cmd *cobra.Command, args []string) error { - ctx := cmd.Context() - routeDomain := args[0] - - // Confirm unless --force - if !force { - confirmed, err := components.RunConfirm( - fmt.Sprintf("Remove route '%s'?", routeDomain), - components.WithDescription("This will stop and remove the associated container."), - ) - if err != nil { - return err - } - if !confirmed { - fmt.Println(styles.Theme.Muted.Render("Cancelled")) - return nil - } - } - - handle, err := resolveControlPlaneForRouteCleanupDomain(ctx, routeDomain) - if err != nil { - return err - } - defer handle.close() - var cleanup *dto.CleanupReport - if remover, ok := handle.plane.(interface { - RemoveRouteWithCleanup(context.Context, string) (*dto.RouteDeleteResponse, error) - }); ok { - resp, err := remover.RemoveRouteWithCleanup(ctx, routeDomain) - if err != nil { - return fmt.Errorf("failed to remove route: %w", err) - } - if resp != nil { - cleanup = resp.Cleanup - } - } else if err := handle.plane.RemoveRoute(ctx, routeDomain); err != nil { - return fmt.Errorf("failed to remove route: %w", err) - } - - return writeRouteRemoveText(cmd.OutOrStdout(), routeDomain, cleanup) - }, - } - - cmd.Flags().BoolVarP(&force, "force", "f", false, "Skip confirmation") - - return cmd -} - -func writeRouteRemoveText(out io.Writer, routeDomain string, cleanup *dto.CleanupReport) error { - if err := cliWriteLine(out, styles.RenderSuccess(fmt.Sprintf("Route removed: %s", routeDomain))); err != nil { - return err - } - if cleanup == nil { - return nil - } - cleanupReport := cleanupReportDTOToDomain(cleanup) - if err := writeRemovedContainers(out, cleanupReport.RemovedContainers); err != nil { - return err - } - if err := writePreservedAttachments(out, cleanupReport.PreservedAttachments); err != nil { - return err - } - return writeCleanupMessages(out, cleanupReport.Warnings, cleanupReport.Hints) -} - -func cleanupReportDTOToDomain(cleanup *dto.CleanupReport) *domain.CleanupReport { - if cleanup == nil { - return nil - } - out := &domain.CleanupReport{ - Domain: cleanup.Domain, - Warnings: append([]string(nil), cleanup.Warnings...), - Hints: append([]string(nil), cleanup.Hints...), - } - for _, item := range cleanup.RemovedContainers { - out.RemovedContainers = append(out.RemovedContainers, domain.CleanupContainer(item)) - } - for _, item := range cleanup.PreservedAttachments { - out.PreservedAttachments = append(out.PreservedAttachments, domain.CleanupAttachment(item)) - } - return out -} - -func writeRemovedContainers(out io.Writer, containers []domain.CleanupContainer) error { - if len(containers) == 0 { - return nil - } - if err := cliWriteLine(out, ""); err != nil { - return err - } - if err := cliWriteLine(out, styles.Theme.Bold.Render("Removed containers:")); err != nil { - return err - } - for _, c := range containers { - label := c.Name - if label == "" { - label = c.ID - } - if id := shortContainerID(c.ID); id != "" && id != label { - label = fmt.Sprintf("%s (%s)", label, id) - } - if err := cliWriteLine(out, " "+label); err != nil { - return err - } - } - return nil -} - -func writePreservedAttachments(out io.Writer, attachments []domain.CleanupAttachment) error { - if len(attachments) == 0 { - return nil - } - if err := cliWriteLine(out, ""); err != nil { - return err - } - if err := cliWriteLine(out, styles.Theme.Bold.Render("Preserved attachments:")); err != nil { - return err - } - for _, a := range attachments { - label := a.Name - if label == "" { - label = a.ContainerID - } - if a.Status != "" { - label = fmt.Sprintf("%s (%s)", label, a.Status) - } - if err := cliWriteLine(out, " "+label); err != nil { - return err - } - } - return nil -} - -func writeCleanupMessages(out io.Writer, warnings, hints []string) error { - for _, warning := range warnings { - if err := cliWriteLine(out, cliRenderWarning(warning)); err != nil { - return err - } - } - for _, hint := range hints { - if err := cliWriteLine(out, cliRenderMuted("Hint: "+hint)); err != nil { - return err - } - } - return nil -} - -// newStatusCmd creates the status command. -func newStatusCmd() *cobra.Command { - return &cobra.Command{ - Use: "status", - Short: "Show Gordon server status", - RunE: func(cmd *cobra.Command, args []string) error { - ctx := context.Background() - - handle, err := resolveControlPlane(configPath) - if err != nil { - return err - } - defer handle.close() - - status, err := handle.plane.GetStatus(ctx) - if err != nil { - return fmt.Errorf("failed to get status: %w", err) - } - - // Display status - fmt.Println(styles.Theme.Title.Render("Gordon Status")) - fmt.Println() - - fmt.Printf("%s %s\n", styles.Theme.Bold.Render("Domain:"), status.RegistryDomain) - fmt.Printf("%s %d\n", styles.Theme.Bold.Render("Registry Port:"), status.RegistryPort) - fmt.Printf("%s %d\n", styles.Theme.Bold.Render("Server Port:"), status.ServerPort) - fmt.Printf("%s %d\n", styles.Theme.Bold.Render("Routes:"), status.Routes) - fmt.Printf("%s %v\n", styles.Theme.Bold.Render("Auto-Route:"), status.AutoRoute) - fmt.Printf("%s %v\n", styles.Theme.Bold.Render("Network Isolation:"), status.NetworkIsolation) - - if len(status.ContainerStatus) > 0 { - fmt.Println() - fmt.Println(styles.Theme.Bold.Render("Container Status:")) - for domain, containerStatus := range status.ContainerStatus { - badge := components.ContainerStatusBadge(containerStatus) - fmt.Printf(" %s: %s\n", domain, badge) - } - } - - return nil - }, - } -} diff --git a/internal/adapters/in/cli/routes_test.go b/internal/adapters/in/cli/routes_test.go deleted file mode 100644 index df6836abb..000000000 --- a/internal/adapters/in/cli/routes_test.go +++ /dev/null @@ -1,1209 +0,0 @@ -package cli - -import ( - "bytes" - "context" - "encoding/json" - "errors" - "io" - "net/http" - "net/http/httptest" - "os" - "path/filepath" - "strconv" - "strings" - "sync/atomic" - "testing" - - "github.com/bnema/gordon/internal/adapters/dto" - climocks "github.com/bnema/gordon/internal/adapters/in/cli/mocks" - "github.com/bnema/gordon/internal/adapters/in/cli/remote" - "github.com/bnema/gordon/internal/adapters/in/cli/ui/components" - "github.com/bnema/gordon/internal/adapters/in/cli/ui/styles" - "github.com/bnema/gordon/internal/domain" - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" -) - -func TestTruncateImage(t *testing.T) { - tests := []struct { - name string - image string - maxLen int - expected string - }{ - // Basic cases - no truncation needed - { - name: "short image fits", - image: "nginx:latest", - maxLen: 20, - expected: "nginx:latest", - }, - { - name: "exact fit", - image: "nginx:latest", - maxLen: 12, - expected: "nginx:latest", - }, - - // Regular tag truncation - { - name: "truncate long tag with ellipsis", - image: "registry.test.com/test:v1234567890", - maxLen: 30, - expected: "registry.test.com/test:v123...", - }, - { - name: "truncate very long image", - image: "registry.example.com/organization/project/image:v1.2.3-beta.4", - maxLen: 35, - expected: "registry.example.com/organizatio...", - }, - - // Digest truncation - { - name: "digest shortened to 12 chars", - image: "myapp@sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4", - maxLen: 50, - expected: "myapp@sha256:a3ed95caeb02", - }, - { - name: "digest truncated with ellipsis when too long", - image: "registry.example.com/org/app@sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4", - maxLen: 35, - expected: "registry.example.com/org/app@sha...", - }, - { - name: "short digest fits", - image: "app@sha256:abc123", - maxLen: 30, - expected: "app@sha256:abc123", - }, - - // Edge cases - { - name: "maxLen zero returns empty", - image: "nginx:latest", - maxLen: 0, - expected: "", - }, - { - name: "maxLen negative returns empty", - image: "nginx:latest", - maxLen: -1, - expected: "", - }, - { - name: "maxLen 3 or less - no ellipsis", - image: "nginx:latest", - maxLen: 3, - expected: "ngi", - }, - { - name: "maxLen 4 - truncate with ellipsis", - image: "nginx:latest", - maxLen: 4, - expected: "n...", - }, - { - name: "empty image", - image: "", - maxLen: 10, - expected: "", - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - result := truncateImage(tt.image, tt.maxLen) - assert.Equal(t, tt.expected, result) - }) - } -} - -func TestHTTPHealthToStatus(t *testing.T) { - tests := []struct { - name string - health *remote.RouteHealth - expected components.Status - }{ - { - name: "nil health returns unknown", - health: nil, - expected: components.StatusUnknown, - }, - { - name: "zero status no error returns unknown", - health: &remote.RouteHealth{HTTPStatus: 0}, - expected: components.StatusUnknown, - }, - { - name: "zero status with error returns error", - health: &remote.RouteHealth{HTTPStatus: 0, Error: "connection refused"}, - expected: components.StatusError, - }, - { - name: "200 returns success", - health: &remote.RouteHealth{HTTPStatus: 200}, - expected: components.StatusSuccess, - }, - { - name: "301 returns success", - health: &remote.RouteHealth{HTTPStatus: 301}, - expected: components.StatusSuccess, - }, - { - name: "500 returns error", - health: &remote.RouteHealth{HTTPStatus: 500}, - expected: components.StatusError, - }, - { - name: "404 returns error", - health: &remote.RouteHealth{HTTPStatus: 404}, - expected: components.StatusError, - }, - { - name: "100 informational returns error", - health: &remote.RouteHealth{HTTPStatus: 100}, - expected: components.StatusError, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - result := httpHealthToStatus(tt.health) - assert.Equal(t, tt.expected, result) - }) - } -} - -func TestGroupRoutesByNetwork(t *testing.T) { - routes := []remote.RouteInfo{ - {Domain: "alpha.dev", Network: "gordon-alpha-dev"}, - {Domain: "beta.dev", Network: "gordon-shared"}, - {Domain: "gamma.dev", Network: "gordon-shared"}, - {Domain: "delta.dev", Network: "gordon-delta-dev"}, - } - - groups, solo := groupRoutesByNetwork(routes) - - assert.Len(t, solo, 2) - assert.Equal(t, "alpha.dev", solo[0].Domain) - assert.Equal(t, "delta.dev", solo[1].Domain) - - assert.Len(t, groups, 1) - assert.Equal(t, "shared", groups[0].name) - assert.Len(t, groups[0].routes, 2) -} - -func TestStripNetworkPrefix(t *testing.T) { - tests := []struct { - network string - expected string - }{ - {"gordon-shared-services", "shared-services"}, - {"gordon-my-app-dev", "my-app-dev"}, - {"custom-network", "custom-network"}, - {"gordon-", ""}, - {"", ""}, - } - - for _, tt := range tests { - t.Run(tt.network, func(t *testing.T) { - assert.Equal(t, tt.expected, stripNetworkPrefix(tt.network)) - }) - } -} - -func TestResolveRoutesExplicitRemote_AllowsAdHocURLWhenSavedRemotesConfigUnreadable(t *testing.T) { - originalRemoteFlag := remoteFlag - originalTokenFlag := tokenFlag - originalInsecureTLSFlag := insecureTLSFlag - t.Cleanup(func() { - remoteFlag = originalRemoteFlag - tokenFlag = originalTokenFlag - insecureTLSFlag = originalInsecureTLSFlag - }) - - remoteFlag = "" - tokenFlag = "" - insecureTLSFlag = false - - configHome := t.TempDir() - t.Setenv("XDG_CONFIG_HOME", configHome) - t.Setenv("HOME", t.TempDir()) - t.Setenv("GORDON_REMOTE", "https://ad-hoc.example.com") - - configPath := filepath.Join(configHome, "gordon", "remotes.toml") - require.NoError(t, os.MkdirAll(configPath, 0o755)) - - resolved, ok, err := resolveRoutesExplicitRemote() - require.True(t, ok) - require.NoError(t, err) - require.NotNil(t, resolved) - assert.Equal(t, "https://ad-hoc.example.com", resolved.URL) - assert.Empty(t, resolved.Name) -} - -func TestResolveRoutesExplicitRemote_ReturnsErrorForUnreadableNamedRemoteConfig(t *testing.T) { - originalRemoteFlag := remoteFlag - originalTokenFlag := tokenFlag - originalInsecureTLSFlag := insecureTLSFlag - t.Cleanup(func() { - remoteFlag = originalRemoteFlag - tokenFlag = originalTokenFlag - insecureTLSFlag = originalInsecureTLSFlag - }) - - remoteFlag = "missing-remote" - tokenFlag = "" - insecureTLSFlag = false - - configHome := t.TempDir() - t.Setenv("XDG_CONFIG_HOME", configHome) - t.Setenv("HOME", t.TempDir()) - t.Setenv("GORDON_REMOTE", "") - - configPath := filepath.Join(configHome, "gordon", "remotes.toml") - require.NoError(t, os.MkdirAll(configPath, 0o755)) - - resolved, ok, err := resolveRoutesExplicitRemote() - require.Error(t, err) - require.False(t, ok) - require.Nil(t, resolved) - assert.Contains(t, err.Error(), "failed to read remotes") -} - -func TestRouteStatusTitle_PreservesProbeFailureError(t *testing.T) { - item := routeStatusItem{ - Domain: "app.example.com", - Image: "app:latest", - ContainerStatus: "running", - HTTPStatus: 0, - HealthError: "connection refused", - } - - expected := components.StatusIcon(styles.IconHTTPStatus, components.StatusError) + " " + - components.StatusIcon(styles.IconContainerStatus, components.ParseStatus("running")) + - " app.example.com" - - assert.Equal(t, expected, routeStatusTitle(item)) -} - -func TestRunRoutesShow_JSONIncludesHealthError(t *testing.T) { - cpMock := climocks.NewMockControlPlane(t) - cpMock.EXPECT().GetRoute(context.Background(), "app.example.com").Return(&domain.Route{Domain: "app.example.com", Image: "app:latest"}, nil).Once() - cpMock.EXPECT().GetHealth(context.Background()).Return(nil, errors.New("probe failed")).Once() - - var out bytes.Buffer - err := runRoutesShow(context.Background(), cpMock, &out, "app.example.com", true) - - require.NoError(t, err) - - var got map[string]any - require.NoError(t, json.Unmarshal(out.Bytes(), &got)) - assert.Equal(t, "app.example.com", got["domain"]) - assert.Equal(t, "app:latest", got["image"]) - assert.Equal(t, "unknown", got["container_status"]) - assert.Equal(t, float64(0), got["http_status"]) - assert.Equal(t, "probe failed", got["health_error"]) -} - -func TestRunRoutesShow_TextIncludesHealthWarning(t *testing.T) { - cpMock := climocks.NewMockControlPlane(t) - cpMock.EXPECT().GetRoute(context.Background(), "app.example.com").Return(&domain.Route{Domain: "app.example.com", Image: "app:latest"}, nil).Once() - cpMock.EXPECT().GetHealth(context.Background()).Return(nil, errors.New("probe failed")).Once() - - var out bytes.Buffer - err := runRoutesShow(context.Background(), cpMock, &out, "app.example.com", false) - - require.NoError(t, err) - - text := stripANSI(out.String()) - assert.Contains(t, text, "Route: app.example.com") - assert.Contains(t, text, "Domain:") - assert.Contains(t, text, "Image:") - assert.Contains(t, text, "Container:") - assert.Contains(t, text, "probe failed") -} - -func TestRunRoutesShow_JSONIncludesRouteProbeFailureAndUnknownContainerStatus(t *testing.T) { - cpMock := climocks.NewMockControlPlane(t) - cpMock.EXPECT().GetRoute(context.Background(), "app.example.com").Return(&domain.Route{Domain: "app.example.com", Image: "app:latest"}, nil).Once() - cpMock.EXPECT().GetHealth(context.Background()).Return(map[string]*remote.RouteHealth{ - "app.example.com": { - HTTPStatus: 503, - ContainerStatus: "", - Error: "probe failed", - }, - }, nil).Once() - - var out bytes.Buffer - err := runRoutesShow(context.Background(), cpMock, &out, "app.example.com", true) - - require.NoError(t, err) - - var got map[string]any - require.NoError(t, json.Unmarshal(out.Bytes(), &got)) - assert.Equal(t, "app.example.com", got["domain"]) - assert.Equal(t, "app:latest", got["image"]) - assert.Equal(t, "unknown", got["container_status"]) - assert.Equal(t, float64(503), got["http_status"]) - assert.Equal(t, "probe failed", got["health_error"]) -} - -func TestRunRoutesShow_TextIncludesRouteProbeFailureAndUnknownContainerStatus(t *testing.T) { - cpMock := climocks.NewMockControlPlane(t) - cpMock.EXPECT().GetRoute(context.Background(), "app.example.com").Return(&domain.Route{Domain: "app.example.com", Image: "app:latest"}, nil).Once() - cpMock.EXPECT().GetHealth(context.Background()).Return(map[string]*remote.RouteHealth{ - "app.example.com": { - HTTPStatus: 503, - ContainerStatus: "", - Error: "probe failed", - }, - }, nil).Once() - - var out bytes.Buffer - err := runRoutesShow(context.Background(), cpMock, &out, "app.example.com", false) - - require.NoError(t, err) - - text := stripANSI(out.String()) - assert.Contains(t, text, "Route: app.example.com") - assert.Contains(t, text, "Container:") - assert.Contains(t, text, "unknown") - assert.Contains(t, text, "probe failed") -} - -func TestCollectRoutesListSections_DefaultModeIncludesLocalThenSortedRemotes(t *testing.T) { - testsDeps := routesListDeps{ - explicitRemote: func() (*remote.ResolvedRemote, bool, error) { - return nil, false, nil - }, - loadLocal: func(context.Context, string) (routeListSection, error) { - return routeListSection{ - Kind: "local", - Name: "local", - Routes: []routeListItem{{ - Domain: "app.local", - Image: "myapp:latest", - }}, - }, nil - }, - listRemotes: func() (map[string]remote.RemoteEntry, string, error) { - return map[string]remote.RemoteEntry{ - "remote-a": {URL: "https://remote-a.example.com"}, - "remote-b": {URL: "https://remote-b.example.com"}, - }, "remote-a", nil - }, - loadRemote: func(_ context.Context, name string, entry remote.RemoteEntry) (routeListSection, error) { - if name == "remote-b" { - return routeListSection{ - Kind: "remote", - Name: name, - URL: entry.URL, - Error: "remote unavailable", - }, nil - } - return routeListSection{ - Kind: "remote", - Name: name, - URL: entry.URL, - Routes: []routeListItem{{ - Domain: "dashboard.example.com", - Image: "dashboard", - }}, - }, nil - }, - } - - sections, err := collectRoutesListSections(context.Background(), "", testsDeps) - - require.NoError(t, err) - require.Len(t, sections, 3) - assert.Equal(t, "local", sections[0].Name) - assert.Equal(t, "remote-a", sections[1].Name) - assert.Equal(t, "remote-b", sections[2].Name) - assert.Equal(t, "dashboard.example.com", sections[1].Routes[0].Domain) - assert.Equal(t, "remote unavailable", sections[2].Error) -} - -func TestCollectRoutesListSections_ExplicitRemoteSkipsAggregate(t *testing.T) { - testsDeps := routesListDeps{ - explicitRemote: func() (*remote.ResolvedRemote, bool, error) { - return &remote.ResolvedRemote{ - Name: "remote-a", - URL: "https://remote-a.example.com", - }, true, nil - }, - loadLocal: func(context.Context, string) (routeListSection, error) { - t.Fatal("loadLocal should not be called when an explicit remote is selected") - return routeListSection{}, nil - }, - listRemotes: func() (map[string]remote.RemoteEntry, string, error) { - t.Fatal("listRemotes should not be called when an explicit remote is selected") - return nil, "", nil - }, - loadRemote: func(_ context.Context, name string, entry remote.RemoteEntry) (routeListSection, error) { - return routeListSection{ - Kind: "remote", - Name: name, - URL: entry.URL, - Routes: []routeListItem{{ - Domain: "test.example.com", - Image: "hello-app", - }}, - }, nil - }, - } - - sections, err := collectRoutesListSections(context.Background(), "", testsDeps) - - require.NoError(t, err) - require.Len(t, sections, 1) - assert.Equal(t, "remote-a", sections[0].Name) - assert.Equal(t, "test.example.com", sections[0].Routes[0].Domain) -} - -func TestCollectRoutesSections_ExplicitRemoteResolutionErrorReturnsError(t *testing.T) { - - t.Run("list", func(t *testing.T) { - var loadLocalCalls atomic.Int32 - var listRemotesCalls atomic.Int32 - var loadRemoteCalls atomic.Int32 - - deps := routesListDeps{ - explicitRemote: func() (*remote.ResolvedRemote, bool, error) { - return nil, false, errors.New("failed to read remotes: boom") - }, - loadLocal: func(context.Context, string) (routeListSection, error) { - loadLocalCalls.Add(1) - return routeListSection{}, nil - }, - listRemotes: func() (map[string]remote.RemoteEntry, string, error) { - listRemotesCalls.Add(1) - return map[string]remote.RemoteEntry{}, "", nil - }, - loadRemote: func(context.Context, string, remote.RemoteEntry) (routeListSection, error) { - loadRemoteCalls.Add(1) - return routeListSection{}, nil - }, - } - - sections, err := collectRoutesListSections(context.Background(), "", deps) - - require.Error(t, err) - require.Nil(t, sections) - assert.Contains(t, err.Error(), "failed to read remotes") - assert.Zero(t, loadLocalCalls.Load()) - assert.Zero(t, listRemotesCalls.Load()) - assert.Zero(t, loadRemoteCalls.Load()) - }) - - t.Run("status", func(t *testing.T) { - var loadLocalCalls atomic.Int32 - var listRemotesCalls atomic.Int32 - var loadRemoteCalls atomic.Int32 - - deps := routesStatusDeps{ - explicitRemote: func() (*remote.ResolvedRemote, bool, error) { - return nil, false, errors.New("failed to read remotes: boom") - }, - loadLocal: func(context.Context, string) (routeStatusSection, error) { - loadLocalCalls.Add(1) - return routeStatusSection{}, nil - }, - listRemotes: func() (map[string]remote.RemoteEntry, string, error) { - listRemotesCalls.Add(1) - return map[string]remote.RemoteEntry{}, "", nil - }, - loadRemote: func(context.Context, string, remote.RemoteEntry) (routeStatusSection, error) { - loadRemoteCalls.Add(1) - return routeStatusSection{}, nil - }, - } - - sections, err := collectRoutesStatusSections(context.Background(), "", deps) - - require.Error(t, err) - require.Nil(t, sections) - assert.Contains(t, err.Error(), "failed to read remotes") - assert.Zero(t, loadLocalCalls.Load()) - assert.Zero(t, listRemotesCalls.Load()) - assert.Zero(t, loadRemoteCalls.Load()) - }) -} - -func TestLoadRoutesListExplicitRemoteSection_PropagatesLoaderError(t *testing.T) { - section := loadRoutesListExplicitRemoteSection(context.Background(), routesListDeps{ - loadRemote: func(context.Context, string, remote.RemoteEntry) (routeListSection, error) { - return routeListSection{}, errors.New("boom") - }, - }, &remote.ResolvedRemote{Name: "remote-a", URL: "https://remote-a.example.com"}) - - assert.Equal(t, "remote", section.Kind) - assert.Equal(t, "remote-a", section.Name) - assert.Equal(t, "https://remote-a.example.com", section.URL) - assert.Equal(t, "boom", section.Error) -} - -func TestLoadRoutesStatusExplicitRemoteSection_PropagatesLoaderError(t *testing.T) { - section := loadRoutesStatusExplicitRemoteSection(context.Background(), routesStatusDeps{ - loadRemote: func(context.Context, string, remote.RemoteEntry) (routeStatusSection, error) { - return routeStatusSection{}, errors.New("boom") - }, - }, &remote.ResolvedRemote{Name: "remote-a", URL: "https://remote-a.example.com"}) - - assert.Equal(t, "remote", section.Kind) - assert.Equal(t, "remote-a", section.Name) - assert.Equal(t, "https://remote-a.example.com", section.URL) - assert.Equal(t, "boom", section.Error) -} - -func TestCollectRoutesListAggregateSections_EncodesLoaderErrors(t *testing.T) { - t.Run("local loader error", func(t *testing.T) { - sections, err := collectRoutesListAggregateSections(context.Background(), "config.toml", routesListDeps{ - loadLocal: func(context.Context, string) (routeListSection, error) { - return routeListSection{Kind: "local", Name: "local"}, errors.New("local failed") - }, - listRemotes: func() (map[string]remote.RemoteEntry, string, error) { - return map[string]remote.RemoteEntry{}, "", nil - }, - loadRemote: func(context.Context, string, remote.RemoteEntry) (routeListSection, error) { - return routeListSection{Kind: "remote", Name: "remote-a", URL: "https://remote-a.example.com"}, nil - }, - }) - - require.NoError(t, err) - require.Len(t, sections, 1) - assert.Equal(t, "local", sections[0].Kind) - assert.Equal(t, "local failed", sections[0].Error) - }) - - t.Run("remote loader error", func(t *testing.T) { - sections, err := collectRoutesListAggregateSections(context.Background(), "config.toml", routesListDeps{ - loadLocal: func(context.Context, string) (routeListSection, error) { - return routeListSection{Kind: "local", Name: "local"}, nil - }, - listRemotes: func() (map[string]remote.RemoteEntry, string, error) { - return map[string]remote.RemoteEntry{"remote-a": {URL: "https://remote-a.example.com"}}, "", nil - }, - loadRemote: func(context.Context, string, remote.RemoteEntry) (routeListSection, error) { - return routeListSection{Kind: "remote", Name: "remote-a", URL: "https://remote-a.example.com"}, errors.New("remote failed") - }, - }) - - require.NoError(t, err) - require.Len(t, sections, 2) - assert.Equal(t, "local", sections[0].Kind) - assert.Equal(t, "remote", sections[1].Kind) - assert.Equal(t, "remote failed", sections[1].Error) - }) -} - -func TestCollectRoutesStatusAggregateSections_EncodesLoaderErrors(t *testing.T) { - t.Run("local loader error", func(t *testing.T) { - sections, err := collectRoutesStatusAggregateSections(context.Background(), "config.toml", routesStatusDeps{ - loadLocal: func(context.Context, string) (routeStatusSection, error) { - return routeStatusSection{Kind: "local", Name: "local"}, errors.New("local failed") - }, - listRemotes: func() (map[string]remote.RemoteEntry, string, error) { - return map[string]remote.RemoteEntry{}, "", nil - }, - loadRemote: func(context.Context, string, remote.RemoteEntry) (routeStatusSection, error) { - return routeStatusSection{Kind: "remote", Name: "remote-a", URL: "https://remote-a.example.com"}, nil - }, - }) - - require.NoError(t, err) - require.Len(t, sections, 1) - assert.Equal(t, "local", sections[0].Kind) - assert.Equal(t, "local failed", sections[0].Error) - }) - - t.Run("remote loader error", func(t *testing.T) { - sections, err := collectRoutesStatusAggregateSections(context.Background(), "config.toml", routesStatusDeps{ - loadLocal: func(context.Context, string) (routeStatusSection, error) { - return routeStatusSection{Kind: "local", Name: "local"}, nil - }, - listRemotes: func() (map[string]remote.RemoteEntry, string, error) { - return map[string]remote.RemoteEntry{"remote-a": {URL: "https://remote-a.example.com"}}, "", nil - }, - loadRemote: func(context.Context, string, remote.RemoteEntry) (routeStatusSection, error) { - return routeStatusSection{Kind: "remote", Name: "remote-a", URL: "https://remote-a.example.com"}, errors.New("remote failed") - }, - }) - - require.NoError(t, err) - require.Len(t, sections, 2) - assert.Equal(t, "local", sections[0].Kind) - assert.Equal(t, "remote", sections[1].Kind) - assert.Equal(t, "remote failed", sections[1].Error) - }) -} - -func TestLoadRoutesStatusLocalSection_DoesNotLeakKernelInitLogs(t *testing.T) { - tmpDir := t.TempDir() - cfgPath := filepath.Join(tmpDir, "gordon.toml") - require.NoError(t, os.WriteFile(cfgPath, []byte(`[server] -gordon_domain = "gordon.local" -data_dir = `+strconv.Quote(filepath.Join(tmpDir, "data"))+` - -[auth] -enabled = true -`), 0o600)) - - stdout, stderr := captureOutput(t, func() { - section, err := loadRoutesStatusLocalSection(context.Background(), cfgPath) - require.NoError(t, err) - assert.Equal(t, "local", section.Kind) - assert.Equal(t, "local", section.Name) - assert.Contains(t, section.Error, "failed to initialize local control plane") - assert.Contains(t, section.Error, "auth.secrets_backend is required") - }) - - combined := stdout + stderr - assert.NotContains(t, combined, "failed to resolve secrets backend") - assert.NotContains(t, combined, "local kernel running in minimal mode") -} - -func captureOutput(t *testing.T, fn func()) (string, string) { - t.Helper() - - oldStdout, oldStderr := os.Stdout, os.Stderr - outR, outW, err := os.Pipe() - require.NoError(t, err) - errR, errW, err := os.Pipe() - require.NoError(t, err) - - os.Stdout = outW - os.Stderr = errW - - stdoutCh := make(chan string, 1) - stderrCh := make(chan string, 1) - go func() { - var buf bytes.Buffer - _, _ = io.Copy(&buf, outR) - stdoutCh <- buf.String() - }() - go func() { - var buf bytes.Buffer - _, _ = io.Copy(&buf, errR) - stderrCh <- buf.String() - }() - - defer func() { - os.Stdout = oldStdout - os.Stderr = oldStderr - }() - - fn() - - require.NoError(t, outW.Close()) - require.NoError(t, errW.Close()) - - return <-stdoutCh, <-stderrCh -} - -func TestRenderRoutesStatusSections_IncludesSectionHeadingsAndErrors(t *testing.T) { - sections := []routeStatusSection{ - { - Kind: "local", - Name: "local", - Routes: []routeStatusItem{{ - Domain: "app.local", - Image: "myapp:latest", - ContainerStatus: "running", - }}, - }, - { - Kind: "remote", - Name: "remote-a", - URL: "https://remote-a.example.com", - Error: "dial tcp timeout", - }, - } - - var out bytes.Buffer - err := renderRoutesStatusSections(&out, sections) - - require.NoError(t, err) - rendered := stripANSI(out.String()) - lines := strings.Split(rendered, "\n") - - lineIndex := func(substr string) int { - for i, line := range lines { - if strings.Contains(line, substr) { - return i - } - } - return -1 - } - - titleIdx := lineIndex("Route Status") - require.NotEqual(t, -1, titleIdx) - - localIdx := lineIndex("Local") - require.NotEqual(t, -1, localIdx) - - routeIdx := lineIndex("app.local") - require.NotEqual(t, -1, routeIdx) - - remoteIdx := lineIndex("Remote: remote-a") - require.NotEqual(t, -1, remoteIdx) - - errorIdx := lineIndex("dial tcp timeout") - require.NotEqual(t, -1, errorIdx) - - assert.Less(t, titleIdx, localIdx) - assert.Less(t, localIdx, routeIdx) - assert.Less(t, routeIdx, remoteIdx) - assert.Less(t, remoteIdx, errorIdx) -} - -func TestRenderRoutesStatusSections_DoesNotAddExtraBlankLineBetweenSections(t *testing.T) { - sections := []routeStatusSection{ - { - Kind: "remote", - Name: "remote-b", - Routes: []routeStatusItem{{ - Domain: "app.example.com", - Image: "app:latest", - ContainerStatus: "running", - }}, - }, - { - Kind: "remote", - Name: "remote-a", - Routes: []routeStatusItem{{ - Domain: "dashboard.example.com", - Image: "dashboard", - ContainerStatus: "running", - }}, - }, - } - - var out bytes.Buffer - err := renderRoutesStatusSections(&out, sections) - - require.NoError(t, err) - rendered := stripANSI(out.String()) - assert.NotContains(t, rendered, "app:latest\n\n\nRemote: remote-a") - assert.Contains(t, rendered, "app:latest\n\nRemote: remote-a") -} - -func TestRenderRoutesListSections_HidesLocalErrorWhenRemoteIsUsable(t *testing.T) { - sections := []routeListSection{ - {Kind: "local", Name: "local", Error: "failed to initialize local control plane"}, - {Kind: "remote", Name: "remote-a", URL: "https://remote-a.example.com"}, - } - - var out bytes.Buffer - err := renderRoutesListSections(&out, sections) - - require.NoError(t, err) - rendered := stripANSI(out.String()) - assert.NotContains(t, rendered, "Local") - assert.NotContains(t, rendered, "failed to initialize local control plane") - assert.Contains(t, rendered, "Remote: remote-a") - assert.Contains(t, rendered, "No routes configured") -} - -func TestLoadRoutesListRemoteSection_UsesInventoryEndpoint(t *testing.T) { - srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - w.Header().Set("Content-Type", "application/json") - switch { - case r.URL.Path == "/admin/routes" && r.URL.RawQuery == "": - _, _ = w.Write([]byte(`{"routes":[{"domain":"app.example.com","image":"app:latest"}]}`)) - case r.URL.Path == "/admin/routes" && r.URL.RawQuery == "detailed=true": - w.WriteHeader(http.StatusInternalServerError) - _, _ = w.Write([]byte(`{"error":"detailed route status unavailable"}`)) - default: - t.Fatalf("unexpected request: %s?%s", r.URL.Path, r.URL.RawQuery) - } - })) - defer srv.Close() - - section, err := loadRoutesListRemoteSection(context.Background(), "prod", remote.RemoteEntry{URL: srv.URL}) - - require.NoError(t, err) - require.Empty(t, section.Error) - require.Len(t, section.Routes, 1) - assert.Equal(t, "app.example.com", section.Routes[0].Domain) - assert.Equal(t, "app:latest", section.Routes[0].Image) -} - -func TestRenderRoutesStatusSections_HidesLocalErrorWhenRemoteIsUsable(t *testing.T) { - sections := []routeStatusSection{ - {Kind: "local", Name: "local", Error: "failed to initialize local control plane"}, - {Kind: "remote", Name: "remote-a", URL: "https://remote-a.example.com"}, - } - - var out bytes.Buffer - err := renderRoutesStatusSections(&out, sections) - - require.NoError(t, err) - rendered := stripANSI(out.String()) - assert.NotContains(t, rendered, "Local") - assert.NotContains(t, rendered, "failed to initialize local control plane") - assert.Contains(t, rendered, "Remote: remote-a") - assert.Contains(t, rendered, "No routes configured") -} - -func TestRenderRoutesStatusSections_KeepsLocalErrorWhenAllRemotesFail(t *testing.T) { - sections := []routeStatusSection{ - {Kind: "local", Name: "local", Error: "failed to initialize local control plane"}, - {Kind: "remote", Name: "remote-a", URL: "https://remote-a.example.com", Error: "dial tcp timeout"}, - } - - var out bytes.Buffer - err := renderRoutesStatusSections(&out, sections) - - require.NoError(t, err) - rendered := stripANSI(out.String()) - assert.Contains(t, rendered, "Local") - assert.Contains(t, rendered, "failed to initialize local control plane") - assert.Contains(t, rendered, "Remote: remote-a") - assert.Contains(t, rendered, "dial tcp timeout") -} - -func TestWriteRouteRemoveTextShowsCleanupReport(t *testing.T) { - var out bytes.Buffer - report := &domain.CleanupReport{ - Domain: "app.example.com", - RemovedContainers: []domain.CleanupContainer{{ - ID: "1234567890abcdef", - Name: "gordon-app.example.com", - }}, - PreservedAttachments: []domain.CleanupAttachment{{ - Name: "postgres", - ContainerID: "attachment-1", - Status: "running", - }}, - Warnings: []string{"volumes preserved"}, - Hints: []string{"run diagnose"}, - } - - err := writeRouteRemoveText(&out, "app.example.com", dto.CleanupReportFromDomain(report)) - require.NoError(t, err) - rendered := stripANSI(out.String()) - assert.Contains(t, rendered, "Route removed: app.example.com") - assert.Contains(t, rendered, "Removed containers:") - assert.Contains(t, rendered, "gordon-app.example.com (1234567890ab)") - assert.Contains(t, rendered, "Preserved attachments:") - assert.Contains(t, rendered, "postgres (running)") - assert.Contains(t, rendered, "volumes preserved") - assert.Contains(t, rendered, "run diagnose") -} - -func TestBuildRouteStatusTree_DeterministicAcrossInputOrder(t *testing.T) { - routes := []routeStatusItem{ - {Domain: "zulu.example", Image: "zulu:v1", Network: "gordon-zulu", ContainerStatus: "running"}, - {Domain: "alpha.example", Image: "alpha:v1", Network: "gordon-alpha", ContainerStatus: "running"}, - {Domain: "mike.example", Image: "solo:v1", ContainerStatus: "running"}, - } - - var outA bytes.Buffer - _, _ = outA.WriteString(stripANSI(buildRouteStatusTree(routes).Render())) - - var outB bytes.Buffer - reordered := []routeStatusItem{routes[2], routes[0], routes[1]} - _, _ = outB.WriteString(stripANSI(buildRouteStatusTree(reordered).Render())) - - assert.Equal(t, outA.String(), outB.String()) -} - -func TestAttachmentsForRouteDiagnosisUsesOwner(t *testing.T) { - attachments := []domain.CleanupAttachment{ - {Name: "unexpected-name", Owner: "app.example.com", ContainerID: "match"}, - {Name: "gordon-other-example-com-postgres", Owner: "other.example.com", ContainerID: "miss"}, - } - - matched := attachmentsForRouteDiagnosis(attachments, "app.example.com") - - require.Len(t, matched, 1) - assert.Equal(t, "match", matched[0].ContainerID) -} - -func TestBuildRouteDiagnosis_VolumesAreScopedToExactRouteContainers(t *testing.T) { - cpMock := climocks.NewMockControlPlane(t) - cpMock.EXPECT().GetRoute(context.Background(), "example.test").Return(&domain.Route{Domain: "example.test", Image: "site:latest"}, nil).Once() - cpMock.EXPECT().ListRoutesWithDetails(context.Background()).Return([]remote.RouteInfo{{ - Domain: "example.test", - ContainerID: "route-1", - ContainerStatus: "running", - }}, nil).Once() - cpMock.EXPECT().GetHealth(context.Background()).Return(nil, nil).Once() - cpMock.EXPECT().ListVolumes(context.Background()).Return([]dto.Volume{ - {Name: "gordon-example-test-data-docs", Containers: []string{"gordon-example.test"}}, - {Name: "gordon-api-example-test-cache", Containers: []string{"gordon-api.example.test"}}, - }, nil).Once() - cpMock.EXPECT().GetConfig(context.Background()).Return(&remote.Config{}, nil).Once() - cpMock.EXPECT().ListOrphanedAttachments(context.Background()).Return(nil, nil).Once() - - diag, err := buildRouteDiagnosis(context.Background(), cpMock, "example.test") - - require.NoError(t, err) - require.Len(t, diag.Volumes, 1) - assert.Equal(t, "gordon-example-test-data-docs", diag.Volumes[0].Name) -} - -func TestBuildRouteDiagnosis_UsesConfiguredVolumePrefixForPreservedVolumes(t *testing.T) { - srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - w.Header().Set("Content-Type", "application/json") - switch r.URL.Path { - case "/admin/routes/example.test": - w.WriteHeader(http.StatusNotFound) - _, _ = w.Write([]byte(`{"error":"route not found"}`)) - case "/admin/routes/example.test/cleanup": - _, _ = w.Write([]byte(`{"domain":"example.test","preserved_volumes":[{"name":"custom-example-test-data-docs"}]}`)) - case "/admin/routes": - require.Equal(t, "detailed=true", r.URL.RawQuery) - _, _ = w.Write([]byte(`{"routes":[]}`)) - case "/admin/health": - _, _ = w.Write([]byte(`{"health":{}}`)) - case "/admin/volumes": - _, _ = w.Write([]byte(`[{"name":"custom-example-test-data-docs","containers":[]}]`)) - case "/admin/attachments/orphans": - _, _ = w.Write([]byte(`{"attachments":[]}`)) - case "/admin/config": - _, _ = w.Write([]byte(`{"server":{"port":80,"registry_port":5000,"registry_domain":"registry.example.test"},"auto_route":{"enabled":false},"network_isolation":{"enabled":true,"prefix":"custom-network"},"volumes":{"auto_create":true,"prefix":"custom","preserve":true},"routes":[],"external_routes":[]}`)) - default: - t.Fatalf("unexpected path: %s", r.URL.Path) - } - })) - defer srv.Close() - - cp := NewRemoteControlPlane(remote.NewClient(srv.URL)) - diag, err := buildRouteDiagnosis(context.Background(), cp, "example.test") - - require.NoError(t, err) - require.Len(t, diag.Volumes, 1) - assert.Equal(t, "custom-example-test-data-docs", diag.Volumes[0].Name) -} - -func TestBuildRouteDiagnosis_IncludesAttachmentVolumesWithDoubleUnderscoreOwnerEncoding(t *testing.T) { - cpMock := climocks.NewMockControlPlane(t) - cpMock.EXPECT().GetRoute(context.Background(), "app.example.test").Return(&domain.Route{Domain: "app.example.test", Image: "app:latest"}, nil).Once() - cpMock.EXPECT().ListRoutesWithDetails(context.Background()).Return([]remote.RouteInfo{{ - Domain: "app.example.test", - ContainerID: "route-1", - ContainerStatus: "running", - Attachments: []dto.Attachment{{ - Name: "postgres", - Image: "postgres:16", - ContainerID: "attachment-1", - Status: "running", - }}, - }}, nil).Once() - cpMock.EXPECT().GetHealth(context.Background()).Return(nil, nil).Once() - cpMock.EXPECT().ListVolumes(context.Background()).Return([]dto.Volume{ - {Name: "gordon-gordon-app__example__test-postgres-var-lib-postgresql", Containers: []string{"gordon-app__example__test-postgres"}}, - }, nil).Once() - cpMock.EXPECT().GetConfig(context.Background()).Return(&remote.Config{}, nil).Once() - cpMock.EXPECT().ListOrphanedAttachments(context.Background()).Return(nil, nil).Once() - - diag, err := buildRouteDiagnosis(context.Background(), cpMock, "app.example.test") - - require.NoError(t, err) - require.Len(t, diag.Volumes, 1) - assert.Equal(t, "gordon-gordon-app__example__test-postgres-var-lib-postgresql", diag.Volumes[0].Name) -} - -func TestBuildRouteDiagnosis_UsesCleanupPreviewForRemovedRouteRuntimeState(t *testing.T) { - srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - w.Header().Set("Content-Type", "application/json") - switch r.URL.Path { - case "/admin/routes/example.test": - w.WriteHeader(http.StatusNotFound) - _, _ = w.Write([]byte(`{"error":"route not found"}`)) - case "/admin/routes/example.test/cleanup": - _, _ = w.Write([]byte(`{"domain":"example.test","orphaned_entities":[{"kind":"route_container","id":"ctr-1","name":"gordon-example.test","status":"running"}],"preserved_attachments":[{"name":"old-postgres","container_id":"att-1","owner":"example.test"}],"preserved_volumes":[{"name":"gordon-example-test-data"}]}`)) - case "/admin/routes": - require.Equal(t, "detailed=true", r.URL.RawQuery) - _, _ = w.Write([]byte(`{"routes":[]}`)) - case "/admin/health": - _, _ = w.Write([]byte(`{"health":{}}`)) - default: - t.Fatalf("unexpected path: %s", r.URL.Path) - } - })) - defer srv.Close() - - cp := NewRemoteControlPlane(remote.NewClient(srv.URL)) - diag, err := buildRouteDiagnosis(context.Background(), cp, "example.test") - - require.NoError(t, err) - assert.False(t, diag.Configured) - require.Len(t, diag.OrphanedEntities, 1) - assert.Equal(t, "route_container", diag.OrphanedEntities[0].Kind) - require.Len(t, diag.OrphanedAttachments, 1) - assert.Equal(t, "old-postgres", diag.OrphanedAttachments[0].Name) - require.Len(t, diag.Volumes, 1) - assert.Equal(t, "gordon-example-test-data", diag.Volumes[0].Name) - assert.Contains(t, diag.Warnings, "route is not configured but runtime state still exists") -} - -func TestBuildRouteDiagnosis_RemoteMissingRouteDoesNotWarnOnExpectedNotFound(t *testing.T) { - srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - w.Header().Set("Content-Type", "application/json") - switch r.URL.Path { - case "/admin/routes/missing.example.test": - w.WriteHeader(http.StatusNotFound) - _, _ = w.Write([]byte(`{"error":"route not found"}`)) - case "/admin/routes/missing.example.test/cleanup": - _, _ = w.Write([]byte(`{"domain":"missing.example.test"}`)) - case "/admin/routes": - require.Equal(t, "detailed=true", r.URL.RawQuery) - _, _ = w.Write([]byte(`{"routes":[]}`)) - case "/admin/health": - _, _ = w.Write([]byte(`{"health":{}}`)) - case "/admin/volumes": - _, _ = w.Write([]byte(`[]`)) - case "/admin/attachments/orphans": - _, _ = w.Write([]byte(`{"attachments":[]}`)) - default: - t.Fatalf("unexpected path: %s", r.URL.Path) - } - })) - defer srv.Close() - - cp := NewRemoteControlPlane(remote.NewClient(srv.URL)) - diag, err := buildRouteDiagnosis(context.Background(), cp, "missing.example.test") - - require.NoError(t, err) - assert.False(t, diag.Configured) - assert.Empty(t, diag.Warnings) -} - -func TestRunRoutePurge_VolumesUsesConfiguredVolumePrefix(t *testing.T) { - srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - w.Header().Set("Content-Type", "application/json") - switch r.URL.Path { - case "/admin/routes/example.test": - w.WriteHeader(http.StatusNotFound) - _, _ = w.Write([]byte(`{"error":"route not found"}`)) - case "/admin/routes/example.test/cleanup": - _, _ = w.Write([]byte(`{"domain":"example.test","preserved_volumes":[{"name":"custom-example-test-data-docs"}]}`)) - case "/admin/routes": - require.Equal(t, "detailed=true", r.URL.RawQuery) - _, _ = w.Write([]byte(`{"routes":[]}`)) - case "/admin/health": - _, _ = w.Write([]byte(`{"health":{}}`)) - case "/admin/volumes": - _, _ = w.Write([]byte(`[{"name":"custom-example-test-data-docs","containers":[]}]`)) - case "/admin/attachments/orphans": - _, _ = w.Write([]byte(`{"attachments":[]}`)) - case "/admin/config": - _, _ = w.Write([]byte(`{"server":{"port":80,"registry_port":5000,"registry_domain":"registry.example.test"},"auto_route":{"enabled":false},"network_isolation":{"enabled":true,"prefix":"custom-network"},"volumes":{"auto_create":true,"prefix":"custom","preserve":true},"routes":[],"external_routes":[]}`)) - default: - t.Fatalf("unexpected path: %s", r.URL.Path) - } - })) - defer srv.Close() - - cp := NewRemoteControlPlane(remote.NewClient(srv.URL)) - report, err := runRoutePurge(context.Background(), cp, "example.test", routePurgeOptions{Volumes: true}) - - require.NoError(t, err) - require.Len(t, report.PreservedVolumes, 1) - assert.Equal(t, "custom-example-test-data-docs", report.PreservedVolumes[0].Name) -} - -func TestBuildRouteDiagnosis_UsesRouteScopedReviewHintForOrphanedAttachments(t *testing.T) { - cpMock := climocks.NewMockControlPlane(t) - cpMock.EXPECT().GetRoute(context.Background(), "app.example.test").Return(&domain.Route{Domain: "app.example.test", Image: "app:latest"}, nil).Once() - cpMock.EXPECT().ListRoutesWithDetails(context.Background()).Return([]remote.RouteInfo{{ - Domain: "app.example.test", - ContainerID: "route-1", - ContainerStatus: "running", - }}, nil).Once() - cpMock.EXPECT().GetHealth(context.Background()).Return(nil, nil).Once() - cpMock.EXPECT().ListVolumes(context.Background()).Return(nil, nil).Once() - cpMock.EXPECT().ListOrphanedAttachments(context.Background()).Return([]domain.CleanupAttachment{{ - Name: "old-postgres", - ContainerID: "attachment-1", - Owner: "app.example.test", - }}, nil).Once() - - diag, err := buildRouteDiagnosis(context.Background(), cpMock, "app.example.test") - - require.NoError(t, err) - assert.Contains(t, diag.Hints, "run 'gordon routes purge app.example.test --attachments' to review orphaned attachment cleanup for this route") -} - -func TestWriteRouteDiagnosisText_UsesNeutralVolumeLabelForConfiguredRoutes(t *testing.T) { - var out bytes.Buffer - diag := &routeDiagnosis{ - Domain: "app.example.test", - Configured: true, - Route: &domain.Route{Domain: "app.example.test", Image: "app:latest"}, - Runtime: &remote.RouteInfo{ - Domain: "app.example.test", - ContainerID: "route-1", - ContainerStatus: "running", - }, - Volumes: []dto.Volume{{Name: "gordon-app-example-test-data"}}, - } - - err := writeRouteDiagnosisText(&out, diag) - require.NoError(t, err) - rendered := stripANSI(out.String()) - assert.Contains(t, rendered, "Volume: gordon-app-example-test-data") - assert.NotContains(t, rendered, "Preserved volume:") -} - -func TestWriteRouteDiagnosisText_ShowsActiveAttachmentsSeparatelyFromOrphans(t *testing.T) { - var out bytes.Buffer - diag := &routeDiagnosis{ - Domain: "app.example.test", - Configured: true, - Route: &domain.Route{Domain: "app.example.test", Image: "app:latest"}, - Runtime: &remote.RouteInfo{ - Domain: "app.example.test", - ContainerID: "route-1", - ContainerStatus: "running", - Attachments: []dto.Attachment{{ - Name: "postgres", - Image: "postgres:16", - ContainerID: "attachment-1", - Status: "running", - }}, - }, - OrphanedAttachments: []domain.CleanupAttachment{{ - Name: "old-postgres", - ContainerID: "attachment-2", - Owner: "app.example.test", - }}, - } - - err := writeRouteDiagnosisText(&out, diag) - require.NoError(t, err) - rendered := stripANSI(out.String()) - assert.Contains(t, rendered, "Active attachments:") - assert.Contains(t, rendered, "postgres (running)") - assert.Contains(t, rendered, "Orphaned attachment: old-postgres") -} - -func TestRunRoutePurge_AttachmentsForceIsScopedToRoute(t *testing.T) { - cpMock := climocks.NewMockControlPlane(t) - orphanedAttachments := []domain.CleanupAttachment{ - {ContainerID: "app-attachment", Owner: "app.example.com", Name: "postgres"}, - {ContainerID: "other-attachment", Owner: "other.example.com", Name: "redis"}, - } - cpMock.EXPECT().GetRoute(context.Background(), "app.example.com").Return(&domain.Route{Domain: "app.example.com", Image: "app:latest"}, nil).Once() - cpMock.EXPECT().ListRoutesWithDetails(context.Background()).Return(nil, nil).Once() - cpMock.EXPECT().GetHealth(context.Background()).Return(nil, nil).Once() - cpMock.EXPECT().ListVolumes(context.Background()).Return(nil, nil).Once() - cpMock.EXPECT().ListOrphanedAttachments(context.Background()).Return(orphanedAttachments, nil).Once() - cpMock.EXPECT().CleanupOrphanedAttachments(context.Background(), "app.example.com", true).Return(&domain.CleanupReport{ - RemovedContainers: []domain.CleanupContainer{{ID: "app-attachment", Name: "postgres"}}, - }, nil).Once() - - report, err := runRoutePurge(context.Background(), cpMock, "app.example.com", routePurgeOptions{Force: true, Attachments: true}) - - require.NoError(t, err) - require.NotNil(t, report) - require.Len(t, report.RemovedContainers, 1) - assert.Equal(t, "app-attachment", report.RemovedContainers[0].ID) -} diff --git a/internal/adapters/in/cli/secrets.go b/internal/adapters/in/cli/secrets.go deleted file mode 100644 index e79315131..000000000 --- a/internal/adapters/in/cli/secrets.go +++ /dev/null @@ -1,369 +0,0 @@ -package cli - -import ( - "context" - "fmt" - "strings" - - "github.com/bnema/gordon/internal/adapters/in/cli/remote" - "github.com/bnema/gordon/internal/adapters/in/cli/ui/components" - "github.com/bnema/gordon/internal/adapters/in/cli/ui/styles" - - "github.com/spf13/cobra" -) - -// newSecretsCmd creates the secrets command group. -func newSecretsCmd() *cobra.Command { - cmd := &cobra.Command{ - Use: "secrets", - Short: "Manage secrets", - Long: `Manage secrets (environment variables) for routes and attachments. - -Secrets are stored per-domain and injected into containers as environment variables. -Use --attachment to target attachment containers (databases, caches, etc.). - -When targeting a remote Gordon instance (via --remote flag or GORDON_REMOTE env var), -these commands operate on the remote server.`, - } - - cmd.AddCommand(newSecretsListCmd()) - cmd.AddCommand(newSecretsSetCmd()) - cmd.AddCommand(newSecretsRemoveCmd()) - - return cmd -} - -// newSecretsListCmd creates the secrets list command. -func newSecretsListCmd() *cobra.Command { - var jsonOut bool - - cmd := &cobra.Command{ - Use: "list ", - Short: "List secrets for a domain", - Long: `List all secret keys configured for a domain. - -Note: Only secret keys are shown, not values (for security). -Attachment secrets (for services like databases) are also displayed. - -Examples: - gordon secrets list app.mydomain.com - gordon --remote https://gordon.mydomain.com secrets list api.mydomain.com`, - Args: cobra.ExactArgs(1), - RunE: func(cmd *cobra.Command, args []string) error { - return runSecretsListCmd(cmd, args, jsonOut) - }, - } - - cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") - - return cmd -} - -// runSecretsListCmd executes the secrets list command. -func runSecretsListCmd(cmd *cobra.Command, args []string, jsonOut bool) error { - ctx := cmd.Context() - secretDomain := args[0] - - handle, err := resolveControlPlaneForRouteDomain(ctx, secretDomain) - if err != nil { - return err - } - defer handle.close() - - keys, attachments, err := fetchSecretsWithAttachments(ctx, handle.plane, secretDomain) - if err != nil { - return err - } - - totalSecrets := len(keys) - for _, att := range attachments { - totalSecrets += len(att.Keys) - } - - if totalSecrets == 0 { - if jsonOut { - return writeJSON(cmd.OutOrStdout(), map[string]any{ - "domain": secretDomain, - "keys": []string{}, - "attachments": []any{}, - }) - } - fmt.Println(styles.Theme.Muted.Render(fmt.Sprintf("No secrets configured for %s", secretDomain))) - return nil - } - - if jsonOut { - if attachments == nil { - attachments = []remote.AttachmentSecrets{} - } - return writeJSON(cmd.OutOrStdout(), map[string]any{ - "domain": secretDomain, - "keys": keys, - "attachments": attachments, - }) - } - - title := fmt.Sprintf("Secrets for %s", secretDomain) - if !handle.isRemote { - title = fmt.Sprintf("Secrets for %s (local)", secretDomain) - } - fmt.Println(styles.Theme.Title.Render(title)) - fmt.Println() - - rows := buildSecretsTableRows(keys, attachments) - - table := components.NewTable( - components.WithColumns([]components.TableColumn{ - {Title: "Key", Width: 45}, - {Title: "Value", Width: 10}, - }), - components.WithRows(rows), - ) - - fmt.Println(table.View()) - return nil -} - -// fetchSecretsWithAttachments retrieves secrets from the selected control plane. -func fetchSecretsWithAttachments(ctx context.Context, cp ControlPlane, secretDomain string) ([]string, []remote.AttachmentSecrets, error) { - result, err := cp.ListSecretsWithAttachments(ctx, secretDomain) - if err != nil { - return nil, nil, fmt.Errorf("failed to list secrets: %w", err) - } - return result.Keys, result.Attachments, nil -} - -// buildSecretsTableRows builds table rows with tree structure for attachments. -func buildSecretsTableRows(keys []string, attachments []remote.AttachmentSecrets) [][]string { - var rows [][]string - - // Domain secrets first - for _, key := range keys { - rows = append(rows, []string{key, styles.Theme.Muted.Render("(hidden)")}) - } - - // Attachment secrets with tree structure - for i, att := range attachments { - isLastAttachment := i == len(attachments)-1 - rows = append(rows, buildAttachmentRows(att, isLastAttachment)...) - } - - return rows -} - -// buildAttachmentRows builds table rows for a single attachment with tree structure. -func buildAttachmentRows(att remote.AttachmentSecrets, isLastAttachment bool) [][]string { - var rows [][]string - - // Attachment header with tree prefix - prefix := styles.IconTreeBranch + styles.IconTreeLine - if isLastAttachment { - prefix = styles.IconTreeLast + styles.IconTreeLine - } - - serviceName := extractServiceName(att.Service) - attachmentHeader := fmt.Sprintf("%s %s", prefix, styles.Theme.Muted.Render(fmt.Sprintf("[%s]", serviceName))) - rows = append(rows, []string{attachmentHeader, ""}) - - // Keys for this attachment with nested tree structure - for j, key := range att.Keys { - isLastKey := j == len(att.Keys)-1 - keyPrefix := getKeyPrefix(isLastAttachment, isLastKey) - rows = append(rows, []string{keyPrefix + " " + key, styles.Theme.Muted.Render("(hidden)")}) - } - - return rows -} - -// extractServiceName extracts a short service name from a container name. -// e.g., "gordon-git-bnema-dev-gitea-postgres" → "gitea-postgres" -func extractServiceName(containerName string) string { - if !strings.HasPrefix(containerName, "gordon-") { - return containerName - } - - parts := strings.SplitN(containerName, "-", 2) - if len(parts) <= 1 { - return containerName - } - - serviceName := parts[1] - allParts := strings.Split(serviceName, "-") - - // Handle service names based on the number of segments explicitly - if len(allParts) < 2 { - // No additional segments; use the service name as-is. - return serviceName - } - if len(allParts) == 2 { - // Exactly two segments; the service name is already in the desired form. - return strings.Join(allParts, "-") - } - - // More than two segments: take the last two as the short service name (e.g., "gitea-postgres"). - return strings.Join(allParts[len(allParts)-2:], "-") -} - -// getKeyPrefix returns the tree prefix for a key based on its position. -func getKeyPrefix(isLastAttachment, isLastKey bool) string { - if isLastAttachment { - // Parent is last, use space continuation - if isLastKey { - return " " + styles.IconTreeLast + styles.IconTreeLine - } - return " " + styles.IconTreeBranch + styles.IconTreeLine - } - // Parent has siblings, use vertical line continuation - if isLastKey { - return styles.IconTreeVert + " " + styles.IconTreeLast + styles.IconTreeLine - } - return styles.IconTreeVert + " " + styles.IconTreeBranch + styles.IconTreeLine -} - -// newSecretsSetCmd creates the secrets set command. -func newSecretsSetCmd() *cobra.Command { - var attachment, fromFile string - var jsonOut bool - - cmd := &cobra.Command{ - Use: "set --from-file ", - Short: "Set secrets for a domain or attachment", - Long: `Set one or more secrets for a domain or an attachment container. - -Secrets are read from a mode 0600 file containing one KEY=value pair per line. - -Use --attachment to target an attachment service (e.g., postgres, redis) instead -of the main domain container. - -Examples: - gordon secrets set app.mydomain.com --from-file ./app.env - gordon secrets set app.mydomain.com --attachment postgres --from-file ./postgres.env`, - Args: cobra.ExactArgs(1), - RunE: func(cmd *cobra.Command, args []string) error { - ctx := cmd.Context() - secretDomain := args[0] - content, err := readProtectedSecretFile(fromFile) - if err != nil { - return fmt.Errorf("read secrets: %w", err) - } - - secrets := make(map[string]string) - for _, pair := range strings.Split(content, "\n") { - parts := strings.SplitN(pair, "=", 2) - if len(parts) != 2 { - return fmt.Errorf("invalid format: %s (expected KEY=value)", pair) - } - secrets[parts[0]] = parts[1] - } - - handle, err := resolveControlPlaneForRouteDomain(ctx, secretDomain) - if err != nil { - return err - } - defer handle.close() - if attachment != "" { - if err := handle.plane.SetAttachmentSecrets(ctx, secretDomain, attachment, secrets); err != nil { - return fmt.Errorf("failed to set secrets: %w", err) - } - } else { - if err := handle.plane.SetSecrets(ctx, secretDomain, secrets); err != nil { - return fmt.Errorf("failed to set secrets: %w", err) - } - } - - target := secretDomain - if attachment != "" { - target = fmt.Sprintf("%s [%s]", secretDomain, attachment) - } - - if jsonOut { - return writeJSON(cmd.OutOrStdout(), map[string]any{ - "domain": secretDomain, "attachment": attachment, "count": len(secrets), "updated": true, - }) - } - if len(secrets) == 1 { - for key := range secrets { - return cliWriteLine(cmd.OutOrStdout(), styles.RenderSuccess(fmt.Sprintf("Secret set: %s on %s", key, target))) - } - } - return cliWriteLine(cmd.OutOrStdout(), styles.RenderSuccess(fmt.Sprintf("Set %d secrets for %s", len(secrets), target))) - }, - } - - cmd.Flags().StringVarP(&attachment, "attachment", "a", "", "Target an attachment service (e.g., postgres, redis)") - cmd.Flags().StringVar(&fromFile, "from-file", "", "Read KEY=value lines from a mode 0600 file") - cmd.Flags().BoolVar(&jsonOut, "json", false, "Output as JSON") - _ = cmd.MarkFlagRequired("from-file") - - return cmd -} - -// newSecretsRemoveCmd creates the secrets remove command. -func newSecretsRemoveCmd() *cobra.Command { - var ( - force bool - attachment string - ) - - cmd := &cobra.Command{ - Use: "remove ", - Short: "Remove a secret", - Long: `Remove a secret from a domain or an attachment container. - -Use --attachment to target an attachment service (e.g., postgres, redis) instead -of the main domain container. - -Examples: - gordon secrets remove app.mydomain.com OLD_API_KEY - gordon secrets remove app.mydomain.com OLD_API_KEY --force - gordon secrets remove app.mydomain.com --attachment postgres POSTGRES_PASSWORD`, - Args: cobra.ExactArgs(2), - RunE: func(cmd *cobra.Command, args []string) error { - ctx := cmd.Context() - secretDomain := args[0] - key := args[1] - - target := secretDomain - if attachment != "" { - target = fmt.Sprintf("%s [%s]", secretDomain, attachment) - } - - // Confirm unless --force - if !force { - confirmed, err := components.RunConfirm( - fmt.Sprintf("Remove secret '%s' from %s?", key, target), - ) - if err != nil { - return err - } - if !confirmed { - fmt.Println(styles.Theme.Muted.Render("Cancelled")) - return nil - } - } - - handle, err := resolveControlPlaneForRouteDomain(ctx, secretDomain) - if err != nil { - return err - } - defer handle.close() - if attachment != "" { - if err := handle.plane.DeleteAttachmentSecret(ctx, secretDomain, attachment, key); err != nil { - return fmt.Errorf("failed to remove secret: %w", err) - } - } else { - if err := handle.plane.DeleteSecret(ctx, secretDomain, key); err != nil { - return fmt.Errorf("failed to remove secret: %w", err) - } - } - - fmt.Println(styles.RenderSuccess(fmt.Sprintf("Secret removed from %s: %s", target, key))) - return nil - }, - } - - cmd.Flags().BoolVarP(&force, "force", "f", false, "Skip confirmation") - cmd.Flags().StringVarP(&attachment, "attachment", "a", "", "Target an attachment service (e.g., postgres, redis)") - - return cmd -} diff --git a/internal/adapters/in/cli/serve.go b/internal/adapters/in/cli/serve.go index 324e13066..e2e2628b9 100644 --- a/internal/adapters/in/cli/serve.go +++ b/internal/adapters/in/cli/serve.go @@ -10,19 +10,19 @@ import ( // newServeCmd creates the serve command. func newServeCmd() *cobra.Command { - var configPath string + var cliConfigPath string cmd := &cobra.Command{ Use: "serve", Short: "Start the Gordon server", Long: `Start the Gordon server, including the registry and proxy components.`, RunE: func(cmd *cobra.Command, args []string) error { - return app.Run(context.Background(), configPath) + return app.Run(context.Background(), cliConfigPath) }, } // Add flags - cmd.Flags().StringVarP(&configPath, "config", "c", "", "Path to config file") + cmd.Flags().StringVarP(&cliConfigPath, "config", "c", "", "Path to config file") return cmd } diff --git a/internal/adapters/in/cli/status.go b/internal/adapters/in/cli/status.go new file mode 100644 index 000000000..b2ba3c4c0 --- /dev/null +++ b/internal/adapters/in/cli/status.go @@ -0,0 +1,89 @@ +package cli + +import ( + "context" + "fmt" + "io" + "sort" + + "github.com/spf13/cobra" + + "github.com/bnema/gordon/internal/adapters/in/cli/ui/components" + "github.com/bnema/gordon/internal/adapters/in/cli/ui/styles" +) + +// newStatusCmd creates the status command. +func newStatusCmd() *cobra.Command { + return &cobra.Command{ + Use: "status", + Short: "Show Gordon server status", + RunE: func(cmd *cobra.Command, args []string) error { + handle, err := resolveControlPlane(cliConfigPath) + if err != nil { + return err + } + defer handle.close() + + return runStatusCmd(cmd.Context(), handle.plane, cmd.OutOrStdout()) + }, + } +} + +func runStatusCmd(ctx context.Context, plane ControlPlane, out io.Writer) error { + status, err := plane.GetStatus(ctx) + if err != nil { + return fmt.Errorf("failed to get status: %w", err) + } + if status == nil { + return fmt.Errorf("failed to get status: empty response") + } + + if err := cliWriteLine(out, styles.Theme.Title.Render("Gordon Status")); err != nil { + return err + } + if err := cliWriteLine(out, ""); err != nil { + return err + } + + if err := cliWriteLine(out, fmt.Sprintf("%s %s", styles.Theme.Bold.Render("Domain:"), status.RegistryDomain)); err != nil { + return err + } + if err := cliWriteLine(out, fmt.Sprintf("%s %d", styles.Theme.Bold.Render("Registry Port:"), status.RegistryPort)); err != nil { + return err + } + if err := cliWriteLine(out, fmt.Sprintf("%s %d", styles.Theme.Bold.Render("Server Port:"), status.ServerPort)); err != nil { + return err + } + if err := cliWriteLine(out, fmt.Sprintf("%s %d", styles.Theme.Bold.Render("Apps:"), status.Apps)); err != nil { + return err + } + if err := cliWriteLine(out, fmt.Sprintf("%s %v", styles.Theme.Bold.Render("Network Isolation:"), status.NetworkIsolation)); err != nil { + return err + } + + return renderStatusFleet(out, status.ContainerStatus) +} + +func renderStatusFleet(out io.Writer, fleet map[string]string) error { + if len(fleet) == 0 { + return nil + } + if err := cliWriteLine(out, ""); err != nil { + return err + } + if err := cliWriteLine(out, styles.Theme.Bold.Render("Container Status:")); err != nil { + return err + } + names := make([]string, 0, len(fleet)) + for name := range fleet { + names = append(names, name) + } + sort.Strings(names) + for _, name := range names { + badge := components.ContainerStatusBadge(fleet[name]) + if err := cliWriteLine(out, fmt.Sprintf(" %s: %s", name, badge)); err != nil { + return err + } + } + return nil +} diff --git a/internal/adapters/in/cli/tls.go b/internal/adapters/in/cli/tls.go index f4296aef6..d52d5d224 100644 --- a/internal/adapters/in/cli/tls.go +++ b/internal/adapters/in/cli/tls.go @@ -16,24 +16,13 @@ import ( var tlsResolveControlPlane = resolveControlPlane func newTLSCmd() *cobra.Command { - cmd := &cobra.Command{ - Use: "tls", - Short: "Inspect TLS certificate status", - } - - cmd.AddCommand(newTLSStatusCmd()) - - return cmd -} - -func newTLSStatusCmd() *cobra.Command { var jsonOut bool cmd := &cobra.Command{ - Use: "status", + Use: "tls", Short: "Show public TLS certificate status", RunE: func(cmd *cobra.Command, _ []string) error { - handle, err := tlsResolveControlPlane(configPath) + handle, err := tlsResolveControlPlane(cliConfigPath) if err != nil { return err } diff --git a/internal/adapters/in/cli/traffic.go b/internal/adapters/in/cli/traffic.go deleted file mode 100644 index 62321cd1c..000000000 --- a/internal/adapters/in/cli/traffic.go +++ /dev/null @@ -1,12 +0,0 @@ -package cli - -import "github.com/spf13/cobra" - -func newTrafficCmd() *cobra.Command { - cmd := &cobra.Command{ - Use: "traffic", - Short: "Inspect remote traffic plane status", - } - cmd.AddCommand(newTrafficStatusCmd()) - return cmd -} diff --git a/internal/adapters/in/cli/traffic_status.go b/internal/adapters/in/cli/traffic_status.go index 558b318dd..3481f037c 100644 --- a/internal/adapters/in/cli/traffic_status.go +++ b/internal/adapters/in/cli/traffic_status.go @@ -13,14 +13,14 @@ import ( var trafficResolveControlPlane = resolveControlPlane -func newTrafficStatusCmd() *cobra.Command { +func newTrafficCmd() *cobra.Command { var jsonOut bool cmd := &cobra.Command{ - Use: "status", + Use: "traffic", Short: "Show traffic entrypoint, router, and counter status", Args: cobra.NoArgs, RunE: func(cmd *cobra.Command, args []string) error { - handle, err := trafficResolveControlPlane(configPath) + handle, err := trafficResolveControlPlane(cliConfigPath) if err != nil { return err } diff --git a/internal/adapters/in/cli/traffic_test.go b/internal/adapters/in/cli/traffic_test.go index 5cb5608fe..a7d1f76ce 100644 --- a/internal/adapters/in/cli/traffic_test.go +++ b/internal/adapters/in/cli/traffic_test.go @@ -4,7 +4,6 @@ import ( "bytes" "context" "encoding/json" - "errors" "net/http" "net/http/httptest" "testing" @@ -13,25 +12,21 @@ import ( "github.com/stretchr/testify/require" "github.com/bnema/gordon/internal/adapters/dto" + "github.com/bnema/gordon/internal/adapters/in/cli/mocks" "github.com/bnema/gordon/internal/adapters/in/cli/remote" "github.com/bnema/gordon/internal/domain" ) func TestTrafficCommandExists(t *testing.T) { cmd := NewRootCmd() - traffic, _, err := cmd.Find([]string{"traffic"}) + traffic, _, err := cmd.Find([]string{"daemon", "traffic"}) require.NoError(t, err) require.NotNil(t, traffic) assert.Equal(t, "traffic", traffic.Name()) - - status, _, err := cmd.Find([]string{"traffic", "status"}) - require.NoError(t, err) - require.NotNil(t, status) - assert.Equal(t, "status", status.Name()) } func TestTrafficStatusRejectsArgs(t *testing.T) { - cmd := newTrafficStatusCmd() + cmd := newTrafficCmd() require.Error(t, cmd.Args(cmd, []string{"extra"})) } @@ -81,12 +76,12 @@ func TestTrafficStatusJSONOutput(t *testing.T) { func TestRunTrafficStatusUsesControlPlane(t *testing.T) { status := &dto.TrafficStatusResponse{LastReloadStatus: "ok"} - cp := &trafficStatusPlane{status: status} + cp := mocks.NewMockControlPlane(t) + cp.EXPECT().GetTrafficStatus(context.Background()).Return(status, nil).Once() var buf bytes.Buffer require.NoError(t, runTrafficStatus(context.Background(), cp, &buf, true)) - assert.True(t, cp.called) assert.Contains(t, buf.String(), `"last_reload_status": "ok"`) } @@ -99,7 +94,7 @@ func TestRemoteTrafficStatusControlPlaneStillRenders(t *testing.T) { })) defer srv.Close() - cp := NewRemoteControlPlane(remote.NewClient(srv.URL)) + cp := remote.NewClient(srv.URL) var buf bytes.Buffer require.NoError(t, runTrafficStatus(context.Background(), cp, &buf, false)) @@ -109,24 +104,3 @@ func TestRemoteTrafficStatusControlPlaneStillRenders(t *testing.T) { assert.Contains(t, output, "ok") assert.Contains(t, output, "udp=3") } - -func TestLocalTrafficStatusReturnsActionableDaemonGuidance(t *testing.T) { - _, err := (&localControlPlane{}).GetTrafficStatus(context.Background()) - require.Error(t, err) - assert.True(t, errors.Is(err, domain.ErrTrafficStatusUnavailable)) - assert.Contains(t, err.Error(), "in-process CLI control plane") - assert.Contains(t, err.Error(), "running Gordon daemon") - assert.Contains(t, err.Error(), "--remote") - assert.Contains(t, err.Error(), "GORDON_REMOTE") -} - -type trafficStatusPlane struct { - ControlPlane - status *dto.TrafficStatusResponse - called bool -} - -func (p *trafficStatusPlane) GetTrafficStatus(context.Context) (*dto.TrafficStatusResponse, error) { - p.called = true - return p.status, nil -} diff --git a/internal/adapters/in/cli/ui_adoption_test.go b/internal/adapters/in/cli/ui_adoption_test.go index 96bcdc1df..1f14cd498 100644 --- a/internal/adapters/in/cli/ui_adoption_test.go +++ b/internal/adapters/in/cli/ui_adoption_test.go @@ -3,31 +3,16 @@ package cli import ( "bytes" "context" - "go/ast" - "go/parser" - "go/token" "io" - "strings" "sync" "testing" + "github.com/stretchr/testify/mock" + "github.com/bnema/gordon/internal/adapters/dto" climocks "github.com/bnema/gordon/internal/adapters/in/cli/mocks" ) -var uiAdoptionHelperCalls = map[string]struct{}{ - "cliWriteLine": {}, - "cliWritef": {}, - "cliRenderTitle": {}, - "cliRenderMuted": {}, - "cliRenderEmptyState": {}, - "cliRenderListItem": {}, - "cliRenderMeta": {}, - "cliRenderSuccess": {}, - "cliRenderWarning": {}, - "cliRenderInfo": {}, -} - // uiAdoptionSeamMu guards mutations of the package-level cliWriteLine and // cliWritef variables. Any test that overrides these seams must hold this // mutex for the duration of the override and restore the originals on cleanup. @@ -54,102 +39,6 @@ func TestPresentationHelpers(t *testing.T) { } } -func TestUIAdoption(t *testing.T) { - for i := range uiAdoptionExpectations { - expect := uiAdoptionExpectations[i] - t.Run(expect.family, func(t *testing.T) { - fset := token.NewFileSet() - fileNode, err := parser.ParseFile(fset, expect.file, nil, parser.AllErrors) - if err != nil { - t.Fatalf("failed to parse %s: %v", expect.file, err) - } - - for _, fnName := range expect.functions { - t.Run(fnName, func(t *testing.T) { - fn := findFuncDecl(fileNode, fnName) - if fn == nil { - t.Fatalf("function %s not found in %s", fnName, expect.file) - } - - hasHelperCall := false - hasSharedUIUsage := false - hasForbiddenRawPrint := false - - ast.Inspect(fn.Body, func(n ast.Node) bool { - call, ok := n.(*ast.CallExpr) - if !ok { - return true - } - - switch fun := call.Fun.(type) { - case *ast.Ident: - if _, ok := uiAdoptionHelperCalls[fun.Name]; ok { - hasHelperCall = true - } - case *ast.SelectorExpr: - if isForbiddenRawPrintCall(call, fun) { - hasForbiddenRawPrint = true - } - pkgIdent, ok := fun.X.(*ast.Ident) - if !ok { - return true - } - if pkgIdent.Name == "styles" || pkgIdent.Name == "components" { - hasSharedUIUsage = true - } - } - - return true - }) - - if !hasHelperCall && !hasSharedUIUsage { - t.Fatalf("%s in %s does not use presentation helpers or shared ui styles/components", fnName, expect.file) - } - - if hasForbiddenRawPrint { - t.Fatalf("%s in %s still uses forbidden raw print calls", fnName, expect.file) - } - }) - } - }) - } -} - -func isForbiddenRawPrintCall(call *ast.CallExpr, sel *ast.SelectorExpr) bool { - pkgIdent, ok := sel.X.(*ast.Ident) - if !ok { - return false - } - - if pkgIdent.Name == "fmt" { - switch sel.Sel.Name { - case "Print", "Printf", "Println": - return true - case "Fprint", "Fprintf", "Fprintln": - if len(call.Args) == 0 { - return true - } - - switch dst := call.Args[0].(type) { - case *ast.SelectorExpr: - if dstPkg, ok := dst.X.(*ast.Ident); ok && dstPkg.Name == "os" && (dst.Sel.Name == "Stdout" || dst.Sel.Name == "Stderr") { - return true - } - case *ast.Ident: - if dst.Name == "out" { - return true - } - } - } - } - - if pkgIdent.Name == "cmd" && strings.HasPrefix(sel.Sel.Name, "Print") { - return true - } - - return false -} - func TestUIAdoptionRuntimeSeams(t *testing.T) { uiAdoptionSeamMu.Lock() defer uiAdoptionSeamMu.Unlock() @@ -190,7 +79,7 @@ func TestUIAdoptionRuntimeSeams(t *testing.T) { } pruneMock := climocks.NewMockimagesClient(t) - pruneMock.EXPECT().ListImages(context.Background()).Return([]dto.Image{}, nil).Once() + pruneMock.EXPECT().PruneImages(context.Background(), mock.Anything).Return(&dto.ImagePruneResponse{Plan: dto.PruneSummary{}}, nil).Once() pruneBuf := new(bytes.Buffer) if err := runImagesPrune(context.Background(), pruneMock, imagesPruneOptions{DryRun: true}, pruneBuf); err != nil { @@ -204,16 +93,3 @@ func TestUIAdoptionRuntimeSeams(t *testing.T) { t.Fatal("expected cliWritef to be called") } } - -func findFuncDecl(file *ast.File, name string) *ast.FuncDecl { - for _, decl := range file.Decls { - fn, ok := decl.(*ast.FuncDecl) - if !ok { - continue - } - if fn.Name.Name == name { - return fn - } - } - return nil -} diff --git a/internal/adapters/in/cli/volumes.go b/internal/adapters/in/cli/volumes.go index 8d806f889..8dfe24de8 100644 --- a/internal/adapters/in/cli/volumes.go +++ b/internal/adapters/in/cli/volumes.go @@ -40,7 +40,7 @@ func newVolumesListCmd() *cobra.Command { Use: "list", Short: "List all volumes", RunE: func(cmd *cobra.Command, _ []string) error { - handle, err := resolveControlPlane(configPath) + handle, err := resolveControlPlane(cliConfigPath) if err != nil { return err } @@ -56,9 +56,14 @@ func newVolumesPruneCmd() *cobra.Command { var opts volumesPruneOptions cmd := &cobra.Command{ Use: "prune", - Short: "Remove orphaned volumes", + Short: "Remove released app-owned volumes", + Long: `Remove app-owned volumes that were explicitly released and are unused. + +Only a volume whose durable ownership record marks it released, whose runtime +labels agree with that record, and which no container mounts is removed. +Attached, retained, legacy-managed, unmanaged, and unknown volumes survive.`, RunE: func(cmd *cobra.Command, _ []string) error { - handle, err := resolveControlPlane(configPath) + handle, err := resolveControlPlane(cliConfigPath) if err != nil { return err } @@ -66,7 +71,7 @@ func newVolumesPruneCmd() *cobra.Command { return runVolumesPrune(cmd.Context(), handle.plane, opts, cmd.OutOrStdout()) }, } - cmd.Flags().BoolVar(&opts.DryRun, "dry-run", false, "show what would be removed") + cmd.Flags().BoolVar(&opts.DryRun, "dry-run", false, "report the plan without deleting") cmd.Flags().BoolVar(&opts.NoConfirm, "no-confirm", false, "skip confirmation prompt") cmd.Flags().BoolVar(&opts.Json, "json", false, "output as JSON") return cmd @@ -133,33 +138,27 @@ func runVolumesPrune(ctx context.Context, client volumesClient, opts volumesPrun return err } - if preview.VolumesRemoved == 0 { - if opts.Json { - return writeJSON(out, preview) - } - return cliWriteLine(out, "No orphaned volumes to remove.") - } - if opts.DryRun { - if opts.Json { - return writeJSON(out, preview) - } - return renderPrunePreview(out, preview) + return renderVolumePrunePlan(out, opts.Json, preview) + } + if preview.Plan.Eligible == 0 { + // A zero-eligibility plan is a normal outcome: the report explains + // why each volume survived. + return renderVolumePrunePlan(out, opts.Json, preview) } if !opts.Json { if err := renderPrunePreview(out, preview); err != nil { return err } - } - - if !opts.NoConfirm && !opts.Json { - confirmed, err := components.RunConfirm("Remove these volumes?") - if err != nil { - return err - } - if !confirmed { - return cliWriteLine(out, "Cancelled.") + if !opts.NoConfirm { + confirmed, err := components.RunConfirm("Remove these volumes?") + if err != nil { + return err + } + if !confirmed { + return cliWriteLine(out, "Cancelled.") + } } } @@ -167,10 +166,25 @@ func runVolumesPrune(ctx context.Context, client volumesClient, opts volumesPrun if err != nil { return err } - if opts.Json { return writeJSON(out, result) } - return cliWriteLine(out, cliRenderSuccess(fmt.Sprintf("Removed %d volumes, reclaimed %s", result.VolumesRemoved, bytesize.Format(result.SpaceReclaimed)))) + if err := cliWriteLine(out, cliRenderSuccess(fmt.Sprintf("Removed %d volumes, reclaimed %s", result.VolumesRemoved, bytesize.Format(result.SpaceReclaimed)))); err != nil { + return err + } + return cliRenderPruneSummary(out, result.Plan) +} + +// renderVolumePrunePlan writes one plan without deleting anything. +func renderVolumePrunePlan(out io.Writer, asJSON bool, resp *dto.VolumePruneResponse) error { + if asJSON { + return writeJSON(out, resp) + } + if resp.Plan.Eligible == 0 { + if err := cliWriteLine(out, "No released volumes to remove."); err != nil { + return err + } + } + return cliRenderPruneSummary(out, resp.Plan) } diff --git a/internal/adapters/in/cli/volumes_test.go b/internal/adapters/in/cli/volumes_test.go index 2aab490c5..fc6fbbcb6 100644 --- a/internal/adapters/in/cli/volumes_test.go +++ b/internal/adapters/in/cli/volumes_test.go @@ -41,13 +41,29 @@ func TestRunVolumesList_JSON(t *testing.T) { func TestRunVolumesPrune_JSONExecutesActualPrune(t *testing.T) { clientMock := climocks.NewMockvolumesClient(t) - resp := &dto.VolumePruneResponse{ + preview := &dto.VolumePruneResponse{Plan: dto.PruneSummary{ + Eligible: 1, + Candidates: []dto.PruneCandidate{{ + Kind: "volume", Ref: "orphan1", Verdict: "eligible", + }}, + }} + result := &dto.VolumePruneResponse{ VolumesRemoved: 1, SpaceReclaimed: 1024, Volumes: []dto.Volume{{Name: "orphan1", Size: 1024}}, + Plan: dto.PruneSummary{ + Applied: true, + Eligible: 1, + Candidates: []dto.PruneCandidate{{ + Kind: "volume", Ref: "orphan1", Verdict: "eligible", + }}, + Deleted: []dto.PruneCandidate{{ + Kind: "volume", Ref: "orphan1", Verdict: "eligible", + }}, + }, } - previewCall := clientMock.EXPECT().PruneVolumes(context.Background(), dto.VolumePruneRequest{DryRun: true}).Return(resp, nil).Once() - pruneCall := clientMock.EXPECT().PruneVolumes(context.Background(), dto.VolumePruneRequest{DryRun: false}).Return(resp, nil).Once() + previewCall := clientMock.EXPECT().PruneVolumes(context.Background(), dto.VolumePruneRequest{DryRun: true}).Return(preview, nil).Once() + pruneCall := clientMock.EXPECT().PruneVolumes(context.Background(), dto.VolumePruneRequest{DryRun: false}).Return(result, nil).Once() mock.InOrder(previewCall, pruneCall) var out bytes.Buffer @@ -59,16 +75,17 @@ func TestRunVolumesPrune_JSONExecutesActualPrune(t *testing.T) { func TestRunVolumesPrune_DryRun(t *testing.T) { clientMock := climocks.NewMockvolumesClient(t) clientMock.EXPECT().PruneVolumes(context.Background(), dto.VolumePruneRequest{DryRun: true}).Return(&dto.VolumePruneResponse{ - VolumesRemoved: 2, - SpaceReclaimed: 4096, - Volumes: []dto.Volume{ - {Name: "orphan1", Size: 2048}, - {Name: "orphan2", Size: 2048}, + Plan: dto.PruneSummary{ + Eligible: 2, + Candidates: []dto.PruneCandidate{ + {Kind: "volume", Ref: "orphan1", Verdict: "eligible"}, + {Kind: "volume", Ref: "orphan2", Verdict: "eligible"}, + }, }, }, nil).Once() var out bytes.Buffer err := runVolumesPrune(context.Background(), clientMock, volumesPruneOptions{DryRun: true}, &out) require.NoError(t, err) - assert.Contains(t, out.String(), "orphan1") + assert.Contains(t, out.String(), "Would delete: 2") } diff --git a/internal/adapters/in/http/admin/bootstrap_test.go b/internal/adapters/in/http/admin/bootstrap_test.go deleted file mode 100644 index a3f2e2040..000000000 --- a/internal/adapters/in/http/admin/bootstrap_test.go +++ /dev/null @@ -1,413 +0,0 @@ -package admin - -import ( - "bytes" - "encoding/json" - "errors" - "io" - "net/http" - "testing" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/mock" - "github.com/stretchr/testify/require" - - "github.com/bnema/gordon/internal/adapters/dto" - inmocks "github.com/bnema/gordon/internal/boundaries/in/mocks" - "github.com/bnema/gordon/internal/domain" -) - -func TestBootstrap_FullWorkflow_HappyPath(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - registrySvc := inmocks.NewMockRegistryService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - d.RegistrySvc = registrySvc - }) - server := newScopedTestServer(t, handler, "admin:routes:write", "admin:secrets:write", "admin:config:write") - - payload := dto.BootstrapRequest{ - Domain: "app.example.com", - Image: "myapp:latest", - Attachments: []string{"postgres:16"}, - Env: map[string]string{"APP_ENV": "prod"}, - AttachmentEnv: map[string]map[string]string{ - "postgres": {"POSTGRES_PASSWORD": "secret"}, - }, - } - - configSvc.EXPECT().GetRegistryDomain().Return("reg.bnema.dev").Once() - configSvc.EXPECT().AddRoute(mock.Anything, domain.Route{Domain: payload.Domain, Image: "reg.bnema.dev/myapp", HTTPS: true}).Return(nil).Once() - configSvc.EXPECT().AddAttachment(mock.Anything, payload.Domain, "postgres:16").Return(nil).Once() - secretSvc.EXPECT().Set(mock.Anything, payload.Domain, map[string]string{"APP_ENV": "prod"}).Return(nil).Once() - secretSvc.EXPECT().SetAttachment(mock.Anything, payload.Domain, "postgres", map[string]string{"POSTGRES_PASSWORD": "secret"}).Return(nil).Once() - - body, err := json.Marshal(payload) - require.NoError(t, err) - - req, err := http.NewRequest(http.MethodPost, server.URL+"/admin/bootstrap", bytes.NewReader(body)) - require.NoError(t, err) - - resp, err := server.Client().Do(req) - require.NoError(t, err) - defer resp.Body.Close() - - assert.Equal(t, http.StatusOK, resp.StatusCode) - registrySvc.AssertNotCalled(t, "GetManifest", mock.Anything, mock.Anything, mock.Anything) - - var response dto.BootstrapResponse - require.NoError(t, json.NewDecoder(resp.Body).Decode(&response)) - assert.Equal(t, payload.Domain, response.Domain) - assert.Equal(t, "reg.bnema.dev/myapp", response.Image) - assert.Equal(t, "push reg.bnema.dev/myapp to trigger deployment", response.Next) - assert.Equal(t, []dto.BootstrapStep{ - {Name: "route", Status: "configured"}, - {Name: "attachment:postgres:16", Status: "created"}, - {Name: "env", Status: "updated"}, - {Name: "attachment_env:postgres", Status: "updated"}, - }, response.Steps) -} - -func TestHandleBootstrap_DefaultsHTTPSWhenOmitted(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - server := newScopedTestServer(t, handler, "admin:routes:write", "admin:config:write") - - payload := dto.BootstrapRequest{Domain: "app.example.com", Image: "myapp:latest"} - configSvc.EXPECT().GetRegistryDomain().Return("reg.bnema.dev").Once() - configSvc.EXPECT().AddRoute(mock.Anything, domain.Route{Domain: payload.Domain, Image: "reg.bnema.dev/myapp", HTTPS: true}).Return(nil).Once() - - body, err := json.Marshal(payload) - require.NoError(t, err) - - req, err := http.NewRequest(http.MethodPost, server.URL+"/admin/bootstrap", bytes.NewReader(body)) - require.NoError(t, err) - - resp, err := server.Client().Do(req) - require.NoError(t, err) - defer resp.Body.Close() - - assert.Equal(t, http.StatusOK, resp.StatusCode) - - var response dto.BootstrapResponse - require.NoError(t, json.NewDecoder(resp.Body).Decode(&response)) - assert.Equal(t, payload.Domain, response.Domain) - assert.Equal(t, "reg.bnema.dev/myapp", response.Image) - assert.Equal(t, "push reg.bnema.dev/myapp to trigger deployment", response.Next) -} - -func TestBootstrap_Idempotent_Rerun(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - server := newScopedTestServer(t, handler, "admin:routes:write", "admin:secrets:write", "admin:config:write") - - payload := dto.BootstrapRequest{ - Domain: "app.example.com", - Image: "myapp:latest", - Attachments: []string{"postgres:16"}, - Env: map[string]string{"APP_ENV": "prod"}, - AttachmentEnv: map[string]map[string]string{ - "postgres": {"POSTGRES_PASSWORD": "secret"}, - }, - } - - configSvc.EXPECT().GetRegistryDomain().Return("reg.bnema.dev").Times(2) - configSvc.EXPECT().AddRoute(mock.Anything, domain.Route{Domain: payload.Domain, Image: "reg.bnema.dev/myapp", HTTPS: true}).Return(nil).Once() - configSvc.EXPECT().AddAttachment(mock.Anything, payload.Domain, "postgres:16").Return(nil).Once() - secretSvc.EXPECT().Set(mock.Anything, payload.Domain, map[string]string{"APP_ENV": "prod"}).Return(nil).Once() - secretSvc.EXPECT().SetAttachment(mock.Anything, payload.Domain, "postgres", map[string]string{"POSTGRES_PASSWORD": "secret"}).Return(nil).Once() - - configSvc.EXPECT().AddRoute(mock.Anything, domain.Route{Domain: payload.Domain, Image: "reg.bnema.dev/myapp", HTTPS: true}).Return(nil).Once() - configSvc.EXPECT().AddAttachment(mock.Anything, payload.Domain, "postgres:16").Return(domain.ErrAttachmentExists).Once() - secretSvc.EXPECT().Set(mock.Anything, payload.Domain, map[string]string{"APP_ENV": "prod"}).Return(nil).Once() - secretSvc.EXPECT().SetAttachment(mock.Anything, payload.Domain, "postgres", map[string]string{"POSTGRES_PASSWORD": "secret"}).Return(nil).Once() - - body, err := json.Marshal(payload) - require.NoError(t, err) - - for i := range 2 { - req, err := http.NewRequest(http.MethodPost, server.URL+"/admin/bootstrap", bytes.NewReader(body)) - require.NoError(t, err) - - resp, err := server.Client().Do(req) - require.NoError(t, err) - defer resp.Body.Close() - - assert.Equal(t, http.StatusOK, resp.StatusCode) - - var response dto.BootstrapResponse - require.NoError(t, json.NewDecoder(resp.Body).Decode(&response)) - assert.Equal(t, payload.Domain, response.Domain) - assert.Equal(t, "reg.bnema.dev/myapp", response.Image) - require.Len(t, response.Steps, 4) - if i == 0 { - assert.Equal(t, []dto.BootstrapStep{ - {Name: "route", Status: "configured"}, - {Name: "attachment:postgres:16", Status: "created"}, - {Name: "env", Status: "updated"}, - {Name: "attachment_env:postgres", Status: "updated"}, - }, response.Steps) - } else { - assert.Equal(t, []dto.BootstrapStep{ - {Name: "route", Status: "configured"}, - {Name: "attachment:postgres:16", Status: "noop"}, - {Name: "env", Status: "updated"}, - {Name: "attachment_env:postgres", Status: "updated"}, - }, response.Steps) - } - } -} - -func TestBootstrap_MissingDomain(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - server := newScopedTestServer(t, handler, "admin:routes:write", "admin:config:write") - - body, err := json.Marshal(dto.BootstrapRequest{Image: "myapp:latest"}) - require.NoError(t, err) - - req, err := http.NewRequest(http.MethodPost, server.URL+"/admin/bootstrap", bytes.NewReader(body)) - require.NoError(t, err) - - resp, err := server.Client().Do(req) - require.NoError(t, err) - defer resp.Body.Close() - - assert.Equal(t, http.StatusBadRequest, resp.StatusCode) - bodyBytes, err := io.ReadAll(resp.Body) - require.NoError(t, err) - assert.JSONEq(t, `{"error":"domain is required"}`+"\n", string(bodyBytes)) -} - -func TestBootstrap_MissingImage(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - server := newScopedTestServer(t, handler, "admin:routes:write", "admin:config:write") - - body, err := json.Marshal(dto.BootstrapRequest{Domain: "app.example.com"}) - require.NoError(t, err) - - req, err := http.NewRequest(http.MethodPost, server.URL+"/admin/bootstrap", bytes.NewReader(body)) - require.NoError(t, err) - - resp, err := server.Client().Do(req) - require.NoError(t, err) - defer resp.Body.Close() - - assert.Equal(t, http.StatusBadRequest, resp.StatusCode) - bodyBytes, err := io.ReadAll(resp.Body) - require.NoError(t, err) - assert.JSONEq(t, `{"error":"image is required"}`+"\n", string(bodyBytes)) -} - -func TestBootstrap_NormalizesImageBeforeStoring(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - server := newScopedTestServer(t, handler, "admin:routes:write", "admin:config:write") - - payload := dto.BootstrapRequest{Domain: "app.example.com", Image: "pitlane:v1"} - configSvc.EXPECT().GetRegistryDomain().Return("reg.bnema.dev").Once() - configSvc.EXPECT().AddRoute(mock.Anything, domain.Route{Domain: payload.Domain, Image: "reg.bnema.dev/pitlane", HTTPS: true}).Return(nil).Once() - - body, err := json.Marshal(payload) - require.NoError(t, err) - - req, err := http.NewRequest(http.MethodPost, server.URL+"/admin/bootstrap", bytes.NewReader(body)) - require.NoError(t, err) - - resp, err := server.Client().Do(req) - require.NoError(t, err) - defer resp.Body.Close() - - assert.Equal(t, http.StatusOK, resp.StatusCode) - - var response dto.BootstrapResponse - require.NoError(t, json.NewDecoder(resp.Body).Decode(&response)) - assert.Equal(t, payload.Domain, response.Domain) - assert.Equal(t, "reg.bnema.dev/pitlane", response.Image) - assert.Equal(t, "push reg.bnema.dev/pitlane to trigger deployment", response.Next) -} - -func TestBootstrap_AddRouteValidationErrors(t *testing.T) { - tests := []struct { - name string - mockErr error - }{ - { - name: "empty domain from service", - mockErr: domain.ErrRouteDomainEmpty, - }, - { - name: "empty image from service", - mockErr: domain.ErrRouteImageEmpty, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - server := newScopedTestServer(t, handler, "admin:routes:write", "admin:config:write") - - payload := dto.BootstrapRequest{Domain: "app.example.com", Image: "myapp:latest"} - configSvc.EXPECT().GetRegistryDomain().Return("reg.bnema.dev").Once() - configSvc.EXPECT().AddRoute(mock.Anything, domain.Route{Domain: payload.Domain, Image: "reg.bnema.dev/myapp", HTTPS: true}).Return(tt.mockErr).Once() - - body, err := json.Marshal(payload) - require.NoError(t, err) - - req, err := http.NewRequest(http.MethodPost, server.URL+"/admin/bootstrap", bytes.NewReader(body)) - require.NoError(t, err) - - resp, err := server.Client().Do(req) - require.NoError(t, err) - defer resp.Body.Close() - - assert.Equal(t, http.StatusBadRequest, resp.StatusCode) - bodyBytes, err := io.ReadAll(resp.Body) - require.NoError(t, err) - assert.JSONEq(t, `{"error":"`+tt.mockErr.Error()+`"}`+"\n", string(bodyBytes)) - }) - } -} - -func TestBootstrap_InvalidRouteDomain(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - server := newScopedTestServer(t, handler, "admin:routes:write", "admin:config:write") - - payload := dto.BootstrapRequest{Domain: "invalid domain", Image: "myapp:latest"} - configSvc.EXPECT().GetRegistryDomain().Return("reg.bnema.dev").Once() - configSvc.EXPECT().AddRoute(mock.Anything, domain.Route{Domain: payload.Domain, Image: "reg.bnema.dev/myapp", HTTPS: true}).Return(domain.ErrRouteDomainInvalid).Once() - - body, err := json.Marshal(payload) - require.NoError(t, err) - - req, err := http.NewRequest(http.MethodPost, server.URL+"/admin/bootstrap", bytes.NewReader(body)) - require.NoError(t, err) - - resp, err := server.Client().Do(req) - require.NoError(t, err) - defer resp.Body.Close() - - assert.Equal(t, http.StatusBadRequest, resp.StatusCode) - bodyBytes, err := io.ReadAll(resp.Body) - require.NoError(t, err) - assert.JSONEq(t, `{"error":"`+domain.ErrRouteDomainInvalid.Error()+`"}`+"\n", string(bodyBytes)) -} - -func TestBootstrap_PartialFailure_AttachmentError(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - server := newScopedTestServer(t, handler, "admin:routes:write", "admin:config:write") - - payload := dto.BootstrapRequest{ - Domain: "app.example.com", - Image: "myapp:latest", - Attachments: []string{"postgres:16"}, - } - - configSvc.EXPECT().GetRegistryDomain().Return("reg.bnema.dev").Once() - configSvc.EXPECT().AddRoute(mock.Anything, domain.Route{Domain: payload.Domain, Image: "reg.bnema.dev/myapp", HTTPS: true}).Return(nil).Once() - configSvc.EXPECT().AddAttachment(mock.Anything, payload.Domain, "postgres:16").Return(errors.New("attachment failed")).Once() - - body, err := json.Marshal(payload) - require.NoError(t, err) - - req, err := http.NewRequest(http.MethodPost, server.URL+"/admin/bootstrap", bytes.NewReader(body)) - require.NoError(t, err) - - resp, err := server.Client().Do(req) - require.NoError(t, err) - defer resp.Body.Close() - - assert.Equal(t, http.StatusInternalServerError, resp.StatusCode) - - var response dto.BootstrapResponse - require.NoError(t, json.NewDecoder(resp.Body).Decode(&response)) - assert.Equal(t, []dto.BootstrapStep{ - {Name: "route", Status: "configured"}, - {Name: "attachment:postgres:16", Status: "failed"}, - }, response.Steps) - secretSvc.AssertNotCalled(t, "Set", mock.Anything, mock.Anything, mock.Anything) - secretSvc.AssertNotCalled(t, "SetAttachment", mock.Anything, mock.Anything, mock.Anything, mock.Anything) -} diff --git a/internal/adapters/in/http/admin/handler.go b/internal/adapters/in/http/admin/handler.go index 22afa00c4..b0933fc32 100644 --- a/internal/adapters/in/http/admin/handler.go +++ b/internal/adapters/in/http/admin/handler.go @@ -18,7 +18,6 @@ import ( "github.com/bnema/gordon/internal/adapters/dto" "github.com/bnema/gordon/internal/boundaries/in" "github.com/bnema/gordon/internal/domain" - "github.com/bnema/gordon/internal/usecase/config" "github.com/bnema/gordon/internal/usecase/registry" "github.com/bnema/gordon/pkg/validation" ) @@ -29,11 +28,6 @@ const maxAdminRequestSize = 1 << 20 // 1MB // maxLogLines is the maximum allowed number of log lines that can be requested. const maxLogLines = 10000 -type registryDeployService interface { - in.RegistryService - in.DeployCoordinator -} - type reloadTrigger interface { Trigger(ctx context.Context) error } @@ -47,77 +41,16 @@ type Handler struct { volumeBackupSvc in.VolumeBackupService imageSvc in.ImageService healthSvc in.HealthService - secretSvc in.SecretService logSvc in.LogService volumeSvc in.VolumeService - registrySvc registryDeployService - previewSvc previewService + registrySvc in.RegistryService reloadTrigger reloadTrigger publicTLSSvc in.PublicTLSService trafficSvc in.TrafficStatusService + appSvc in.AppService log zerowrap.Logger } -// Type aliases for API responses using shared DTO types. -type routeInfoResponse = dto.RouteInfo -type attachmentResponse = dto.Attachment -type routeResponse = dto.Route - -// toAttachmentResponse converts a domain.Attachment to a dto.Attachment. -func toAttachmentResponse(a domain.Attachment) dto.Attachment { - return dto.Attachment{ - Name: a.Name, - Image: a.Image, - ContainerID: a.ContainerID, - Status: a.Status, - Network: a.Network, - } -} - -// toRouteInfoResponse converts a domain.RouteInfo to a dto.RouteInfo. -func toRouteInfoResponse(r domain.RouteInfo) dto.RouteInfo { - attachments := make([]dto.Attachment, 0, len(r.Attachments)) - for _, a := range r.Attachments { - attachments = append(attachments, toAttachmentResponse(a)) - } - return dto.RouteInfo{ - Domain: r.Domain, - Image: r.Image, - ContainerID: r.ContainerID, - ContainerStatus: r.ContainerStatus, - Network: r.Network, - Attachments: attachments, - } -} - -// toRouteResponse converts a domain.Route to a dto.Route. -func toRouteResponse(r domain.Route) dto.Route { - return dto.Route{ - Domain: r.Domain, - Image: r.Image, - HTTPS: r.HTTPS, - } -} - -func mergeConfiguredRouteDetails(configured []domain.Route, detailed []domain.RouteInfo) []domain.RouteInfo { - detailedByDomain := make(map[string]domain.RouteInfo, len(detailed)) - for _, info := range detailed { - detailedByDomain[info.Domain] = info - } - - merged := make([]domain.RouteInfo, 0, len(configured)) - for _, route := range configured { - info, ok := detailedByDomain[route.Domain] - if !ok { - info = domain.RouteInfo{Domain: route.Domain, Image: route.Image} - } else if info.Image == "" { - info.Image = route.Image - } - merged = append(merged, info) - } - return merged -} - func toBackupJobResponse(job domain.BackupJob) dto.BackupJob { var startedAt *time.Time if !job.StartedAt.IsZero() { @@ -132,8 +65,9 @@ func toBackupJobResponse(job domain.BackupJob) dto.BackupJob { return dto.BackupJob{ ID: job.ID, - Domain: job.Domain, - DBName: job.DBName, + App: job.App, + Service: job.Service, + Database: job.DBName, Schedule: string(job.Schedule), Type: string(job.Type), Status: string(job.Status), @@ -157,32 +91,20 @@ func toVolumeBackupJobResponse(job domain.VolumeBackupJob) dto.VolumeBackupJob { } return dto.VolumeBackupJob{ - ID: job.ID, - Domain: job.Domain, - ContainerName: job.ContainerName, - ContainerID: job.ContainerID, - VolumeName: job.VolumeName, - MountPath: job.MountPath, - Compression: job.Metadata["compression"], - Type: string(job.Type), - Status: string(job.Status), - StartedAt: startedAt, - CompletedAt: completedAt, - SizeBytes: job.SizeBytes, - ArtifactRef: job.ArtifactRef, - Error: job.Error, - } -} - -func toDatabaseInfoResponse(db domain.DBInfo) dto.DatabaseInfo { - return dto.DatabaseInfo{ - Type: string(db.Type), - Name: db.Name, - Version: db.Version, - Host: db.Host, - Port: db.Port, - ContainerID: db.ContainerID, - ImageName: db.ImageName, + ID: job.ID, + App: job.App, + Service: job.Service, + VolumeName: job.VolumeName, + RuntimeVolumeName: job.RuntimeVolumeName, + MountPath: job.MountPath, + Compression: job.Metadata["compression"], + Type: string(job.Type), + Status: string(job.Status), + StartedAt: startedAt, + CompletedAt: completedAt, + SizeBytes: job.SizeBytes, + ArtifactRef: job.ArtifactRef, + Error: job.Error, } } @@ -192,18 +114,17 @@ type HandlerDeps struct { AuthSvc in.AuthService ContainerSvc in.ContainerService HealthSvc in.HealthService - SecretSvc in.SecretService LogSvc in.LogService - RegistrySvc registryDeployService + RegistrySvc in.RegistryService Log zerowrap.Logger BackupSvc in.BackupService VolumeBackupSvc in.VolumeBackupService - PreviewSvc previewService ImageSvc in.ImageService VolumeSvc in.VolumeService ReloadTrigger reloadTrigger PublicTLSSvc in.PublicTLSService TrafficSvc in.TrafficStatusService + AppSvc in.AppService } // NewHandler creates a new admin HTTP handler. @@ -216,14 +137,13 @@ func NewHandler(deps HandlerDeps) *Handler { volumeBackupSvc: deps.VolumeBackupSvc, imageSvc: deps.ImageSvc, healthSvc: deps.HealthSvc, - secretSvc: deps.SecretSvc, logSvc: deps.LogSvc, volumeSvc: deps.VolumeSvc, registrySvc: deps.RegistrySvc, - previewSvc: deps.PreviewSvc, reloadTrigger: deps.ReloadTrigger, publicTLSSvc: deps.PublicTLSSvc, trafficSvc: deps.TrafficSvc, + appSvc: deps.AppSvc, log: deps.Log, } } @@ -263,21 +183,22 @@ type routeHandler func(w http.ResponseWriter, r *http.Request, path string) // matchRoute returns the handler for a given path, or false if not found. func (h *Handler) matchRoute(path string) (routeHandler, bool) { + // Retired legacy mutations answer 410 Gone before any other match. + if isRetiredMutation(path) { + return h.handleRetiredMutation, true + } // Exact match routes exactRoutes := map[string]routeHandler{ - "/networks": func(w http.ResponseWriter, r *http.Request, _ string) { h.handleNetworks(w, r) }, - "/status": func(w http.ResponseWriter, r *http.Request, _ string) { h.handleStatus(w, r) }, - "/health": func(w http.ResponseWriter, r *http.Request, _ string) { h.handleHealth(w, r) }, - "/bootstrap": func(w http.ResponseWriter, r *http.Request, _ string) { h.handleBootstrap(w, r) }, - "/reload": func(w http.ResponseWriter, r *http.Request, _ string) { h.handleReload(w, r) }, - "/config": func(w http.ResponseWriter, r *http.Request, _ string) { h.handleConfig(w, r) }, - "/auth/verify": func(w http.ResponseWriter, r *http.Request, _ string) { h.handleAuthVerify(w, r) }, - "/volumes": func(w http.ResponseWriter, r *http.Request, _ string) { h.handleListVolumes(w, r) }, - "/volumes/prune": func(w http.ResponseWriter, r *http.Request, _ string) { h.handlePruneVolumes(w, r) }, - "/attachments/orphans": func(w http.ResponseWriter, r *http.Request, _ string) { h.handleAttachmentOrphans(w, r) }, - "/attachments/prune": func(w http.ResponseWriter, r *http.Request, _ string) { h.handleAttachmentPrune(w, r) }, - "/tls/status": func(w http.ResponseWriter, r *http.Request, _ string) { h.handleTLSStatus(w, r) }, - "/traffic/status": func(w http.ResponseWriter, r *http.Request, _ string) { h.handleTrafficStatus(w, r) }, + "/networks": func(w http.ResponseWriter, r *http.Request, _ string) { h.handleNetworks(w, r) }, + "/status": func(w http.ResponseWriter, r *http.Request, _ string) { h.handleStatus(w, r) }, + "/health": func(w http.ResponseWriter, r *http.Request, _ string) { h.handleHealth(w, r) }, + "/reload": func(w http.ResponseWriter, r *http.Request, _ string) { h.handleReload(w, r) }, + "/config": func(w http.ResponseWriter, r *http.Request, _ string) { h.handleConfig(w, r) }, + "/auth/verify": func(w http.ResponseWriter, r *http.Request, _ string) { h.handleAuthVerify(w, r) }, + "/volumes": func(w http.ResponseWriter, r *http.Request, _ string) { h.handleListVolumes(w, r) }, + "/volumes/prune": func(w http.ResponseWriter, r *http.Request, _ string) { h.handlePruneVolumes(w, r) }, + "/tls/status": func(w http.ResponseWriter, r *http.Request, _ string) { h.handleTLSStatus(w, r) }, + "/traffic/status": func(w http.ResponseWriter, r *http.Request, _ string) { h.handleTrafficStatus(w, r) }, } if handler, ok := exactRoutes[path]; ok { return handler, true @@ -288,21 +209,11 @@ func (h *Handler) matchRoute(path string) (routeHandler, bool) { prefix string handler routeHandler }{ + {"/apps", h.handleApps}, {"/backups", h.handleBackups}, - {"/attachments/by-image", h.handleAttachmentsByImage}, - {"/attachments", h.handleAttachmentsConfig}, - {"/routes/by-image", h.handleRoutesByImage}, - {"/routes", h.handleRoutes}, - {"/secrets", h.handleSecrets}, - {"/deploy-intent", h.handleDeployIntent}, - {"/deploy", h.handleDeploy}, - {"/restart", h.handleRestart}, {"/tags", h.handleTags}, {"/images", h.handleImages}, {"/logs", h.handleLogs}, - {"/autoroute/allowed-domains", h.handleAutoRouteAllowedDomains}, - {"/previews", h.handlePreviewList}, - {"/preview", h.handlePreviewAction}, } for _, route := range prefixRoutes { if path == route.prefix || strings.HasPrefix(path, route.prefix+"/") { @@ -313,288 +224,6 @@ func (h *Handler) matchRoute(path string) (routeHandler, bool) { return nil, false } -func (h *Handler) handleAttachmentOrphans(w http.ResponseWriter, r *http.Request) { - ctx := r.Context() - if r.Method != http.MethodGet { - h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") - return - } - if !HasAccess(ctx, domain.AdminResourceConfig, domain.AdminActionRead) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for config:read") - return - } - if h.containerSvc == nil { - log := zerowrap.FromCtx(ctx) - log.Error().Msg("container service not available for orphaned attachment listing") - h.sendError(w, http.StatusInternalServerError, "container service not available") - return - } - attachments, err := h.containerSvc.ListOrphanedAttachments(ctx) - if err != nil { - log := zerowrap.FromCtx(ctx) - log.Error().Err(err).Msg("failed to list orphaned attachments") - h.sendError(w, http.StatusInternalServerError, "failed to list orphaned attachments") - return - } - h.sendJSON(w, http.StatusOK, map[string]any{"attachments": attachments}) -} - -func (h *Handler) handleAttachmentPrune(w http.ResponseWriter, r *http.Request) { - ctx := r.Context() - if r.Method != http.MethodPost { - h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") - return - } - if !HasAccess(ctx, domain.AdminResourceConfig, domain.AdminActionWrite) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for config:write") - return - } - if h.containerSvc == nil { - log := zerowrap.FromCtx(ctx) - log.Error().Msg("container service not available for orphaned attachment cleanup") - h.sendError(w, http.StatusInternalServerError, "container service not available") - return - } - stop := r.URL.Query().Get("stop") == "true" - owner := r.URL.Query().Get("owner") - var report *domain.CleanupReport - var err error - report, err = h.containerSvc.CleanupOrphanedAttachments(ctx, owner, stop) - if err != nil { - log := zerowrap.FromCtx(ctx) - log.Error().Err(err).Str("owner", owner).Bool("stop", stop).Msg("failed to cleanup orphaned attachments") - h.sendError(w, http.StatusInternalServerError, "failed to cleanup orphaned attachments") - return - } - h.sendJSON(w, http.StatusOK, report) -} - -// handleAttachmentsByImage handles GET /admin/attachments/by-image/{image} endpoint. -// Returns all attachment targets associated with the given image name. -func (h *Handler) handleAttachmentsByImage(w http.ResponseWriter, r *http.Request, path string) { - if r.Method != http.MethodGet { - h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") - return - } - - ctx := r.Context() - - if !HasAccess(ctx, domain.AdminResourceConfig, domain.AdminActionRead) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for config:read") - return - } - - imageName := strings.TrimPrefix(path, "/attachments/by-image/") - if imageName == "" || imageName == "/attachments/by-image" { - h.sendError(w, http.StatusBadRequest, "image name required in path") - return - } - - imageName, err := url.PathUnescape(imageName) - if err != nil { - h.sendError(w, http.StatusBadRequest, "invalid image name encoding") - return - } - - targets := h.configSvc.FindAttachmentTargetsByImage(ctx, imageName) - - h.sendJSON(w, http.StatusOK, dto.AttachmentTargetsByImageResponse{ - Image: imageName, - Targets: targets, - }) -} - -// handleDeployIntent handles /admin/deploy-intent/:image endpoint. -// It registers a deploy intent, suppressing event-based deploys for the image. -func (h *Handler) handleDeployIntent(w http.ResponseWriter, r *http.Request, path string) { - if r.Method != http.MethodPost { - h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") - return - } - - ctx := r.Context() - if !HasAccess(ctx, domain.AdminResourceConfig, domain.AdminActionWrite) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for config:write") - return - } - - if h.registrySvc == nil { - h.sendError(w, http.StatusServiceUnavailable, "registry service unavailable") - return - } - - rawName := strings.TrimPrefix(path, "/deploy-intent/") - if rawName == "" || rawName == "/deploy-intent" { - h.sendError(w, http.StatusBadRequest, "image name required") - return - } - - imageName, err := url.PathUnescape(rawName) - if err != nil { - h.sendError(w, http.StatusBadRequest, "invalid image name encoding") - return - } - - log := zerowrap.FromCtx(ctx) - log.Info().Str("image", imageName).Msg("deploy intent registered, suppressing image.pushed events") - - h.registrySvc.SuppressDeployEvent(imageName) - - h.sendJSON(w, http.StatusOK, map[string]string{ - "status": "ok", - "image": imageName, - }) -} - -func (h *Handler) handleBootstrap(w http.ResponseWriter, r *http.Request) { - if r.Method != http.MethodPost { - h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") - return - } - - ctx := r.Context() - log := zerowrap.FromCtx(ctx) - - if !HasAccess(ctx, domain.AdminResourceRoutes, domain.AdminActionWrite) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for routes:write") - return - } - - r.Body = http.MaxBytesReader(w, r.Body, maxAdminRequestSize) - - var req dto.BootstrapRequest - if err := json.NewDecoder(r.Body).Decode(&req); err != nil { - log.Warn().Err(err).Msg("invalid bootstrap JSON") - h.sendError(w, http.StatusBadRequest, "invalid JSON") - return - } - - if req.Domain == "" { - h.sendError(w, http.StatusBadRequest, "domain is required") - return - } - - if req.Image == "" { - h.sendError(w, http.StatusBadRequest, "image is required") - return - } - - normalizedImage, err := config.NormalizeBootstrapImage(req.Image, h.configSvc.GetRegistryDomain()) - if err != nil { - h.sendError(w, http.StatusBadRequest, err.Error()) - return - } - - if err := h.validateBootstrapPermissions(ctx, req); err != nil { - h.sendError(w, http.StatusForbidden, err.Error()) - return - } - - resp := dto.BootstrapResponse{ - Domain: req.Domain, - Image: normalizedImage, - Next: fmt.Sprintf("push %s to trigger deployment", normalizedImage), - } - addStep := func(name, status string) { - resp.Steps = append(resp.Steps, dto.BootstrapStep{Name: name, Status: status}) - } - - err = h.configSvc.AddRoute(ctx, domain.Route{Domain: req.Domain, Image: normalizedImage, HTTPS: true}) - switch { - case err == nil: - addStep("route", "configured") - case errors.Is(err, domain.ErrRouteDomainEmpty), errors.Is(err, domain.ErrRouteDomainInvalid), errors.Is(err, domain.ErrRouteImageEmpty): - addStep("route", "failed") - h.sendError(w, http.StatusBadRequest, err.Error()) - return - case errors.Is(err, domain.ErrRouteConflict): - addStep("route", "failed") - h.sendError(w, http.StatusConflict, err.Error()) - return - default: - addStep("route", "failed") - log.Error().Err(err).Str("domain", req.Domain).Str("image", req.Image).Msg("failed to bootstrap route") - h.sendJSON(w, http.StatusInternalServerError, resp) - return - } - - if err := h.bootstrapAttachments(ctx, h.configSvc, req.Domain, req.Attachments, &resp.Steps); err != nil { - log.Error().Err(err).Str("domain", req.Domain).Msg("failed to bootstrap attachments") - h.sendJSON(w, http.StatusInternalServerError, resp) - return - } - - if err := h.bootstrapSecrets(ctx, h.secretSvc, req.Domain, req.Env, &resp.Steps); err != nil { - log.Error().Err(err).Str("domain", req.Domain).Msg("failed to bootstrap env") - h.sendJSON(w, http.StatusInternalServerError, resp) - return - } - - if err := h.bootstrapAttachmentSecrets(ctx, h.secretSvc, req.Domain, req.AttachmentEnv, &resp.Steps); err != nil { - log.Error().Err(err).Str("domain", req.Domain).Msg("failed to bootstrap attachment env") - h.sendJSON(w, http.StatusInternalServerError, resp) - return - } - - h.sendJSON(w, http.StatusOK, resp) -} - -func (h *Handler) validateBootstrapPermissions(ctx context.Context, req dto.BootstrapRequest) error { - if (len(req.Env) > 0 || len(req.AttachmentEnv) > 0) && !HasAccess(ctx, domain.AdminResourceSecrets, domain.AdminActionWrite) { - return errors.New("insufficient permissions for secrets:write") - } - if len(req.Attachments) > 0 && !HasAccess(ctx, domain.AdminResourceConfig, domain.AdminActionWrite) { - return errors.New("insufficient permissions for config:write") - } - - return nil -} - -func (h *Handler) bootstrapAttachments(ctx context.Context, configSvc in.ConfigService, domainName string, attachments []string, steps *[]dto.BootstrapStep) error { - for _, attachment := range attachments { - err := configSvc.AddAttachment(ctx, domainName, attachment) - switch { - case err == nil: - *steps = append(*steps, dto.BootstrapStep{Name: "attachment:" + attachment, Status: "created"}) - case errors.Is(err, domain.ErrAttachmentExists): - *steps = append(*steps, dto.BootstrapStep{Name: "attachment:" + attachment, Status: "noop"}) - default: - *steps = append(*steps, dto.BootstrapStep{Name: "attachment:" + attachment, Status: "failed"}) - return err - } - } - - return nil -} - -func (h *Handler) bootstrapSecrets(ctx context.Context, secretSvc in.SecretService, domainName string, env map[string]string, steps *[]dto.BootstrapStep) error { - if len(env) == 0 { - return nil - } - - if err := secretSvc.Set(ctx, domainName, env); err != nil { - *steps = append(*steps, dto.BootstrapStep{Name: "env", Status: "failed"}) - return err - } - - *steps = append(*steps, dto.BootstrapStep{Name: "env", Status: "updated"}) - return nil -} - -func (h *Handler) bootstrapAttachmentSecrets(ctx context.Context, secretSvc in.SecretService, domainName string, attachmentEnv map[string]map[string]string, steps *[]dto.BootstrapStep) error { - for service, env := range attachmentEnv { - if err := secretSvc.SetAttachment(ctx, domainName, service, env); err != nil { - *steps = append(*steps, dto.BootstrapStep{Name: "attachment_env:" + service, Status: "failed"}) - return err - } - - *steps = append(*steps, dto.BootstrapStep{Name: "attachment_env:" + service, Status: "updated"}) - } - - return nil -} - -// sendJSON sends a JSON response. func (h *Handler) sendJSON(w http.ResponseWriter, status int, data any) { w.Header().Set("Content-Type", "application/json") w.WriteHeader(status) @@ -609,667 +238,38 @@ func (h *Handler) sendError(w http.ResponseWriter, status int, message string) { } // handleRoutes handles /admin/routes endpoints. -func (h *Handler) handleRoutes(w http.ResponseWriter, r *http.Request, path string) { - // Parse domain from path if present - routeDomain := strings.TrimPrefix(path, "/routes/") - if routeDomain == "/routes" { - routeDomain = "" - } - - switch r.Method { - case http.MethodGet: - h.handleRoutesGet(w, r, routeDomain) - case http.MethodPost: - h.handleRoutesPost(w, r) - case http.MethodPut: - h.handleRoutesPut(w, r, routeDomain) - case http.MethodDelete: - h.handleRoutesDelete(w, r, routeDomain) - default: - h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") - } -} - -func (h *Handler) handleRoutesGet(w http.ResponseWriter, r *http.Request, routeDomain string) { - ctx := r.Context() - - // Check read permission - if !HasAccess(ctx, domain.AdminResourceRoutes, domain.AdminActionRead) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for routes:read") - return - } - - if routeDomain == "" { - if r.URL.Query().Get("detailed") == "true" { - if err := h.containerSvc.SyncContainers(ctx); err != nil { - h.log.Error().Err(err).Msg("failed to sync containers before detailed route listing") - h.sendError(w, http.StatusInternalServerError, "failed to list routes") - return - } - configured := h.configSvc.GetRoutes(ctx) - routes := mergeConfiguredRouteDetails(configured, h.containerSvc.ListRoutesWithDetails(ctx)) - response := make([]routeInfoResponse, 0, len(routes)) - for _, route := range routes { - response = append(response, toRouteInfoResponse(route)) - } - h.sendJSON(w, http.StatusOK, dto.RoutesDetailResponse{Routes: response}) - return - } - - routes := h.configSvc.GetRoutes(ctx) - response := make([]routeResponse, 0, len(routes)) - for _, route := range routes { - response = append(response, toRouteResponse(route)) - } - h.sendJSON(w, http.StatusOK, dto.RoutesResponse{Routes: response}) - return - } - - if parentDomain, ok := strings.CutSuffix(routeDomain, "/cleanup"); ok { - if parentDomain == "" { - h.sendError(w, http.StatusBadRequest, "domain required in path") - return - } - h.handleRouteCleanupPreview(w, r, parentDomain) - return - } - - if parentDomain, ok := strings.CutSuffix(routeDomain, "/attachments"); ok { - if parentDomain == "" { - h.sendError(w, http.StatusBadRequest, "domain required in path") - return - } - attachments := h.containerSvc.ListAttachments(ctx, parentDomain) - response := make([]attachmentResponse, 0, len(attachments)) - for _, attachment := range attachments { - response = append(response, toAttachmentResponse(attachment)) - } - h.sendJSON(w, http.StatusOK, dto.AttachmentsResponse{Attachments: response}) - return - } - - route, err := h.configSvc.GetRoute(ctx, routeDomain) - if err != nil { - h.sendError(w, http.StatusNotFound, "route not found") - return - } - h.sendJSON(w, http.StatusOK, toRouteResponse(*route)) -} - -func (h *Handler) handleRouteCleanupPreview(w http.ResponseWriter, r *http.Request, routeDomain string) { - ctx := r.Context() - log := zerowrap.FromCtx(ctx) - - if !HasAccess(ctx, domain.AdminResourceConfig, domain.AdminActionRead) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for config:read") - return - } - if !HasAccess(ctx, domain.AdminResourceVolumes, domain.AdminActionRead) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for volumes:read") - return - } - - previewer, ok := any(h.containerSvc).(interface { - PreviewRemovedRouteCleanup(context.Context, string) (*domain.CleanupReport, error) - }) - if !ok { - log.Error().Str("domain", routeDomain).Msg("route cleanup preview unavailable") - h.sendError(w, http.StatusInternalServerError, "route cleanup preview unavailable") - return - } - - report, err := previewer.PreviewRemovedRouteCleanup(ctx, routeDomain) - if err != nil { - switch { - case errors.Is(err, domain.ErrRouteDomainInvalid), errors.Is(err, domain.ErrRouteDomainEmpty): - h.sendError(w, http.StatusBadRequest, "invalid route domain") - default: - log.Error().Err(err).Str("domain", routeDomain).Msg("failed to inspect route cleanup state") - h.sendError(w, http.StatusInternalServerError, "failed to inspect route cleanup state") - } - return - } - - if h.volumeSvc == nil { - log.Error().Str("domain", routeDomain).Msg("volume service not available for route cleanup preview") - h.sendError(w, http.StatusServiceUnavailable, "volume service not available") - return - } - - volumes, err := h.volumeSvc.ListVolumes(ctx) - if err != nil { - log.Error().Err(err).Str("domain", routeDomain).Msg("failed to list volumes for route cleanup preview") - h.sendError(w, http.StatusInternalServerError, "failed to inspect route cleanup state") - return - } - report.PreservedVolumes = append( - report.PreservedVolumes, - matchingRouteCleanupVolumes(routeDomain, h.routeCleanupVolumePrefix(), volumes, report.PreservedAttachments)..., - ) - - h.sendJSON(w, http.StatusOK, dto.CleanupReportFromDomain(report)) -} - -func (h *Handler) routeCleanupVolumePrefix() string { - const defaultPrefix = "gordon" - if h.configSvc == nil { - return defaultPrefix - } - if volumeCfg, ok := any(h.configSvc).(interface{ GetVolumeConfig() (bool, string, bool) }); ok { - _, prefix, _ := volumeCfg.GetVolumeConfig() - if prefix != "" { - return prefix - } - } - return defaultPrefix -} - -type routeCleanupVolumeScope struct { - containerNames map[string]struct{} - volumePrefixes []string -} - -func newRouteCleanupVolumeScope(routeDomain, volumePrefix string, attachments []domain.CleanupAttachment) routeCleanupVolumeScope { - if volumePrefix == "" { - volumePrefix = "gordon" - } - scope := routeCleanupVolumeScope{ - containerNames: make(map[string]struct{}), - volumePrefixes: []string{volumePrefix + "-" + strings.ReplaceAll(routeDomain, ".", "-") + "-"}, - } - for _, name := range []string{ - fmt.Sprintf("gordon-%s", routeDomain), - fmt.Sprintf("gordon-%s-new", routeDomain), - fmt.Sprintf("gordon-%s-next", routeDomain), - } { - scope.containerNames[name] = struct{}{} - } - for _, attachment := range attachments { - if attachment.Name == "" { - continue - } - scope.containerNames[attachment.Name] = struct{}{} - scope.volumePrefixes = append(scope.volumePrefixes, volumePrefix+"-"+attachment.Name+"-") - for _, name := range routeAttachmentContainerNamesForCleanup(routeDomain, attachment.Name) { - scope.containerNames[name] = struct{}{} - scope.volumePrefixes = append(scope.volumePrefixes, volumePrefix+"-"+name+"-") - } - } - return scope -} - -func routeAttachmentContainerNamesForCleanup(routeDomain, serviceName string) []string { - if serviceName == "" { - return nil - } - return []string{ - fmt.Sprintf("gordon-%s-%s", domain.SanitizeDomainForContainer(routeDomain), serviceName), - fmt.Sprintf("gordon-%s-%s", domain.SanitizeDomainForContainerLegacy(routeDomain), serviceName), - } -} - -func matchingRouteCleanupVolumes(routeDomain, volumePrefix string, volumes []*domain.VolumeInfo, attachments []domain.CleanupAttachment) []domain.CleanupVolume { - scope := newRouteCleanupVolumeScope(routeDomain, volumePrefix, attachments) - matched := make([]domain.CleanupVolume, 0) - for _, volume := range volumes { - if volume == nil || !scope.matches(volume) { - continue - } - matched = append(matched, domain.CleanupVolume{ - Name: volume.Name, - ContainerPath: "", - Reason: "volume preserved for explicit cleanup review", - }) - } - return matched -} - -func (s routeCleanupVolumeScope) matches(volume *domain.VolumeInfo) bool { - if volume == nil { - return false - } - for _, containerName := range volume.Containers { - if _, ok := s.containerNames[containerName]; ok { - return true - } - } - for _, prefix := range s.volumePrefixes { - if strings.HasPrefix(volume.Name, prefix) { - return true - } - } - return false -} - -func (h *Handler) handleRoutesPost(w http.ResponseWriter, r *http.Request) { - ctx := r.Context() - log := zerowrap.FromCtx(ctx) - - // Check write permission - if !HasAccess(ctx, domain.AdminResourceRoutes, domain.AdminActionWrite) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for routes:write") - return - } - - // Limit request body size - r.Body = http.MaxBytesReader(w, r.Body, maxAdminRequestSize) - - var req struct { - Domain string `json:"domain"` - Image string `json:"image"` - HTTPS *bool `json:"https"` - } - if err := json.NewDecoder(r.Body).Decode(&req); err != nil { - log.Warn().Err(err).Msg("invalid route JSON") - h.sendError(w, http.StatusBadRequest, "invalid JSON") - return - } - - route := domain.Route{Domain: req.Domain, Image: req.Image, HTTPS: true} - if req.HTTPS != nil { - route.HTTPS = *req.HTTPS - } - - if err := h.configSvc.AddRoute(ctx, route); err != nil { - log.Error().Err(err).Str("domain", route.Domain).Msg("failed to add route") - switch { - case errors.Is(err, domain.ErrRouteDomainEmpty), errors.Is(err, domain.ErrRouteDomainInvalid), errors.Is(err, domain.ErrRouteImageEmpty): - h.sendError(w, http.StatusBadRequest, err.Error()) - case errors.Is(err, domain.ErrRouteConflict): - h.sendError(w, http.StatusConflict, err.Error()) - default: - h.sendError(w, http.StatusInternalServerError, "failed to add route") - } - return - } - - log.Info().Str("domain", route.Domain).Str("image", route.Image).Msg("route added") - h.sendJSON(w, http.StatusCreated, toRouteResponse(route)) -} - -func (h *Handler) handleRoutesPut(w http.ResponseWriter, r *http.Request, routeDomain string) { - ctx := r.Context() - log := zerowrap.FromCtx(ctx) - - // Check write permission - if !HasAccess(ctx, domain.AdminResourceRoutes, domain.AdminActionWrite) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for routes:write") - return - } - - if routeDomain == "" { - h.sendError(w, http.StatusBadRequest, "domain required in path") - return - } - - // Limit request body size - r.Body = http.MaxBytesReader(w, r.Body, maxAdminRequestSize) - - var req struct { - Image string `json:"image"` - HTTPS *bool `json:"https"` - } - if err := json.NewDecoder(r.Body).Decode(&req); err != nil { - log.Warn().Err(err).Msg("invalid route JSON") - h.sendError(w, http.StatusBadRequest, "invalid JSON") - return - } - - storedRoute, err := h.configSvc.GetRoute(ctx, routeDomain) - if err != nil { - if errors.Is(err, domain.ErrRouteNotFound) { - h.sendError(w, http.StatusNotFound, "route not found") - return - } - log.Error().Err(err).Str("domain", routeDomain).Msg("failed to load route") - h.sendError(w, http.StatusInternalServerError, "failed to update route") - return - } - if storedRoute == nil { - h.sendError(w, http.StatusNotFound, "route not found") - return - } - - route := *storedRoute - route.Domain = routeDomain - route.Image = req.Image - if req.HTTPS != nil { - route.HTTPS = *req.HTTPS - } - - if err := h.configSvc.UpdateRoute(ctx, route); err != nil { - log.Error().Err(err).Str("domain", routeDomain).Msg("failed to update route") - switch { - case errors.Is(err, domain.ErrRouteNotFound): - h.sendError(w, http.StatusNotFound, "route not found") - case errors.Is(err, domain.ErrRouteDomainEmpty), errors.Is(err, domain.ErrRouteDomainInvalid), errors.Is(err, domain.ErrRouteImageEmpty): - h.sendError(w, http.StatusBadRequest, err.Error()) - default: - h.sendError(w, http.StatusInternalServerError, "failed to update route") - } - return - } - - log.Info().Str("domain", route.Domain).Str("image", route.Image).Msg("route updated") - h.sendJSON(w, http.StatusOK, toRouteResponse(route)) -} - -func (h *Handler) handleRoutesDelete(w http.ResponseWriter, r *http.Request, routeDomain string) { - ctx := r.Context() - log := zerowrap.FromCtx(ctx) - - // Check write permission - if !HasAccess(ctx, domain.AdminResourceRoutes, domain.AdminActionWrite) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for routes:write") - return - } - - if routeDomain == "" { - h.sendError(w, http.StatusBadRequest, "domain required in path") - return - } - - if err := h.configSvc.RemoveRoute(ctx, routeDomain); err != nil { - switch { - case errors.Is(err, domain.ErrRouteNotFound): - log.Debug().Str("domain", routeDomain).Msg("route already absent, reconciling runtime state") - case errors.Is(err, domain.ErrRouteDomainEmpty), errors.Is(err, domain.ErrRouteDomainInvalid): - log.Error().Err(err).Str("domain", routeDomain).Msg("failed to remove route") - h.sendError(w, http.StatusBadRequest, "invalid route domain") - return - default: - log.Error().Err(err).Str("domain", routeDomain).Msg("failed to remove route") - h.sendError(w, http.StatusInternalServerError, "failed to remove route") - return - } - } - - var cleanup *dto.CleanupReport - if h.containerSvc != nil { - report, err := h.containerSvc.ReconcileRemovedRoute(ctx, routeDomain) - if err != nil { - log.Error().Err(err).Str("domain", routeDomain).Msg("failed to cleanup removed route runtime state") - h.sendError(w, http.StatusInternalServerError, "route removed but runtime cleanup failed") - return - } - cleanup = dto.CleanupReportFromDomain(report) - } - - log.Info().Str("domain", routeDomain).Msg("route removed") - h.sendJSON(w, http.StatusOK, dto.RouteDeleteResponse{Status: "removed", Cleanup: cleanup}) -} - -// handleRoutesByImage handles GET /admin/routes/by-image/{image} endpoint. -// Returns all routes associated with the given image name. -func (h *Handler) handleRoutesByImage(w http.ResponseWriter, r *http.Request, path string) { - if r.Method != http.MethodGet { - h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") - return - } - - ctx := r.Context() - - if !HasAccess(ctx, domain.AdminResourceRoutes, domain.AdminActionRead) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for routes:read") - return - } - - rawImageName := strings.TrimPrefix(path, "/routes/by-image/") - if rawImageName == "" || rawImageName == "/routes/by-image" { - h.sendError(w, http.StatusBadRequest, "image name required in path") - return - } - - imageName, err := url.PathUnescape(rawImageName) - if err != nil { - h.sendError(w, http.StatusBadRequest, "invalid image name encoding") - return - } - - routes := h.configSvc.FindRoutesByImage(ctx, imageName) - - response := make([]routeResponse, 0, len(routes)) - for _, route := range routes { - response = append(response, toRouteResponse(route)) - } - - h.sendJSON(w, http.StatusOK, dto.RoutesByImageResponse{ - Image: imageName, - Routes: response, - }) -} - func (h *Handler) handleNetworks(w http.ResponseWriter, r *http.Request) { if r.Method != http.MethodGet { h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") - return - } - - ctx := r.Context() - - if !HasAccess(ctx, domain.AdminResourceStatus, domain.AdminActionRead) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for status:read") - return - } - - networks, err := h.containerSvc.ListNetworks(ctx) - if err != nil { - h.sendError(w, http.StatusInternalServerError, "failed to list networks") - return - } - - response := make([]dto.Network, 0, len(networks)) - for _, network := range networks { - if network == nil { - continue - } - response = append(response, dto.Network{ - Name: network.Name, - Driver: network.Driver, - Containers: append([]string{}, network.Containers...), - }) - } - - h.sendJSON(w, http.StatusOK, dto.NetworksResponse{Networks: response}) -} - -// handleSecrets handles /admin/secrets endpoints. -func (h *Handler) handleSecrets(w http.ResponseWriter, r *http.Request, path string) { - // Parse path: /secrets/{domain} or /secrets/{domain}/{key} - parts := strings.Split(strings.TrimPrefix(path, "/secrets/"), "/") - if len(parts) == 0 || parts[0] == "" { - h.sendError(w, http.StatusBadRequest, "domain required") - return - } - - secretDomain := parts[0] - - // Check for attachment sub-path: /secrets/{domain}/attachments/{service}[/{key}] - if len(parts) >= 2 && parts[1] == "attachments" { - switch r.Method { - case http.MethodPost: - // Expected path: /secrets/{domain}/attachments/{service} - if len(parts) != 3 { - h.sendError(w, http.StatusBadRequest, "invalid attachment path: expected /secrets/{domain}/attachments/{service}") - return - } - service := parts[2] - h.handleAttachmentSecrets(w, r, secretDomain, service, "") - return - case http.MethodDelete: - // Expected path: /secrets/{domain}/attachments/{service}/{key} - if len(parts) != 4 { - h.sendError(w, http.StatusBadRequest, "invalid attachment path: expected /secrets/{domain}/attachments/{service}/{key}") - return - } - service := parts[2] - attachmentKey := parts[3] - h.handleAttachmentSecrets(w, r, secretDomain, service, attachmentKey) - return - default: - h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") - return - } - } - - secretKey := "" - if len(parts) > 1 { - secretKey = parts[1] - } - - switch r.Method { - case http.MethodGet: - h.handleSecretsGet(w, r, secretDomain) - case http.MethodPost: - h.handleSecretsPost(w, r, secretDomain) - case http.MethodDelete: - h.handleSecretsDelete(w, r, secretDomain, secretKey) - default: - h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") - } -} - -// handleSecretsPost handles POST /admin/secrets/{domain} - set secrets. -func (h *Handler) handleSecretsPost(w http.ResponseWriter, r *http.Request, secretDomain string) { - ctx := r.Context() - log := zerowrap.FromCtx(ctx) - - if !HasAccess(ctx, domain.AdminResourceSecrets, domain.AdminActionWrite) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for secrets:write") - return - } - - r.Body = http.MaxBytesReader(w, r.Body, maxAdminRequestSize) - - var data map[string]string - if err := json.NewDecoder(r.Body).Decode(&data); err != nil { - log.Warn().Err(err).Msg("invalid secrets JSON") - h.sendError(w, http.StatusBadRequest, "invalid JSON") - return - } - - if err := h.secretSvc.Set(ctx, secretDomain, data); err != nil { - log.Error().Err(err).Str("domain", secretDomain).Msg("failed to set secrets") - h.sendError(w, http.StatusBadRequest, "invalid domain") - return - } - - log.Info().Str("domain", secretDomain).Int("count", len(data)).Msg("secrets set") - h.sendJSON(w, http.StatusOK, dto.SecretsStatusResponse{Status: "updated"}) -} - -// handleSecretsDelete handles DELETE /admin/secrets/{domain}/{key} - delete a secret. -func (h *Handler) handleSecretsDelete(w http.ResponseWriter, r *http.Request, secretDomain, secretKey string) { - ctx := r.Context() - log := zerowrap.FromCtx(ctx) - - if !HasAccess(ctx, domain.AdminResourceSecrets, domain.AdminActionWrite) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for secrets:write") - return - } - if secretKey == "" { - h.sendError(w, http.StatusBadRequest, "key required in path") - return - } - - if err := h.secretSvc.Delete(ctx, secretDomain, secretKey); err != nil { - log.Error().Err(err).Str("domain", secretDomain).Str("key", secretKey).Msg("failed to delete secret") - h.sendError(w, http.StatusBadRequest, "invalid domain") - return - } - - log.Info().Str("domain", secretDomain).Str("key", secretKey).Msg("secret deleted") - h.sendJSON(w, http.StatusOK, dto.SecretsStatusResponse{Status: "deleted"}) -} - -// handleSecretsGet handles GET /admin/secrets/{domain} - list secrets. -func (h *Handler) handleSecretsGet(w http.ResponseWriter, r *http.Request, secretDomain string) { - ctx := r.Context() - log := zerowrap.FromCtx(ctx) - - // Check read permission - if !HasAccess(ctx, domain.AdminResourceSecrets, domain.AdminActionRead) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for secrets:read") - return - } - - // List secrets for domain (names only, not values) including attachments - keys, attachments, err := h.secretSvc.ListKeysWithAttachments(ctx, secretDomain) - if err != nil { - log.Error().Err(err).Str("domain", secretDomain).Msg("failed to list secrets") - h.sendError(w, http.StatusBadRequest, "invalid domain") - return - } - - // Convert attachments to DTO format - var attachmentDTOs []dto.AttachmentSecretsResponse - for _, att := range attachments { - attachmentDTOs = append(attachmentDTOs, dto.AttachmentSecretsResponse{ - Service: att.Service, - Keys: att.Keys, - }) - } - - h.sendJSON(w, http.StatusOK, dto.SecretsListResponse{ - Domain: secretDomain, - Keys: keys, - Attachments: attachmentDTOs, - }) -} - -// handleAttachmentSecrets handles /admin/secrets/{domain}/attachments/{service}[/{key}] endpoints. -func (h *Handler) handleAttachmentSecrets(w http.ResponseWriter, r *http.Request, secretDomain, service, key string) { - ctx := r.Context() - log := zerowrap.FromCtx(ctx) - - switch r.Method { - case http.MethodPost: - if !HasAccess(ctx, domain.AdminResourceSecrets, domain.AdminActionWrite) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for secrets:write") - return - } - - r.Body = http.MaxBytesReader(w, r.Body, maxAdminRequestSize) - - var data map[string]string - if err := json.NewDecoder(r.Body).Decode(&data); err != nil { - log.Warn().Err(err).Msg("invalid attachment secrets JSON") - h.sendError(w, http.StatusBadRequest, "invalid JSON") - return - } - - if err := h.secretSvc.SetAttachment(ctx, secretDomain, service, data); err != nil { - log.Error().Err(err).Str("domain", secretDomain).Str("service", service).Msg("failed to set attachment secrets") - h.sendError(w, http.StatusBadRequest, "invalid domain or service") - return - } + return + } - log.Info().Str("domain", secretDomain).Str("service", service).Int("count", len(data)).Msg("attachment secrets set") - h.sendJSON(w, http.StatusOK, dto.SecretsStatusResponse{Status: "updated"}) + ctx := r.Context() - case http.MethodDelete: - if !HasAccess(ctx, domain.AdminResourceSecrets, domain.AdminActionWrite) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for secrets:write") - return - } + if !HasAccess(ctx, domain.AdminResourceStatus, domain.AdminActionRead) { + h.sendError(w, http.StatusForbidden, "insufficient permissions for status:read") + return + } - if key == "" { - h.sendError(w, http.StatusBadRequest, "key required in path") - return - } + networks, err := h.containerSvc.ListNetworks(ctx) + if err != nil { + h.sendError(w, http.StatusInternalServerError, "failed to list networks") + return + } - if err := h.secretSvc.DeleteAttachment(ctx, secretDomain, service, key); err != nil { - log.Error().Err(err).Str("domain", secretDomain).Str("service", service).Str("key", key).Msg("failed to delete attachment secret") - h.sendError(w, http.StatusBadRequest, "invalid domain or service") - return + response := make([]dto.Network, 0, len(networks)) + for _, network := range networks { + if network == nil { + continue } - - log.Info().Str("domain", secretDomain).Str("service", service).Str("key", key).Msg("attachment secret deleted") - h.sendJSON(w, http.StatusOK, dto.SecretsStatusResponse{Status: "deleted"}) - - default: - h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") + response = append(response, dto.Network{ + Name: network.Name, + Driver: network.Driver, + Containers: append([]string{}, network.Containers...), + }) } + + h.sendJSON(w, http.StatusOK, dto.NetworksResponse{Networks: response}) } // handleHealth handles /admin/health endpoint. @@ -1308,6 +308,9 @@ func (h *Handler) handleHealth(w http.ResponseWriter, r *http.Request) { } // handleStatus handles /admin/status endpoint. +// Reports installation identity plus the app fleet summary from +// desired/active state (no container inspection). Per-service detail +// lives under `apps show APP`. func (h *Handler) handleStatus(w http.ResponseWriter, r *http.Request) { if r.Method != http.MethodGet { h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") @@ -1322,25 +325,21 @@ func (h *Handler) handleStatus(w http.ResponseWriter, r *http.Request) { return } - routes := h.configSvc.GetRoutes(ctx) - - // Get container statuses + // App fleet summary from desired/active state. statuses := make(map[string]string) - for _, route := range routes { - status := "unknown" - container, ok := h.containerSvc.Get(ctx, route.Domain) - if ok && container != nil { - status = container.Status + if h.appSvc != nil { + if apps, err := h.appSvc.List(ctx); err == nil { + for _, app := range apps { + statuses[app.App] = appStatusLabel(app) + } } - statuses[route.Domain] = status } status := dto.StatusResponse{ - Routes: len(routes), + Apps: len(statuses), RegistryDomain: h.configSvc.GetRegistryDomain(), RegistryPort: h.configSvc.GetRegistryPort(), ServerPort: h.configSvc.GetServerPort(), - AutoRoute: h.configSvc.IsAutoRouteEnabled(), NetworkIsolation: h.configSvc.IsNetworkIsolationEnabled(), ContainerStatuses: statuses, } @@ -1348,6 +347,20 @@ func (h *Handler) handleStatus(w http.ResponseWriter, r *http.Request) { h.sendJSON(w, http.StatusOK, status) } +// appStatusLabel renders one app's fleet status from its summary. +func appStatusLabel(app in.AppSummary) string { + if app.Stopped { + return "stopped" + } + if app.Active == "" { + return "pending" + } + if app.Converged { + return "active" + } + return "deploying" +} + // handleBackups handles /admin/backups endpoints. func (h *Handler) handleBackups(w http.ResponseWriter, r *http.Request, path string) { path = strings.TrimSuffix(path, "/") @@ -1369,18 +382,13 @@ func (h *Handler) handleBackups(w http.ResponseWriter, r *http.Request, path str suffix := strings.TrimPrefix(path, "/backups/") parts := strings.Split(suffix, "/") if len(parts) == 0 || parts[0] == "" { - h.sendError(w, http.StatusBadRequest, "domain required in path") + h.sendError(w, http.StatusBadRequest, "app required in path") return } backupDomain := parts[0] if len(parts) == 1 { - h.handleBackupsDomain(w, r, backupDomain) - return - } - - if len(parts) == 2 && parts[1] == "detect" { - h.handleBackupsDetect(w, r, backupDomain) + h.handleBackupsApp(w, r, backupDomain) return } @@ -1452,7 +460,7 @@ func (h *Handler) handleVolumeBackupsDomain(w http.ResponseWriter, r *http.Reque h.sendError(w, http.StatusBadRequest, "invalid JSON") return } - jobs, err := h.volumeBackupSvc.RunVolumeBackups(ctx, backupDomain, req.Volume) + jobs, err := h.volumeBackupSvc.RunVolumeBackups(ctx, backupDomain, req.Service, req.Volume) backups := mapVolumeBackupJobsResponse(jobs) if err != nil { if len(backups) > 0 { @@ -1487,25 +495,25 @@ func (h *Handler) handleBackupsStatus(w http.ResponseWriter, r *http.Request) { h.sendJSON(w, http.StatusOK, dto.BackupsResponse{Backups: mapBackupJobsResponse(jobs)}) } -func (h *Handler) handleBackupsDomain(w http.ResponseWriter, r *http.Request, backupDomain string) { +func (h *Handler) handleBackupsApp(w http.ResponseWriter, r *http.Request, app string) { switch r.Method { case http.MethodGet: - h.handleBackupsDomainList(w, r, backupDomain) + h.handleBackupsAppList(w, r, app) case http.MethodPost: - h.handleBackupsDomainRun(w, r, backupDomain) + h.handleBackupsAppRun(w, r, app) default: h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") } } -func (h *Handler) handleBackupsDomainList(w http.ResponseWriter, r *http.Request, backupDomain string) { +func (h *Handler) handleBackupsAppList(w http.ResponseWriter, r *http.Request, app string) { ctx := r.Context() if !HasAccess(ctx, domain.AdminResourceStatus, domain.AdminActionRead) { h.sendError(w, http.StatusForbidden, "insufficient permissions for status:read") return } - jobs, err := h.backupSvc.ListBackups(ctx, backupDomain) + jobs, err := h.backupSvc.ListBackups(ctx, app) if err != nil { h.sendError(w, http.StatusInternalServerError, "failed to list backups") return @@ -1514,7 +522,7 @@ func (h *Handler) handleBackupsDomainList(w http.ResponseWriter, r *http.Request h.sendJSON(w, http.StatusOK, dto.BackupsResponse{Backups: mapBackupJobsResponse(jobs)}) } -func (h *Handler) handleBackupsDomainRun(w http.ResponseWriter, r *http.Request, backupDomain string) { +func (h *Handler) handleBackupsAppRun(w http.ResponseWriter, r *http.Request, app string) { ctx := r.Context() if !HasAccess(ctx, domain.AdminResourceConfig, domain.AdminActionWrite) { h.sendError(w, http.StatusForbidden, "insufficient permissions for config:write") @@ -1528,15 +536,16 @@ func (h *Handler) handleBackupsDomainRun(w http.ResponseWriter, r *http.Request, return } - result, err := h.backupSvc.RunBackup(ctx, backupDomain, req.DB) + result, err := h.backupSvc.RunBackup(ctx, app, req.Service, req.Database) if err != nil { log := zerowrap.FromCtx(ctx) - log.Error().Err(err).Str("domain", backupDomain).Msg("backup run failed") + log.Error().Err(err).Str("app", app).Msg("backup run failed") h.sendError(w, http.StatusInternalServerError, "failed to run backup") return } log := zerowrap.FromCtx(ctx) - log.Info().Str("domain", backupDomain).Str("db", req.DB).Str("job_id", result.Job.ID).Msg("backup completed via admin API") + log.Info().Str("app", app).Str("service", req.Service).Str("database", req.Database). + Str("job_id", result.Job.ID).Msg("backup completed via admin API") job := toBackupJobResponse(result.Job) h.sendJSON(w, http.StatusOK, dto.BackupRunResponse{ @@ -1545,31 +554,6 @@ func (h *Handler) handleBackupsDomainRun(w http.ResponseWriter, r *http.Request, }) } -func (h *Handler) handleBackupsDetect(w http.ResponseWriter, r *http.Request, backupDomain string) { - ctx := r.Context() - if r.Method != http.MethodGet { - h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") - return - } - if !HasAccess(ctx, domain.AdminResourceStatus, domain.AdminActionRead) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for status:read") - return - } - - dbs, err := h.backupSvc.DetectDatabases(ctx, backupDomain) - if err != nil { - h.sendError(w, http.StatusInternalServerError, "failed to detect databases") - return - } - - response := make([]dto.DatabaseInfo, 0, len(dbs)) - for _, db := range dbs { - response = append(response, toDatabaseInfoResponse(db)) - } - - h.sendJSON(w, http.StatusOK, dto.BackupDetectResponse{Databases: response}) -} - func mapBackupJobsResponse(jobs []domain.BackupJob) []dto.BackupJob { response := make([]dto.BackupJob, 0, len(jobs)) for _, job := range jobs { @@ -1637,12 +621,6 @@ func (h *Handler) handleConfig(w http.ResponseWriter, r *http.Request) { return } - routes := h.configSvc.GetRoutes(ctx) - routeResponses := make([]dto.Route, 0, len(routes)) - for _, route := range routes { - routeResponses = append(routeResponses, toRouteResponse(route)) - } - externalRoutes := h.configSvc.GetExternalRoutes() externalResponses := make([]dto.ExternalRoute, 0, len(externalRoutes)) for domain := range externalRoutes { @@ -1657,14 +635,10 @@ func (h *Handler) handleConfig(w http.ResponseWriter, r *http.Request) { RegistryPort: h.configSvc.GetRegistryPort(), RegistryDomain: h.configSvc.GetRegistryDomain(), }, - AutoRoute: dto.AutoRouteConfig{ - Enabled: h.configSvc.IsAutoRouteEnabled(), - }, NetworkIsolation: dto.NetworkIsolationConfig{ Enabled: h.configSvc.IsNetworkIsolationEnabled(), Prefix: h.configSvc.GetNetworkPrefix(), }, - Routes: routeResponses, ExternalRoutes: externalResponses, } if volumeCfg, ok := any(h.configSvc).(interface{ GetVolumeConfig() (bool, string, bool) }); ok { @@ -1676,126 +650,6 @@ func (h *Handler) handleConfig(w http.ResponseWriter, r *http.Request) { // handleDeploy handles /admin/deploy/:domain endpoint. // POST triggers a deployment for the specified domain. -func (h *Handler) handleDeploy(w http.ResponseWriter, r *http.Request, path string) { - if r.Method != http.MethodPost { - h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") - return - } - - ctx := r.Context() - log := zerowrap.FromCtx(ctx) - - // Check write permission - if !HasAccess(ctx, domain.AdminResourceConfig, domain.AdminActionWrite) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for config:write") - return - } - - // Parse domain from path - deployDomain := strings.TrimPrefix(path, "/deploy/") - if deployDomain == "" || deployDomain == "/deploy" { - h.sendError(w, http.StatusBadRequest, "domain required in path") - return - } - if err := validation.ValidateDomainParam(deployDomain); err != nil { - h.sendError(w, http.StatusBadRequest, "invalid domain") - return - } - - // Get the route for this domain - route, err := h.configSvc.GetRoute(ctx, deployDomain) - if err != nil { - h.sendError(w, http.StatusNotFound, "route not found") - return - } - - // Deploy is an internal server-side action: pull from local registry path. - container, err := h.containerSvc.Deploy(domain.WithInternalDeploy(ctx), *route) - if err != nil { - log.Error().Err(err).Str("domain", deployDomain).Msg("failed to deploy container") - if deployErr, ok := errors.AsType[*domain.DeployFailureError](err); ok { - response := dto.DeployErrorResponse{ - Error: deployErr.Error(), - Cause: deployErr.Cause, - Hint: deployErr.Hint, - } - if HasAccess(ctx, domain.AdminResourceLogs, domain.AdminActionRead) { - response.Logs = domain.RedactSecretLines(deployErr.Logs) - } - h.sendJSON(w, http.StatusInternalServerError, response) - return - } - h.sendError(w, http.StatusInternalServerError, "failed to deploy container") - return - } - - // Clear deploy event suppression now that the explicit deploy has completed. - // This re-enables event-based deploys for future direct docker pushes. - if route.Image != "" && h.registrySvc != nil { - // Use the registry package's image name normaliser so digest-form refs - // and multi-segment paths are handled correctly. - imageName := registry.ExtractImageName(route.Image) - h.registrySvc.ClearDeployEventSuppression(imageName) - } - - log.Info().Str("domain", deployDomain).Str("container_id", container.ID).Msg("container deployed via admin API") - h.sendJSON(w, http.StatusOK, dto.DeployResponse{ - Status: "deployed", - ContainerID: container.ID, - Domain: deployDomain, - }) -} - -// handleRestart handles /admin/restart/:domain endpoint. -func (h *Handler) handleRestart(w http.ResponseWriter, r *http.Request, path string) { - if r.Method != http.MethodPost { - h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") - return - } - - ctx := r.Context() - log := zerowrap.FromCtx(ctx) - - if !HasAccess(ctx, domain.AdminResourceConfig, domain.AdminActionWrite) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for config:write") - return - } - - restartDomain := strings.TrimPrefix(path, "/restart/") - if restartDomain == "" || restartDomain == "/restart" { - h.sendError(w, http.StatusBadRequest, "domain required in path") - return - } - if err := validation.ValidateDomainParam(restartDomain); err != nil { - h.sendError(w, http.StatusBadRequest, "invalid domain") - return - } - - withAttachments := r.URL.Query().Get("attachments") == "true" - - // Use a detached context so the restart completes even if the HTTP client - // disconnects. Podman restart can take 30+ seconds (SIGTERM + wait + start). - restartCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), 2*time.Minute) - defer cancel() - - if err := h.containerSvc.Restart(restartCtx, restartDomain, withAttachments); err != nil { - log.Error().Err(err).Str("domain", restartDomain).Msg("failed to restart container") - if errors.Is(err, domain.ErrContainerNotFound) { - h.sendError(w, http.StatusNotFound, "container not found") - return - } - h.sendError(w, http.StatusInternalServerError, "failed to restart container") - return - } - - log.Info().Str("domain", restartDomain).Bool("with_attachments", withAttachments).Msg("container restarted via admin API") - h.sendJSON(w, http.StatusOK, dto.RestartResponse{ - Status: "restarted", - Domain: restartDomain, - }) -} - -// handleTags handles /admin/tags/:repository endpoint. func (h *Handler) handleTags(w http.ResponseWriter, r *http.Request, path string) { if r.Method != http.MethodGet { h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") @@ -1888,8 +742,9 @@ func (h *Handler) handleLogs(w http.ResponseWriter, r *http.Request, path string } if logDomain != "" { - if err := validation.ValidateDomainParam(logDomain); err != nil { - h.sendError(w, http.StatusBadRequest, "invalid domain") + parts := strings.Split(logDomain, "/") + if len(parts) != 2 || validation.ValidateDomainParam(parts[0]) != nil || validation.ValidateDomainParam(parts[1]) != nil { + h.sendError(w, http.StatusBadRequest, "invalid app/service reference") return } } @@ -2047,233 +902,7 @@ func (h *Handler) streamContainerLogs(w http.ResponseWriter, r *http.Request, lo } } -// handleAttachmentsConfig handles /admin/attachments endpoints for config-level attachments. -func (h *Handler) handleAttachmentsConfig(w http.ResponseWriter, r *http.Request, path string) { - // Parse target (domain or group) from path - target := strings.TrimPrefix(path, "/attachments/") - if target == "/attachments" { - target = "" - } - - switch r.Method { - case http.MethodGet: - h.handleAttachmentsConfigGet(w, r, target) - case http.MethodPost: - h.handleAttachmentsConfigPost(w, r, target) - case http.MethodDelete: - h.handleAttachmentsConfigDelete(w, r, target) - default: - h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") - } -} - -func (h *Handler) handleAttachmentsConfigGet(w http.ResponseWriter, r *http.Request, target string) { - ctx := r.Context() - - // Check read permission - if !HasAccess(ctx, domain.AdminResourceConfig, domain.AdminActionRead) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for config:read") - return - } - - if target == "" { - // List all attachments - attachments := h.configSvc.GetAllAttachments(ctx) - h.sendJSON(w, http.StatusOK, dto.AttachmentsConfigResponse{Attachments: attachments}) - return - } - - // List attachments for specific target - images, err := h.configSvc.GetAttachmentsFor(ctx, target) - if err != nil { - if errors.Is(err, domain.ErrAttachmentNotFound) { - h.sendError(w, http.StatusNotFound, "no attachments found for target") - return - } - h.sendError(w, http.StatusInternalServerError, "failed to get attachments") - return - } - - h.sendJSON(w, http.StatusOK, dto.AttachmentConfigResponse{Target: target, Images: images}) -} - -func (h *Handler) handleAttachmentsConfigPost(w http.ResponseWriter, r *http.Request, target string) { - ctx := r.Context() - log := zerowrap.FromCtx(ctx) - - // Check write permission - if !HasAccess(ctx, domain.AdminResourceConfig, domain.AdminActionWrite) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for config:write") - return - } - - if target == "" { - h.sendError(w, http.StatusBadRequest, "target (domain or group) required in path") - return - } - - // Limit request body size - r.Body = http.MaxBytesReader(w, r.Body, maxAdminRequestSize) - - var req dto.AttachmentAddRequest - if err := json.NewDecoder(r.Body).Decode(&req); err != nil { - log.Warn().Err(err).Msg("invalid attachment JSON") - h.sendError(w, http.StatusBadRequest, "invalid JSON") - return - } - - if err := h.configSvc.AddAttachment(ctx, target, req.Image); err != nil { - log.Error().Err(err).Str("target", target).Str("image", req.Image).Msg("failed to add attachment") - switch { - case errors.Is(err, domain.ErrAttachmentExists): - h.sendError(w, http.StatusConflict, "attachment already exists") - case errors.Is(err, domain.ErrAttachmentImageEmpty): - h.sendError(w, http.StatusBadRequest, err.Error()) - case errors.Is(err, domain.ErrAttachmentTargetEmpty): - h.sendError(w, http.StatusBadRequest, err.Error()) - default: - h.sendError(w, http.StatusInternalServerError, "failed to add attachment") - } - return - } - - log.Info().Str("target", target).Str("image", req.Image).Msg("attachment added") - h.sendJSON(w, http.StatusCreated, dto.AttachmentStatusResponse{Status: "added"}) -} - -func (h *Handler) handleAttachmentsConfigDelete(w http.ResponseWriter, r *http.Request, target string) { - ctx := r.Context() - log := zerowrap.FromCtx(ctx) - - // Check write permission - if !HasAccess(ctx, domain.AdminResourceConfig, domain.AdminActionWrite) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for config:write") - return - } - - if target == "" { - h.sendError(w, http.StatusBadRequest, "target (domain or group) required in path") - return - } - - // Parse image from path: /attachments/{target}/{image} - // target at this point contains "{domain}/{image}" or just "{domain}" - parts := strings.SplitN(target, "/", 2) - if len(parts) != 2 { - h.sendError(w, http.StatusBadRequest, "image required in path: /attachments/{target}/{image}") - return - } - - domainOrGroup := parts[0] - image := parts[1] - - if err := h.configSvc.RemoveAttachment(ctx, domainOrGroup, image); err != nil { - log.Error().Err(err).Str("target", domainOrGroup).Str("image", image).Msg("failed to remove attachment") - switch { - case errors.Is(err, domain.ErrAttachmentNotFound): - h.sendError(w, http.StatusNotFound, "attachment not found") - case errors.Is(err, domain.ErrAttachmentImageEmpty): - h.sendError(w, http.StatusBadRequest, err.Error()) - case errors.Is(err, domain.ErrAttachmentTargetEmpty): - h.sendError(w, http.StatusBadRequest, err.Error()) - default: - h.sendError(w, http.StatusInternalServerError, "failed to remove attachment") - } - return - } - - log.Info().Str("target", domainOrGroup).Str("image", image).Msg("attachment removed") - h.sendJSON(w, http.StatusOK, dto.AttachmentStatusResponse{Status: "removed"}) -} - -func (h *Handler) handleAutoRouteAllowedDomains(w http.ResponseWriter, r *http.Request, path string) { - switch r.Method { - case http.MethodGet: - h.handleAutoRouteAllowedDomainsGet(w, r) - case http.MethodPost: - h.handleAutoRouteAllowedDomainsPost(w, r) - case http.MethodDelete: - h.handleAutoRouteAllowedDomainsDelete(w, r, path) - default: - h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") - } -} - -func (h *Handler) handleAutoRouteAllowedDomainsGet(w http.ResponseWriter, r *http.Request) { - ctx := r.Context() - if !HasAccess(ctx, domain.AdminResourceConfig, domain.AdminActionRead) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for config:read") - return - } - - domains, err := h.configSvc.GetAutoRouteAllowedDomains(ctx) - if err != nil { - h.sendError(w, http.StatusInternalServerError, "failed to get auto-route allowed domains") - return - } - - h.sendJSON(w, http.StatusOK, dto.AutoRouteAllowedDomainsResponse{Domains: domains}) -} - -func (h *Handler) handleAutoRouteAllowedDomainsPost(w http.ResponseWriter, r *http.Request) { - ctx := r.Context() - log := zerowrap.FromCtx(ctx) - if !HasAccess(ctx, domain.AdminResourceConfig, domain.AdminActionWrite) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for config:write") - return - } - - r.Body = http.MaxBytesReader(w, r.Body, maxAdminRequestSize) - var req dto.AutoRouteAllowedDomainRequest - if err := json.NewDecoder(r.Body).Decode(&req); err != nil { - log.Warn().Err(err).Msg("invalid auto-route allowlist JSON") - h.sendError(w, http.StatusBadRequest, "invalid JSON") - return - } - - if err := h.configSvc.AddAutoRouteAllowedDomain(ctx, req.Pattern); err != nil { - if errors.Is(err, domain.ErrInvalidDomainPattern) { - h.sendError(w, http.StatusBadRequest, err.Error()) - } else { - log.Error().Err(err).Str("pattern", req.Pattern).Msg("failed to add auto-route allowed domain") - h.sendError(w, http.StatusInternalServerError, "failed to add auto-route allowed domain") - } - return - } - - h.sendJSON(w, http.StatusCreated, dto.AutoRouteStatusResponse{Status: "added"}) -} - -func (h *Handler) handleAutoRouteAllowedDomainsDelete(w http.ResponseWriter, r *http.Request, path string) { - ctx := r.Context() - log := zerowrap.FromCtx(ctx) - if !HasAccess(ctx, domain.AdminResourceConfig, domain.AdminActionWrite) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for config:write") - return - } - - raw := strings.TrimPrefix(path, "/autoroute/allowed-domains/") - if raw == path || raw == "" || raw == "/" { - h.sendError(w, http.StatusBadRequest, "missing domain pattern") - return - } - pattern, err := url.PathUnescape(raw) - if err != nil || strings.TrimSpace(pattern) == "" { - h.sendError(w, http.StatusBadRequest, "invalid domain pattern") - return - } - - if err := h.configSvc.RemoveAutoRouteAllowedDomain(ctx, pattern); err != nil { - log.Error().Err(err).Str("pattern", pattern).Msg("failed to remove auto-route allowed domain") - h.sendError(w, http.StatusInternalServerError, "failed to remove auto-route allowed domain") - return - } - - h.sendJSON(w, http.StatusOK, dto.AutoRouteStatusResponse{Status: "removed"}) -} - -// handleAuthVerify handles /admin/auth/verify endpoint. -// Validates authentication session and returns token status. +// handleAuthVerify handles /admin/auth/verify. func (h *Handler) handleAuthVerify(w http.ResponseWriter, r *http.Request) { ctx := r.Context() diff --git a/internal/adapters/in/http/admin/handler_apps.go b/internal/adapters/in/http/admin/handler_apps.go new file mode 100644 index 000000000..1b2aa4304 --- /dev/null +++ b/internal/adapters/in/http/admin/handler_apps.go @@ -0,0 +1,670 @@ +package admin + +import ( + "context" + "encoding/json" + "errors" + "io" + "net/http" + "strings" + + "github.com/bnema/gordon/internal/adapters/dto" + "github.com/bnema/gordon/internal/adapters/in/appmanifest" + "github.com/bnema/gordon/internal/boundaries/in" + "github.com/bnema/gordon/internal/domain" +) + +// handleApps dispatches /admin/apps/* reads and lifecycle mutations. +func (h *Handler) handleApps(w http.ResponseWriter, r *http.Request, path string) { + if path == "/apps/apply" { + h.handleAppApply(w, r) + return + } + if app, key, ok := splitOpLookup(path); ok { + h.handleAppOpLookup(w, r, app, key) + return + } + rest := strings.TrimPrefix(path, "/apps") + if rest == "" || rest == "/" { + if r.Method != http.MethodGet { + h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") + return + } + h.handleAppList(w, r) + return + } + parts := strings.Split(strings.Trim(rest, "/"), "/") + if parts[0] == "" { + h.sendError(w, http.StatusBadRequest, "app required in path") + return + } + h.dispatchAppSubroute(w, r, parts) +} + +// splitOpLookup parses /apps/{app}/operations/by-key/{key}. +func splitOpLookup(path string) (string, string, bool) { + rest, ok := strings.CutPrefix(path, "/apps/") + if !ok || !strings.Contains(rest, "/operations/by-key/") { + return "", "", false + } + app, key, _ := strings.Cut(rest, "/") + key, ok = strings.CutPrefix(key, "operations/by-key/") + if !ok || app == "" || key == "" || strings.Contains(key, "/") { + return "", "", false + } + return app, key, true +} + +// dispatchAppSubroute routes one app's subpaths by method + shape. +func (h *Handler) dispatchAppSubroute(w http.ResponseWriter, r *http.Request, parts []string) { + app := parts[0] + if len(parts) == 1 { + if r.Method != http.MethodGet { + h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") + return + } + h.handleAppShow(w, r, app) + return + } + if len(parts) == 2 { + if parts[1] == "secrets" && r.Method == http.MethodGet { + h.handleAppSecretsList(w, r, app) + } else { + h.dispatchAppVerb(w, r, app, parts[1]) + } + return + } + if len(parts) == 3 && parts[1] == "secrets" { + h.dispatchAppSecrets(w, r, app, parts[2]) + return + } + h.sendError(w, http.StatusNotFound, "route not found") +} + +// dispatchAppVerb routes single-segment verbs (diff/deploy/stop/...). +func (h *Handler) dispatchAppVerb(w http.ResponseWriter, r *http.Request, app, verb string) { + switch { + case verb == "diff" && r.Method == http.MethodGet: + h.handleAppDiff(w, r, app) + case verb == "deploy" && r.Method == http.MethodPost: + h.handleAppDeploy(w, r, app) + case verb == "restart" && r.Method == http.MethodPost: + h.handleAppRestart(w, r, app) + case (verb == "stop" || verb == "start" || verb == "remove") && r.Method == http.MethodPost: + h.handleAppLifecycle(w, r, app, verb) + default: + h.sendError(w, http.StatusNotFound, "route not found") + } +} + +// dispatchAppSecrets routes the secrets set/delete pair. +func (h *Handler) dispatchAppSecrets(w http.ResponseWriter, r *http.Request, app, action string) { + if r.Method != http.MethodPost { + h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") + return + } + switch action { + case "set": + h.handleAppSecretsSet(w, r, app) + case "delete": + h.handleAppSecretsDelete(w, r, app) + default: + h.sendError(w, http.StatusNotFound, "route not found") + } +} + +// appService returns the port or 503 when the daemon engine is unwired. +func (h *Handler) appService(w http.ResponseWriter) (in.AppService, bool) { + if h.appSvc == nil { + h.sendError(w, http.StatusServiceUnavailable, "app engine not available") + return nil, false + } + return h.appSvc, true +} + +// handleAppApply validates a manifest and persists desired state. +func (h *Handler) handleAppApply(w http.ResponseWriter, r *http.Request) { + ctx := r.Context() + if r.Method != http.MethodPost { + h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") + return + } + if !HasAccess(ctx, domain.AdminResourceApps, domain.AdminActionWrite) { + h.sendError(w, http.StatusForbidden, "insufficient permissions for apps:write") + return + } + svc, ok := h.appService(w) + if !ok { + return + } + r.Body = http.MaxBytesReader(w, r.Body, maxAdminRequestSize) + var req dto.AppApplyRequest + if err := json.NewDecoder(r.Body).Decode(&req); err != nil { + h.sendAppError(w, http.StatusBadRequest, "invalid-request", "invalid JSON", "", "") + return + } + if strings.TrimSpace(req.ManifestTOML) == "" { + h.sendAppError(w, http.StatusBadRequest, "invalid-manifest", "manifest_toml is required", "", "") + return + } + spec, _, err := appmanifest.Parse([]byte(req.ManifestTOML), "") + if err != nil { + h.sendAppError(w, http.StatusBadRequest, "invalid-manifest", err.Error(), "", "") + return + } + applyResult, dryResult, err := svc.Apply(ctx, spec, []byte(req.ManifestTOML), req.DryRun) + if err != nil { + h.sendAppOpError(w, err) + return + } + if dryResult != nil { + h.sendJSON(w, http.StatusOK, dto.AppApplyResponse{ + App: spec.Name, + Diff: toAppDiffSection(dryResult.Diff), + }) + return + } + h.sendJSON(w, http.StatusOK, dto.AppApplyResponse{ + App: applyResult.App, + FormerRevision: applyResult.FormerRevision, + ResultingRevision: applyResult.ResultingRevision, + Pending: applyResult.Pending, + Noop: applyResult.Noop, + Diff: toAppDiffSection(applyResult.Diff), + Intent: applyResult.IntentID, + }) +} + +// handleAppList returns one summary per known app. +func (h *Handler) handleAppList(w http.ResponseWriter, r *http.Request) { + ctx := r.Context() + if !HasAccess(ctx, domain.AdminResourceApps, domain.AdminActionRead) { + h.sendError(w, http.StatusForbidden, "insufficient permissions for apps:read") + return + } + svc, ok := h.appService(w) + if !ok { + return + } + summaries, err := svc.List(ctx) + if err != nil { + h.sendAppOpError(w, err) + return + } + items := make([]dto.AppSummaryDTO, 0, len(summaries)) + for _, s := range summaries { + items = append(items, dto.AppSummaryDTO{ + App: s.App, Desired: s.Desired, DesiredStatus: s.DesiredStatus, + Active: s.Active, Converged: s.Converged, Pending: s.Pending, + Stopped: s.Stopped, LastOutcome: s.LastOutcome, + }) + } + if items == nil { + items = []dto.AppSummaryDTO{} + } + h.sendJSON(w, http.StatusOK, map[string]any{"apps": items}) +} + +// handleAppShow inspects desired + active + intent. +func (h *Handler) handleAppShow(w http.ResponseWriter, r *http.Request, app string) { + ctx := r.Context() + if !HasAccess(ctx, domain.AdminResourceApps, domain.AdminActionRead) { + h.sendError(w, http.StatusForbidden, "insufficient permissions for apps:read") + return + } + svc, ok := h.appService(w) + if !ok { + return + } + detail, err := svc.Show(ctx, app) + if err != nil { + h.sendAppOpError(w, err) + return + } + services := map[string]dto.AppActiveServiceDTO{} + for name, view := range detail.Services { + services[name] = dto.AppActiveServiceDTO{ + EffectiveRevision: view.EffectiveRevision, + Digest: view.Digest, + Container: view.Container, + RestartUnsafe: view.RestartUnsafe, + } + } + resp := dto.AppShowResponse{ + App: detail.App, + Desired: dto.AppDesiredDTO{Revision: detail.DesiredRevision, Status: detail.DesiredStatus, Pending: detail.Pending}, + Active: dto.AppActiveDTO{ + Converged: detail.Converged, + ConvergedRevision: detail.ConvergedRevision, + Services: services, + }, + Intent: dto.AppIntentDTO{Stopped: detail.Stopped}, + Retained: dto.AppRetainedDTO{ + Volumes: detail.Retained.Volumes, + Secrets: detail.Retained.Secrets, + Images: detail.Retained.Images, + }, + } + if detail.LastOp != "" { + resp.LastOp = &dto.AppLastOpDTO{ + Op: detail.LastOp, Kind: detail.LastOpKind, + Outcome: detail.LastOutcome, StartedAt: detail.LastOpStartedAt, + } + } + h.sendJSON(w, http.StatusOK, resp) +} + +// handleAppDiff returns the normalized desired-vs-active diff. +func (h *Handler) handleAppDiff(w http.ResponseWriter, r *http.Request, app string) { + ctx := r.Context() + if !HasAccess(ctx, domain.AdminResourceApps, domain.AdminActionRead) { + h.sendError(w, http.StatusForbidden, "insufficient permissions for apps:read") + return + } + svc, ok := h.appService(w) + if !ok { + return + } + diff, err := svc.Diff(ctx, app) + if err != nil { + h.sendAppOpError(w, err) + return + } + h.sendJSON(w, http.StatusOK, dto.AppDiffResponse{App: app, Diff: toAppDiffSection(diff)}) +} + +// handleAppDeploy activates a captured revision. +func (h *Handler) handleAppDeploy(w http.ResponseWriter, r *http.Request, app string) { + ctx := r.Context() + if !HasAccess(ctx, domain.AdminResourceApps, domain.AdminActionWrite) { + h.sendError(w, http.StatusForbidden, "insufficient permissions for apps:write") + return + } + svc, ok := h.appService(w) + if !ok { + return + } + r.Body = http.MaxBytesReader(w, r.Body, maxAdminRequestSize) + var req dto.AppDeployRequest + if r.Body != nil { + raw, err := io.ReadAll(r.Body) + if err != nil { + h.sendAppError(w, http.StatusBadRequest, "invalid-request", "unreadable body", "", "") + return + } + if len(raw) > 0 { + if err := json.Unmarshal(raw, &req); err != nil { + h.sendAppError(w, http.StatusBadRequest, "invalid-request", "invalid JSON", "", "") + return + } + } + } + key, ok := appIdempotencyKey(w, r) + if !ok { + return + } + op, err := svc.Deploy(ctx, app, req.Revision, req.Service, req.All, key) + if err != nil && (op == nil || isMappedPreflightError(err, op)) { + h.sendAppOpError(w, err) + return + } + status := http.StatusOK + switch { + case err != nil: + status = http.StatusConflict + case op != nil && !op.Terminal(): + // Newly owned: the claim is durable and its effects run on the + // daemon context, so answer 202 with the running operation and let + // the client observe completion through operations/by-key. A + // terminal replay instead answers 200 with the stored outcome. + status = http.StatusAccepted + } + h.sendJSON(w, status, h.appMutationResponse(ctx, svc, app, op, false)) +} + +// handleAppLifecycle runs stop/start/remove verbs. +func (h *Handler) handleAppLifecycle(w http.ResponseWriter, r *http.Request, app, verb string) { + ctx := r.Context() + if !HasAccess(ctx, domain.AdminResourceApps, domain.AdminActionWrite) { + h.sendError(w, http.StatusForbidden, "insufficient permissions for apps:write") + return + } + svc, ok := h.appService(w) + if !ok { + return + } + key, ok := appIdempotencyKey(w, r) + if !ok { + return + } + var domainOp *domain.AppOperation + var err error + switch verb { + case "stop": + domainOp, err = svc.Stop(ctx, app, key) + case "start": + domainOp, err = svc.Start(ctx, app, key) + case "remove": + domainOp, err = svc.Remove(ctx, app, key) + default: + h.sendError(w, http.StatusNotFound, "route not found") + return + } + if err != nil && domainOp == nil { + h.sendAppOpError(w, err) + return + } + status := http.StatusOK + if err != nil { + status = http.StatusConflict + } + h.sendJSON(w, status, h.appMutationResponse(ctx, svc, app, domainOp, false)) +} + +// handleAppRestart restarts from pinned digests. +func (h *Handler) handleAppRestart(w http.ResponseWriter, r *http.Request, app string) { + ctx := r.Context() + if !HasAccess(ctx, domain.AdminResourceApps, domain.AdminActionWrite) { + h.sendError(w, http.StatusForbidden, "insufficient permissions for apps:write") + return + } + svc, ok := h.appService(w) + if !ok { + return + } + key, ok := appIdempotencyKey(w, r) + if !ok { + return + } + query := r.URL.Query() + all := query.Get("all") == "true" + op, err := svc.Restart(ctx, app, query.Get("service"), all, key) + if err != nil && op == nil { + h.sendAppOpError(w, err) + return + } + status := http.StatusOK + if err != nil { + status = http.StatusConflict + } + h.sendJSON(w, status, h.appMutationResponse(ctx, svc, app, op, false)) +} + +func appIdempotencyKey(w http.ResponseWriter, r *http.Request) (string, bool) { + key := strings.TrimSpace(r.Header.Get("Idempotency-Key")) + if key == "" || len(key) > 128 { + http.Error(w, "valid Idempotency-Key header required", http.StatusBadRequest) + return "", false + } + for _, char := range key { + if char < '!' || char > '~' { + http.Error(w, "valid Idempotency-Key header required", http.StatusBadRequest) + return "", false + } + } + return key, true +} + +// handleAppOpLookup serves GET /apps/{app}/operations/{key}. +func (h *Handler) handleAppOpLookup(w http.ResponseWriter, r *http.Request, app, key string) { + ctx := r.Context() + if !HasAccess(ctx, domain.AdminResourceApps, domain.AdminActionRead) { + h.sendError(w, http.StatusForbidden, "insufficient permissions for apps:read") + return + } + svc, ok := h.appService(w) + if !ok { + return + } + op, err := svc.OperationByKey(ctx, app, key) + if err != nil { + h.sendAppOpError(w, err) + return + } + // Application diagnostics require the logs read scope; an apps-read + // actor receives the stable error only. The lookup answers with the + // journal alone: no app read is added to a journal query. + includeDiagnostics := HasAccess(ctx, domain.AdminResourceLogs, domain.AdminActionRead) + h.sendJSON(w, http.StatusOK, toAppDeployResponse(app, op, includeDiagnostics)) +} + +func (h *Handler) handleAppSecretsList(w http.ResponseWriter, r *http.Request, app string) { + ctx := r.Context() + if !HasAccess(ctx, domain.AdminResourceApps, domain.AdminActionRead) { + h.sendError(w, http.StatusForbidden, "insufficient permissions for apps:read") + return + } + svc, ok := h.appService(w) + if !ok { + return + } + entries, err := svc.ListSecrets(ctx, app, r.URL.Query().Get("service")) + if err != nil { + h.sendAppOpError(w, err) + return + } + result := make([]dto.AppSecretMetadataDTO, 0, len(entries)) + for _, entry := range entries { + result = append(result, dto.AppSecretMetadataDTO{Service: entry.Service, Key: entry.Key, Name: entry.Name, Source: entry.Source, Presence: entry.Presence}) + } + h.sendJSON(w, http.StatusOK, result) +} + +// handleAppSecretsSet writes secret values for pre-registered names. +func (h *Handler) handleAppSecretsSet(w http.ResponseWriter, r *http.Request, app string) { + ctx := r.Context() + if !HasAccess(ctx, domain.AdminResourceApps, domain.AdminActionWrite) { + h.sendError(w, http.StatusForbidden, "insufficient permissions for apps:write") + return + } + svc, ok := h.appService(w) + if !ok { + return + } + r.Body = http.MaxBytesReader(w, r.Body, maxAdminRequestSize) + var req dto.AppSecretSetRequest + if err := json.NewDecoder(r.Body).Decode(&req); err != nil { + h.sendAppError(w, http.StatusBadRequest, "invalid-request", "invalid JSON", "", "") + return + } + if err := svc.SetSecrets(ctx, app, req.Service, req.Secrets); err != nil { + h.sendAppOpError(w, err) + return + } + h.sendJSON(w, http.StatusOK, map[string]string{"status": "updated"}) +} + +// handleAppSecretsDelete removes one secret value (refused when referenced). +func (h *Handler) handleAppSecretsDelete(w http.ResponseWriter, r *http.Request, app string) { + ctx := r.Context() + if !HasAccess(ctx, domain.AdminResourceApps, domain.AdminActionWrite) { + h.sendError(w, http.StatusForbidden, "insufficient permissions for apps:write") + return + } + svc, ok := h.appService(w) + if !ok { + return + } + r.Body = http.MaxBytesReader(w, r.Body, maxAdminRequestSize) + var req dto.AppSecretDeleteRequest + if err := json.NewDecoder(r.Body).Decode(&req); err != nil { + h.sendAppError(w, http.StatusBadRequest, "invalid-request", "invalid JSON", "", "") + return + } + if err := svc.DeleteSecret(ctx, app, req.Service, req.Key); err != nil { + h.sendAppOpError(w, err) + return + } + h.sendJSON(w, http.StatusOK, map[string]string{"status": "deleted"}) +} + +// sendAppError writes the v3 error envelope (never carries logs). +func (h *Handler) sendAppError(w http.ResponseWriter, status int, code, message, cause, hint string) { + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(status) + _ = json.NewEncoder(w).Encode(dto.AppError{Error: code, Message: message, Cause: cause, Hint: hint}) +} + +func isMappedPreflightError(err error, op *domain.AppOperation) bool { + // A non-terminal journal is a running replay, never a preflight + // failure: its stored operation is the response regardless of which + // steps have started, so the mapped error envelope must not replace + // it. Only an absent or terminal journal can map a preflight error. + if op != nil && !op.Terminal() { + return false + } + for _, step := range op.Steps { + if strings.HasPrefix(step.ID, "service.") { + return false + } + } + return errors.Is(err, domain.ErrInvalidAppSpec) || + errors.Is(err, domain.ErrBindPolicy) || + errors.Is(err, domain.ErrDevicePolicy) || + isPreflightErrorRest(err) +} + +// isPreflightErrorRest covers the remaining preflight-mappable sentinels. +// Split from isMappedPreflightError to keep its complexity within budget. +func isPreflightErrorRest(err error) bool { + return errors.Is(err, domain.ErrRuntimeUnsupported) || + errors.Is(err, domain.ErrAppNotFound) || + errors.Is(err, domain.ErrAppReservationConflict) || + errors.Is(err, domain.ErrAppImageUnresolvable) || + errors.Is(err, domain.ErrAppSecretMissing) || + errors.Is(err, domain.ErrAppUnmanagedImageVolume) || + errors.Is(err, domain.ErrAppRevisionNotFound) || + errors.Is(err, domain.ErrAppIntentNotFound) || + errors.Is(err, domain.ErrAppOperationNotFound) || + errors.Is(err, domain.ErrAppStateConflict) +} + +// sendAppOpError maps domain errors to status + envelope. +func (h *Handler) sendAppOpError(w http.ResponseWriter, err error) { + switch { + case errors.Is(err, domain.ErrInvalidAppSpec): + h.sendAppError(w, http.StatusBadRequest, "invalid-manifest", err.Error(), "", "") + case errors.Is(err, domain.ErrBindPolicy): + h.sendAppError(w, http.StatusBadRequest, "bind-policy-violation", err.Error(), "", "check the administrative mount authorization") + case errors.Is(err, domain.ErrDevicePolicy): + h.sendAppError(w, http.StatusBadRequest, "device-policy-violation", err.Error(), "", "check the administrative device authorization") + case errors.Is(err, domain.ErrRuntimeUnsupported): + h.sendAppError(w, http.StatusBadRequest, "runtime-unsupported", err.Error(), "", "use an engine with native CDI support") + case errors.Is(err, domain.ErrAppReservationConflict): + h.sendAppError(w, http.StatusConflict, "reservation-conflict", err.Error(), "", "") + case errors.Is(err, domain.ErrAppImageUnresolvable): + h.sendAppError(w, http.StatusBadRequest, "image-unresolvable", err.Error(), "", "") + case errors.Is(err, domain.ErrAppSecretMissing): + h.sendAppError(w, http.StatusBadRequest, "secret-missing", err.Error(), "", "set the secret, then redeploy") + case errors.Is(err, domain.ErrAppUnmanagedImageVolume): + h.sendAppError(w, http.StatusBadRequest, "unmanaged-image-volume", err.Error(), "", "") + case errors.Is(err, domain.ErrAppServiceScope): + h.sendAppError(w, http.StatusBadRequest, "service-scope-required", err.Error(), "", "pass --service NAME or --all") + case errors.Is(err, domain.ErrAppNotFound): + h.sendAppError(w, http.StatusNotFound, "app-not-found", err.Error(), "", "check the app name") + case errors.Is(err, domain.ErrAppRevisionNotFound), + errors.Is(err, domain.ErrAppIntentNotFound), + errors.Is(err, domain.ErrAppOperationNotFound): + h.sendAppError(w, http.StatusNotFound, "not-found", err.Error(), "", "") + case errors.Is(err, domain.ErrAppStateConflict): + h.sendAppError(w, http.StatusConflict, "state-conflict", err.Error(), "", "") + default: + h.sendAppError(w, http.StatusInternalServerError, "internal", "app operation failed", "", "") + } +} + +// toAppDiffSection maps the domain diff to its DTO shape. +func toAppDiffSection(diff domain.AppDiff) dto.AppDiffSection { + return dto.AppDiffSection{Added: diff.Added, Removed: diff.Removed, Changed: diff.Changed} +} + +// toAppDeployResponse maps a journaled operation to its DTO shape. +// toAppDeployResponse maps an operation journal record to the wire DTO. +// includeDiagnostics must be true only for callers holding the logs read +// scope: application output is never returned on a mutation response. +func toAppDeployResponse(app string, op *domain.AppOperation, includeDiagnostics bool) dto.AppDeployResponse { + resp := dto.AppDeployResponse{App: app} + if op == nil { + return resp + } + resp.Op = op.Op + resp.Revision = op.InputRevision + resp.Outcome = op.Outcome + // A non-terminal operation is still running; a terminal one reports + // its persisted outcome as the stable status. + resp.Status = dto.AppStatusRunning + if op.Terminal() { + resp.Status = op.Outcome + } + resp.Services = map[string]dto.AppServiceResultDTO{} + for _, step := range op.Steps { + service, ok := strings.CutPrefix(step.ID, "service.") + if !ok { + continue + } + service, _, _ = strings.Cut(service, ".") + entry := resp.Services[service] + switch step.State { + case domain.AppStepSucceeded: + entry.Result = "deployed" + if strings.HasPrefix(step.Detail, domain.AppServiceUnchanged+":") { + entry.Result = domain.AppServiceUnchanged + } + case domain.AppStepFailed: + entry.Result = "failed" + entry.Error = step.Error + default: + entry.Result = "pending" + } + entry.Before = step.Before + entry.After = step.After + if entry.EffectiveRevision == "" { + entry.EffectiveRevision = op.InputRevision + } + resp.Services[service] = entry + } + for _, warning := range op.Warnings { + resp.CleanupWarnings = append(resp.CleanupWarnings, dto.AppCleanupWarningDTO{ + Service: warning.Service, Leftover: warning.Leftover, Detail: warning.Detail, + }) + } + resp.Steps = make([]dto.AppStepDTO, 0, len(op.Steps)) + for _, step := range op.Steps { + entry := dto.AppStepDTO{ + ID: step.ID, State: step.State, Detail: step.Detail, + Error: step.Error, Before: step.Before, After: step.After, + } + if includeDiagnostics { + entry.Diagnostics = step.Diagnostics + } + resp.Steps = append(resp.Steps, entry) + } + return resp +} + +// appMutationResponse maps one operation journal to the wire DTO and +// attaches current app state. Effective revisions, convergence, and owned +// resources come from desired + ACTIVE + ownership, never from the +// journal. An app that no longer exists (removal) omits both sections. +func (h *Handler) appMutationResponse(ctx context.Context, svc in.AppService, app string, op *domain.AppOperation, includeDiagnostics bool) dto.AppDeployResponse { + resp := toAppDeployResponse(app, op, includeDiagnostics) + detail, err := svc.Show(ctx, app) + if err != nil { + return resp + } + effective := &dto.AppEffectiveDTO{ + Converged: detail.Converged, + ConvergedRevision: detail.ConvergedRevision, + Services: map[string]string{}, + } + for name, view := range detail.Services { + effective.Services[name] = view.EffectiveRevision + } + resp.Effective = effective + resp.Retained = &dto.AppRetainedDTO{ + Volumes: detail.Retained.Volumes, + Secrets: detail.Retained.Secrets, + Images: detail.Retained.Images, + } + return resp +} diff --git a/internal/adapters/in/http/admin/handler_apps_test.go b/internal/adapters/in/http/admin/handler_apps_test.go new file mode 100644 index 000000000..c4ad71d77 --- /dev/null +++ b/internal/adapters/in/http/admin/handler_apps_test.go @@ -0,0 +1,683 @@ +package admin + +import ( + "bytes" + "encoding/json" + "fmt" + "io" + "net/http" + "net/http/httptest" + "sort" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/dto" + in "github.com/bnema/gordon/internal/boundaries/in" + inmocks "github.com/bnema/gordon/internal/boundaries/in/mocks" + "github.com/bnema/gordon/internal/domain" +) + +func appsTestHandler(t *testing.T, appSvc *inmocks.MockAppService) *Handler { + t.Helper() + return NewHandler(HandlerDeps{ + ConfigSvc: inmocks.NewMockConfigService(t), + AuthSvc: inmocks.NewMockAuthService(t), + ContainerSvc: inmocks.NewMockContainerService(t), + HealthSvc: inmocks.NewMockHealthService(t), + Log: testLogger(), + ReloadTrigger: noopReloadTrigger{}, + AppSvc: appSvc, + }) +} + +func appsRequest(t *testing.T, handler *Handler, method, target string, body any, scopes ...string) *httptest.ResponseRecorder { + t.Helper() + var reader io.Reader + if body != nil { + raw, err := json.Marshal(body) + require.NoError(t, err) + reader = bytes.NewReader(raw) + } + req := httptest.NewRequest(method, target, reader) + if method != http.MethodGet { + req.Header.Set("Idempotency-Key", "test-operation-key") + } + req = req.WithContext(ctxWithScopes(scopes...)) + rec := httptest.NewRecorder() + handler.ServeHTTP(rec, req) + return rec +} + +// appsServerDo sends a request through a test server backed by the apps +// handler and returns the real HTTP status code and body. The scopes are +// injected by newScopedTestServer because the server, not the client, owns +// the request context. +func appsServerDo(t *testing.T, srv *httptest.Server, method, target string, body any) (int, []byte) { + t.Helper() + var reader io.Reader + if body != nil { + raw, err := json.Marshal(body) + require.NoError(t, err) + reader = bytes.NewReader(raw) + } + req, err := http.NewRequest(method, srv.URL+target, reader) + require.NoError(t, err) + if method != http.MethodGet { + req.Header.Set("Idempotency-Key", "test-operation-key") + } + resp, err := srv.Client().Do(req) + require.NoError(t, err) + defer func() { require.NoError(t, resp.Body.Close()) }() + assert.Equal(t, "application/json", resp.Header.Get("Content-Type")) + raw, err := io.ReadAll(resp.Body) + require.NoError(t, err) + return resp.StatusCode, raw +} + +const validAppManifest = ` +name = "blog" +[services.web] +image = "img:1" +[[services.web.http]] +host = "blog.example.com" +port = 8080 +` + +func TestHandler_AppApply_Persists(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + appSvc.EXPECT().Apply(mock.Anything, mock.Anything, mock.Anything, false).Return( + &in.AppApplyResult{App: "blog", ResultingRevision: "rev-1", Pending: true, IntentID: "apply-1"}, + nil, nil, + ).Once() + + rec := appsRequest(t, handler, http.MethodPost, "/admin/apps/apply", + dto.AppApplyRequest{ManifestTOML: validAppManifest}, "admin:apps:write") + require.Equal(t, http.StatusOK, rec.Code) + var resp dto.AppApplyResponse + require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &resp)) + assert.Equal(t, "blog", resp.App) + assert.Equal(t, "rev-1", resp.ResultingRevision) + assert.True(t, resp.Pending) +} + +func TestHandler_AppApply_RejectsBadManifest(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + rec := appsRequest(t, handler, http.MethodPost, "/admin/apps/apply", + dto.AppApplyRequest{ManifestTOML: "name = \"Bad!\"\n"}, "admin:apps:write") + assert.Equal(t, http.StatusBadRequest, rec.Code) + var envelope dto.AppError + require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &envelope)) + assert.Equal(t, "invalid-manifest", envelope.Error) +} + +func TestHandler_AppDeploy_MapsOp(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + op := &domain.AppOperation{ + Op: "op-1", Kind: "deploy", App: "blog", InputRevision: "rev-1", + Outcome: domain.AppOutcomeSuccess, + Steps: []domain.AppOperationStep{ + {ID: "preflight", State: domain.AppStepSucceeded}, + {ID: "service.web.replace", State: domain.AppStepSucceeded, Before: "c-old", After: "c-new"}, + }, + } + appSvc.EXPECT().Deploy(mock.Anything, "blog", "", "", false, "test-operation-key").Return(op, nil).Once() + appSvc.EXPECT().Show(mock.Anything, "blog").Return(&in.AppDetail{ + App: "blog", Converged: true, ConvergedRevision: "rev-1", + Services: map[string]in.AppServiceView{"web": {EffectiveRevision: "rev-1", Container: "c-new"}}, + Retained: in.AppRetainedView{Volumes: []string{"gordon-blog--web--vol--data"}, Secrets: []string{"gordon/apps/app-1/web/database-url"}}, + }, nil).Once() + + rec := appsRequest(t, handler, http.MethodPost, "/admin/apps/blog/deploy", + dto.AppDeployRequest{}, "admin:apps:write") + require.Equal(t, http.StatusOK, rec.Code) + var resp dto.AppDeployResponse + require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &resp)) + assert.Equal(t, "op-1", resp.Op) + assert.Equal(t, "success", resp.Outcome) + require.Contains(t, resp.Services, "web") + assert.Equal(t, "deployed", resp.Services["web"].Result) + assert.Equal(t, "c-new", resp.Services["web"].After) + // Effective and retained state comes from the app read model, never + // from the journal outcome. + require.NotNil(t, resp.Effective) + assert.True(t, resp.Effective.Converged) + assert.Equal(t, "rev-1", resp.Effective.Services["web"]) + require.NotNil(t, resp.Retained) + assert.Equal(t, []string{"gordon-blog--web--vol--data"}, resp.Retained.Volumes) +} + +// TestHandler_AppDeploy_MapsCleanupWarnings proves bounded leftovers of a +// successful mutation reach the wire instead of being dropped. +func TestHandler_AppDeploy_MapsCleanupWarnings(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + op := &domain.AppOperation{ + Op: "op-1", Kind: "deploy", App: "blog", InputRevision: "rev-1", + Outcome: domain.AppOutcomeSuccess, + Steps: []domain.AppOperationStep{{ID: "service.web.replace", State: domain.AppStepSucceeded}}, + Warnings: []domain.AppOperationWarning{{ + Service: "web", Leftover: "ctr-old", Detail: "remove: still present", + }}, + } + appSvc.EXPECT().Deploy(mock.Anything, "blog", "", "", false, "test-operation-key").Return(op, nil).Once() + appSvc.EXPECT().Show(mock.Anything, "blog").Return(&in.AppDetail{App: "blog"}, nil).Once() + + rec := appsRequest(t, handler, http.MethodPost, "/admin/apps/blog/deploy", + dto.AppDeployRequest{}, "admin:apps:write") + require.Equal(t, http.StatusOK, rec.Code) + var resp dto.AppDeployResponse + require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &resp)) + require.Len(t, resp.CleanupWarnings, 1) + assert.Equal(t, "web", resp.CleanupWarnings[0].Service) + assert.Equal(t, "ctr-old", resp.CleanupWarnings[0].Leftover) + assert.Equal(t, "remove: still present", resp.CleanupWarnings[0].Detail) +} + +// TestHandler_AppDeploy_RunningReturns202 proves a newly owned deploy whose +// effects are still executing answers 202 Accepted with the durable +// operation identity and a stable running status, so the client can recover +// completion through GET operations/by-key. The claim and its journal are +// already persisted when the handler answers. +func TestHandler_AppDeploy_RunningReturns202(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + running := &domain.AppOperation{ + Op: "op-run", Kind: "deploy", App: "blog", InputRevision: "rev-1", + Steps: []domain.AppOperationStep{ + {ID: "preflight", State: domain.AppStepSucceeded}, + {ID: "service.web.replace", State: domain.AppStepPending}, + }, + } + appSvc.EXPECT().Deploy(mock.Anything, "blog", "", "", false, "test-operation-key").Return(running, nil).Once() + appSvc.EXPECT().Show(mock.Anything, "blog").Return(&in.AppDetail{App: "blog", Services: map[string]in.AppServiceView{}}, nil).Once() + + srv := newScopedTestServer(t, handler, "admin:apps:write") + status, body := appsServerDo(t, srv, http.MethodPost, "/admin/apps/blog/deploy", dto.AppDeployRequest{}) + require.Equal(t, http.StatusAccepted, status) + + var resp dto.AppDeployResponse + require.NoError(t, json.Unmarshal(body, &resp)) + assert.Equal(t, "op-run", resp.Op) + assert.Equal(t, "running", resp.Status) + assert.Empty(t, resp.Outcome, "a running operation has no terminal outcome") + require.Contains(t, resp.Services, "web") + assert.Equal(t, "pending", resp.Services["web"].Result) +} + +// TestHandler_AppDeploy_TerminalReplayReturnsOK proves a duplicate idempotency +// key whose stored operation already reached a successful terminal outcome +// replays with 200 and the stored status instead of 202. +func TestHandler_AppDeploy_TerminalReplayReturnsOK(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + terminal := &domain.AppOperation{ + Op: "op-done", Kind: "deploy", App: "blog", InputRevision: "rev-1", + Outcome: domain.AppOutcomeSuccess, + Steps: []domain.AppOperationStep{{ID: "service.web.replace", State: domain.AppStepSucceeded}}, + } + appSvc.EXPECT().Deploy(mock.Anything, "blog", "", "", false, "test-operation-key").Return(terminal, nil).Once() + appSvc.EXPECT().Show(mock.Anything, "blog").Return(&in.AppDetail{App: "blog", Services: map[string]in.AppServiceView{}}, nil).Once() + + srv := newScopedTestServer(t, handler, "admin:apps:write") + status, body := appsServerDo(t, srv, http.MethodPost, "/admin/apps/blog/deploy", dto.AppDeployRequest{}) + require.Equal(t, http.StatusOK, status) + + var resp dto.AppDeployResponse + require.NoError(t, json.Unmarshal(body, &resp)) + assert.Equal(t, "op-done", resp.Op) + assert.Equal(t, "success", resp.Status) + assert.Equal(t, "success", resp.Outcome) +} + +// TestHandler_AppDeploy_TerminalFailureReplayIsVisible proves a replay of a +// failed operation keeps its conflict mapping while still surfacing the +// stored failure and per-service result, never a 202. +func TestHandler_AppDeploy_TerminalFailureReplayIsVisible(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + failed := &domain.AppOperation{ + Op: "op-fail", Kind: "deploy", App: "blog", InputRevision: "rev-1", + Outcome: domain.AppOutcomeFailed, + Steps: []domain.AppOperationStep{{ + ID: "service.web.replace", State: domain.AppStepFailed, Error: "boom", + }}, + } + replayErr := fmt.Errorf("deployment: operation op-fail previously ended with outcome failed: %w", domain.ErrAppStateConflict) + appSvc.EXPECT().Deploy(mock.Anything, "blog", "", "", false, "test-operation-key").Return(failed, replayErr).Once() + appSvc.EXPECT().Show(mock.Anything, "blog").Return(&in.AppDetail{App: "blog", Services: map[string]in.AppServiceView{}}, nil).Once() + + srv := newScopedTestServer(t, handler, "admin:apps:write") + status, body := appsServerDo(t, srv, http.MethodPost, "/admin/apps/blog/deploy", dto.AppDeployRequest{}) + require.Equal(t, http.StatusConflict, status) + + var resp dto.AppDeployResponse + require.NoError(t, json.Unmarshal(body, &resp)) + assert.Equal(t, "op-fail", resp.Op) + assert.Equal(t, "failed", resp.Status) + assert.Equal(t, "failed", resp.Outcome) + require.Contains(t, resp.Services, "web") + assert.Equal(t, "failed", resp.Services["web"].Result) + assert.Equal(t, "boom", resp.Services["web"].Error) +} + +// assertRunningReplayJournal pins the replay contract shared by a running +// replay before and after any service step has started: 409, the operation +// journal DTO with a running status and no terminal outcome, and never the +// mapped preflight error envelope. The top-level key set is asserted so both +// cases are proven to share one wire shape. +func assertRunningReplayJournal(t *testing.T, status int, body []byte, opID, lastStep string) { + t.Helper() + require.Equal(t, http.StatusConflict, status) + + var shape map[string]json.RawMessage + require.NoError(t, json.Unmarshal(body, &shape)) + _, hasError := shape["error"] + assert.False(t, hasError, "a running replay must not degrade to the preflight error envelope") + keys := make([]string, 0, len(shape)) + for key := range shape { + keys = append(keys, key) + } + sort.Strings(keys) + assert.Equal(t, + []string{"app", "effective", "op", "outcome", "retained", "revision", "services", "status", "steps"}, + keys, "a running replay always returns the operation journal shape") + + var resp dto.AppDeployResponse + require.NoError(t, json.Unmarshal(body, &resp)) + assert.Equal(t, opID, resp.Op) + assert.Equal(t, "running", resp.Status) + assert.Empty(t, resp.Outcome, "a running operation has no terminal outcome") + found := false + for _, step := range resp.Steps { + if step.ID == lastStep { + found = true + assert.Equal(t, domain.AppStepPending, step.State) + } + } + assert.True(t, found, "the journal must expose step %q", lastStep) +} + +// TestHandler_AppDeploy_RunningReplayBeforeServiceSteps proves a same-key +// replay of an operation still in flight answers 409 with the stored journal +// even before any service step has started. A non-terminal claim is a running +// journal, not a preflight failure, so the mapped error envelope never +// replaces it. +func TestHandler_AppDeploy_RunningReplayBeforeServiceSteps(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + running := &domain.AppOperation{ + Op: "op-run-pre", Kind: "deploy", App: "blog", InputRevision: "rev-1", + Steps: []domain.AppOperationStep{{ID: "preflight", State: domain.AppStepPending}}, + } + replayErr := fmt.Errorf("deployment: operation op-run-pre has not reached a terminal outcome: %w", domain.ErrAppStateConflict) + appSvc.EXPECT().Deploy(mock.Anything, "blog", "", "", false, "test-operation-key").Return(running, replayErr).Once() + appSvc.EXPECT().Show(mock.Anything, "blog").Return(&in.AppDetail{App: "blog", Services: map[string]in.AppServiceView{}}, nil).Once() + + srv := newScopedTestServer(t, handler, "admin:apps:write") + status, body := appsServerDo(t, srv, http.MethodPost, "/admin/apps/blog/deploy", dto.AppDeployRequest{}) + + assertRunningReplayJournal(t, status, body, "op-run-pre", "preflight") +} + +// TestHandler_AppDeploy_RunningReplayAfterServiceSteps is the after-steps +// twin: the same running replay, now with a pending service step, must answer +// the identical 409 journal shape. Only the journal contents may differ from +// the before-steps case, never the envelope. +func TestHandler_AppDeploy_RunningReplayAfterServiceSteps(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + running := &domain.AppOperation{ + Op: "op-run-post", Kind: "deploy", App: "blog", InputRevision: "rev-1", + Steps: []domain.AppOperationStep{ + {ID: "preflight", State: domain.AppStepSucceeded}, + {ID: "service.web.replace", State: domain.AppStepPending}, + }, + } + replayErr := fmt.Errorf("deployment: operation op-run-post has not reached a terminal outcome: %w", domain.ErrAppStateConflict) + appSvc.EXPECT().Deploy(mock.Anything, "blog", "", "", false, "test-operation-key").Return(running, replayErr).Once() + appSvc.EXPECT().Show(mock.Anything, "blog").Return(&in.AppDetail{App: "blog", Services: map[string]in.AppServiceView{}}, nil).Once() + + srv := newScopedTestServer(t, handler, "admin:apps:write") + status, body := appsServerDo(t, srv, http.MethodPost, "/admin/apps/blog/deploy", dto.AppDeployRequest{}) + + assertRunningReplayJournal(t, status, body, "op-run-post", "service.web.replace") +} + +func TestHandler_AppDeploy_ServiceScopeRequiredIs400(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + scopeErr := fmt.Errorf("app blog has several services (db, web): pass --service NAME or --all: %w", domain.ErrAppServiceScope) + appSvc.EXPECT().Deploy(mock.Anything, "blog", "", "", false, "test-operation-key").Return(nil, scopeErr).Once() + + srv := newScopedTestServer(t, handler, "admin:apps:write") + status, body := appsServerDo(t, srv, http.MethodPost, "/admin/apps/blog/deploy", dto.AppDeployRequest{}) + require.Equal(t, http.StatusBadRequest, status) + + var resp dto.AppError + require.NoError(t, json.Unmarshal(body, &resp)) + assert.Equal(t, "service-scope-required", resp.Error) + assert.Contains(t, resp.Message, "(db, web)") +} + +func TestHandler_AppDeploy_PassesAll(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + appSvc.EXPECT().Deploy(mock.Anything, "blog", "", "", true, "test-operation-key").Return(nil, domain.ErrAppStateConflict).Once() + + srv := newScopedTestServer(t, handler, "admin:apps:write") + status, _ := appsServerDo(t, srv, http.MethodPost, "/admin/apps/blog/deploy", dto.AppDeployRequest{All: true}) + require.Equal(t, http.StatusConflict, status) +} + +func TestHandler_AppRestart_PassesServiceAndAll(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + appSvc.EXPECT().Restart(mock.Anything, "blog", "", true, "test-operation-key").Return(nil, domain.ErrAppStateConflict).Once() + appSvc.EXPECT().Restart(mock.Anything, "blog", "web", false, "test-operation-key").Return(nil, domain.ErrAppStateConflict).Once() + + srv := newScopedTestServer(t, handler, "admin:apps:write") + status, _ := appsServerDo(t, srv, http.MethodPost, "/admin/apps/blog/restart?all=true", nil) + require.Equal(t, http.StatusConflict, status) + status, _ = appsServerDo(t, srv, http.MethodPost, "/admin/apps/blog/restart?service=web", nil) + require.Equal(t, http.StatusConflict, status) +} + +// TestHandler_AppDeploy_RejectsMissingIdempotencyKey keeps the request +// validation contract: a deploy without a valid key never reaches the +// service. +func TestHandler_AppDeploy_RejectsMissingIdempotencyKey(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + srv := newScopedTestServer(t, handler, "admin:apps:write") + req, err := http.NewRequest(http.MethodPost, srv.URL+"/admin/apps/blog/deploy", nil) + require.NoError(t, err) + resp, err := srv.Client().Do(req) + require.NoError(t, err) + defer func() { require.NoError(t, resp.Body.Close()) }() + + require.Equal(t, http.StatusBadRequest, resp.StatusCode) +} + +func TestHandler_AppList_RequiresScope(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + rec := appsRequest(t, handler, http.MethodGet, "/admin/apps", nil, "admin:status:read") + assert.Equal(t, http.StatusForbidden, rec.Code) + + appSvc.EXPECT().List(mock.Anything).Return(nil, nil).Once() + rec = appsRequest(t, handler, http.MethodGet, "/admin/apps", nil, "admin:apps:read") + assert.Equal(t, http.StatusOK, rec.Code) +} + +func TestHandler_AppSecretsSet_Forwards(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + appSvc.EXPECT().SetSecrets(mock.Anything, "blog", "web", map[string]string{"DATABASE_URL": "v"}).Return(nil).Once() + rec := appsRequest(t, handler, http.MethodPost, "/admin/apps/blog/secrets/set", + dto.AppSecretSetRequest{Service: "web", Secrets: map[string]string{"DATABASE_URL": "v"}}, "admin:apps:write") + assert.Equal(t, http.StatusOK, rec.Code) +} + +// TestHandler_AppLifecycle_UnknownAppIsNotFound proves a lifecycle +// mutation of a name with no app identity is a 404, not a 500 or a silent +// success. +func TestHandler_AppLifecycle_UnknownAppIsNotFound(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + appSvc.EXPECT().Stop(mock.Anything, "ghost", "test-operation-key"). + Return(nil, fmt.Errorf("deployment: app %q does not exist: %w", "ghost", domain.ErrAppNotFound)).Once() + + rec := appsRequest(t, handler, http.MethodPost, "/admin/apps/ghost/stop", nil, "admin:apps:write") + require.Equal(t, http.StatusNotFound, rec.Code) + var envelope dto.AppError + require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &envelope)) + assert.Equal(t, "app-not-found", envelope.Error) +} + +// TestHandler_AppLifecycle_ReplayConflictCarriesJournal proves a replayed +// key that never finished returns 409 with the stored journal, so the +// caller can inspect what was recorded instead of guessing. +func TestHandler_AppLifecycle_ReplayConflictCarriesJournal(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + op := &domain.AppOperation{ + Op: "test-operation-key", Kind: "remove", App: "blog", + Steps: []domain.AppOperationStep{{ID: "service.web.remove", State: domain.AppStepPending}}, + } + appSvc.EXPECT().Remove(mock.Anything, "blog", "test-operation-key"). + Return(op, domain.ErrAppStateConflict).Once() + // A removal that already retired the incarnation has no app read + // model left, so the response carries the journal alone. + appSvc.EXPECT().Show(mock.Anything, "blog"). + Return(nil, domain.ErrAppNotFound).Once() + + rec := appsRequest(t, handler, http.MethodPost, "/admin/apps/blog/remove", nil, "admin:apps:write") + require.Equal(t, http.StatusConflict, rec.Code) + var resp dto.AppDeployResponse + require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &resp)) + assert.Equal(t, "test-operation-key", resp.Op) + assert.Nil(t, resp.Effective) + assert.Nil(t, resp.Retained) +} + +func TestHandler_AppShow_MapsReadModel(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + started := time.Date(2026, 1, 2, 3, 4, 5, 0, time.UTC) + appSvc.EXPECT().Show(mock.Anything, "blog").Return(&in.AppDetail{ + App: "blog", + DesiredRevision: "rev-2", + DesiredStatus: "pending", + Pending: true, + Converged: false, + ConvergedRevision: "rev-1", + Stopped: true, + Services: map[string]in.AppServiceView{ + "web": {EffectiveRevision: "rev-1", Digest: "sha256:x", Container: "ctr-1", RestartUnsafe: true}, + }, + Retained: in.AppRetainedView{ + Volumes: []string{"gordon-blog--web--vol--data"}, + Secrets: []string{"gordon/apps/app-1/web/database-url"}, + Images: []string{"registry.example.com/blog/web:1.4.2"}, + }, + LastOp: "op-9", + LastOpKind: "deploy", + LastOutcome: "partial", + LastOpStartedAt: started, + }, nil).Once() + + rec := appsRequest(t, handler, http.MethodGet, "/admin/apps/blog", nil, "admin:apps:read") + require.Equal(t, http.StatusOK, rec.Code) + var resp dto.AppShowResponse + require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &resp)) + + assert.Equal(t, "rev-2", resp.Desired.Revision) + assert.True(t, resp.Desired.Pending) + assert.False(t, resp.Active.Converged) + assert.Equal(t, "rev-1", resp.Active.ConvergedRevision) + assert.True(t, resp.Intent.Stopped) + assert.True(t, resp.Active.Services["web"].RestartUnsafe, "restart safety must reach the wire") + assert.Equal(t, "ctr-1", resp.Active.Services["web"].Container) + assert.Equal(t, []string{"gordon-blog--web--vol--data"}, resp.Retained.Volumes) + assert.Equal(t, []string{"registry.example.com/blog/web:1.4.2"}, resp.Retained.Images) + require.NotNil(t, resp.LastOp) + assert.Equal(t, "op-9", resp.LastOp.Op) + assert.Equal(t, "deploy", resp.LastOp.Kind) + assert.Equal(t, "partial", resp.LastOp.Outcome) + assert.Equal(t, started, resp.LastOp.StartedAt) +} + +// TestHandler_AppShow_UnknownAppIsNotFound proves showing a name with no +// app identity is 404, not an empty success. +func TestHandler_AppShow_UnknownAppIsNotFound(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + appSvc.EXPECT().Show(mock.Anything, "ghost"). + Return(nil, fmt.Errorf("apps: app %q does not exist: %w", "ghost", domain.ErrAppNotFound)).Once() + + rec := appsRequest(t, handler, http.MethodGet, "/admin/apps/ghost", nil, "admin:apps:read") + require.Equal(t, http.StatusNotFound, rec.Code) + var envelope dto.AppError + require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &envelope)) + assert.Equal(t, "app-not-found", envelope.Error) +} + +func TestHandler_AppList_MapsReadModel(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + appSvc.EXPECT().List(mock.Anything).Return([]in.AppSummary{{ + App: "blog", Desired: "rev-2", DesiredStatus: "pending", + Active: "rev-1", Converged: false, Pending: true, LastOutcome: "partial", + }}, nil).Once() + + rec := appsRequest(t, handler, http.MethodGet, "/admin/apps", nil, "admin:apps:read") + require.Equal(t, http.StatusOK, rec.Code) + var envelope struct { + Apps []dto.AppSummaryDTO `json:"apps"` + } + require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &envelope)) + require.Len(t, envelope.Apps, 1) + assert.Equal(t, "rev-2", envelope.Apps[0].Desired) + assert.Equal(t, "pending", envelope.Apps[0].DesiredStatus) + assert.True(t, envelope.Apps[0].Pending) + assert.Equal(t, "partial", envelope.Apps[0].LastOutcome) +} + +func TestHandler_AppOpLookup_Recovers(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + op := &domain.AppOperation{Op: "op-9", Kind: "deploy", App: "blog", InputRevision: "rev-1", Outcome: "success"} + appSvc.EXPECT().OperationByKey(mock.Anything, "blog", "op-9").Return(op, nil).Once() + rec := appsRequest(t, handler, http.MethodGet, "/admin/apps/blog/operations/by-key/op-9", nil, "admin:apps:read") + require.Equal(t, http.StatusOK, rec.Code) + var resp dto.AppDeployResponse + require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &resp)) + assert.Equal(t, "op-9", resp.Op) +} + +// TestHandler_AppOpLookup_GatesDiagnosticsByLogsScope proves failure +// diagnostics are returned only to callers holding the logs read scope; an +// apps-read actor receives the stable error alone. +func TestHandler_AppOpLookup_GatesDiagnosticsByLogsScope(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + op := &domain.AppOperation{ + Op: "op-9", Kind: "deploy", App: "blog", InputRevision: "rev-1", Outcome: "failed", + Steps: []domain.AppOperationStep{{ + ID: "service.web.replace", State: domain.AppStepFailed, + Error: "deployment: readiness failed", Diagnostics: []string{"SECRET=supersecret"}, + }}, + } + appSvc.EXPECT().OperationByKey(mock.Anything, "blog", "op-9").Return(op, nil).Twice() + + appsRead := appsRequest(t, handler, http.MethodGet, "/admin/apps/blog/operations/by-key/op-9", nil, "admin:apps:read") + require.Equal(t, http.StatusOK, appsRead.Code) + assert.NotContains(t, appsRead.Body.String(), "supersecret", "apps-read must not receive application diagnostics") + assert.NotContains(t, appsRead.Body.String(), "diagnostics") + + logsRead := appsRequest(t, handler, http.MethodGet, "/admin/apps/blog/operations/by-key/op-9", + nil, "admin:apps:read", "admin:logs:read") + require.Equal(t, http.StatusOK, logsRead.Code) + var resp dto.AppDeployResponse + require.NoError(t, json.Unmarshal(logsRead.Body.Bytes(), &resp)) + require.Len(t, resp.Steps, 1) + require.Len(t, resp.Steps[0].Diagnostics, 1) + assert.Equal(t, "SECRET=supersecret", resp.Steps[0].Diagnostics[0]) +} + +// TestHandler_AppApply_DryRunOmitsRevision proves a dry-run apply never +// reports a revision: nothing was persisted, so resulting_revision is +// omitted instead of echoing the app name. +func TestHandler_AppApply_DryRunOmitsRevision(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + appSvc.EXPECT().Apply(mock.Anything, mock.Anything, mock.Anything, true).Return( + nil, + &in.AppDryRunResult{App: "blog", Valid: true}, + nil, + ).Once() + + rec := appsRequest(t, handler, http.MethodPost, "/admin/apps/apply", + dto.AppApplyRequest{ManifestTOML: validAppManifest, DryRun: true}, "admin:apps:write") + require.Equal(t, http.StatusOK, rec.Code) + var resp dto.AppApplyResponse + require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &resp)) + assert.Equal(t, "blog", resp.App) + assert.Empty(t, resp.ResultingRevision) + assert.NotContains(t, rec.Body.String(), "resulting_revision") +} + +func TestHandler_AppSecretsList_MissingAppIs404(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + appSvc.EXPECT().ListSecrets(mock.Anything, "ghost", "").Return( + nil, + fmt.Errorf("apps: app %q does not exist: %w", "ghost", domain.ErrAppNotFound), + ).Once() + + rec := appsRequest(t, handler, http.MethodGet, "/admin/apps/ghost/secrets", nil, "admin:apps:read") + require.Equal(t, http.StatusNotFound, rec.Code) + assert.Contains(t, rec.Body.String(), "app-not-found") +} + +func TestHandler_AppApply_MapsDevicePolicyViolation(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + appSvc.EXPECT().Apply(mock.Anything, mock.Anything, mock.Anything, false).Return( + nil, nil, + fmt.Errorf("%w: app %q service %q device %q: refused by administrative device policy", domain.ErrDevicePolicy, "blog", "web", "test_gpu"), + ).Once() + + rec := appsRequest(t, handler, http.MethodPost, "/admin/apps/apply", + dto.AppApplyRequest{ManifestTOML: validAppManifest}, "admin:apps:write") + assert.Equal(t, http.StatusBadRequest, rec.Code) + var envelope dto.AppError + require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &envelope)) + assert.Equal(t, "device-policy-violation", envelope.Error) +} + +func TestHandler_AppApply_MapsRuntimeUnsupported(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + appSvc.EXPECT().Apply(mock.Anything, mock.Anything, mock.Anything, false).Return( + nil, nil, + fmt.Errorf("%w: podman 4.9.5 below minimum 5.4", domain.ErrRuntimeUnsupported), + ).Once() + + rec := appsRequest(t, handler, http.MethodPost, "/admin/apps/apply", + dto.AppApplyRequest{ManifestTOML: validAppManifest}, "admin:apps:write") + assert.Equal(t, http.StatusBadRequest, rec.Code) + var envelope dto.AppError + require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &envelope)) + assert.Equal(t, "runtime-unsupported", envelope.Error) +} diff --git a/internal/adapters/in/http/admin/handler_retired.go b/internal/adapters/in/http/admin/handler_retired.go new file mode 100644 index 000000000..3cd734ac9 --- /dev/null +++ b/internal/adapters/in/http/admin/handler_retired.go @@ -0,0 +1,49 @@ +package admin + +import ( + "net/http" + + "github.com/bnema/gordon/internal/adapters/dto" +) + +// retiredMutationPaths are legacy endpoints removed by the v3 +// declarative-apps cutover. Only the 410 Gone rejection dispatcher below +// remains; no old business handlers or compatibility behavior is kept. +var retiredMutationPaths = []string{ + "/deploy", + "/restart", + "/deploy-intent", + "/routes", + "/routes/by-image", + "/attachments", + "/attachments/by-image", + "/attachments/orphans", + "/attachments/prune", + "/bootstrap", + "/preview", + "/previews", + "/autoroute/allowed-domains", + "/secrets", +} + +// isRetiredMutation reports whether a request path targets a removed +// legacy mutation endpoint. +func isRetiredMutation(path string) bool { + for _, prefix := range retiredMutationPaths { + if path == prefix || len(path) > len(prefix) && path[:len(prefix)+1] == prefix+"/" { + return true + } + } + return false +} + +// handleRetiredMutation answers removed endpoints with 410 Gone and the +// explicit endpoint-retired error. All methods on retired prefixes retire +// together: no old business handlers or compatibility behavior is kept. +func (h *Handler) handleRetiredMutation(w http.ResponseWriter, _ *http.Request, _ string) { + h.sendJSON(w, http.StatusGone, dto.AppError{ + Error: "endpoint-retired", + Message: "this endpoint was removed by the v3 declarative-apps cutover; use /admin/apps/* instead", + Hint: "see gordon apps --help", + }) +} diff --git a/internal/adapters/in/http/admin/handler_retired_test.go b/internal/adapters/in/http/admin/handler_retired_test.go new file mode 100644 index 000000000..24948721f --- /dev/null +++ b/internal/adapters/in/http/admin/handler_retired_test.go @@ -0,0 +1,62 @@ +package admin + +import ( + "encoding/json" + "net/http" + "net/http/httptest" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/dto" + inmocks "github.com/bnema/gordon/internal/boundaries/in/mocks" +) + +func retiredTestHandler(t *testing.T) *Handler { + t.Helper() + return NewHandler(HandlerDeps{ + ConfigSvc: inmocks.NewMockConfigService(t), + AuthSvc: inmocks.NewMockAuthService(t), + ContainerSvc: inmocks.NewMockContainerService(t), + HealthSvc: inmocks.NewMockHealthService(t), + Log: testLogger(), + ReloadTrigger: noopReloadTrigger{}, + }) +} + +func TestHandler_RetiredMutations_Return410(t *testing.T) { + handler := retiredTestHandler(t) + targets := []struct { + method string + path string + }{ + {http.MethodPost, "/admin/deploy/example.com"}, + {http.MethodPost, "/admin/restart/example.com"}, + {http.MethodPost, "/admin/deploy-intent"}, + {http.MethodPost, "/admin/routes"}, + {http.MethodGet, "/admin/routes"}, + {http.MethodDelete, "/admin/routes/example.com"}, + {http.MethodPost, "/admin/attachments"}, + {http.MethodGet, "/admin/attachments/by-image/x"}, + {http.MethodPost, "/admin/attachments/prune"}, + {http.MethodGet, "/admin/attachments/orphans"}, + {http.MethodPost, "/admin/bootstrap"}, + {http.MethodPost, "/admin/preview/blog"}, + {http.MethodGet, "/admin/previews"}, + {http.MethodPost, "/admin/autoroute/allowed-domains"}, + {http.MethodGet, "/admin/secrets/example.com"}, + {http.MethodPost, "/admin/secrets/example.com"}, + {http.MethodDelete, "/admin/secrets/example.com/KEY"}, + } + for _, target := range targets { + req := httptest.NewRequest(target.method, target.path, nil) + req = req.WithContext(ctxWithScopes("admin:*:*")) + rec := httptest.NewRecorder() + handler.ServeHTTP(rec, req) + require.Equal(t, http.StatusGone, rec.Code, "%s %s", target.method, target.path) + var envelope dto.AppError + require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &envelope), target.path) + assert.Equal(t, "endpoint-retired", envelope.Error, target.path) + } +} diff --git a/internal/adapters/in/http/admin/handler_test.go b/internal/adapters/in/http/admin/handler_test.go index b6f84eaa9..b2b4d6f41 100644 --- a/internal/adapters/in/http/admin/handler_test.go +++ b/internal/adapters/in/http/admin/handler_test.go @@ -9,8 +9,6 @@ import ( "io" "net/http" "net/http/httptest" - "net/url" - "strings" "testing" "time" @@ -29,18 +27,6 @@ type stubImageService struct { pruneFunc func(context.Context, domain.ImagePruneOptions) (domain.ImagePruneReport, error) } -type previewContainerService struct { - *inmocks.MockContainerService - preview func(context.Context, string) (*domain.CleanupReport, error) -} - -func (s *previewContainerService) PreviewRemovedRouteCleanup(ctx context.Context, routeDomain string) (*domain.CleanupReport, error) { - if s.preview == nil { - return nil, nil - } - return s.preview(ctx, routeDomain) -} - type noopReloadTrigger struct{} func (noopReloadTrigger) Trigger(context.Context) error { return nil } @@ -100,7 +86,6 @@ func newTestHandler(t *testing.T, opts ...func(*HandlerDeps)) *Handler { AuthSvc: inmocks.NewMockAuthService(t), ContainerSvc: inmocks.NewMockContainerService(t), HealthSvc: inmocks.NewMockHealthService(t), - SecretSvc: inmocks.NewMockSecretService(t), Log: testLogger(), ReloadTrigger: noopReloadTrigger{}, } @@ -170,18 +155,15 @@ func TestHandler_VolumesPrune_RequiresVolumesWriteScope(t *testing.T) { // Routes endpoint tests -func TestHandler_RoutesGet_RequiresReadScope(t *testing.T) { +func TestHandler_Status_RequiresReadScope(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) authSvc := inmocks.NewMockAuthService(t) containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) handler := newTestHandler(t, func(d *HandlerDeps) { d.ConfigSvc = configSvc d.AuthSvc = authSvc d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - d.RegistrySvc = inmocks.NewMockRegistryService(t) }) tests := []struct { @@ -190,13 +172,8 @@ func TestHandler_RoutesGet_RequiresReadScope(t *testing.T) { wantStatus int }{ { - name: "routes read access granted", - scopes: []string{"admin:routes:read"}, - wantStatus: http.StatusOK, - }, - { - name: "routes wildcard access granted", - scopes: []string{"admin:routes:*"}, + name: "status read access granted", + scopes: []string{"admin:status:read"}, wantStatus: http.StatusOK, }, { @@ -204,19 +181,9 @@ func TestHandler_RoutesGet_RequiresReadScope(t *testing.T) { scopes: []string{"admin:*:*"}, wantStatus: http.StatusOK, }, - { - name: "only write access denied", - scopes: []string{"admin:routes:write"}, - wantStatus: http.StatusForbidden, - }, { name: "wrong resource denied", - scopes: []string{"admin:secrets:read"}, - wantStatus: http.StatusForbidden, - }, - { - name: "no scopes denied", - scopes: []string{}, + scopes: []string{"admin:routes:read"}, wantStatus: http.StatusForbidden, }, } @@ -224,10 +191,13 @@ func TestHandler_RoutesGet_RequiresReadScope(t *testing.T) { for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { if tt.wantStatus == http.StatusOK { - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{}).Maybe() + configSvc.EXPECT().GetRegistryDomain().Return("registry.example.com").Maybe() + configSvc.EXPECT().GetRegistryPort().Return(5000).Maybe() + configSvc.EXPECT().GetServerPort().Return(8080).Maybe() + configSvc.EXPECT().IsNetworkIsolationEnabled().Return(false).Maybe() } - req := httptest.NewRequest("GET", "/admin/routes", nil) + req := httptest.NewRequest("GET", "/admin/status", nil) req = req.WithContext(ctxWithScopes(tt.scopes...)) rec := httptest.NewRecorder() @@ -238,60 +208,82 @@ func TestHandler_RoutesGet_RequiresReadScope(t *testing.T) { } } -func TestHandler_RoutesPost_RequiresWriteScope(t *testing.T) { +// Config endpoint tests + +func TestHandler_Config_HidesSensitiveInventoryByDefault(t *testing.T) { + configSvc := inmocks.NewMockConfigService(t) + handler := newTestHandler(t, func(d *HandlerDeps) { d.ConfigSvc = configSvc }) + + configSvc.EXPECT().GetServerPort().Return(8080).Once() + configSvc.EXPECT().GetRegistryPort().Return(5000).Once() + configSvc.EXPECT().GetRegistryDomain().Return("registry.example.com").Once() + configSvc.EXPECT().IsNetworkIsolationEnabled().Return(false).Once() + configSvc.EXPECT().GetNetworkPrefix().Return("gordon").Once() + configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{"db.example.com": "http://10.0.0.5:5432"}).Once() + + server := newScopedTestServer(t, handler, "admin:config:read") + resp, err := http.Get(server.URL + "/admin/config") + require.NoError(t, err) + defer resp.Body.Close() + bodyBytes, err := io.ReadAll(resp.Body) + require.NoError(t, err) + + require.Equal(t, http.StatusOK, resp.StatusCode) + body := string(bodyBytes) + var raw map[string]any + require.NoError(t, json.Unmarshal([]byte(body), &raw)) + serverConfig := raw["server"].(map[string]any) + assert.NotContains(t, serverConfig, "data_dir") + assert.NotContains(t, body, "/var/lib/gordon") + assert.NotContains(t, body, "10.0.0.5") +} + +func TestHandler_Config_RequiresReadScope(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) authSvc := inmocks.NewMockAuthService(t) containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) handler := newTestHandler(t, func(d *HandlerDeps) { d.ConfigSvc = configSvc d.AuthSvc = authSvc d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc }) - routeJSON := `{"domain": "app.example.com", "image": "myapp:latest"}` - tests := []struct { name string scopes []string wantStatus int }{ { - name: "routes write access granted", - scopes: []string{"admin:routes:write"}, - wantStatus: http.StatusCreated, - }, - { - name: "routes wildcard access granted", - scopes: []string{"admin:routes:*"}, - wantStatus: http.StatusCreated, + name: "config read access granted", + scopes: []string{"admin:config:read"}, + wantStatus: http.StatusOK, }, { name: "all admin access granted", scopes: []string{"admin:*:*"}, - wantStatus: http.StatusCreated, - }, - { - name: "only read access denied", - scopes: []string{"admin:routes:read"}, - wantStatus: http.StatusForbidden, + wantStatus: http.StatusOK, }, { name: "wrong resource denied", - scopes: []string{"admin:secrets:write"}, + scopes: []string{"admin:routes:read"}, wantStatus: http.StatusForbidden, }, } for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { - if tt.wantStatus == http.StatusCreated { - configSvc.EXPECT().AddRoute(mock.Anything, mock.AnythingOfType("domain.Route")).Return(nil).Maybe() + if tt.wantStatus == http.StatusOK { + configSvc.EXPECT().GetServerPort().Return(8080).Maybe() + configSvc.EXPECT().GetRegistryPort().Return(5000).Maybe() + configSvc.EXPECT().GetRegistryDomain().Return("registry.example.com").Maybe() + configSvc.EXPECT().GetDataDir().Return("/var/lib/gordon").Maybe() + configSvc.EXPECT().IsNetworkIsolationEnabled().Return(false).Maybe() + configSvc.EXPECT().GetNetworkPrefix().Return("gordon").Maybe() + configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{}).Maybe() } - req := httptest.NewRequest("POST", "/admin/routes", bytes.NewBufferString(routeJSON)) + req := httptest.NewRequest("GET", "/admin/config", nil) req = req.WithContext(ctxWithScopes(tt.scopes...)) rec := httptest.NewRecorder() @@ -302,1790 +294,141 @@ func TestHandler_RoutesPost_RequiresWriteScope(t *testing.T) { } } -func TestHandler_RoutesPost_InvalidRouteDomain(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - server := newScopedTestServer(t, handler, "admin:routes:write") - - configSvc.EXPECT().AddRoute(mock.Anything, domain.Route{Domain: "localhost", Image: "myapp:latest", HTTPS: true}).Return(domain.ErrRouteDomainInvalid).Once() - - req, err := http.NewRequest(http.MethodPost, server.URL+"/admin/routes", bytes.NewBufferString(`{"domain":"localhost","image":"myapp:latest"}`)) - require.NoError(t, err) - - resp, err := server.Client().Do(req) - require.NoError(t, err) - defer resp.Body.Close() - - assert.Equal(t, http.StatusBadRequest, resp.StatusCode) - body, err := io.ReadAll(resp.Body) - require.NoError(t, err) - assert.JSONEq(t, `{"error":"`+domain.ErrRouteDomainInvalid.Error()+`"}`, string(body)) -} +// Reload endpoint tests -func TestHandler_RoutesPut_RequiresWriteScope(t *testing.T) { +func TestHandler_Reload_RequiresWriteScope(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) authSvc := inmocks.NewMockAuthService(t) containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) + reloadTrigger := &reloadTriggerRecorder{} handler := newTestHandler(t, func(d *HandlerDeps) { d.ConfigSvc = configSvc d.AuthSvc = authSvc d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc + d.ReloadTrigger = reloadTrigger }) - routeJSON := `{"image": "myapp:v2"}` - tests := []struct { name string scopes []string wantStatus int }{ { - name: "routes write access granted", - scopes: []string{"admin:routes:write"}, + name: "config write access granted", + scopes: []string{"admin:config:write"}, + wantStatus: http.StatusOK, + }, + { + name: "all admin access granted", + scopes: []string{"admin:*:*"}, wantStatus: http.StatusOK, }, { name: "only read access denied", - scopes: []string{"admin:routes:read"}, + scopes: []string{"admin:config:read"}, wantStatus: http.StatusForbidden, }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - if tt.wantStatus == http.StatusOK { - configSvc.EXPECT().GetRoute(mock.Anything, "app.example.com").Return(&domain.Route{Domain: "app.example.com", Image: "myapp:v1", HTTPS: true}, nil).Maybe() - configSvc.EXPECT().UpdateRoute(mock.Anything, mock.AnythingOfType("domain.Route")).Return(nil).Maybe() - } - - req := httptest.NewRequest("PUT", "/admin/routes/app.example.com", bytes.NewBufferString(routeJSON)) - req = req.WithContext(ctxWithScopes(tt.scopes...)) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, tt.wantStatus, rec.Code) - }) - } -} - -func TestHandler_RoutesPut_InvalidRouteDomain(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - server := newScopedTestServer(t, handler, "admin:routes:write") - - configSvc.EXPECT().GetRoute(mock.Anything, "localhost").Return(&domain.Route{Domain: "localhost", Image: "myapp:v1", HTTPS: true}, nil).Once() - configSvc.EXPECT().UpdateRoute(mock.Anything, domain.Route{Domain: "localhost", Image: "myapp:v2", HTTPS: true}).Return(domain.ErrRouteDomainInvalid).Once() - - req, err := http.NewRequest(http.MethodPut, server.URL+"/admin/routes/localhost", bytes.NewBufferString(`{"image":"myapp:v2"}`)) - require.NoError(t, err) - - resp, err := server.Client().Do(req) - require.NoError(t, err) - defer resp.Body.Close() - - assert.Equal(t, http.StatusBadRequest, resp.StatusCode) - body, err := io.ReadAll(resp.Body) - require.NoError(t, err) - assert.JSONEq(t, `{"error":"`+domain.ErrRouteDomainInvalid.Error()+`"}`, string(body)) -} - -func TestHandler_RoutesPost_AllowsRouteCreationWithoutRegistryManifest(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - registrySvc := inmocks.NewMockRegistryService(t) - route := domain.Route{Domain: "app.example.com", Image: "myapp:latest", HTTPS: true} - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - d.RegistrySvc = registrySvc - }) - - configSvc.EXPECT().AddRoute(mock.Anything, route).Return(nil).Once() - - server := newScopedTestServer(t, handler, "admin:routes:write") - req, err := http.NewRequest(http.MethodPost, server.URL+"/admin/routes", bytes.NewBufferString(`{"domain":"app.example.com","image":"myapp:latest"}`)) - require.NoError(t, err) - - resp, err := server.Client().Do(req) - require.NoError(t, err) - defer resp.Body.Close() - - body, err := io.ReadAll(resp.Body) - require.NoError(t, err) - assert.Equal(t, http.StatusCreated, resp.StatusCode) - assert.JSONEq(t, `{"domain":"app.example.com","image":"myapp:latest","https":true}`, string(body)) - registrySvc.AssertNotCalled(t, "GetManifest", mock.Anything, mock.Anything, mock.Anything) -} - -func TestHandleRoutesPost_DefaultsHTTPSWhenOmitted(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - configSvc.EXPECT().AddRoute(mock.Anything, domain.Route{Domain: "app.example.com", Image: "myapp:latest", HTTPS: true}).Return(nil).Once() - - server := newScopedTestServer(t, handler, "admin:routes:write") - req, err := http.NewRequest(http.MethodPost, server.URL+"/admin/routes", bytes.NewBufferString(`{"domain":"app.example.com","image":"myapp:latest"}`)) - require.NoError(t, err) - - resp, err := server.Client().Do(req) - require.NoError(t, err) - defer resp.Body.Close() - - body, err := io.ReadAll(resp.Body) - require.NoError(t, err) - assert.Equal(t, http.StatusCreated, resp.StatusCode) - assert.JSONEq(t, `{"domain":"app.example.com","image":"myapp:latest","https":true}`, string(body)) -} - -func TestHandleRoutesPut_PreservesStoredHTTPSWhenOmitted(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - registrySvc := inmocks.NewMockRegistryService(t) - route := domain.Route{Domain: "app.example.com", Image: "myapp:v2", HTTPS: false} - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - d.RegistrySvc = registrySvc - }) - - configSvc.EXPECT().GetRoute(mock.Anything, "app.example.com").Return(&domain.Route{Domain: "app.example.com", Image: "myapp:latest", HTTPS: false}, nil).Once() - configSvc.EXPECT().UpdateRoute(mock.Anything, route).Return(nil).Once() - - req := httptest.NewRequest("PUT", "/admin/routes/app.example.com", bytes.NewBufferString(`{"image":"myapp:v2"}`)) - req = req.WithContext(ctxWithScopes("admin:routes:write")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusOK, rec.Code) - assert.JSONEq(t, `{"domain":"app.example.com","image":"myapp:v2","https":false}`, rec.Body.String()) - registrySvc.AssertNotCalled(t, "GetManifest", mock.Anything, mock.Anything, mock.Anything) -} - -func TestHandleRoutesPut_OverridesStoredHTTPSWhenProvided(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - configSvc.EXPECT().GetRoute(mock.Anything, "app.example.com").Return(&domain.Route{Domain: "app.example.com", Image: "myapp:latest", HTTPS: false}, nil).Once() - configSvc.EXPECT().UpdateRoute(mock.Anything, domain.Route{Domain: "app.example.com", Image: "myapp:v2", HTTPS: true}).Return(nil).Once() - - req := httptest.NewRequest("PUT", "/admin/routes/app.example.com", bytes.NewBufferString(`{"image":"myapp:v2","https":true}`)) - req = req.WithContext(ctxWithScopes("admin:routes:write")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusOK, rec.Code) - assert.JSONEq(t, `{"domain":"app.example.com","image":"myapp:v2","https":true}`, rec.Body.String()) -} - -func TestHandler_AttachmentsByImageGet_OneTarget(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - configSvc.EXPECT().FindAttachmentTargetsByImage(mock.Anything, "postgres:16").Return([]string{"app.example.com"}).Once() - - req := httptest.NewRequest("GET", "/admin/attachments/by-image/postgres:16", nil) - req = req.WithContext(ctxWithScopes("admin:config:read")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusOK, rec.Code) - assert.JSONEq(t, `{"image":"postgres:16","targets":["app.example.com"]}`, rec.Body.String()) -} - -func TestHandler_AttachmentsByImageGet_MultipleTargets(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - configSvc.EXPECT().FindAttachmentTargetsByImage(mock.Anything, "postgres").Return([]string{"app.example.com", "workers"}).Once() - - req := httptest.NewRequest("GET", "/admin/attachments/by-image/postgres", nil) - req = req.WithContext(ctxWithScopes("admin:config:read")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusOK, rec.Code) - assert.JSONEq(t, `{"image":"postgres","targets":["app.example.com","workers"]}`, rec.Body.String()) -} - -func TestHandler_AttachmentsByImageGet_URLDecodedImage(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - configSvc.EXPECT().FindAttachmentTargetsByImage(mock.Anything, "registry/org/image:tag").Return([]string{"app.example.com"}).Once() - - req := httptest.NewRequest("GET", "/admin/attachments/by-image/registry%2Forg%2Fimage:tag", nil) - req = req.WithContext(ctxWithScopes("admin:config:read")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusOK, rec.Code) - assert.JSONEq(t, `{"image":"registry/org/image:tag","targets":["app.example.com"]}`, rec.Body.String()) -} - -func TestHandler_AttachmentsByImageGet_EmptyResultSet(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - configSvc.EXPECT().FindAttachmentTargetsByImage(mock.Anything, "missing:latest").Return([]string{}).Once() - - req := httptest.NewRequest("GET", "/admin/attachments/by-image/missing:latest", nil) - req = req.WithContext(ctxWithScopes("admin:config:read")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusOK, rec.Code) - assert.JSONEq(t, `{"image":"missing:latest","targets":[]}`, rec.Body.String()) -} - -func TestHandler_AttachmentsByImageGet_MissingImagePath(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - req := httptest.NewRequest("GET", "/admin/attachments/by-image", nil) - req = req.WithContext(ctxWithScopes("admin:config:read")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusBadRequest, rec.Code) - assert.JSONEq(t, `{"error":"image name required in path"}`, rec.Body.String()) -} - -func TestHandler_AttachmentsByImageGet_InvalidEncoding(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - req := httptest.NewRequest("GET", "/admin/attachments/by-image/placeholder", nil) - req.URL = &url.URL{Path: "/admin/attachments/by-image/%zz"} - req = req.WithContext(ctxWithScopes("admin:config:read")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusBadRequest, rec.Code) - assert.JSONEq(t, `{"error":"invalid image name encoding"}`, rec.Body.String()) -} - -func TestHandler_AttachmentsByImageGet_RequiresReadScope(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - req := httptest.NewRequest("GET", "/admin/attachments/by-image/postgres:16", nil) - req = req.WithContext(ctxWithScopes("admin:routes:read")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusForbidden, rec.Code) - assert.JSONEq(t, `{"error":"insufficient permissions for config:read"}`, rec.Body.String()) -} - -func TestHandler_RoutesByImageGet_URLDecodedImage(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - configSvc.EXPECT().FindRoutesByImage(mock.Anything, "registry/org/image:tag").Return([]domain.Route{{ - Domain: "app.example.com", - Image: "registry/org/image:tag", - }}).Once() - - req := httptest.NewRequest("GET", "/admin/routes/by-image/registry%2Forg%2Fimage:tag", nil) - req = req.WithContext(ctxWithScopes("admin:routes:read")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusOK, rec.Code) - assert.JSONEq(t, `{"image":"registry/org/image:tag","routes":[{"domain":"app.example.com","image":"registry/org/image:tag","https":false}]}`, rec.Body.String()) -} - -func TestHandler_RoutesDelete_RequiresWriteScope(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - tests := []struct { - name string - scopes []string - wantStatus int - }{ - { - name: "routes write access granted", - scopes: []string{"admin:routes:write"}, - wantStatus: http.StatusOK, - }, - { - name: "only read access denied", - scopes: []string{"admin:routes:read"}, - wantStatus: http.StatusForbidden, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - if tt.wantStatus == http.StatusOK { - configSvc.EXPECT().RemoveRoute(mock.Anything, "app.example.com").Return(nil).Once() - containerSvc.EXPECT().ReconcileRemovedRoute(mock.Anything, "app.example.com").Return(&domain.CleanupReport{Domain: "app.example.com"}, nil).Once() - } - - req := httptest.NewRequest("DELETE", "/admin/routes/app.example.com", nil) - req = req.WithContext(ctxWithScopes(tt.scopes...)) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, tt.wantStatus, rec.Code) - }) - } -} - -func TestHandler_RoutesDelete_ReconcilesRuntimeWhenRouteAlreadyMissing(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - report := &domain.CleanupReport{Domain: "app.example.com"} - - configSvc.EXPECT().RemoveRoute(mock.Anything, "app.example.com").Return(domain.ErrRouteNotFound).Once() - containerSvc.EXPECT().ReconcileRemovedRoute(mock.Anything, "app.example.com").Return(report, nil).Once() - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - server := newScopedTestServer(t, handler, "admin:routes:write") - - req, err := http.NewRequest(http.MethodDelete, server.URL+"/admin/routes/app.example.com", nil) - require.NoError(t, err) - respHTTP, err := http.DefaultClient.Do(req) - require.NoError(t, err) - defer respHTTP.Body.Close() - - require.Equal(t, http.StatusOK, respHTTP.StatusCode) - var resp dto.RouteDeleteResponse - require.NoError(t, json.NewDecoder(respHTTP.Body).Decode(&resp)) - require.NotNil(t, resp.Cleanup) - assert.Equal(t, "app.example.com", resp.Cleanup.Domain) -} - -func TestHandler_RoutesDelete_ReturnsServerErrorWhenRuntimeCleanupFails(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - configSvc.EXPECT().RemoveRoute(mock.Anything, "app.example.com").Return(nil).Once() - containerSvc.EXPECT().ReconcileRemovedRoute(mock.Anything, "app.example.com").Return(nil, errors.New("boom")).Once() - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - server := newScopedTestServer(t, handler, "admin:routes:write") - - req, err := http.NewRequest(http.MethodDelete, server.URL+"/admin/routes/app.example.com", nil) - require.NoError(t, err) - resp, err := http.DefaultClient.Do(req) - require.NoError(t, err) - defer resp.Body.Close() - body, err := io.ReadAll(resp.Body) - require.NoError(t, err) - - assert.Equal(t, http.StatusInternalServerError, resp.StatusCode) - assert.Contains(t, string(body), "route removed but runtime cleanup failed") -} - -func TestHandler_RoutesDelete_ReconcilesRuntimeAndReturnsCleanupReport(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - report := &domain.CleanupReport{ - Domain: "app.example.com", - RemovedContainers: []domain.CleanupContainer{{ - ID: "container-1", - Name: "gordon-app.example.com", - }}, - PreservedAttachments: []domain.CleanupAttachment{{ - Name: "postgres", - ContainerID: "attachment-1", - Status: "running", - }}, - } - - removeCall := configSvc.EXPECT().RemoveRoute(mock.Anything, "app.example.com").Return(nil).Once() - cleanupCall := containerSvc.EXPECT().ReconcileRemovedRoute(mock.Anything, "app.example.com").Return(report, nil).Once() - mock.InOrder(removeCall, cleanupCall) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - server := newScopedTestServer(t, handler, "admin:routes:write") - - req, err := http.NewRequest(http.MethodDelete, server.URL+"/admin/routes/app.example.com", nil) - require.NoError(t, err) - respHTTP, err := http.DefaultClient.Do(req) - require.NoError(t, err) - defer respHTTP.Body.Close() - - require.Equal(t, http.StatusOK, respHTTP.StatusCode) - var resp dto.RouteDeleteResponse - require.NoError(t, json.NewDecoder(respHTTP.Body).Decode(&resp)) - assert.Equal(t, "removed", resp.Status) - require.NotNil(t, resp.Cleanup) - assert.Equal(t, "app.example.com", resp.Cleanup.Domain) - require.Len(t, resp.Cleanup.RemovedContainers, 1) - assert.Equal(t, "container-1", resp.Cleanup.RemovedContainers[0].ID) - require.Len(t, resp.Cleanup.PreservedAttachments, 1) - assert.Equal(t, "attachment-1", resp.Cleanup.PreservedAttachments[0].ContainerID) -} - -// Secrets endpoint tests - -func TestHandler_SecretsGet_RequiresReadScope(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - tests := []struct { - name string - scopes []string - wantStatus int - }{ - { - name: "secrets read access granted", - scopes: []string{"admin:secrets:read"}, - wantStatus: http.StatusOK, - }, - { - name: "secrets wildcard access granted", - scopes: []string{"admin:secrets:*"}, - wantStatus: http.StatusOK, - }, - { - name: "all admin access granted", - scopes: []string{"admin:*:*"}, - wantStatus: http.StatusOK, - }, - { - name: "only write access denied", - scopes: []string{"admin:secrets:write"}, - wantStatus: http.StatusForbidden, - }, - { - name: "wrong resource denied", - scopes: []string{"admin:routes:read"}, - wantStatus: http.StatusForbidden, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - if tt.wantStatus == http.StatusOK { - secretSvc.EXPECT().ListKeysWithAttachments(mock.Anything, "app.example.com").Return([]string{}, nil, nil).Maybe() - } - - req := httptest.NewRequest("GET", "/admin/secrets/app.example.com", nil) - req = req.WithContext(ctxWithScopes(tt.scopes...)) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, tt.wantStatus, rec.Code) - }) - } -} - -func TestHandler_SecretsPost_RequiresWriteScope(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - secretsJSON := `{"API_KEY": "secret123"}` - - tests := []struct { - name string - scopes []string - wantStatus int - }{ - { - name: "secrets write access granted", - scopes: []string{"admin:secrets:write"}, - wantStatus: http.StatusOK, - }, - { - name: "only read access denied", - scopes: []string{"admin:secrets:read"}, - wantStatus: http.StatusForbidden, - }, - { - name: "wrong resource denied", - scopes: []string{"admin:routes:write"}, - wantStatus: http.StatusForbidden, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - if tt.wantStatus == http.StatusOK { - secretSvc.EXPECT().Set(mock.Anything, "app.example.com", mock.AnythingOfType("map[string]string")).Return(nil).Maybe() - } - - req := httptest.NewRequest("POST", "/admin/secrets/app.example.com", bytes.NewBufferString(secretsJSON)) - req = req.WithContext(ctxWithScopes(tt.scopes...)) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, tt.wantStatus, rec.Code) - }) - } -} - -func TestHandler_SecretsDelete_RequiresWriteScope(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - tests := []struct { - name string - scopes []string - wantStatus int - }{ - { - name: "secrets write access granted", - scopes: []string{"admin:secrets:write"}, - wantStatus: http.StatusOK, - }, - { - name: "only read access denied", - scopes: []string{"admin:secrets:read"}, - wantStatus: http.StatusForbidden, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - if tt.wantStatus == http.StatusOK { - secretSvc.EXPECT().Delete(mock.Anything, "app.example.com", "API_KEY").Return(nil).Maybe() - } - - req := httptest.NewRequest("DELETE", "/admin/secrets/app.example.com/API_KEY", nil) - req = req.WithContext(ctxWithScopes(tt.scopes...)) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, tt.wantStatus, rec.Code) - }) - } -} - -// Attachment secrets endpoint tests - -func TestHandler_AttachmentSecretsPost(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - secretSvc.EXPECT().SetAttachment(mock.Anything, "app.example.com", "redis", mock.AnythingOfType("map[string]string")).Return(nil) - - body := `{"REDIS_URL": "redis://localhost:6379"}` - req := httptest.NewRequest("POST", "/admin/secrets/app.example.com/attachments/redis", bytes.NewBufferString(body)) - req = req.WithContext(ctxWithScopes("admin:secrets:write")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusOK, rec.Code) - - var response struct { - Status string `json:"status"` - } - assert.NoError(t, json.NewDecoder(rec.Body).Decode(&response)) - assert.Equal(t, "updated", response.Status) -} - -func TestHandler_AttachmentSecretsDelete(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - secretSvc.EXPECT().DeleteAttachment(mock.Anything, "app.example.com", "redis", "REDIS_URL").Return(nil) - - req := httptest.NewRequest("DELETE", "/admin/secrets/app.example.com/attachments/redis/REDIS_URL", nil) - req = req.WithContext(ctxWithScopes("admin:secrets:write")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusOK, rec.Code) - - var response struct { - Status string `json:"status"` - } - assert.NoError(t, json.NewDecoder(rec.Body).Decode(&response)) - assert.Equal(t, "deleted", response.Status) -} - -func TestHandler_Bootstrap_HappyPath(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - registrySvc := inmocks.NewMockRegistryService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - d.RegistrySvc = registrySvc - }) - - configSvc.EXPECT().GetRegistryDomain().Return("reg.bnema.dev").Once() - configSvc.EXPECT().AddRoute(mock.Anything, domain.Route{Domain: "app.example.com", Image: "reg.bnema.dev/myapp", HTTPS: true}).Return(nil).Once() - configSvc.EXPECT().AddAttachment(mock.Anything, "app.example.com", "postgres:16").Return(nil).Once() - secretSvc.EXPECT().Set(mock.Anything, "app.example.com", map[string]string{"APP_ENV": "prod"}).Return(nil).Once() - secretSvc.EXPECT().SetAttachment(mock.Anything, "app.example.com", "postgres", map[string]string{"POSTGRES_PASSWORD": "secret"}).Return(nil).Once() - - body, err := json.Marshal(dto.BootstrapRequest{ - Domain: "app.example.com", - Image: "myapp:latest", - Attachments: []string{"postgres:16"}, - Env: map[string]string{"APP_ENV": "prod"}, - AttachmentEnv: map[string]map[string]string{ - "postgres": {"POSTGRES_PASSWORD": "secret"}, - }, - }) - assert.NoError(t, err) - - req := httptest.NewRequest("POST", "/admin/bootstrap", bytes.NewReader(body)) - req = req.WithContext(ctxWithScopes("admin:routes:write", "admin:secrets:write", "admin:config:write")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusOK, rec.Code) - registrySvc.AssertNotCalled(t, "GetManifest", mock.Anything, mock.Anything, mock.Anything) - - var response dto.BootstrapResponse - assert.NoError(t, json.NewDecoder(rec.Body).Decode(&response)) - assert.Equal(t, "app.example.com", response.Domain) - assert.Equal(t, "reg.bnema.dev/myapp", response.Image) - assert.Equal(t, "push reg.bnema.dev/myapp to trigger deployment", response.Next) - assert.Equal(t, []dto.BootstrapStep{ - {Name: "route", Status: "configured"}, - {Name: "attachment:postgres:16", Status: "created"}, - {Name: "env", Status: "updated"}, - {Name: "attachment_env:postgres", Status: "updated"}, - }, response.Steps) -} - -func TestHandler_Bootstrap_RequiresPermissions(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - tests := []struct { - name string - scopes []string - body string - wantStatus int - }{ - {name: "route only needs routes permission", scopes: []string{"admin:routes:write"}, body: `{"domain":"app.example.com","image":"myapp:latest"}`, wantStatus: http.StatusOK}, - {name: "env requires secrets permission", scopes: []string{"admin:routes:write"}, body: `{"domain":"app.example.com","image":"myapp:latest","env":{"APP_ENV":"prod"}}`, wantStatus: http.StatusForbidden}, - {name: "attachments require config permission", scopes: []string{"admin:routes:write"}, body: `{"domain":"app.example.com","image":"myapp:latest","attachments":["postgres:16"]}`, wantStatus: http.StatusForbidden}, - {name: "missing routes permission", scopes: []string{"admin:secrets:write", "admin:config:write"}, body: `{"domain":"app.example.com","image":"myapp:latest"}`, wantStatus: http.StatusForbidden}, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - if tt.wantStatus == http.StatusOK || strings.Contains(tt.body, `"env"`) || strings.Contains(tt.body, `"attachments"`) { - configSvc.EXPECT().GetRegistryDomain().Return("reg.bnema.dev").Once() - } - if tt.wantStatus == http.StatusOK { - configSvc.EXPECT().AddRoute(mock.Anything, domain.Route{Domain: "app.example.com", Image: "reg.bnema.dev/myapp", HTTPS: true}).Return(nil).Once() - } - - req := httptest.NewRequest("POST", "/admin/bootstrap", bytes.NewBufferString(tt.body)) - req = req.WithContext(ctxWithScopes(tt.scopes...)) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, tt.wantStatus, rec.Code) - }) - } -} - -// Status endpoint tests - -func TestHandler_Status_RequiresReadScope(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - tests := []struct { - name string - scopes []string - wantStatus int - }{ - { - name: "status read access granted", - scopes: []string{"admin:status:read"}, - wantStatus: http.StatusOK, - }, - { - name: "all admin access granted", - scopes: []string{"admin:*:*"}, - wantStatus: http.StatusOK, - }, - { - name: "wrong resource denied", - scopes: []string{"admin:routes:read"}, - wantStatus: http.StatusForbidden, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - if tt.wantStatus == http.StatusOK { - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{}).Maybe() - configSvc.EXPECT().GetRegistryDomain().Return("registry.example.com").Maybe() - configSvc.EXPECT().GetRegistryPort().Return(5000).Maybe() - configSvc.EXPECT().GetServerPort().Return(8080).Maybe() - configSvc.EXPECT().IsAutoRouteEnabled().Return(true).Maybe() - configSvc.EXPECT().IsNetworkIsolationEnabled().Return(false).Maybe() - } - - req := httptest.NewRequest("GET", "/admin/status", nil) - req = req.WithContext(ctxWithScopes(tt.scopes...)) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, tt.wantStatus, rec.Code) - }) - } -} - -// Config endpoint tests - -func TestHandler_Config_HidesSensitiveInventoryByDefault(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - handler := newTestHandler(t, func(d *HandlerDeps) { d.ConfigSvc = configSvc }) - - configSvc.EXPECT().GetServerPort().Return(8080).Once() - configSvc.EXPECT().GetRegistryPort().Return(5000).Once() - configSvc.EXPECT().GetRegistryDomain().Return("registry.example.com").Once() - configSvc.EXPECT().IsAutoRouteEnabled().Return(true).Once() - configSvc.EXPECT().IsNetworkIsolationEnabled().Return(false).Once() - configSvc.EXPECT().GetNetworkPrefix().Return("gordon").Once() - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{}).Once() - configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{"db.example.com": "http://10.0.0.5:5432"}).Once() - - server := newScopedTestServer(t, handler, "admin:config:read") - resp, err := http.Get(server.URL + "/admin/config") - require.NoError(t, err) - defer resp.Body.Close() - bodyBytes, err := io.ReadAll(resp.Body) - require.NoError(t, err) - - require.Equal(t, http.StatusOK, resp.StatusCode) - body := string(bodyBytes) - var raw map[string]any - require.NoError(t, json.Unmarshal([]byte(body), &raw)) - serverConfig := raw["server"].(map[string]any) - assert.NotContains(t, serverConfig, "data_dir") - assert.NotContains(t, body, "/var/lib/gordon") - assert.NotContains(t, body, "10.0.0.5") -} - -func TestHandler_Config_RequiresReadScope(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - tests := []struct { - name string - scopes []string - wantStatus int - }{ - { - name: "config read access granted", - scopes: []string{"admin:config:read"}, - wantStatus: http.StatusOK, - }, - { - name: "all admin access granted", - scopes: []string{"admin:*:*"}, - wantStatus: http.StatusOK, - }, - { - name: "wrong resource denied", - scopes: []string{"admin:routes:read"}, - wantStatus: http.StatusForbidden, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - if tt.wantStatus == http.StatusOK { - configSvc.EXPECT().GetServerPort().Return(8080).Maybe() - configSvc.EXPECT().GetRegistryPort().Return(5000).Maybe() - configSvc.EXPECT().GetRegistryDomain().Return("registry.example.com").Maybe() - configSvc.EXPECT().GetDataDir().Return("/var/lib/gordon").Maybe() - configSvc.EXPECT().IsAutoRouteEnabled().Return(true).Maybe() - configSvc.EXPECT().IsNetworkIsolationEnabled().Return(false).Maybe() - configSvc.EXPECT().GetNetworkPrefix().Return("gordon").Maybe() - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{}).Maybe() - configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{}).Maybe() - } - - req := httptest.NewRequest("GET", "/admin/config", nil) - req = req.WithContext(ctxWithScopes(tt.scopes...)) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, tt.wantStatus, rec.Code) - }) - } -} - -// Reload endpoint tests - -func TestHandler_Reload_RequiresWriteScope(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - reloadTrigger := &reloadTriggerRecorder{} - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - d.ReloadTrigger = reloadTrigger - }) - - tests := []struct { - name string - scopes []string - wantStatus int - }{ - { - name: "config write access granted", - scopes: []string{"admin:config:write"}, - wantStatus: http.StatusOK, - }, - { - name: "all admin access granted", - scopes: []string{"admin:*:*"}, - wantStatus: http.StatusOK, - }, - { - name: "only read access denied", - scopes: []string{"admin:config:read"}, - wantStatus: http.StatusForbidden, - }, - { - name: "wrong resource denied", - scopes: []string{"admin:routes:write"}, - wantStatus: http.StatusForbidden, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - reloadTrigger.calls = 0 - - req := httptest.NewRequest("POST", "/admin/reload", nil) - req = req.WithContext(ctxWithScopes(tt.scopes...)) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, tt.wantStatus, rec.Code) - if tt.wantStatus == http.StatusOK { - assert.Equal(t, 1, reloadTrigger.calls) - } else { - assert.Zero(t, reloadTrigger.calls) - } - }) - } -} - -func TestHandler_Reload_ReturnsServerErrorOnTriggerFailure(t *testing.T) { - reloadTrigger := &reloadTriggerRecorder{err: errors.New("reload failed")} - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ReloadTrigger = reloadTrigger - }) - - req := httptest.NewRequest("POST", "/admin/reload", nil) - req = req.WithContext(ctxWithScopes("admin:config:write")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusInternalServerError, rec.Code) - assert.Equal(t, 1, reloadTrigger.calls) -} - -func TestHandler_Reload_UsesDetachedContextWithDeadline(t *testing.T) { - reloadTrigger := &reloadTriggerRecorder{} - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ReloadTrigger = reloadTrigger - }) - - reqCtx, cancel := context.WithCancel(ctxWithScopes("admin:config:write")) - cancel() - - req := httptest.NewRequest("POST", "/admin/reload", nil) - req = req.WithContext(reqCtx) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusOK, rec.Code) - assert.Equal(t, 1, reloadTrigger.calls) - require.NoError(t, reloadTrigger.ctxErrAtCall) - assert.True(t, reloadTrigger.hasDeadline) -} - -// Functional tests - -func TestHandler_RoutesGet_ReturnsRoutes(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - expectedRoutes := []domain.Route{ - {Domain: "app1.example.com", Image: "app1:latest"}, - {Domain: "app2.example.com", Image: "app2:v1.0"}, - } - configSvc.EXPECT().GetRoutes(mock.Anything).Return(expectedRoutes) - - req := httptest.NewRequest("GET", "/admin/routes", nil) - req = req.WithContext(ctxWithScopes("admin:routes:read")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusOK, rec.Code) - - var response struct { - Routes []struct { - Domain string `json:"domain"` - Image string `json:"image"` - HTTPS bool `json:"https"` - } `json:"routes"` - } - err := json.NewDecoder(rec.Body).Decode(&response) - assert.NoError(t, err) - - assert.Len(t, response.Routes, 2) -} - -func TestHandler_RoutesGet_DetailedSyncsAndIncludesConfiguredRouteWithoutContainer(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{{ - Domain: "app.example.com", - Image: "app:latest", - }}).Once() - syncCall := containerSvc.EXPECT().SyncContainers(mock.Anything).Return(nil).Once() - listCall := containerSvc.EXPECT().ListRoutesWithDetails(mock.Anything).Return(nil).Once() - mock.InOrder(syncCall, listCall) - - req := httptest.NewRequest("GET", "/admin/routes?detailed=true", nil) - req = req.WithContext(ctxWithScopes("admin:routes:read")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusOK, rec.Code) - - var response dto.RoutesDetailResponse - err := json.NewDecoder(rec.Body).Decode(&response) - require.NoError(t, err) - require.Len(t, response.Routes, 1) - assert.Equal(t, "app.example.com", response.Routes[0].Domain) - assert.Equal(t, "app:latest", response.Routes[0].Image) - assert.Empty(t, response.Routes[0].ContainerID) - assert.Empty(t, response.Routes[0].ContainerStatus) -} - -func TestHandler_RoutesGet_SingleRoute(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - configSvc.EXPECT().GetRoute(mock.Anything, "app.example.com").Return(&domain.Route{ - Domain: "app.example.com", Image: "app:latest", - }, nil) - - req := httptest.NewRequest("GET", "/admin/routes/app.example.com", nil) - req = req.WithContext(ctxWithScopes("admin:routes:read")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusOK, rec.Code) - - var route struct { - Domain string `json:"domain"` - Image string `json:"image"` - HTTPS bool `json:"https"` - } - err := json.NewDecoder(rec.Body).Decode(&route) - assert.NoError(t, err) - assert.Equal(t, "app.example.com", route.Domain) - assert.Equal(t, "app:latest", route.Image) -} - -func TestHandler_RoutesGet_NotFound(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - configSvc.EXPECT().GetRoute(mock.Anything, "nonexistent.example.com").Return(nil, domain.ErrRouteNotFound) - - req := httptest.NewRequest("GET", "/admin/routes/nonexistent.example.com", nil) - req = req.WithContext(ctxWithScopes("admin:routes:read")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusNotFound, rec.Code) -} - -func TestHandler_RouteCleanupPreview_RequiresRoutesConfigAndVolumesReadScopes(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - volumeSvc := inmocks.NewMockVolumeService(t) - secretSvc := inmocks.NewMockSecretService(t) - containerMock := inmocks.NewMockContainerService(t) - containerSvc := &previewContainerService{ - MockContainerService: containerMock, - preview: func(_ context.Context, routeDomain string) (*domain.CleanupReport, error) { - return &domain.CleanupReport{ - Domain: routeDomain, - OrphanedEntities: []domain.CleanupOrphanedEntity{{Kind: "route_container", ID: "ctr-1", Status: "running"}}, - }, nil - }, - } - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.VolumeSvc = volumeSvc - d.SecretSvc = secretSvc - }) - - volumeSvc.EXPECT().ListVolumes(mock.Anything).Return([]*domain.VolumeInfo{{Name: "gordon-app-example-com-data"}}, nil).Once() - - req := httptest.NewRequest("GET", "/admin/routes/app.example.com/cleanup", nil) - req = req.WithContext(ctxWithScopes("admin:routes:read", "admin:config:read", "admin:volumes:read")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusOK, rec.Code) - var report dto.CleanupReport - require.NoError(t, json.NewDecoder(rec.Body).Decode(&report)) - require.Len(t, report.OrphanedEntities, 1) - assert.Equal(t, "route_container", report.OrphanedEntities[0].Kind) - require.Len(t, report.PreservedVolumes, 1) - assert.Equal(t, "gordon-app-example-com-data", report.PreservedVolumes[0].Name) - - for _, scopes := range [][]string{ - {"admin:routes:read", "admin:volumes:read"}, - {"admin:routes:read", "admin:config:read"}, - } { - req := httptest.NewRequest("GET", "/admin/routes/app.example.com/cleanup", nil) - req = req.WithContext(ctxWithScopes(scopes...)) - rec := httptest.NewRecorder() - handler.ServeHTTP(rec, req) - assert.Equal(t, http.StatusForbidden, rec.Code) - } -} - -func TestHandler_RoutesPost_MissingFields(t *testing.T) { - tests := []struct { - name string - body string - mockErr error - }{ - { - name: "missing domain", - body: `{"image": "app:latest"}`, - mockErr: domain.ErrRouteDomainEmpty, - }, - { - name: "missing image", - body: `{"domain": "app.example.com"}`, - mockErr: domain.ErrRouteImageEmpty, - }, { - name: "empty object", - body: `{}`, - mockErr: domain.ErrRouteDomainEmpty, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - registrySvc := inmocks.NewMockRegistryService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - d.RegistrySvc = registrySvc - }) - - configSvc.EXPECT().AddRoute(mock.Anything, mock.Anything).Return(tt.mockErr) - - req := httptest.NewRequest("POST", "/admin/routes", bytes.NewBufferString(tt.body)) - req = req.WithContext(ctxWithScopes("admin:routes:write")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusBadRequest, rec.Code) - }) - } -} - -func TestHandler_RoutesPut_MissingDomain(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - req := httptest.NewRequest("PUT", "/admin/routes/", bytes.NewBufferString(`{"image": "app:latest"}`)) - req = req.WithContext(ctxWithScopes("admin:routes:write")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusBadRequest, rec.Code) -} - -func TestHandler_RoutesDelete_MissingDomain(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - req := httptest.NewRequest("DELETE", "/admin/routes/", nil) - req = req.WithContext(ctxWithScopes("admin:routes:write")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusBadRequest, rec.Code) -} - -func TestHandler_Secrets_MissingDomain(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - req := httptest.NewRequest("GET", "/admin/secrets/", nil) - req = req.WithContext(ctxWithScopes("admin:secrets:read")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusBadRequest, rec.Code) -} - -func TestHandler_SecretsDelete_MissingKey(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - req := httptest.NewRequest("DELETE", "/admin/secrets/app.example.com", nil) - req = req.WithContext(ctxWithScopes("admin:secrets:write")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusBadRequest, rec.Code) -} - -func TestHandler_MethodNotAllowed(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - tests := []struct { - method string - path string - }{ - {"DELETE", "/admin/status"}, - {"PUT", "/admin/status"}, - {"GET", "/admin/reload"}, - {"DELETE", "/admin/reload"}, - {"POST", "/admin/config"}, - {"DELETE", "/admin/config"}, - } - - for _, tt := range tests { - t.Run(tt.method+" "+tt.path, func(t *testing.T) { - req := httptest.NewRequest(tt.method, tt.path, nil) - req = req.WithContext(ctxWithScopes("admin:*:*")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusMethodNotAllowed, rec.Code) - }) - } -} - -func TestHandler_Restart_InvalidDomain(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - tests := []struct { - name string - path string - status int - }{ - {"path traversal", "/admin/restart/../../etc/passwd", http.StatusBadRequest}, - {"empty domain", "/admin/restart/", http.StatusBadRequest}, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - req := httptest.NewRequest("POST", tt.path, nil) - req = req.WithContext(ctxWithScopes("admin:config:write")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, tt.status, rec.Code) - }) - } -} - -func TestHandler_Deploy_ConfigWriteOnlyOmitsDeployFailureLogs(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - route := &domain.Route{Domain: "app.example.com", Image: "registry.local/myapp:latest"} - deployErr := &domain.DeployFailureError{ - Summary: "failed to deploy", - Cause: "image pull failed", - Hint: "push the image before retrying", - Logs: []string{"PASSWORD=hunter2", "TOKEN=abc123"}, - } - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - configSvc.EXPECT().GetRoute(mock.Anything, "app.example.com").Return(route, nil).Once() - containerSvc.EXPECT().Deploy(mock.Anything, *route).Return(nil, deployErr).Once() - - server := newScopedTestServer(t, handler, "admin:config:write") - resp, err := http.Post(server.URL+"/admin/deploy/app.example.com", "application/json", nil) - require.NoError(t, err) - defer resp.Body.Close() - bodyBytes, err := io.ReadAll(resp.Body) - require.NoError(t, err) - body := string(bodyBytes) - - assert.Equal(t, http.StatusInternalServerError, resp.StatusCode) - assert.JSONEq(t, `{"error":"failed to deploy","cause":"image pull failed","hint":"push the image before retrying"}`, body) - assert.NotContains(t, body, "hunter2") - assert.NotContains(t, body, "abc123") -} - -func TestHandler_Deploy_StructuredDeployFailure(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - route := &domain.Route{Domain: "app.example.com", Image: "registry.local/myapp:latest"} - deployErr := &domain.DeployFailureError{ - Summary: "failed to deploy", - Cause: "image pull failed", - Hint: "push the image before retrying", - Logs: []string{ - "pulling manifest PASSWORD=hunter2", - `{"TOKEN":"abc123"}`, - }, - Err: errors.New("manifest unknown: internal registry detail"), - } - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - s := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - handler.ServeHTTP(w, r.WithContext(ctxWithScopes("admin:config:write", "admin:logs:read"))) - })) - defer s.Close() - - configSvc.EXPECT().GetRoute(mock.Anything, "app.example.com").Return(route, nil).Once() - containerSvc.EXPECT().Deploy(mock.Anything, *route).Return(nil, deployErr).Once() - - req, err := http.NewRequestWithContext(context.Background(), http.MethodPost, s.URL+"/admin/deploy/app.example.com", nil) - assert.NoError(t, err) - resp, err := s.Client().Do(req) - assert.NoError(t, err) - defer resp.Body.Close() - body, err := io.ReadAll(resp.Body) - assert.NoError(t, err) - - assert.Equal(t, http.StatusInternalServerError, resp.StatusCode) - assert.JSONEq(t, `{"error":"failed to deploy","cause":"image pull failed","hint":"push the image before retrying","logs":["pulling manifest PASSWORD=[REDACTED]","{\"TOKEN\":\"[REDACTED]\"}"]}`, string(body)) - assert.NotContains(t, string(body), "internal registry detail") - assert.NotContains(t, string(body), "hunter2") - assert.NotContains(t, string(body), "abc123") -} - -func TestHandler_Deploy_GenericErrorStillUsesLegacyBody(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - route := &domain.Route{Domain: "app.example.com", Image: "registry.local/myapp:latest"} - deployErr := errors.New("manifest unknown: internal registry detail") + name: "wrong resource denied", + scopes: []string{"admin:routes:write"}, + wantStatus: http.StatusForbidden, + }, + } - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - s := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - handler.ServeHTTP(w, r.WithContext(ctxWithScopes("admin:config:write"))) - })) - defer s.Close() + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + reloadTrigger.calls = 0 - configSvc.EXPECT().GetRoute(mock.Anything, "app.example.com").Return(route, nil).Once() - containerSvc.EXPECT().Deploy(mock.Anything, *route).Return(nil, deployErr).Once() + req := httptest.NewRequest("POST", "/admin/reload", nil) + req = req.WithContext(ctxWithScopes(tt.scopes...)) + rec := httptest.NewRecorder() - req, err := http.NewRequestWithContext(context.Background(), http.MethodPost, s.URL+"/admin/deploy/app.example.com", nil) - assert.NoError(t, err) - resp, err := s.Client().Do(req) - assert.NoError(t, err) - defer resp.Body.Close() - body, err := io.ReadAll(resp.Body) - assert.NoError(t, err) + handler.ServeHTTP(rec, req) - assert.Equal(t, http.StatusInternalServerError, resp.StatusCode) - assert.JSONEq(t, `{"error":"failed to deploy container"}`, string(body)) - assert.NotContains(t, string(body), "internal registry detail") + assert.Equal(t, tt.wantStatus, rec.Code) + if tt.wantStatus == http.StatusOK { + assert.Equal(t, 1, reloadTrigger.calls) + } else { + assert.Zero(t, reloadTrigger.calls) + } + }) + } } -func TestHandler_Restart_ContainerNotFound(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) +func TestHandler_Reload_ReturnsServerErrorOnTriggerFailure(t *testing.T) { + reloadTrigger := &reloadTriggerRecorder{err: errors.New("reload failed")} handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc + d.ReloadTrigger = reloadTrigger }) - containerSvc.EXPECT().Restart(mock.Anything, "nonexistent.example.com", false).Return(domain.ErrContainerNotFound) - - req := httptest.NewRequest("POST", "/admin/restart/nonexistent.example.com", nil) + req := httptest.NewRequest("POST", "/admin/reload", nil) req = req.WithContext(ctxWithScopes("admin:config:write")) rec := httptest.NewRecorder() handler.ServeHTTP(rec, req) - assert.Equal(t, http.StatusNotFound, rec.Code) + assert.Equal(t, http.StatusInternalServerError, rec.Code) + assert.Equal(t, 1, reloadTrigger.calls) } -func TestHandler_Restart_GenericError(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) +func TestHandler_Reload_UsesDetachedContextWithDeadline(t *testing.T) { + reloadTrigger := &reloadTriggerRecorder{} handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc + d.ReloadTrigger = reloadTrigger }) - containerSvc.EXPECT().Restart(mock.Anything, "test.example.com", false).Return(fmt.Errorf("docker daemon error: secret details")) + reqCtx, cancel := context.WithCancel(ctxWithScopes("admin:config:write")) + cancel() - req := httptest.NewRequest("POST", "/admin/restart/test.example.com", nil) - req = req.WithContext(ctxWithScopes("admin:config:write")) + req := httptest.NewRequest("POST", "/admin/reload", nil) + req = req.WithContext(reqCtx) rec := httptest.NewRecorder() handler.ServeHTTP(rec, req) - assert.Equal(t, http.StatusInternalServerError, rec.Code) - // Verify internal error details are NOT leaked - assert.NotContains(t, rec.Body.String(), "docker daemon error") - assert.NotContains(t, rec.Body.String(), "secret details") + assert.Equal(t, http.StatusOK, rec.Code) + assert.Equal(t, 1, reloadTrigger.calls) + require.NoError(t, reloadTrigger.ctxErrAtCall) + assert.True(t, reloadTrigger.hasDeadline) } -func TestHandler_Restart_Success(t *testing.T) { +// Functional tests + +func TestHandler_MethodNotAllowed(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) authSvc := inmocks.NewMockAuthService(t) containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) handler := newTestHandler(t, func(d *HandlerDeps) { d.ConfigSvc = configSvc d.AuthSvc = authSvc d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc }) tests := []struct { - name string - path string - withAttachments bool + method string + path string }{ - { - name: "without attachments", - path: "/admin/restart/test.example.com", - withAttachments: false, - }, - { - name: "with attachments", - path: "/admin/restart/test.example.com?attachments=true", - withAttachments: true, - }, + {"DELETE", "/admin/status"}, + {"PUT", "/admin/status"}, + {"GET", "/admin/reload"}, + {"DELETE", "/admin/reload"}, + {"POST", "/admin/config"}, + {"DELETE", "/admin/config"}, } for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - containerSvc.EXPECT().Restart(mock.Anything, "test.example.com", tt.withAttachments).Return(nil) - - req := httptest.NewRequest("POST", tt.path, nil) - req = req.WithContext(ctxWithScopes("admin:config:write")) + t.Run(tt.method+" "+tt.path, func(t *testing.T) { + req := httptest.NewRequest(tt.method, tt.path, nil) + req = req.WithContext(ctxWithScopes("admin:*:*")) rec := httptest.NewRecorder() handler.ServeHTTP(rec, req) - assert.Equal(t, http.StatusOK, rec.Code) - var response struct { - Status string `json:"status"` - Domain string `json:"domain"` - } - assert.NoError(t, json.NewDecoder(rec.Body).Decode(&response)) - assert.Equal(t, "restarted", response.Status) - assert.Equal(t, "test.example.com", response.Domain) + assert.Equal(t, http.StatusMethodNotAllowed, rec.Code) }) } } @@ -2094,13 +437,11 @@ func TestHandler_Tags_InvalidRepository(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) authSvc := inmocks.NewMockAuthService(t) containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) handler := newTestHandler(t, func(d *HandlerDeps) { d.ConfigSvc = configSvc d.AuthSvc = authSvc d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc }) tests := []struct { @@ -2129,14 +470,12 @@ func TestHandler_Tags_Success(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) authSvc := inmocks.NewMockAuthService(t) containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) registrySvc := inmocks.NewMockRegistryService(t) handler := newTestHandler(t, func(d *HandlerDeps) { d.ConfigSvc = configSvc d.AuthSvc = authSvc d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc d.RegistrySvc = registrySvc }) @@ -2162,14 +501,12 @@ func TestHandler_Tags_Error(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) authSvc := inmocks.NewMockAuthService(t) containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) registrySvc := inmocks.NewMockRegistryService(t) handler := newTestHandler(t, func(d *HandlerDeps) { d.ConfigSvc = configSvc d.AuthSvc = authSvc d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc d.RegistrySvc = registrySvc }) @@ -2189,14 +526,12 @@ func TestHandler_Tags_URLDecodedRepository(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) authSvc := inmocks.NewMockAuthService(t) containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) registrySvc := inmocks.NewMockRegistryService(t) handler := newTestHandler(t, func(d *HandlerDeps) { d.ConfigSvc = configSvc d.AuthSvc = authSvc d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc d.RegistrySvc = registrySvc }) @@ -2212,109 +547,15 @@ func TestHandler_Tags_URLDecodedRepository(t *testing.T) { assert.JSONEq(t, `{"repository":"repo/app","tags":["latest"]}`, rec.Body.String()) } -func TestHandler_AttachmentSecretsPost_RequiresWriteScope(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - body := `{"REDIS_URL": "redis://localhost:6379"}` - req := httptest.NewRequest("POST", "/admin/secrets/app.example.com/attachments/redis", bytes.NewBufferString(body)) - req = req.WithContext(ctxWithScopes("admin:secrets:read")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusForbidden, rec.Code) -} - -func TestHandler_AttachmentSecretsDelete_RequiresWriteScope(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - req := httptest.NewRequest("DELETE", "/admin/secrets/app.example.com/attachments/redis/REDIS_URL", nil) - req = req.WithContext(ctxWithScopes("admin:secrets:read")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusForbidden, rec.Code) -} - -func TestHandler_AttachmentSecretsDelete_RequiresKey(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - req := httptest.NewRequest("DELETE", "/admin/secrets/app.example.com/attachments/redis", nil) - req = req.WithContext(ctxWithScopes("admin:secrets:write")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusBadRequest, rec.Code) - assert.Contains(t, rec.Body.String(), "invalid attachment path") -} - -func TestHandler_AttachmentSecretsPost_InvalidJSON(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - body := `{not valid json` - req := httptest.NewRequest("POST", "/admin/secrets/app.example.com/attachments/redis", bytes.NewBufferString(body)) - req = req.WithContext(ctxWithScopes("admin:secrets:write")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusBadRequest, rec.Code) - assert.Contains(t, rec.Body.String(), "invalid JSON") -} - func TestHandler_NotFound(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) authSvc := inmocks.NewMockAuthService(t) containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) handler := newTestHandler(t, func(d *HandlerDeps) { d.ConfigSvc = configSvc d.AuthSvc = authSvc d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc }) req := httptest.NewRequest("GET", "/admin/unknown", nil) @@ -2330,16 +571,14 @@ func TestHandler_BackupsStatus(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) authSvc := inmocks.NewMockAuthService(t) containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) backupSvc := inmocks.NewMockBackupService(t) - backupSvc.EXPECT().Status(mock.Anything).Return([]domain.BackupJob{{Domain: "app.example.com", DBName: "postgres", Status: domain.BackupStatusCompleted, FilePath: "/var/lib/gordon/backups/private.bak"}}, nil) + backupSvc.EXPECT().Status(mock.Anything).Return([]domain.BackupJob{{App: "shop", Service: "api", DBName: "orders", Status: domain.BackupStatusCompleted, FilePath: "/var/lib/gordon/backups/private.bak"}}, nil) handler := newTestHandler(t, func(d *HandlerDeps) { d.ConfigSvc = configSvc d.AuthSvc = authSvc d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc d.BackupSvc = backupSvc }) @@ -2350,56 +589,53 @@ func TestHandler_BackupsStatus(t *testing.T) { handler.ServeHTTP(rec, req) assert.Equal(t, http.StatusOK, rec.Code) - assert.Contains(t, rec.Body.String(), "app.example.com") + assert.Contains(t, rec.Body.String(), "shop") + assert.Contains(t, rec.Body.String(), "orders") assert.NotContains(t, rec.Body.String(), "file_path") } -func TestHandler_BackupsListDomain(t *testing.T) { +func TestHandler_BackupsListApp(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) authSvc := inmocks.NewMockAuthService(t) containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) backupSvc := inmocks.NewMockBackupService(t) - backupSvc.EXPECT().ListBackups(mock.Anything, "app.example.com").Return([]domain.BackupJob{{Domain: "app.example.com", DBName: "postgres", Status: domain.BackupStatusCompleted}}, nil) + backupSvc.EXPECT().ListBackups(mock.Anything, "shop").Return([]domain.BackupJob{{App: "shop", Service: "api", DBName: "orders", Status: domain.BackupStatusCompleted}}, nil) handler := newTestHandler(t, func(d *HandlerDeps) { d.ConfigSvc = configSvc d.AuthSvc = authSvc d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc d.BackupSvc = backupSvc }) - req := httptest.NewRequest("GET", "/admin/backups/app.example.com", nil) + req := httptest.NewRequest("GET", "/admin/backups/shop", nil) req = req.WithContext(ctxWithScopes("admin:status:read")) rec := httptest.NewRecorder() handler.ServeHTTP(rec, req) assert.Equal(t, http.StatusOK, rec.Code) - assert.Contains(t, rec.Body.String(), "app.example.com") + assert.Contains(t, rec.Body.String(), "shop") } -func TestHandler_BackupsRunDomain(t *testing.T) { +func TestHandler_BackupsRunApp(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) authSvc := inmocks.NewMockAuthService(t) containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) backupSvc := inmocks.NewMockBackupService(t) - backupSvc.EXPECT().RunBackup(mock.Anything, "app.example.com", "postgres").Return(&domain.BackupResult{Job: domain.BackupJob{Domain: "app.example.com", DBName: "postgres", Status: domain.BackupStatusCompleted}}, nil) + backupSvc.EXPECT().RunBackup(mock.Anything, "shop", "api", "orders").Return(&domain.BackupResult{Job: domain.BackupJob{App: "shop", Service: "api", DBName: "orders", Status: domain.BackupStatusCompleted}}, nil) handler := newTestHandler(t, func(d *HandlerDeps) { d.ConfigSvc = configSvc d.AuthSvc = authSvc d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc d.BackupSvc = backupSvc }) - body := bytes.NewBufferString(`{"db":"postgres"}`) - req := httptest.NewRequest("POST", "/admin/backups/app.example.com", body) + body := bytes.NewBufferString(`{"service":"api","database":"orders"}`) + req := httptest.NewRequest("POST", "/admin/backups/shop", body) req = req.WithContext(ctxWithScopes("admin:config:write")) rec := httptest.NewRecorder() @@ -2412,15 +648,15 @@ func TestHandler_BackupsRunDomain(t *testing.T) { func TestHandler_VolumeBackupsRunDomain_ReturnsPartialResults(t *testing.T) { volumeBackupSvc := inmocks.NewMockVolumeBackupService(t) runErr := errors.New("one volume failed") - jobs := []domain.VolumeBackupJob{{ID: "v1", Domain: "app.example.com", VolumeName: "gordon-app-data", Status: domain.BackupStatusCompleted}} - volumeBackupSvc.EXPECT().RunVolumeBackups(mock.Anything, "app.example.com", "gordon-app-data").Return(jobs, runErr) + jobs := []domain.VolumeBackupJob{{ID: "v1", App: "shop", Service: "api", VolumeName: "data", Status: domain.BackupStatusCompleted}} + volumeBackupSvc.EXPECT().RunVolumeBackups(mock.Anything, "shop", "api", "data").Return(jobs, runErr) handler := newTestHandler(t, func(d *HandlerDeps) { d.VolumeBackupSvc = volumeBackupSvc }) - body := bytes.NewBufferString(`{"volume":"gordon-app-data"}`) - req := httptest.NewRequest("POST", "/admin/backups/volumes/app.example.com", body) + body := bytes.NewBufferString(`{"service":"api","volume":"data"}`) + req := httptest.NewRequest("POST", "/admin/backups/volumes/shop", body) req = req.WithContext(ctxWithScopes("admin:config:write")) rec := httptest.NewRecorder() @@ -2437,13 +673,13 @@ func TestHandler_VolumeBackupsRunDomain_ReturnsPartialResults(t *testing.T) { func TestHandler_VolumeBackupsRunDomain_ReturnsServerErrorWithoutResults(t *testing.T) { volumeBackupSvc := inmocks.NewMockVolumeBackupService(t) - volumeBackupSvc.EXPECT().RunVolumeBackups(mock.Anything, "app.example.com", "").Return(nil, errors.New("runtime unavailable")) + volumeBackupSvc.EXPECT().RunVolumeBackups(mock.Anything, "shop", "", "").Return(nil, errors.New("runtime unavailable")) handler := newTestHandler(t, func(d *HandlerDeps) { d.VolumeBackupSvc = volumeBackupSvc }) - req := httptest.NewRequest("POST", "/admin/backups/volumes/app.example.com", bytes.NewBufferString(`{}`)) + req := httptest.NewRequest("POST", "/admin/backups/volumes/shop", bytes.NewBufferString(`{}`)) req = req.WithContext(ctxWithScopes("admin:config:write")) rec := httptest.NewRecorder() @@ -2453,25 +689,23 @@ func TestHandler_VolumeBackupsRunDomain_ReturnsServerErrorWithoutResults(t *test assert.Contains(t, rec.Body.String(), "failed to run volume backups") } -func TestHandler_BackupsRunDomain_ChunkedBodyIsDecoded(t *testing.T) { +func TestHandler_BackupsRunApp_ChunkedBodyIsDecoded(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) authSvc := inmocks.NewMockAuthService(t) containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) backupSvc := inmocks.NewMockBackupService(t) - backupSvc.EXPECT().RunBackup(mock.Anything, "app.example.com", "postgres").Return(&domain.BackupResult{Job: domain.BackupJob{Domain: "app.example.com", DBName: "postgres", Status: domain.BackupStatusCompleted}}, nil) + backupSvc.EXPECT().RunBackup(mock.Anything, "shop", "api", "orders").Return(&domain.BackupResult{Job: domain.BackupJob{App: "shop", Service: "api", DBName: "orders", Status: domain.BackupStatusCompleted}}, nil) handler := newTestHandler(t, func(d *HandlerDeps) { d.ConfigSvc = configSvc d.AuthSvc = authSvc d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc d.BackupSvc = backupSvc }) - body := bytes.NewBufferString(`{"db":"postgres"}`) - req := httptest.NewRequest("POST", "/admin/backups/app.example.com", body) + body := bytes.NewBufferString(`{"service":"api","database":"orders"}`) + req := httptest.NewRequest("POST", "/admin/backups/shop", body) req.ContentLength = -1 req = req.WithContext(ctxWithScopes("admin:config:write")) rec := httptest.NewRecorder() @@ -2519,38 +753,10 @@ func TestHandler_LogsRequireLogsReadAndRedact(t *testing.T) { }) } -func TestHandler_BackupsDetectDomain(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - backupSvc := inmocks.NewMockBackupService(t) - - backupSvc.EXPECT().DetectDatabases(mock.Anything, "app.example.com").Return([]domain.DBInfo{{Type: domain.DBTypePostgreSQL, Name: "postgres", Port: 5432}}, nil) - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - d.BackupSvc = backupSvc - }) - - req := httptest.NewRequest("GET", "/admin/backups/app.example.com/detect", nil) - req = req.WithContext(ctxWithScopes("admin:status:read")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusOK, rec.Code) - assert.Contains(t, rec.Body.String(), "postgres") -} - func TestHandler_ImagesGet_ReturnsMappedList(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) authSvc := inmocks.NewMockAuthService(t) containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) createdAt := time.Date(2026, time.February, 8, 14, 30, 0, 0, time.UTC) imageSvc := &stubImageService{ @@ -2580,7 +786,6 @@ func TestHandler_ImagesGet_ReturnsMappedList(t *testing.T) { d.ConfigSvc = configSvc d.AuthSvc = authSvc d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc d.ImageSvc = imageSvc }) @@ -2618,7 +823,6 @@ func TestHandler_ImagesPrune_AcceptsOptionalKeepLast(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) authSvc := inmocks.NewMockAuthService(t) containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) tests := []struct { name string @@ -2647,7 +851,6 @@ func TestHandler_ImagesPrune_AcceptsOptionalKeepLast(t *testing.T) { d.ConfigSvc = configSvc d.AuthSvc = authSvc d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc d.ImageSvc = imageSvc }) @@ -2670,7 +873,6 @@ func TestHandler_Images_ErrorMappingForServiceFailures(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) authSvc := inmocks.NewMockAuthService(t) containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) imageSvc := &stubImageService{ listImagesFunc: func(context.Context) ([]domain.ImageInfo, error) { @@ -2682,7 +884,6 @@ func TestHandler_Images_ErrorMappingForServiceFailures(t *testing.T) { d.ConfigSvc = configSvc d.AuthSvc = authSvc d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc d.ImageSvc = imageSvc }) @@ -2701,7 +902,6 @@ func TestHandler_Images_ErrorMappingForServiceFailures(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) authSvc := inmocks.NewMockAuthService(t) containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) imageSvc := &stubImageService{ pruneFunc: func(context.Context, domain.ImagePruneOptions) (domain.ImagePruneReport, error) { @@ -2713,7 +913,6 @@ func TestHandler_Images_ErrorMappingForServiceFailures(t *testing.T) { d.ConfigSvc = configSvc d.AuthSvc = authSvc d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc d.ImageSvc = imageSvc }) @@ -2734,13 +933,11 @@ func TestHandler_ImagesPrune_ValidationAndAvailability(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) authSvc := inmocks.NewMockAuthService(t) containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) handler := newTestHandler(t, func(d *HandlerDeps) { d.ConfigSvc = configSvc d.AuthSvc = authSvc d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc d.ImageSvc = &stubImageService{} }) @@ -2758,13 +955,11 @@ func TestHandler_ImagesPrune_ValidationAndAvailability(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) authSvc := inmocks.NewMockAuthService(t) containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) handler := newTestHandler(t, func(d *HandlerDeps) { d.ConfigSvc = configSvc d.AuthSvc = authSvc d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc }) req := httptest.NewRequest("GET", "/admin/images", nil) @@ -2782,7 +977,6 @@ func TestHandler_ImagesPrune_ScopeDefaults(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) authSvc := inmocks.NewMockAuthService(t) containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) tests := []struct { name string @@ -2840,7 +1034,6 @@ func TestHandler_ImagesPrune_ScopeDefaults(t *testing.T) { d.ConfigSvc = configSvc d.AuthSvc = authSvc d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc d.ImageSvc = imageSvc }) @@ -2865,13 +1058,11 @@ func TestHandler_Images_Authorization(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) authSvc := inmocks.NewMockAuthService(t) containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) handler := newTestHandler(t, func(d *HandlerDeps) { d.ConfigSvc = configSvc d.AuthSvc = authSvc d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc d.ImageSvc = &stubImageService{} }) @@ -2985,133 +1176,3 @@ func TestHandler_TLSStatus_POSTReturns405(t *testing.T) { assert.Equal(t, http.StatusMethodNotAllowed, resp.StatusCode) } - -func TestHandler_AttachmentOrphans_ReturnsOrphanedAttachments(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - attachments := []domain.CleanupAttachment{{ContainerID: "attachment-1", Name: "postgres", Owner: "app.example.com"}} - containerSvc.EXPECT().ListOrphanedAttachments(mock.Anything).Return(attachments, nil).Once() - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - req := httptest.NewRequest("GET", "/admin/attachments/orphans", nil) - req = req.WithContext(ctxWithScopes("admin:config:read")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - require.Equal(t, http.StatusOK, rec.Code) - var resp struct { - Attachments []domain.CleanupAttachment `json:"attachments"` - } - require.NoError(t, json.NewDecoder(rec.Body).Decode(&resp)) - require.Len(t, resp.Attachments, 1) - assert.Equal(t, "app.example.com", resp.Attachments[0].Owner) -} - -func TestHandler_AttachmentPrune_StopsOnlyWhenRequested(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - report := &domain.CleanupReport{RemovedContainers: []domain.CleanupContainer{{ID: "attachment-1"}}} - containerSvc.EXPECT().CleanupOrphanedAttachments(mock.Anything, "", true).Return(report, nil).Once() - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - req := httptest.NewRequest("POST", "/admin/attachments/prune?stop=true", nil) - req = req.WithContext(ctxWithScopes("admin:config:write")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - require.Equal(t, http.StatusOK, rec.Code) - var got domain.CleanupReport - require.NoError(t, json.NewDecoder(rec.Body).Decode(&got)) - require.Len(t, got.RemovedContainers, 1) - assert.Equal(t, "attachment-1", got.RemovedContainers[0].ID) -} - -func TestHandler_AttachmentOrphans_ReturnsServerErrorOnListFailure(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - containerSvc.EXPECT().ListOrphanedAttachments(mock.Anything).Return(nil, errors.New("boom")).Once() - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - req := httptest.NewRequest("GET", "/admin/attachments/orphans", nil) - req = req.WithContext(ctxWithScopes("admin:config:read")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusInternalServerError, rec.Code) -} - -func TestHandler_AttachmentOrphans_PostReturns405(t *testing.T) { - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ContainerSvc = inmocks.NewMockContainerService(t) - }) - req := httptest.NewRequest("POST", "/admin/attachments/orphans", nil) - req = req.WithContext(ctxWithScopes("admin:config:read")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusMethodNotAllowed, rec.Code) -} - -func TestHandler_AttachmentPrune_ReturnsServerErrorOnCleanupFailure(t *testing.T) { - configSvc := inmocks.NewMockConfigService(t) - authSvc := inmocks.NewMockAuthService(t) - containerSvc := inmocks.NewMockContainerService(t) - secretSvc := inmocks.NewMockSecretService(t) - containerSvc.EXPECT().CleanupOrphanedAttachments(mock.Anything, "", true).Return(nil, errors.New("boom")).Once() - - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ConfigSvc = configSvc - d.AuthSvc = authSvc - d.ContainerSvc = containerSvc - d.SecretSvc = secretSvc - }) - - req := httptest.NewRequest("POST", "/admin/attachments/prune?stop=true", nil) - req = req.WithContext(ctxWithScopes("admin:config:write")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusInternalServerError, rec.Code) -} - -func TestHandler_AttachmentPrune_GetReturns405(t *testing.T) { - handler := newTestHandler(t, func(d *HandlerDeps) { - d.ContainerSvc = inmocks.NewMockContainerService(t) - }) - req := httptest.NewRequest("GET", "/admin/attachments/prune", nil) - req = req.WithContext(ctxWithScopes("admin:config:write")) - rec := httptest.NewRecorder() - - handler.ServeHTTP(rec, req) - - assert.Equal(t, http.StatusMethodNotAllowed, rec.Code) -} diff --git a/internal/adapters/in/http/admin/handler_volumes.go b/internal/adapters/in/http/admin/handler_volumes.go index ce5450377..ec92aed16 100644 --- a/internal/adapters/in/http/admin/handler_volumes.go +++ b/internal/adapters/in/http/admin/handler_volumes.go @@ -2,6 +2,7 @@ package admin import ( "encoding/json" + "errors" "net/http" "github.com/bnema/zerowrap" @@ -84,6 +85,10 @@ func (h *Handler) handlePruneVolumes(w http.ResponseWriter, r *http.Request) { if err != nil { log := zerowrap.FromCtx(ctx) log.Error().Err(err).Msg("failed to prune volumes") + if errors.Is(err, domain.ErrPruneDisabled) { + h.sendJSON(w, http.StatusConflict, dto.AppError{Error: "prune-disabled", Message: err.Error()}) + return + } h.sendError(w, http.StatusInternalServerError, "failed to prune volumes") return } @@ -100,5 +105,6 @@ func (h *Handler) handlePruneVolumes(w http.ResponseWriter, r *http.Request) { VolumesRemoved: report.VolumesRemoved, SpaceReclaimed: report.SpaceReclaimed, Volumes: vols, + Plan: toPruneSummary(report.Plan), }) } diff --git a/internal/adapters/in/http/admin/images.go b/internal/adapters/in/http/admin/images.go index b7ef41f6b..661438c0f 100644 --- a/internal/adapters/in/http/admin/images.go +++ b/internal/adapters/in/http/admin/images.go @@ -34,6 +34,7 @@ func toImagePruneResponse(report domain.ImagePruneReport) dto.ImagePruneResponse BlobsRemoved: report.Registry.BlobsRemoved, SpaceReclaimed: report.Registry.SpaceReclaimed, }, + Plan: toPruneSummary(report.Plan), } } @@ -111,6 +112,9 @@ func (h *Handler) handleImagesPrune(w http.ResponseWriter, r *http.Request) { if req.PruneRegistry != nil { opts.PruneRegistry = *req.PruneRegistry } + if req.DryRun != nil { + opts.DryRun = *req.DryRun + } if !opts.PruneDangling && !opts.PruneRegistry { h.sendError(w, http.StatusBadRequest, "at least one prune scope must be enabled") return @@ -119,6 +123,10 @@ func (h *Handler) handleImagesPrune(w http.ResponseWriter, r *http.Request) { report, err := h.imageSvc.Prune(ctx, opts) if err != nil { log.Error().Err(err).Int("keep_last", opts.KeepLast).Msg("image prune failed") + if errors.Is(err, domain.ErrPruneDisabled) { + h.sendJSON(w, http.StatusConflict, dto.AppError{Error: "prune-disabled", Message: err.Error()}) + return + } h.sendError(w, http.StatusInternalServerError, "failed to prune images") return } diff --git a/internal/adapters/in/http/admin/local_authority.go b/internal/adapters/in/http/admin/local_authority.go new file mode 100644 index 000000000..9f541edd0 --- /dev/null +++ b/internal/adapters/in/http/admin/local_authority.go @@ -0,0 +1,148 @@ +package admin + +import ( + "context" + "net/http" + "path" + "strings" + + "github.com/bnema/gordon/internal/domain" +) + +// LocalAuthoritySubject is the principal carried on requests served over the +// owner-only admin Unix socket. Filesystem access to the 0600 socket is what +// establishes this identity; the scopes below still gate each route. +const LocalAuthoritySubject = "local-owner" + +// localAuthorityScopes are the least-privilege scopes required by CLI +// commands which share the local and remote ControlPlane. The path-and-method +// allowlist below is the second boundary: config:write, for example, permits +// reload and prune operations but cannot reach unrelated config mutations. +var localAuthorityScopes = []string{ + domain.AdminScopeApps(domain.AdminActionRead, domain.AdminActionWrite), + domain.AdminScopeStatus(domain.AdminActionRead), + domain.AdminScopeConfig(domain.AdminActionRead, domain.AdminActionWrite), + domain.AdminScopeLogs(domain.AdminActionRead), + domain.AdminScopeVolumes(domain.AdminActionRead, domain.AdminActionWrite), +} + +// LocalAuthority returns the handler for the owner-only admin Unix socket. +// Filesystem ownership authenticates the caller; this wrapper then restricts +// that identity to the canonical endpoints used by local CLI commands. +func (h *Handler) LocalAuthority() http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + // Classify and dispatch on one canonical path: the socket bypasses the + // ServeMux that would otherwise clean "." and ".." segments, so a + // non-canonical path must never be allowlisted here and interpreted + // differently by the router below. + if !strings.HasPrefix(r.URL.Path, "/admin/") && r.URL.Path != "/admin" { + sendForbidden(w, "local administration is limited to app endpoints") + return + } + trimmed := strings.TrimPrefix(r.URL.Path, "/admin") + cleaned := path.Clean(trimmed) + if cleaned != trimmed || !localPathAllowed(r.Method, cleaned) { + sendForbidden(w, "endpoint is not available to the local owner") + return + } + + scopes := make([]string, len(localAuthorityScopes)) + copy(scopes, localAuthorityScopes) + + ctx := context.WithValue(r.Context(), domain.ContextKeyScopes, scopes) + ctx = context.WithValue(ctx, domain.ContextKeySubject, LocalAuthoritySubject) + h.handleAdminRoutes(w, r.WithContext(ctx)) + }) +} + +func localPathAllowed(method, cleanedPath string) bool { + if localExactPathAllowed(method, cleanedPath) { + return true + } + parts := strings.Split(strings.TrimPrefix(cleanedPath, "/"), "/") + prefixRules := map[string]func(string, []string) bool{ + "apps": localAppPathAllowed, + "backups": localBackupPathAllowed, + "logs": localLogPathAllowed, + "tags": localTagPathAllowed, + } + rule, ok := prefixRules[parts[0]] + return ok && rule(method, parts) +} + +func localLogPathAllowed(method string, parts []string) bool { + return (len(parts) == 1 || len(parts) == 3 && parts[1] != "" && parts[2] != "") && method == http.MethodGet +} + +func localTagPathAllowed(method string, parts []string) bool { + return len(parts) == 2 && parts[1] != "" && method == http.MethodGet +} + +func localExactPathAllowed(method, cleanedPath string) bool { + allowed := map[string]string{ + "/status": http.MethodGet, + "/tls/status": http.MethodGet, + "/traffic/status": http.MethodGet, + "/config": http.MethodGet, + "/networks": http.MethodGet, + "/volumes": http.MethodGet, + "/volumes/prune": http.MethodPost, + "/images": http.MethodGet, + "/images/prune": http.MethodPost, + "/reload": http.MethodPost, + "/backups": http.MethodGet, + "/backups/status": http.MethodGet, + "/backups/volumes": http.MethodGet, + "/backups/volumes/status": http.MethodGet, + } + return allowed[cleanedPath] == method +} + +func localAppPathAllowed(method string, parts []string) bool { + appRoutes := map[string]string{ + "": http.MethodGet, + "apply": http.MethodPost, + "{app}": http.MethodGet, + "{app}/diff": http.MethodGet, + "{app}/deploy": http.MethodPost, + "{app}/restart": http.MethodPost, + "{app}/stop": http.MethodPost, + "{app}/start": http.MethodPost, + "{app}/remove": http.MethodPost, + "{app}/secrets": http.MethodGet, + "{app}/secrets/set": http.MethodPost, + "{app}/secrets/delete": http.MethodPost, + "{app}/operations/by-key/{key}": http.MethodGet, + } + return appRoutes[localAppRouteShape(parts)] == method +} + +func localAppRouteShape(parts []string) string { + if len(parts) == 1 { + return "" + } + if len(parts) == 2 && parts[1] == "apply" { + return "apply" + } + if len(parts) < 2 || parts[1] == "" { + return "invalid" + } + shaped := append([]string{"{app}"}, parts[2:]...) + if len(shaped) == 4 && shaped[1] == "operations" && shaped[2] == "by-key" && shaped[3] != "" { + shaped[3] = "{key}" + } + return strings.Join(shaped, "/") +} + +func localBackupPathAllowed(method string, parts []string) bool { + if len(parts) == 2 { + if parts[1] == "volumes" { + return method == http.MethodPost + } + return parts[1] != "" && parts[1] != "status" && (method == http.MethodGet || method == http.MethodPost) + } + if len(parts) == 3 && parts[1] == "volumes" && parts[2] != "" && parts[2] != "status" { + return method == http.MethodGet || method == http.MethodPost + } + return false +} diff --git a/internal/adapters/in/http/admin/local_authority_test.go b/internal/adapters/in/http/admin/local_authority_test.go new file mode 100644 index 000000000..77d4b5ddd --- /dev/null +++ b/internal/adapters/in/http/admin/local_authority_test.go @@ -0,0 +1,146 @@ +package admin + +import ( + "context" + "net/http" + "net/http/httptest" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + in "github.com/bnema/gordon/internal/boundaries/in" + inmocks "github.com/bnema/gordon/internal/boundaries/in/mocks" + "github.com/bnema/gordon/internal/domain" +) + +func localRequest(t *testing.T, handler http.Handler, method, target string) *httptest.ResponseRecorder { + t.Helper() + req := httptest.NewRequest(method, target, nil) + rec := httptest.NewRecorder() + handler.ServeHTTP(rec, req) + return rec +} + +func TestLocalAuthorityInjectsLeastPrivilegeScopes(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + handler := appsTestHandler(t, appSvc) + + appSvc.EXPECT().List(mock.MatchedBy(func(ctx context.Context) bool { + assert.Equal(t, LocalAuthoritySubject, GetSubject(ctx)) + for _, access := range []struct{ resource, action string }{ + {domain.AdminResourceApps, domain.AdminActionRead}, + {domain.AdminResourceApps, domain.AdminActionWrite}, + {domain.AdminResourceStatus, domain.AdminActionRead}, + {domain.AdminResourceConfig, domain.AdminActionRead}, + {domain.AdminResourceConfig, domain.AdminActionWrite}, + {domain.AdminResourceLogs, domain.AdminActionRead}, + {domain.AdminResourceVolumes, domain.AdminActionRead}, + {domain.AdminResourceVolumes, domain.AdminActionWrite}, + } { + assert.True(t, HasAccess(ctx, access.resource, access.action), "%s:%s", access.resource, access.action) + } + assert.False(t, HasAccess(ctx, domain.AdminResourceRoutes, domain.AdminActionRead)) + assert.False(t, HasAccess(ctx, "secrets", domain.AdminActionRead)) + assert.False(t, HasAccess(ctx, domain.AdminResourceStatus, domain.AdminActionWrite)) + assert.False(t, HasAccess(ctx, domain.AdminResourceLogs, domain.AdminActionWrite)) + assert.False(t, HasAccess(ctx, domain.AdminResourceAll, domain.AdminActionAll)) + return true + })).Return([]in.AppSummary(nil), nil).Once() + + rec := localRequest(t, handler.LocalAuthority(), http.MethodGet, "/admin/apps") + require.Equal(t, http.StatusOK, rec.Code) +} + +func TestLocalAuthorityCanonicalAllowlist(t *testing.T) { + t.Parallel() + allowed := []struct{ method, path string }{ + {http.MethodGet, "/status"}, + {http.MethodGet, "/tls/status"}, + {http.MethodGet, "/traffic/status"}, + {http.MethodGet, "/config"}, + {http.MethodGet, "/networks"}, + {http.MethodGet, "/volumes"}, + {http.MethodPost, "/volumes/prune"}, + {http.MethodGet, "/images"}, + {http.MethodPost, "/images/prune"}, + {http.MethodPost, "/reload"}, + {http.MethodGet, "/tags/repository"}, + {http.MethodGet, "/logs"}, + {http.MethodGet, "/logs/blog/web"}, + {http.MethodGet, "/backups"}, + {http.MethodGet, "/backups/status"}, + {http.MethodGet, "/backups/shop"}, + {http.MethodPost, "/backups/shop"}, + {http.MethodGet, "/backups/volumes"}, + {http.MethodPost, "/backups/volumes"}, + {http.MethodGet, "/backups/volumes/status"}, + {http.MethodGet, "/backups/volumes/blog.example.com"}, + {http.MethodPost, "/backups/volumes/blog.example.com"}, + {http.MethodGet, "/apps"}, + {http.MethodPost, "/apps/apply"}, + {http.MethodGet, "/apps/blog"}, + {http.MethodGet, "/apps/blog/diff"}, + {http.MethodPost, "/apps/blog/deploy"}, + {http.MethodPost, "/apps/blog/restart"}, + {http.MethodPost, "/apps/blog/stop"}, + {http.MethodPost, "/apps/blog/start"}, + {http.MethodPost, "/apps/blog/remove"}, + {http.MethodPost, "/apps/blog/secrets/set"}, + {http.MethodPost, "/apps/blog/secrets/delete"}, + {http.MethodGet, "/apps/blog/operations/by-key/request-1"}, + } + for _, tc := range allowed { + t.Run(tc.method+" "+tc.path, func(t *testing.T) { + assert.True(t, localPathAllowed(tc.method, tc.path)) + }) + } +} + +func TestLocalAuthorityDeniesNearMissesAndNonCanonicalPaths(t *testing.T) { + t.Parallel() + denied := []struct{ method, path string }{ + {http.MethodPost, "/status"}, + {http.MethodGet, "/reload"}, + {http.MethodPost, "/config"}, + {http.MethodGet, "/volumes/prune"}, + {http.MethodGet, "/images/prune"}, + {http.MethodGet, "/auth/verify"}, + {http.MethodGet, "/auth/tokens"}, + {http.MethodPost, "/auth/tokens"}, + {http.MethodGet, "/health"}, + {http.MethodGet, "/appsX"}, + {http.MethodGet, "/secrets"}, + {http.MethodGet, "/secrets/blog.example.com"}, + {http.MethodPost, "/secrets/blog.example.com"}, + {http.MethodDelete, "/secrets/blog"}, + {http.MethodGet, "/images/anything"}, + {http.MethodGet, "/tags/repository/extra"}, + {http.MethodGet, "/logs/blog"}, + {http.MethodGet, "/backups/volumes/status/extra"}, + {http.MethodGet, "/apps/blog/operations/by-key/key/extra"}, + {http.MethodGet, "/"}, + } + for _, tc := range denied { + t.Run(tc.method+" "+tc.path, func(t *testing.T) { + assert.False(t, localPathAllowed(tc.method, tc.path)) + }) + } +} + +func TestLocalAuthorityRejectsNonAdminAndTraversalRequests(t *testing.T) { + appSvc := inmocks.NewMockAppService(t) + local := appsTestHandler(t, appSvc).LocalAuthority() + for _, target := range []string{ + "/apps", + "/admin/auth/tokens", + "/admin/apps/", + "/admin/apps/../config", + "/admin/apps/..%2fconfig", + "/admin/logs/../../config", + } { + rec := localRequest(t, local, http.MethodGet, target) + assert.Equal(t, http.StatusForbidden, rec.Code, target) + } +} diff --git a/internal/adapters/in/http/admin/preview_handler.go b/internal/adapters/in/http/admin/preview_handler.go deleted file mode 100644 index af444b20a..000000000 --- a/internal/adapters/in/http/admin/preview_handler.go +++ /dev/null @@ -1,167 +0,0 @@ -package admin - -import ( - "context" - "encoding/json" - "errors" - "net/http" - "strings" - "time" - - "github.com/bnema/zerowrap" - - "github.com/bnema/gordon/internal/domain" - "github.com/bnema/gordon/pkg/validation" -) - -// previewService is a minimal interface for preview management used by the admin handler. -type previewService interface { - List(ctx context.Context) ([]domain.PreviewRoute, error) - Delete(ctx context.Context, name string) error - Extend(ctx context.Context, name string, ttl time.Duration) error -} - -// previewListResponse is the JSON payload returned by GET /admin/previews. -type previewListResponse struct { - Previews []domain.PreviewRoute `json:"previews"` -} - -// previewExtendRequest is the JSON body for PATCH /admin/preview/{name}. -type previewExtendRequest struct { - TTL string `json:"ttl"` // e.g. "24h", "30m" -} - -// handlePreviewList handles GET /admin/previews. -// Returns a JSON array of all active preview environments. -func (h *Handler) handlePreviewList(w http.ResponseWriter, r *http.Request, _ string) { - if r.Method != http.MethodGet { - h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") - return - } - - ctx := r.Context() - - if !HasAccess(ctx, domain.AdminResourceStatus, domain.AdminActionRead) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for status:read") - return - } - - if h.previewSvc == nil { - h.sendError(w, http.StatusServiceUnavailable, "preview service not available") - return - } - - previews, err := h.previewSvc.List(ctx) - if err != nil { - log := zerowrap.FromCtx(ctx) - log.Error().Err(err).Msg("failed to list previews") - h.sendError(w, http.StatusInternalServerError, "failed to list previews") - return - } - - if previews == nil { - previews = []domain.PreviewRoute{} - } - - h.sendJSON(w, http.StatusOK, previewListResponse{Previews: previews}) -} - -// handlePreviewAction handles DELETE and PATCH on /admin/preview/{name}. -// -// DELETE /admin/preview/{name} — tear down the preview -// PATCH /admin/preview/{name} — extend TTL (body: {"ttl":"24h"}) -func (h *Handler) handlePreviewAction(w http.ResponseWriter, r *http.Request, path string) { - ctx := r.Context() - log := zerowrap.FromCtx(ctx) - - if h.previewSvc == nil { - h.sendError(w, http.StatusServiceUnavailable, "preview service not available") - return - } - - if !HasAccess(ctx, domain.AdminResourceConfig, domain.AdminActionWrite) { - h.sendError(w, http.StatusForbidden, "insufficient permissions for config:write") - return - } - - name := strings.TrimPrefix(path, "/preview/") - if name == "" || name == "/preview" { - h.sendError(w, http.StatusBadRequest, "preview name required in path") - return - } - if err := validation.ValidateDomainParam(name); err != nil { - h.sendError(w, http.StatusBadRequest, "invalid domain") - return - } - - switch r.Method { - case http.MethodDelete: - - if err := h.previewSvc.Delete(ctx, name); err != nil { - log.Error().Err(err).Str("name", name).Msg("failed to delete preview") - if errors.Is(err, domain.ErrPreviewNotFound) { - h.sendError(w, http.StatusNotFound, "preview not found") - } else { - h.sendError(w, http.StatusInternalServerError, "failed to delete preview") - } - return - } - - log.Info().Str("name", name).Msg("preview deleted via admin API") - h.sendJSON(w, http.StatusOK, map[string]string{"status": "deleted", "name": name}) - - case http.MethodPatch: - h.handlePreviewExtend(w, r, name) - - default: - h.sendError(w, http.StatusMethodNotAllowed, "method not allowed") - } -} - -func (h *Handler) handlePreviewExtend(w http.ResponseWriter, r *http.Request, name string) { - ctx := r.Context() - log := zerowrap.FromCtx(ctx) - - r.Body = http.MaxBytesReader(w, r.Body, maxAdminRequestSize) - - var req previewExtendRequest - if err := json.NewDecoder(r.Body).Decode(&req); err != nil { - log.Warn().Err(err).Msg("invalid preview extend JSON") - h.sendError(w, http.StatusBadRequest, "invalid JSON") - return - } - - if req.TTL == "" { - h.sendError(w, http.StatusBadRequest, "ttl is required") - return - } - - ttl, err := time.ParseDuration(req.TTL) - if err != nil { - h.sendError(w, http.StatusBadRequest, "invalid ttl: use Go duration format, e.g. 24h or 30m") - return - } - - if ttl <= 0 { - h.sendError(w, http.StatusBadRequest, "TTL must be positive") - return - } - const maxPreviewTTL = 30 * 24 * time.Hour // 30 days - if ttl > maxPreviewTTL { - h.sendError(w, http.StatusBadRequest, "TTL exceeds maximum of 30 days") - return - } - - if err := h.previewSvc.Extend(ctx, name, ttl); err != nil { - log.Error().Err(err).Str("name", name).Dur("ttl", ttl).Msg("failed to extend preview TTL") - if errors.Is(err, domain.ErrPreviewNotFound) { - h.sendError(w, http.StatusNotFound, "preview not found") - } else { - h.sendError(w, http.StatusInternalServerError, "failed to extend preview") - } - return - } - - log.Info().Str("name", name).Dur("ttl", ttl).Msg("preview TTL extended via admin API") - h.sendJSON(w, http.StatusOK, map[string]string{"status": "extended", "name": name, "ttl": req.TTL}) -} diff --git a/internal/adapters/in/http/admin/prune.go b/internal/adapters/in/http/admin/prune.go new file mode 100644 index 000000000..4e9d93a06 --- /dev/null +++ b/internal/adapters/in/http/admin/prune.go @@ -0,0 +1,11 @@ +package admin + +import ( + "github.com/bnema/gordon/internal/adapters/dto" + "github.com/bnema/gordon/internal/domain" +) + +// toPruneSummary maps the domain prune report to its wire form. +func toPruneSummary(report domain.PruneReport) dto.PruneSummary { + return dto.PruneSummaryFromDomain(report) +} diff --git a/internal/adapters/in/http/admin/prune_test.go b/internal/adapters/in/http/admin/prune_test.go new file mode 100644 index 000000000..ad6f39aec --- /dev/null +++ b/internal/adapters/in/http/admin/prune_test.go @@ -0,0 +1,112 @@ +package admin + +import ( + "bytes" + "context" + "encoding/json" + "net/http" + "net/http/httptest" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/dto" + inmocks "github.com/bnema/gordon/internal/boundaries/in/mocks" + "github.com/bnema/gordon/internal/domain" +) + +// TestHandler_ImagesPrune_ExposesPerCandidateVerdicts proves the wire +// response carries the same plan shape for a dry run and an execution, +// including the reasons protected and unknown candidates were skipped. +func TestHandler_ImagesPrune_ExposesPerCandidateVerdicts(t *testing.T) { + plan := domain.PruneReport{ + Applied: false, + Candidates: []domain.PruneCandidateReport{ + {Kind: domain.PruneResourceRegistryTag, Ref: "app:v1", Verdict: domain.PruneVerdictEligible, + Reasons: []domain.PruneReason{domain.PruneReasonEligibleRetention}}, + {Kind: domain.PruneResourceRuntimeImage, Ref: "sha256:active", Verdict: domain.PruneVerdictProtected, + Reasons: []domain.PruneReason{domain.PruneReasonProtectedContainerUse}}, + {Kind: domain.PruneResourceRuntimeImage, Ref: "sha256:opaque", Verdict: domain.PruneVerdictUnknown, + Reasons: []domain.PruneReason{domain.PruneReasonUnknownProvenance}}, + }, + Gaps: []domain.InventoryGap{{ + Source: domain.InventorySourceRuntimeContainers, + Reason: domain.PruneReasonUnknownContainerUse, + Detail: "list containers failed", + }}, + Failures: []domain.PruneFailure{{ + Kind: domain.PruneResourceRegistryTag, Ref: "app:v9", Err: "storage unavailable", + }}, + } + + var dryRunSeen bool + imageSvc := &stubImageService{ + pruneFunc: func(_ context.Context, opts domain.ImagePruneOptions) (domain.ImagePruneReport, error) { + dryRunSeen = opts.DryRun + return domain.ImagePruneReport{Plan: plan}, nil + }, + } + + handler := newTestHandler(t, func(d *HandlerDeps) { + d.ImageSvc = imageSvc + }) + + rec := httptest.NewRecorder() + req := httptest.NewRequest(http.MethodPost, "/admin/images/prune", bytes.NewBufferString(`{"dry_run": true}`)) + req = req.WithContext(ctxWithScopes("admin:config:write")) + handler.ServeHTTP(rec, req) + + require.Equal(t, http.StatusOK, rec.Code) + assert.True(t, dryRunSeen, "the dry-run flag must reach the use case") + + var resp dto.ImagePruneResponse + require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &resp)) + assert.False(t, resp.Plan.Applied) + assert.Equal(t, 1, resp.Plan.Eligible) + assert.Equal(t, 1, resp.Plan.Protected) + assert.Equal(t, 1, resp.Plan.Unknown) + require.Len(t, resp.Plan.Candidates, 3) + assert.Equal(t, "protected-container-use", resp.Plan.Candidates[1].Reasons[0]) + require.Len(t, resp.Plan.Failures, 1) + assert.Equal(t, "storage unavailable", resp.Plan.Failures[0].Error) + require.Len(t, resp.Plan.Gaps, 1) + assert.Equal(t, "runtime-containers", resp.Plan.Gaps[0].Source) +} + +// TestHandler_VolumesPrune_ExposesPerCandidateVerdicts proves the volume +// endpoint returns the same shared plan shape, including a zero-deletion +// success. +func TestHandler_VolumesPrune_ExposesPerCandidateVerdicts(t *testing.T) { + report := &domain.VolumePruneReport{ + Plan: domain.PruneReport{ + Applied: true, + Candidates: []domain.PruneCandidateReport{ + {Kind: domain.PruneResourceVolume, Ref: "gordon-shop--web--vol--data", Verdict: domain.PruneVerdictProtected, + Reasons: []domain.PruneReason{domain.PruneReasonProtectedOwnership}}, + }, + }, + } + + volumeSvc := inmocks.NewMockVolumeService(t) + volumeSvc.EXPECT().PruneVolumes(mock.Anything, false).Return(report, []*domain.VolumeInfo{}, nil).Once() + handler := newTestHandler(t, func(d *HandlerDeps) { + d.VolumeSvc = volumeSvc + }) + + rec := httptest.NewRecorder() + req := httptest.NewRequest(http.MethodPost, "/admin/volumes/prune", bytes.NewBufferString(`{"dry_run": false}`)) + req = req.WithContext(ctxWithScopes("admin:volumes:write")) + handler.ServeHTTP(rec, req) + + require.Equal(t, http.StatusOK, rec.Code) + var resp dto.VolumePruneResponse + require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &resp)) + assert.True(t, resp.Plan.Applied) + assert.Zero(t, resp.VolumesRemoved) + assert.Equal(t, 1, resp.Plan.Protected) + assert.Zero(t, resp.Plan.Eligible, "a protected candidate is not eligible") + require.Len(t, resp.Plan.Candidates, 1) + assert.Equal(t, "protected-ownership", resp.Plan.Candidates[0].Reasons[0]) +} diff --git a/internal/adapters/in/http/httphelper/localhost.go b/internal/adapters/in/http/httphelper/localhost.go index 4aa590ab7..6cdeafbd1 100644 --- a/internal/adapters/in/http/httphelper/localhost.go +++ b/internal/adapters/in/http/httphelper/localhost.go @@ -9,14 +9,15 @@ import ( // IsLocalhostRequest reports whether the request originates from localhost. // SECURITY: Uses RemoteAddr (server-set) instead of Host header (client-spoofable). func IsLocalhostRequest(r *http.Request) bool { + addr, ok := remoteAddr(r) + return ok && addr.IsLoopback() +} + +func remoteAddr(r *http.Request) (netip.Addr, bool) { host, _, err := net.SplitHostPort(r.RemoteAddr) if err != nil { - // RemoteAddr has no port (unusual but possible in tests/Unix sockets). host = r.RemoteAddr } addr, err := netip.ParseAddr(host) - if err != nil { - return false - } - return addr.IsLoopback() + return addr, err == nil } diff --git a/internal/adapters/in/http/middleware/auth.go b/internal/adapters/in/http/middleware/auth.go index e3943293c..eef62e6f5 100644 --- a/internal/adapters/in/http/middleware/auth.go +++ b/internal/adapters/in/http/middleware/auth.go @@ -13,6 +13,7 @@ import ( "github.com/bnema/gordon/internal/adapters/dto" "github.com/bnema/gordon/internal/adapters/in/http/httphelper" + "github.com/bnema/gordon/internal/adapters/in/http/registry/route" "github.com/bnema/gordon/internal/boundaries/in" "github.com/bnema/gordon/internal/domain" ) @@ -235,47 +236,26 @@ func actionFromMethod(method string) string { } } -// extractRepoName extracts the repository name from a registry path. -// Returns empty string if the path is not a valid registry path or should be allowed. -func extractRepoName(path string) (repoName string, shouldAllow bool) { - if !strings.HasPrefix(path, "/v2/") { - return "", true // Not a registry path, allow - } - - pathParts := strings.Split(strings.TrimPrefix(path, "/v2/"), "/") - if len(pathParts) < 2 { - return "", true // Malformed path or /v2/ root, allow - } - - // Find the boundary between repo name and route (manifests, blobs, tags) - var repoNameParts []string - for i, part := range pathParts { - if part == "manifests" || part == "blobs" || part == "tags" || part == "_catalog" { - repoNameParts = pathParts[:i] - break - } - } - if len(repoNameParts) == 0 { - return "", true // Special route like /v2/token, allow - } - - return strings.Join(repoNameParts, "/"), false -} - // checkScopeAccess verifies the token has permission for the requested operation. -// Maps HTTP method to registry action and checks if any token scope grants access. +// It parses the path with the same parser the registry handler dispatches +// on, so authorization and dispatch always name the same repository; a path +// the handler would reject is denied rather than allowed. func checkScopeAccess(r *http.Request, claims *domain.TokenClaims, log zerowrap.Logger) bool { action := actionFromMethod(r.Method) - repoName, shouldAllow := extractRepoName(r.URL.Path) - if shouldAllow { + op, err := route.Parse(r.URL.Path) + if err != nil { + log.Debug().Err(err).Str("path", r.URL.Path).Msg("malformed registry path denied") + return false + } + if !op.RequiresRepositoryAuth() { return true } // Delegate matching to domain layer - if domain.ScopesGrantRegistryAccess(claims.Scopes, repoName, action) { + if domain.ScopesGrantRegistryAccess(claims.Scopes, op.Repository, action) { log.Debug(). - Str("repo", repoName). + Str("repo", op.Repository). Str("action", action). Strs("scopes", claims.Scopes). Msg("scope access granted") @@ -283,7 +263,7 @@ func checkScopeAccess(r *http.Request, claims *domain.TokenClaims, log zerowrap. } log.Debug(). - Str("repo", repoName). + Str("repo", op.Repository). Str("action", action). Strs("scopes", claims.Scopes). Msg("no scope grants access") diff --git a/internal/adapters/in/http/middleware/auth_test.go b/internal/adapters/in/http/middleware/auth_test.go index d823abb16..e8952a790 100644 --- a/internal/adapters/in/http/middleware/auth_test.go +++ b/internal/adapters/in/http/middleware/auth_test.go @@ -217,7 +217,7 @@ func TestCheckScopeAccess_RepoNameExtraction(t *testing.T) { }, { name: "simple repo with blobs", - path: "/v2/myrepo/blobs/sha256:abc123", + path: "/v2/myrepo/blobs/sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", scopes: []string{"repository:myrepo:pull"}, want: true, }, @@ -257,6 +257,53 @@ func TestCheckScopeAccess_RepoNameExtraction(t *testing.T) { } } +func TestCheckScopeAccess_ReservedComponentsCannotRedirectRepository(t *testing.T) { + log := testLogger() + tests := []struct { + name string + method string + path string + scopes []string + want bool + }{ + { + name: "unauthorized repo behind reserved first component", + method: http.MethodGet, + path: "/v2/manifests/victim/manifests/latest", + scopes: []string{"repository:allowed:pull"}, + want: false, + }, + { + name: "authorized prefix cannot reach a nested repository", + method: http.MethodGet, + path: "/v2/allowed/tags/victim/manifests/latest", + scopes: []string{"repository:allowed:pull"}, + want: false, + }, + { + name: "pull token cannot PUT a reserved-looking repository", + method: http.MethodPut, + path: "/v2/manifests/victim/manifests/latest", + scopes: []string{"repository:manifests/victim:pull"}, + want: false, + }, + { + name: "exact nested repository is authorized", + method: http.MethodGet, + path: "/v2/manifests/victim/manifests/latest", + scopes: []string{"repository:manifests/victim:pull"}, + want: true, + }, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + claims := &domain.TokenClaims{Scopes: tt.scopes} + req := httptest.NewRequest(tt.method, tt.path, nil) + assert.Equal(t, tt.want, checkScopeAccess(req, claims, log)) + }) + } +} + func TestCheckScopeAccess_WildcardMatching(t *testing.T) { log := testLogger() @@ -318,10 +365,10 @@ func TestCheckScopeAccess_SpecialRoutes(t *testing.T) { want: true, }, { - name: "non-v2 path is allowed", + name: "non-v2 path is denied", path: "/healthz", scopes: []string{"repository:myrepo:pull"}, - want: true, + want: false, }, { name: "catalog path is handled", diff --git a/internal/adapters/in/http/middleware/cidr.go b/internal/adapters/in/http/middleware/cidr.go index 3cb819ab3..5436e9456 100644 --- a/internal/adapters/in/http/middleware/cidr.go +++ b/internal/adapters/in/http/middleware/cidr.go @@ -70,18 +70,45 @@ func RegistryCIDRAllowlist(allowedNets, trustedNets []*net.IPNet, log zerowrap.L // helper can map it to tlsPort. Hosts with no explicit port omit the TLS port from // the public URL; hosts with an unknown explicit port preserve it as-is. func HTTPSRedirect(proxyNets []*net.IPNet, httpPort, tlsPort int, forceAll bool, log zerowrap.Logger, isHostAllowed func(string) bool) func(http.Handler) http.Handler { + return HTTPSRedirectWithEligibility(proxyNets, httpPort, tlsPort, forceAll, log, isHostAllowed, nil, nil) +} + +// HTTPSRedirectWithEligibility is HTTPSRedirect with an extra opt-out and +// an enforced-TLS predicate: +// - isRedirectEligible reports whether a KNOWN host should actually be +// redirected (e.g. app hosts declared tls=never stay plain HTTP). +// A known but ineligible host passes through to next instead of +// redirecting; unknown/invalid hosts still get 400. Nil behaves like +// HTTPSRedirect (every allowed host is eligible). +// - isTLSAlways reports whether a host requires HTTPS unconditionally +// (tls=always). Such a host is redirected when an HTTPS endpoint +// exists and otherwise refused, never served over plaintext. Nil +// disables the check. +func HTTPSRedirectWithEligibility(proxyNets []*net.IPNet, httpPort, tlsPort int, forceAll bool, log zerowrap.Logger, isHostAllowed func(string) bool, isRedirectEligible func(string) bool, isTLSAlways func(string) bool) func(http.Handler) http.Handler { + if tlsPort == 0 && isTLSAlways == nil { + return func(next http.Handler) http.Handler { return next } + } + if !forceAll && len(proxyNets) == 0 && isTLSAlways == nil { + log.Info(). + Str(zerowrap.FieldLayer, "adapter"). + Str(zerowrap.FieldAdapter, "http"). + Msg("HTTP→HTTPS redirect disabled: proxy_allowed_ips is empty and force_https_redirect is false; set either to enable redirects") + return func(next http.Handler) http.Handler { return next } + } return func(next http.Handler) http.Handler { - if tlsPort == 0 { - return next - } - if !forceAll && len(proxyNets) == 0 { - log.Info(). - Str(zerowrap.FieldLayer, "adapter"). - Str(zerowrap.FieldAdapter, "http"). - Msg("HTTP→HTTPS redirect disabled: proxy_allowed_ips is empty and force_https_redirect is false; set either to enable redirects") - return next - } return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + // A tls=always host must never be served over plaintext. It is + // redirected when an HTTPS endpoint exists and refused when it + // does not, regardless of proxy/redirect configuration. + if isTLSAlways != nil && isTLSAlwaysForHost(r.Host, isTLSAlways) { + handleTLSAlways(w, r, httpPort, tlsPort, isHostAllowed, log) + return + } + + if tlsPort == 0 { + next.ServeHTTP(w, r) + return + } if !forceAll { remoteIP := httphelper.ExtractRemoteIP(r.RemoteAddr) if httphelper.IsTrustedOrLocal(remoteIP, proxyNets) { @@ -91,6 +118,12 @@ func HTTPSRedirect(proxyNets []*net.IPNet, httpPort, tlsPort int, forceAll bool, } target, ok := httpsRedirectTarget(r.Host, r.RequestURI, httpPort, tlsPort, isHostAllowed) + if ok && isRedirectEligible != nil && !isRedirectEligibleForHost(r.Host, isRedirectEligible) { + // Known host that opted out of TLS (e.g. tls=never): + // serve plain HTTP through the normal chain. + next.ServeHTTP(w, r) + return + } if !ok { http.Error(w, "Bad Request", http.StatusBadRequest) return @@ -105,8 +138,45 @@ func HTTPSRedirect(proxyNets []*net.IPNet, httpPort, tlsPort int, forceAll bool, } } -// httpsRedirectTarget derives the HTTPS redirect URL from a validated request Host. -func httpsRedirectTarget(host, requestURI string, httpPort, tlsPort int, isHostAllowed func(string) bool) (string, bool) { +// handleTLSAlways redirects or refuses a request to a tls=always host. It +// never forwards: the backend is not reached over plaintext. +func handleTLSAlways(w http.ResponseWriter, r *http.Request, httpPort, tlsPort int, isHostAllowed func(string) bool, log zerowrap.Logger) { + if tlsPort == 0 { + log.Warn().Str("host", r.Host).Msg("plaintext refused: host requires TLS but no HTTPS endpoint is configured") + http.Error(w, "Misdirected Request", http.StatusMisdirectedRequest) + return + } + target, ok := httpsRedirectTarget(r.Host, r.RequestURI, httpPort, tlsPort, isHostAllowed) + if !ok { + http.Error(w, "Bad Request", http.StatusBadRequest) + return + } + http.Redirect(w, r, target, http.StatusPermanentRedirect) +} + +// isTLSAlwaysForHost reports whether the canonical request Host requires +// HTTPS. Invalid hosts are never treated as always. +func isTLSAlwaysForHost(host string, isTLSAlways func(string) bool) bool { + canonical, ok := canonicalRedirectHost(host) + if !ok { + return false + } + return isTLSAlways(canonical) +} + +// isRedirectEligibleForHost reports whether the canonical request Host +// is eligible for redirect. Invalid hosts are never eligible. +func isRedirectEligibleForHost(host string, isRedirectEligible func(string) bool) bool { + canonical, ok := canonicalRedirectHost(host) + if !ok { + return false + } + return isRedirectEligible(canonical) +} + +// canonicalRedirectHost lowercases, strips ports/trailing dots and +// validates the request Host, returning the canonical route domain. +func canonicalRedirectHost(host string) (string, bool) { host = strings.ToLower(strings.TrimSpace(host)) if host == "" { return "", false @@ -121,7 +191,15 @@ func httpsRedirectTarget(host, requestURI string, httpPort, tlsPort int, isHostA return "", false } hostname = strings.TrimSuffix(hostname, ".") - canonicalHost, ok := domain.CanonicalRouteDomain(hostname) + return domain.CanonicalRouteDomain(hostname) +} + +// httpsRedirectTarget derives the HTTPS redirect URL from a validated request Host. +func httpsRedirectTarget(host, requestURI string, httpPort, tlsPort int, isHostAllowed func(string) bool) (string, bool) { + raw := strings.ToLower(strings.TrimSpace(host)) + _, portStr, splitErr := net.SplitHostPort(raw) + hasPort := splitErr == nil + canonicalHost, ok := canonicalRedirectHost(host) if !ok { return "", false } @@ -135,7 +213,7 @@ func httpsRedirectTarget(host, requestURI string, httpPort, tlsPort int, isHostA } canonicalAuthority := canonicalHost - if portStr != "" { + if hasPort { canonicalAuthority = net.JoinHostPort(canonicalHost, portStr) } return fmt.Sprintf("https://%s%s", httphelper.HTTPSAuthority(canonicalAuthority, httpPort, tlsPort), path), true diff --git a/internal/adapters/in/http/middleware/cidr_test.go b/internal/adapters/in/http/middleware/cidr_test.go index 6a598ffb5..11385675d 100644 --- a/internal/adapters/in/http/middleware/cidr_test.go +++ b/internal/adapters/in/http/middleware/cidr_test.go @@ -117,7 +117,7 @@ func TestHTTPSRedirect_NoPortHost_OmitsTLSPort(t *testing.T) { w.WriteHeader(http.StatusOK) }) - handler := HTTPSRedirect(nil, 8088, 8443, true, log, func(host string) bool { return host == "o2.bnema.dev" })(ok) + handler := HTTPSRedirect(nil, 8088, 8443, true, log, func(host string) bool { return host == "app.example.com" })(ok) srv := httptest.NewServer(handler) defer srv.Close() @@ -125,13 +125,13 @@ func TestHTTPSRedirect_NoPortHost_OmitsTLSPort(t *testing.T) { client.CheckRedirect = func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse } req, _ := http.NewRequest(http.MethodGet, srv.URL+"/", nil) - req.Host = "o2.bnema.dev" + req.Host = "app.example.com" resp, err := client.Do(req) require.NoError(t, err) defer resp.Body.Close() assert.Equal(t, http.StatusPermanentRedirect, resp.StatusCode) - assert.Equal(t, "https://o2.bnema.dev/", resp.Header.Get("Location")) + assert.Equal(t, "https://app.example.com/", resp.Header.Get("Location")) } func TestHTTPSRedirect_HTTPListenerPort_MapsToTLSPort(t *testing.T) { @@ -140,7 +140,7 @@ func TestHTTPSRedirect_HTTPListenerPort_MapsToTLSPort(t *testing.T) { w.WriteHeader(http.StatusOK) }) - handler := HTTPSRedirect(nil, 8088, 8443, true, log, func(host string) bool { return host == "o2.bnema.dev" })(ok) + handler := HTTPSRedirect(nil, 8088, 8443, true, log, func(host string) bool { return host == "app.example.com" })(ok) srv := httptest.NewServer(handler) defer srv.Close() @@ -148,13 +148,13 @@ func TestHTTPSRedirect_HTTPListenerPort_MapsToTLSPort(t *testing.T) { client.CheckRedirect = func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse } req, _ := http.NewRequest(http.MethodGet, srv.URL+"/", nil) - req.Host = "o2.bnema.dev:8088" + req.Host = "app.example.com:8088" resp, err := client.Do(req) require.NoError(t, err) defer resp.Body.Close() assert.Equal(t, http.StatusPermanentRedirect, resp.StatusCode) - assert.Equal(t, "https://o2.bnema.dev:8443/", resp.Header.Get("Location")) + assert.Equal(t, "https://app.example.com:8443/", resp.Header.Get("Location")) } func TestHTTPSRedirect_UnknownExplicitPort_IsPreserved(t *testing.T) { @@ -163,7 +163,7 @@ func TestHTTPSRedirect_UnknownExplicitPort_IsPreserved(t *testing.T) { w.WriteHeader(http.StatusOK) }) - handler := HTTPSRedirect(nil, 8088, 8443, true, log, func(host string) bool { return host == "o2.bnema.dev" })(ok) + handler := HTTPSRedirect(nil, 8088, 8443, true, log, func(host string) bool { return host == "app.example.com" })(ok) srv := httptest.NewServer(handler) defer srv.Close() @@ -171,13 +171,13 @@ func TestHTTPSRedirect_UnknownExplicitPort_IsPreserved(t *testing.T) { client.CheckRedirect = func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse } req, _ := http.NewRequest(http.MethodGet, srv.URL+"/", nil) - req.Host = "o2.bnema.dev:9999" + req.Host = "app.example.com:9999" resp, err := client.Do(req) require.NoError(t, err) defer resp.Body.Close() assert.Equal(t, http.StatusPermanentRedirect, resp.StatusCode) - assert.Equal(t, "https://o2.bnema.dev:9999/", resp.Header.Get("Location")) + assert.Equal(t, "https://app.example.com:9999/", resp.Header.Get("Location")) } func TestHTTPSRedirect_TrailingDotHostRedirectsToCanonicalHost(t *testing.T) { @@ -185,7 +185,7 @@ func TestHTTPSRedirect_TrailingDotHostRedirectsToCanonicalHost(t *testing.T) { ok := http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { w.WriteHeader(http.StatusOK) }) - handler := HTTPSRedirect(nil, 8088, 8443, true, log, func(host string) bool { return host == "o2.bnema.dev" })(ok) + handler := HTTPSRedirect(nil, 8088, 8443, true, log, func(host string) bool { return host == "app.example.com" })(ok) srv := httptest.NewServer(handler) defer srv.Close() @@ -194,13 +194,13 @@ func TestHTTPSRedirect_TrailingDotHostRedirectsToCanonicalHost(t *testing.T) { req, err := http.NewRequest(http.MethodGet, srv.URL+"/path", nil) require.NoError(t, err) - req.Host = "O2.Bnema.Dev.:8088" + req.Host = "App.Example.Com.:8088" resp, err := client.Do(req) require.NoError(t, err) defer resp.Body.Close() assert.Equal(t, http.StatusPermanentRedirect, resp.StatusCode) - assert.Equal(t, "https://o2.bnema.dev:8443/path", resp.Header.Get("Location")) + assert.Equal(t, "https://app.example.com:8443/path", resp.Header.Get("Location")) } func TestHTTPSRedirect_RejectsInvalidHost(t *testing.T) { @@ -208,13 +208,13 @@ func TestHTTPSRedirect_RejectsInvalidHost(t *testing.T) { ok := http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { w.WriteHeader(http.StatusOK) }) - handler := HTTPSRedirect(nil, 8088, 8443, true, log, func(host string) bool { return host == "o2.bnema.dev" })(ok) + handler := HTTPSRedirect(nil, 8088, 8443, true, log, func(host string) bool { return host == "app.example.com" })(ok) tests := []struct { name string host string }{ - {name: "invalid port", host: "o2.bnema.dev:abcd"}, + {name: "invalid port", host: "app.example.com:abcd"}, {name: "localhost", host: "localhost"}, {name: "ipv6", host: "[::1]:8088"}, @@ -238,6 +238,126 @@ func TestHTTPSRedirect_RejectsInvalidHost(t *testing.T) { } } +// TestHTTPSRedirect_TLSAlwaysNeverServesPlaintext proves a tls=always host +// is redirected when an HTTPS endpoint exists and refused when it does not: +// the backend is never reached over plaintext. +func TestHTTPSRedirect_TLSAlwaysNeverServesPlaintext(t *testing.T) { + log := testLogger() + served := false + backend := http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + served = true + w.WriteHeader(http.StatusOK) + }) + isHostAllowed := func(host string) bool { return host == "secure.example.com" } + isTLSAlways := func(host string) bool { return host == "secure.example.com" } + + redirected := HTTPSRedirectWithEligibility(nil, 8088, 8443, false, log, isHostAllowed, nil, isTLSAlways)(backend) + srv := httptest.NewServer(redirected) + defer srv.Close() + client := srv.Client() + client.CheckRedirect = func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse } + + req, err := http.NewRequest(http.MethodGet, srv.URL+"/", nil) + require.NoError(t, err) + req.Host = "secure.example.com" + resp, err := client.Do(req) + require.NoError(t, err) + defer resp.Body.Close() + assert.Equal(t, http.StatusPermanentRedirect, resp.StatusCode) + assert.Equal(t, "https://secure.example.com/", resp.Header.Get("Location")) + assert.False(t, served, "the plaintext backend must never be reached") + + // No HTTPS endpoint configured: the request is refused, not served. + served = false + refused := HTTPSRedirectWithEligibility(nil, 8088, 0, false, log, isHostAllowed, nil, isTLSAlways)(backend) + srv2 := httptest.NewServer(refused) + defer srv2.Close() + req2, err := http.NewRequest(http.MethodGet, srv2.URL+"/", nil) + require.NoError(t, err) + req2.Host = "secure.example.com" + refusedClient := srv2.Client() + refusedClient.CheckRedirect = func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse } + resp2, err := refusedClient.Do(req2) + require.NoError(t, err) + defer resp2.Body.Close() + assert.Equal(t, http.StatusMisdirectedRequest, resp2.StatusCode) + assert.False(t, served, "plaintext must not be served without an HTTPS endpoint") +} + +// TestHTTPSRedirect_TLSAlwaysIgnoresDisabledRedirectConfig proves tls=always +// still redirects when redirects are otherwise disabled for untrusted +// clients. +func TestHTTPSRedirect_TLSAlwaysIgnoresDisabledRedirectConfig(t *testing.T) { + log := testLogger() + served := false + backend := http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + served = true + w.WriteHeader(http.StatusOK) + }) + isHostAllowed := func(host string) bool { return host == "secure.example.com" } + isTLSAlways := func(host string) bool { return host == "secure.example.com" } + + handler := HTTPSRedirectWithEligibility(nil, 8088, 8443, false, log, isHostAllowed, nil, isTLSAlways)(backend) + srv := httptest.NewServer(handler) + defer srv.Close() + client := srv.Client() + client.CheckRedirect = func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse } + + req, err := http.NewRequest(http.MethodGet, srv.URL+"/", nil) + require.NoError(t, err) + req.Host = "secure.example.com" + resp, err := client.Do(req) + require.NoError(t, err) + defer resp.Body.Close() + assert.Equal(t, http.StatusPermanentRedirect, resp.StatusCode) + assert.False(t, served) +} + +func TestHTTPSRedirect_IneligibleKnownHostPassesThrough(t *testing.T) { + log := testLogger() + ok := http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(http.StatusOK) + }) + // plain.example.com is known but opted out of TLS (tls=never). + handler := HTTPSRedirectWithEligibility(nil, 8088, 8443, true, log, + func(host string) bool { return host == "plain.example.com" || host == "app.example.com" }, + func(host string) bool { return host != "plain.example.com" }, nil)(ok) + + srv := httptest.NewServer(handler) + defer srv.Close() + client := srv.Client() + client.CheckRedirect = func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse } + + // Ineligible known host: served plain HTTP, no redirect. + req, err := http.NewRequest(http.MethodGet, srv.URL+"/", nil) + require.NoError(t, err) + req.Host = "plain.example.com" + resp, err := client.Do(req) + require.NoError(t, err) + defer resp.Body.Close() + assert.Equal(t, http.StatusOK, resp.StatusCode) + assert.Empty(t, resp.Header.Get("Location")) + + // Eligible known host: still redirected. + req, err = http.NewRequest(http.MethodGet, srv.URL+"/", nil) + require.NoError(t, err) + req.Host = "app.example.com" + resp2, err := client.Do(req) + require.NoError(t, err) + defer resp2.Body.Close() + assert.Equal(t, http.StatusPermanentRedirect, resp2.StatusCode) + assert.Equal(t, "https://app.example.com/", resp2.Header.Get("Location")) + + // Unknown host: still 400. + req, err = http.NewRequest(http.MethodGet, srv.URL+"/", nil) + require.NoError(t, err) + req.Host = "evil.example.com" + resp3, err := client.Do(req) + require.NoError(t, err) + defer resp3.Body.Close() + assert.Equal(t, http.StatusBadRequest, resp3.StatusCode) +} + func TestProxyCIDRAllowlist(t *testing.T) { log := testLogger() redirect := http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { diff --git a/internal/adapters/in/http/onboarding/handler_test.go b/internal/adapters/in/http/onboarding/handler_test.go index f36b969f1..14993446f 100644 --- a/internal/adapters/in/http/onboarding/handler_test.go +++ b/internal/adapters/in/http/onboarding/handler_test.go @@ -111,24 +111,24 @@ func TestHandler_OnboardingPage(t *testing.T) { func TestHandler_OnboardingPage_DefaultHTTPSURLOmitsInternalTLSPort(t *testing.T) { srv, _, _ := newTestServer(t) - body := getOnboardingBody(t, srv, "o2.bnema.dev") + body := getOnboardingBody(t, srv, "app.example.com") - assert.Contains(t, body, "https://o2.bnema.dev/") + assert.Contains(t, body, "https://app.example.com/") assert.NotContains(t, body, ":8443") } func TestHandler_OnboardingPage_ExplicitHTTPPortMapsToTLSPort(t *testing.T) { srv, _, _ := newTestServer(t) - body := getOnboardingBody(t, srv, "o2.bnema.dev:8088") + body := getOnboardingBody(t, srv, "app.example.com:8088") - assert.Contains(t, body, "https://o2.bnema.dev:8443/") + assert.Contains(t, body, "https://app.example.com:8443/") } func TestHandler_OnboardingPage_ExplicitNonHTTPPortIsPreserved(t *testing.T) { srv, _, _ := newTestServer(t) - body := getOnboardingBody(t, srv, "o2.bnema.dev:9999") + body := getOnboardingBody(t, srv, "app.example.com:9999") - assert.Contains(t, body, "https://o2.bnema.dev:9999/") + assert.Contains(t, body, "https://app.example.com:9999/") } func TestHandler_OnboardingPage_IPHostHidesGoToSiteLink(t *testing.T) { @@ -140,7 +140,7 @@ func TestHandler_OnboardingPage_IPHostHidesGoToSiteLink(t *testing.T) { func TestHandler_OnboardingPage_IncludesFingerprintAndClientImportCopy(t *testing.T) { srv, _, _ := newTestServer(t) - body := getOnboardingBody(t, srv, "o2.bnema.dev") + body := getOnboardingBody(t, srv, "app.example.com") // Fingerprint must be visible assert.Contains(t, body, testFingerprint) @@ -156,9 +156,9 @@ func TestHandler_OnboardingPage_IncludesFingerprintAndClientImportCopy(t *testin func TestHandler_OnboardingPage_GoToSiteLinkIsPlainLink(t *testing.T) { srv, _, _ := newTestServer(t) - body := getOnboardingBody(t, srv, "o2.bnema.dev") + body := getOnboardingBody(t, srv, "app.example.com") - assert.Contains(t, body, `Go to site →`) + assert.Contains(t, body, `Go to site →`) assert.NotContains(t, body, "onclick=") assert.NotContains(t, body, "document.cookie") } diff --git a/internal/adapters/in/http/proxy/handler.go b/internal/adapters/in/http/proxy/handler.go index 68dd9cfdb..1e3b68a7a 100644 --- a/internal/adapters/in/http/proxy/handler.go +++ b/internal/adapters/in/http/proxy/handler.go @@ -25,10 +25,17 @@ type Handler struct { appTransport http.RoundTripper h2cTransport http.RoundTripper registryTransport http.RoundTripper - activeConns atomic.Int64 + // registryForwarding permits registry-domain requests to reach the + // internal registry through this public proxy. It is enabled only + // when registry authentication is on: with auth disabled the + // registry is a local-only service reached by direct loopback + // connection, never through public ingress. + registryForwarding bool + activeConns atomic.Int64 } -// NewHandler creates a new proxy HTTP handler. +// NewHandler creates a new proxy HTTP handler. Registry forwarding is +// denied until explicitly enabled with WithRegistryForwarding. func NewHandler(proxySvc in.ProxyService, trustedNets []*net.IPNet, log zerowrap.Logger) *Handler { return &Handler{ proxySvc: proxySvc, @@ -40,6 +47,14 @@ func NewHandler(proxySvc in.ProxyService, trustedNets []*net.IPNet, log zerowrap } } +// WithRegistryForwarding enables forwarding public registry-domain requests +// to the internal registry. Callers pass the installation auth state: a +// disabled-auth registry must stay unreachable through public ingress. +func (h *Handler) WithRegistryForwarding(enabled bool) *Handler { + h.registryForwarding = enabled + return h +} + // ServeHTTP handles incoming HTTP requests and proxies them to the appropriate backend. func (h *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) { cfg := h.proxySvc.ProxyConfig() @@ -80,6 +95,15 @@ func (h *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) { // Check if this is the registry domain if h.proxySvc.IsRegistryDomain(host) { + if !h.registryForwarding { + // Auth is disabled: the registry is local-only. Forwarding + // here would present the proxy's loopback peer to the + // registry, which trusts loopback, exposing anonymous + // reads and writes on public ingress. + log.Warn().Str(zerowrap.FieldHost, host).Msg("registry forwarding denied: authentication disabled") + proxyError(w, "404 page not found", http.StatusNotFound) + return + } log.Debug().Msg("routing request to registry") h.forwardToRegistry(w, r, cfg.RegistryPort) return diff --git a/internal/adapters/in/http/proxy/handler_test.go b/internal/adapters/in/http/proxy/handler_test.go index d39a56dcb..e954e64c6 100644 --- a/internal/adapters/in/http/proxy/handler_test.go +++ b/internal/adapters/in/http/proxy/handler_test.go @@ -52,7 +52,7 @@ func TestHandler_RoutesToRegistry(t *testing.T) { proxySvc.EXPECT().TrackRegistryRequest().Return() proxySvc.EXPECT().ReleaseRegistryRequest().Return() - handler := NewHandler(proxySvc, nil, testLogger()) + handler := NewHandler(proxySvc, nil, testLogger()).WithRegistryForwarding(true) req := httptest.NewRequest(http.MethodGet, "http://registry.example.com/v2/", nil) req.Host = "registry.example.com" @@ -62,6 +62,25 @@ func TestHandler_RoutesToRegistry(t *testing.T) { assert.True(t, w.Code == http.StatusBadGateway || w.Code == http.StatusServiceUnavailable) } +// TestHandler_DeniesRegistryForwardingWhenAuthDisabled proves a public +// registry-domain request cannot reach the internal registry unless +// forwarding was explicitly enabled (auth on). +func TestHandler_DeniesRegistryForwardingWhenAuthDisabled(t *testing.T) { + proxySvc := inmocks.NewMockProxyService(t) + proxySvc.EXPECT().ProxyConfig().Return(in.ProxyServiceConfig{RegistryPort: 5000}) + proxySvc.EXPECT().IsRegistryDomain("registry.example.com").Return(true) + + handler := NewHandler(proxySvc, nil, testLogger()) + + req := httptest.NewRequest(http.MethodGet, "http://registry.example.com/v2/", nil) + req.Host = "registry.example.com" + w := httptest.NewRecorder() + handler.ServeHTTP(w, req) + + assert.Equal(t, http.StatusNotFound, w.Code) + proxySvc.AssertNotCalled(t, "TrackRegistryRequest") +} + func TestHandler_NormalizesRequestHostForLookup(t *testing.T) { tests := []struct { name string diff --git a/internal/adapters/in/http/registry/handler.go b/internal/adapters/in/http/registry/handler.go index a767c91c3..1416ac0a1 100644 --- a/internal/adapters/in/http/registry/handler.go +++ b/internal/adapters/in/http/registry/handler.go @@ -9,12 +9,12 @@ import ( "io" "net/http" "strconv" - "strings" "sync/atomic" "github.com/bnema/zerowrap" "github.com/bnema/gordon/internal/adapters/dto" + "github.com/bnema/gordon/internal/adapters/in/http/registry/route" "github.com/bnema/gordon/internal/boundaries/in" "github.com/bnema/gordon/internal/domain" "github.com/bnema/gordon/pkg/manifest" @@ -95,6 +95,10 @@ func (h *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) { h.handleRegistryRoutes(w, r) } +// handleRegistryRoutes parses the request path once and dispatches on the +// resulting operation. Authorization middleware parses the same path with +// the same parser, so the repository that is authorized is exactly the +// repository the handler accesses. func (h *Handler) handleRegistryRoutes(w http.ResponseWriter, r *http.Request) { ctx := zerowrap.CtxWithFields(r.Context(), map[string]any{ zerowrap.FieldLayer: "adapter", @@ -105,65 +109,45 @@ func (h *Handler) handleRegistryRoutes(w http.ResponseWriter, r *http.Request) { }) r = r.WithContext(ctx) - path := r.URL.Path - - // Route manifest operations: /v2/{name}/manifests/{reference} - if strings.Contains(path, "/manifests/") { - h.handleManifestRoutes(w, r) + op, err := route.Parse(r.URL.Path) + if err != nil { + var parseErr *route.ParseError + if errors.As(err, &parseErr) { + h.sendRegistryError(w, http.StatusBadRequest, parseErr.Code, parseErr.Error()) + return + } + h.sendRegistryError(w, http.StatusNotFound, "NOT_FOUND", "route not found") return } - // Route blob operations: /v2/{name}/blobs/{digest} - if strings.Contains(path, "/blobs/") && !strings.Contains(path, "/uploads/") { + // Methods are validated per operation kind after the parsed values are + // installed, so handlers only read the path values the parser produced. + switch op.Kind { + case route.KindBase: + h.handleBase(w, r) + case route.KindManifest: + r.SetPathValue("name", op.Repository) + r.SetPathValue("reference", op.Reference) + h.handleManifestRoutes(w, r) + case route.KindBlob: + r.SetPathValue("name", op.Repository) + r.SetPathValue("digest", op.Digest) h.handleBlobRoutes(w, r) - return - } - - // Route blob upload operations: /v2/{name}/blobs/uploads/ - if strings.Contains(path, "/blobs/uploads/") { + case route.KindUpload: + r.SetPathValue("name", op.Repository) + if op.UploadID != "" { + r.SetPathValue("uuid", op.UploadID) + } h.handleBlobUploadRoutes(w, r) - return - } - - // Route tag list operations: /v2/{name}/tags/list - if strings.Contains(path, "/tags/list") { + case route.KindTagList: + r.SetPathValue("name", op.Repository) h.handleTagListRoutes(w, r) - return - } - - // Base endpoint: /v2/ - if path == "/v2/" { - h.handleBase(w, r) - return + default: + h.sendRegistryError(w, http.StatusNotFound, "NOT_FOUND", "route not found") } - h.sendRegistryError(w, http.StatusNotFound, "NOT_FOUND", "route not found") - } func (h *Handler) handleManifestRoutes(w http.ResponseWriter, r *http.Request) { - // Parse path: /v2/{name}/manifests/{reference} - parts := strings.Split(strings.TrimPrefix(r.URL.Path, "/v2/"), "/") - if len(parts) < 3 || parts[len(parts)-2] != "manifests" { - h.sendRegistryError(w, http.StatusNotFound, "NOT_FOUND", "route not found") - return - } - - reference := parts[len(parts)-1] - name := strings.Join(parts[:len(parts)-2], "/") - - // Validate inputs to prevent path traversal - if err := validation.ValidateRepositoryName(name); err != nil { - h.sendRegistryError(w, http.StatusBadRequest, "NAME_INVALID", err.Error()) - return - } - if err := validation.ValidateReference(reference); err != nil { - h.sendRegistryError(w, http.StatusBadRequest, "TAG_INVALID", err.Error()) - return - } - - r.SetPathValue("name", name) - r.SetPathValue("reference", reference) - switch r.Method { case "HEAD", "GET": h.handleGetManifest(w, r) @@ -175,29 +159,6 @@ func (h *Handler) handleManifestRoutes(w http.ResponseWriter, r *http.Request) { } func (h *Handler) handleBlobRoutes(w http.ResponseWriter, r *http.Request) { - // Parse path: /v2/{name}/blobs/{digest} - parts := strings.Split(strings.TrimPrefix(r.URL.Path, "/v2/"), "/") - if len(parts) < 3 || parts[len(parts)-2] != "blobs" { - h.sendRegistryError(w, http.StatusNotFound, "NOT_FOUND", "route not found") - return - } - - digest := parts[len(parts)-1] - name := strings.Join(parts[:len(parts)-2], "/") - - // Validate inputs to prevent path traversal - if err := validation.ValidateRepositoryName(name); err != nil { - h.sendRegistryError(w, http.StatusBadRequest, "NAME_INVALID", err.Error()) - return - } - if err := validation.ValidateDigest(digest); err != nil { - h.sendRegistryError(w, http.StatusBadRequest, "DIGEST_INVALID", err.Error()) - return - } - - r.SetPathValue("name", name) - r.SetPathValue("digest", digest) - switch r.Method { case "HEAD", "GET": h.handleGetBlob(w, r) @@ -207,26 +168,7 @@ func (h *Handler) handleBlobRoutes(w http.ResponseWriter, r *http.Request) { } func (h *Handler) handleBlobUploadRoutes(w http.ResponseWriter, r *http.Request) { - // Parse path: /v2/{name}/blobs/uploads/{uuid?} - path := strings.TrimPrefix(r.URL.Path, "/v2/") - uploadIndex := strings.Index(path, "/blobs/uploads/") - if uploadIndex == -1 { - h.sendRegistryError(w, http.StatusNotFound, "NOT_FOUND", "route not found") - return - } - - name := path[:uploadIndex] - uploadPart := path[uploadIndex+15:] // len("/blobs/uploads/") = 15 - - // Validate repository name to prevent path traversal - if err := validation.ValidateRepositoryName(name); err != nil { - h.sendRegistryError(w, http.StatusBadRequest, "NAME_INVALID", err.Error()) - return - } - - r.SetPathValue("name", name) - - if uploadPart == "" { + if r.PathValue("uuid") == "" { // POST /v2/{name}/blobs/uploads/ switch r.Method { case "POST": @@ -234,41 +176,18 @@ func (h *Handler) handleBlobUploadRoutes(w http.ResponseWriter, r *http.Request) default: h.sendRegistryError(w, http.StatusMethodNotAllowed, "METHOD_NOT_ALLOWED", "method not allowed") } - } else { - // Validate UUID to prevent path traversal - if err := validation.ValidateUUID(uploadPart); err != nil { - h.sendRegistryError(w, http.StatusBadRequest, "BLOB_UPLOAD_INVALID", err.Error()) - return - } - // PATCH/PUT /v2/{name}/blobs/uploads/{uuid} - r.SetPathValue("uuid", uploadPart) - switch r.Method { - case "PATCH", "PUT": - h.handleBlobUpload(w, r) - default: - h.sendRegistryError(w, http.StatusMethodNotAllowed, "METHOD_NOT_ALLOWED", "method not allowed") - } - } -} - -func (h *Handler) handleTagListRoutes(w http.ResponseWriter, r *http.Request) { - // Parse path: /v2/{name}/tags/list - path := strings.TrimPrefix(r.URL.Path, "/v2/") - if !strings.HasSuffix(path, "/tags/list") { - h.sendRegistryError(w, http.StatusNotFound, "NOT_FOUND", "route not found") return } - - name := strings.TrimSuffix(path, "/tags/list") - - // Validate repository name to prevent path traversal - if err := validation.ValidateRepositoryName(name); err != nil { - h.sendRegistryError(w, http.StatusBadRequest, "NAME_INVALID", err.Error()) - return + // PATCH/PUT /v2/{name}/blobs/uploads/{uuid} + switch r.Method { + case "PATCH", "PUT": + h.handleBlobUpload(w, r) + default: + h.sendRegistryError(w, http.StatusMethodNotAllowed, "METHOD_NOT_ALLOWED", "method not allowed") } +} - r.SetPathValue("name", name) - +func (h *Handler) handleTagListRoutes(w http.ResponseWriter, r *http.Request) { switch r.Method { case "GET": h.handleListTags(w, r) @@ -374,6 +293,10 @@ func (h *Handler) handlePutManifest(w http.ResponseWriter, r *http.Request) { h.sendRegistryError(w, http.StatusBadRequest, "DIGEST_INVALID", "manifest digest does not match content") return } + if errors.Is(err, domain.ErrManifestBlobUnknown) { + h.sendRegistryError(w, http.StatusBadRequest, "MANIFEST_BLOB_UNKNOWN", "manifest references unknown content") + return + } h.sendRegistryError(w, http.StatusInternalServerError, "MANIFEST_INVALID", "failed to store manifest") return } @@ -457,10 +380,15 @@ func (h *Handler) handleBlobUpload(w http.ResponseWriter, r *http.Request) { // copy buffer (~32KB) regardless of chunk size. length, err := h.registrySvc.AppendBlobChunk(ctx, name, uuid, r.Body, r.ContentLength, maxBlobSize) if err != nil { + if errors.Is(err, domain.ErrUploadNotFound) { + log.Warn().Str("uuid", uuid).Msg("blob upload not found for repository") + h.sendRegistryError(w, http.StatusNotFound, "BLOB_UPLOAD_UNKNOWN", "blob upload unknown") + return + } var maxBytesErr *http.MaxBytesError if errors.As(err, &maxBytesErr) { log.Warn().Int64("max_size", maxBlobChunkSize).Msg("blob chunk too large") - if !h.cancelUploadAfterError(w, ctx, uuid, "failed to cancel oversized blob upload") { + if !h.cancelUploadAfterError(w, ctx, name, uuid, "failed to cancel oversized blob upload") { return } h.sendRegistryError(w, http.StatusRequestEntityTooLarge, "SIZE_INVALID", "blob chunk exceeds maximum size") @@ -468,7 +396,7 @@ func (h *Handler) handleBlobUpload(w http.ResponseWriter, r *http.Request) { } if errors.Is(err, domain.ErrBlobSizeExceeded) { log.Warn().Int64("max_size", maxBlobSize).Msg("blob upload too large") - if !h.cancelUploadAfterError(w, ctx, uuid, "failed to cancel oversized blob upload") { + if !h.cancelUploadAfterError(w, ctx, name, uuid, "failed to cancel oversized blob upload") { return } h.sendRegistryError(w, http.StatusRequestEntityTooLarge, "SIZE_INVALID", "blob exceeds maximum size") @@ -481,9 +409,13 @@ func (h *Handler) handleBlobUpload(w http.ResponseWriter, r *http.Request) { // If this is the final chunk (PUT request with digest), finalize the upload if r.Method == "PUT" && digest != "" { - if err := h.registrySvc.FinishUpload(ctx, uuid, digest); err != nil { + if err := h.registrySvc.FinishUpload(ctx, name, uuid, digest); err != nil { log.Error().Err(err).Str("digest", digest).Msg("failed to finalize blob upload") - if !h.cancelUploadAfterError(w, ctx, uuid, "failed to cancel invalid blob upload") { + if errors.Is(err, domain.ErrUploadNotFound) { + h.sendRegistryError(w, http.StatusNotFound, "BLOB_UPLOAD_UNKNOWN", "blob upload unknown") + return + } + if !h.cancelUploadAfterError(w, ctx, name, uuid, "failed to cancel invalid blob upload") { return } h.sendRegistryError(w, http.StatusBadRequest, "DIGEST_INVALID", "digest mismatch") @@ -505,8 +437,8 @@ func (h *Handler) handleBlobUpload(w http.ResponseWriter, r *http.Request) { w.WriteHeader(http.StatusAccepted) } -func (h *Handler) cancelUploadAfterError(w http.ResponseWriter, ctx context.Context, uuid, message string) bool { - if err := h.registrySvc.CancelUpload(ctx, uuid); err != nil { +func (h *Handler) cancelUploadAfterError(w http.ResponseWriter, ctx context.Context, name, uuid, message string) bool { + if err := h.registrySvc.CancelUpload(ctx, name, uuid); err != nil { log := zerowrap.FromCtx(ctx) log.Error().Err(err).Str("uuid", uuid).Msg(message) h.sendRegistryError(w, http.StatusInternalServerError, "UNKNOWN", "failed to cancel upload") diff --git a/internal/adapters/in/http/registry/handler_test.go b/internal/adapters/in/http/registry/handler_test.go index 0d8cf1de8..15104b98b 100644 --- a/internal/adapters/in/http/registry/handler_test.go +++ b/internal/adapters/in/http/registry/handler_test.go @@ -134,6 +134,30 @@ func TestHandler_GetManifest_NestedName(t *testing.T) { assert.Equal(t, http.StatusOK, rec.Code) } +// TestHandler_ReservedLookingRepositoryIsServedExactly proves the handler +// dispatches on the same repository the parser resolves from the last +// route marker, so authorization cannot name a different repository than +// the one accessed. +func TestHandler_ReservedLookingRepositoryIsServedExactly(t *testing.T) { + registrySvc := inmocks.NewMockRegistryService(t) + + handler := NewHandler(registrySvc, testLogger(), DefaultMaxBlobChunkSize) + + registrySvc.EXPECT().GetManifest(mock.Anything, "manifests/victim", "latest").Return(&domain.Manifest{ + Name: "manifests/victim", + Reference: "latest", + ContentType: "application/vnd.docker.distribution.manifest.v2+json", + Data: []byte(`{"schemaVersion": 2}`), + }, nil) + + req := httptest.NewRequest("GET", "/v2/manifests/victim/manifests/latest", nil) + rec := httptest.NewRecorder() + + handler.ServeHTTP(rec, req) + + assert.Equal(t, http.StatusOK, rec.Code) +} + func TestHandler_PutManifest_Success(t *testing.T) { registrySvc := inmocks.NewMockRegistryService(t) @@ -316,7 +340,7 @@ func TestHandler_BlobUpload_PUT_Finalize(t *testing.T) { chunkData := []byte("final chunk") registrySvc.EXPECT().AppendBlobChunk(mock.Anything, "myapp", "550e8400-e29b-41d4-a716-446655440000", mock.Anything, mock.Anything, mock.Anything).Return(int64(len(chunkData)), nil) - registrySvc.EXPECT().FinishUpload(mock.Anything, "550e8400-e29b-41d4-a716-446655440000", "sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4").Return(nil) + registrySvc.EXPECT().FinishUpload(mock.Anything, "myapp", "550e8400-e29b-41d4-a716-446655440000", "sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4").Return(nil) req := httptest.NewRequest("PUT", "/v2/myapp/blobs/uploads/550e8400-e29b-41d4-a716-446655440000?digest=sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4", bytes.NewReader(chunkData)) rec := httptest.NewRecorder() @@ -465,7 +489,7 @@ func TestHandler_BlobUpload_PATCH_RespectsConfiguredMaxBlobChunkSize(t *testing. _, _ = io.ReadAll(data) }). Return(int64(0), &http.MaxBytesError{Limit: 5}) - registrySvc.EXPECT().CancelUpload(mock.Anything, "550e8400-e29b-41d4-a716-446655440000").Return(nil) + registrySvc.EXPECT().CancelUpload(mock.Anything, "myapp", "550e8400-e29b-41d4-a716-446655440000").Return(nil) req := httptest.NewRequest("PATCH", "/v2/myapp/blobs/uploads/550e8400-e29b-41d4-a716-446655440000", bytes.NewReader([]byte("chunk content"))) rec := httptest.NewRecorder() @@ -508,7 +532,7 @@ func TestHandler_BlobUpload_PATCH_MaxBlobSizeExceededReturnsSizeInvalidAndCancel int64(6), int64(10), ).Return(int64(0), domain.ErrBlobSizeExceeded) - registrySvc.EXPECT().CancelUpload(mock.Anything, "550e8400-e29b-41d4-a716-446655440000").Return(nil) + registrySvc.EXPECT().CancelUpload(mock.Anything, "myapp", "550e8400-e29b-41d4-a716-446655440000").Return(nil) req := httptest.NewRequest("PATCH", "/v2/myapp/blobs/uploads/550e8400-e29b-41d4-a716-446655440000", bytes.NewReader([]byte("123456"))) rec := httptest.NewRecorder() @@ -531,7 +555,7 @@ func TestHandler_BlobUpload_PATCH_ContentLengthTooLargeReturnsSizeInvalidAndCanc int64(11), int64(10), ).Return(int64(0), domain.ErrBlobSizeExceeded) - registrySvc.EXPECT().CancelUpload(mock.Anything, "550e8400-e29b-41d4-a716-446655440000").Return(nil) + registrySvc.EXPECT().CancelUpload(mock.Anything, "myapp", "550e8400-e29b-41d4-a716-446655440000").Return(nil) req := httptest.NewRequest("PATCH", "/v2/myapp/blobs/uploads/550e8400-e29b-41d4-a716-446655440000", bytes.NewReader([]byte("12345678901"))) rec := httptest.NewRecorder() @@ -555,7 +579,7 @@ func TestHandler_BlobUpload_PUT_FinalizeUnderMaxBlobSize(t *testing.T) { int64(len(chunkData)), int64(10), ).Return(int64(len(chunkData)), nil) - registrySvc.EXPECT().FinishUpload(mock.Anything, "550e8400-e29b-41d4-a716-446655440000", "sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4").Return(nil) + registrySvc.EXPECT().FinishUpload(mock.Anything, "myapp", "550e8400-e29b-41d4-a716-446655440000", "sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4").Return(nil) req := httptest.NewRequest("PUT", "/v2/myapp/blobs/uploads/550e8400-e29b-41d4-a716-446655440000?digest=sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4", bytes.NewReader(chunkData)) rec := httptest.NewRecorder() diff --git a/internal/adapters/in/http/registry/route/route.go b/internal/adapters/in/http/registry/route/route.go new file mode 100644 index 000000000..1bd11b4c4 --- /dev/null +++ b/internal/adapters/in/http/registry/route/route.go @@ -0,0 +1,227 @@ +// Package route parses Docker Registry V2 request paths into one validated +// operation shared by authorization and dispatch. Authorization and the +// handler must never derive the repository independently: a mismatch lets a +// request be authorized for one repository and served from another. +package route + +import ( + "errors" + "fmt" + "strings" + + "github.com/bnema/gordon/pkg/validation" +) + +// Parsing errors. ErrNotFound means the path is not a registry route; +// ErrInvalid means the route shape is recognized but a component is +// malformed (bad repository, reference, digest, or upload id). +var ( + ErrNotFound = errors.New("registry route not found") + ErrInvalid = errors.New("registry route invalid") +) + +// Registry error codes for malformed components, matching the Docker +// Registry V2 error format the handler returns. +const ( + CodeNameInvalid = "NAME_INVALID" + CodeTagInvalid = "TAG_INVALID" + CodeDigestInvalid = "DIGEST_INVALID" + CodeBlobUploadInvalid = "BLOB_UPLOAD_INVALID" +) + +// ParseError is a malformed-component error carrying the registry error +// code the handler must return. It wraps ErrInvalid for errors.Is. +type ParseError struct { + Code string + Err error +} + +// Error implements error. +func (e *ParseError) Error() string { return e.Err.Error() } + +// Unwrap exposes the wrapped sentinel. +func (e *ParseError) Unwrap() error { return ErrInvalid } + +// invalid builds a ParseError for a malformed component. +func invalid(code, format string, args ...any) error { + return &ParseError{Code: code, Err: fmt.Errorf(format, args...)} +} + +// Kind classifies a registry operation. +type Kind string + +// Registry operation kinds. +const ( + KindBase Kind = "base" + KindCatalog Kind = "catalog" + KindManifest Kind = "manifest" + KindBlob Kind = "blob" + KindUpload Kind = "upload" + KindTagList Kind = "tags" +) + +const ( + manifestsSegment = "/manifests/" + blobsSegment = "/blobs/" + uploadsSegment = "/blobs/uploads/" + tagsSegment = "/tags/list" +) + +// Operation is one parsed registry request. Repository is the exact +// repository the request addresses (empty for protocol roots), so callers +// authorize and dispatch on the same value. +type Operation struct { + Kind Kind + Repository string + // Reference is the manifest tag or digest (manifest kind). + Reference string + // Digest is the blob digest (blob kind). + Digest string + // UploadID is the upload session id; empty marks a new upload (upload kind). + UploadID string +} + +// RequiresRepositoryAuth reports whether the operation must be authorized +// against a repository scope. Only protocol roots are exempt; every route +// that names a repository is scoped. +func (o Operation) RequiresRepositoryAuth() bool { + return o.Kind != KindBase && o.Kind != KindCatalog +} + +// Parse classifies a registry request path. It resolves the repository +// from the LAST route marker, matching the served route exactly, and +// validates every component with the shared validation grammar. +func Parse(path string) (Operation, error) { + if path == "/v2" || path == "/v2/" { + return Operation{Kind: KindBase}, nil + } + if !strings.HasPrefix(path, "/v2/") { + return Operation{}, ErrNotFound + } + rest := strings.TrimPrefix(path, "/v2/") + switch rest { + case "": + return Operation{Kind: KindBase}, nil + case "_catalog": + return Operation{Kind: KindCatalog}, nil + } + + parsers := map[Kind]func(string) (Operation, bool, error){ + KindTagList: parseTags, + KindUpload: parseUpload, + KindManifest: parseManifest, + KindBlob: parseBlob, + } + markers := []struct { + segment string + kind Kind + }{ + {tagsSegment, KindTagList}, + {uploadsSegment, KindUpload}, + {manifestsSegment, KindManifest}, + {blobsSegment, KindBlob}, + } + // The served route is the marker that appears LAST in the path, so a + // reserved-looking segment earlier in the path can never shorten or + // redirect the repository. + best := -1 + bestKind := Kind("") + for _, marker := range markers { + if idx := strings.LastIndex(rest, marker.segment); idx > best { + best = idx + bestKind = marker.kind + } + } + if parse, ok := parsers[bestKind]; ok { + if op, matched, err := parse(rest); matched { + return op, err + } + } + return Operation{}, ErrNotFound +} + +// parseTags matches /{repo}/tags/list. +func parseTags(rest string) (Operation, bool, error) { + repo, ok := strings.CutSuffix(rest, tagsSegment) + if !ok { + return Operation{}, false, nil + } + op, err := repositoryOperation(KindTagList, repo) + return op, true, err +} + +// parseUpload matches /{repo}/blobs/uploads/{uuid?}. +func parseUpload(rest string) (Operation, bool, error) { + idx := strings.LastIndex(rest, uploadsSegment) + if idx < 0 { + return Operation{}, false, nil + } + op, err := repositoryOperation(KindUpload, rest[:idx]) + if err != nil { + return Operation{}, true, err + } + uploadID := rest[idx+len(uploadsSegment):] + if uploadID == "" { + return op, true, nil + } + if err := validation.ValidateUUID(uploadID); err != nil { + return Operation{}, true, invalid(CodeBlobUploadInvalid, "%s", err) + } + op.UploadID = uploadID + return op, true, nil +} + +// parseManifest matches /{repo}/manifests/{reference}. +func parseManifest(rest string) (Operation, bool, error) { + idx := strings.LastIndex(rest, manifestsSegment) + if idx < 0 { + return Operation{}, false, nil + } + reference := rest[idx+len(manifestsSegment):] + if reference == "" || strings.Contains(reference, "/") { + return Operation{}, true, ErrNotFound + } + if err := validation.ValidateReference(reference); err != nil { + return Operation{}, true, invalid(CodeTagInvalid, "%s", err) + } + op, err := repositoryOperation(KindManifest, rest[:idx]) + if err != nil { + return Operation{}, true, err + } + op.Reference = reference + return op, true, nil +} + +// parseBlob matches /{repo}/blobs/{digest}. +func parseBlob(rest string) (Operation, bool, error) { + idx := strings.LastIndex(rest, blobsSegment) + if idx < 0 { + return Operation{}, false, nil + } + digest := rest[idx+len(blobsSegment):] + if digest == "" || strings.Contains(digest, "/") { + return Operation{}, true, ErrNotFound + } + if err := validation.ValidateDigest(digest); err != nil { + return Operation{}, true, invalid(CodeDigestInvalid, "%s", err) + } + op, err := repositoryOperation(KindBlob, rest[:idx]) + if err != nil { + return Operation{}, true, err + } + op.Digest = digest + return op, true, nil +} + +// repositoryOperation builds a repository-scoped operation, rejecting an +// empty or malformed repository so a reserved-looking segment can never be +// mistaken for a protocol root. +func repositoryOperation(kind Kind, repo string) (Operation, error) { + if repo == "" || strings.HasSuffix(repo, "/") { + return Operation{}, ErrNotFound + } + if err := validation.ValidateRepositoryName(repo); err != nil { + return Operation{}, invalid(CodeNameInvalid, "%s", err) + } + return Operation{Kind: kind, Repository: repo}, nil +} diff --git a/internal/adapters/in/http/registry/route/route_test.go b/internal/adapters/in/http/registry/route/route_test.go new file mode 100644 index 000000000..d09584510 --- /dev/null +++ b/internal/adapters/in/http/registry/route/route_test.go @@ -0,0 +1,125 @@ +package route_test + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/in/http/registry/route" +) + +// validDigest is a syntactically valid sha256 digest. +const validDigest = "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + +// validUUID is a syntactically valid v4 UUID. +const validUUID = "12345678-1234-4123-8123-123456789012" + +func TestParse_Routes(t *testing.T) { + tests := []struct { + path string + want route.Operation + }{ + {"/v2/", route.Operation{Kind: route.KindBase}}, + {"/v2", route.Operation{Kind: route.KindBase}}, + {"/v2/_catalog", route.Operation{Kind: route.KindCatalog}}, + { + "/v2/myrepo/manifests/latest", + route.Operation{Kind: route.KindManifest, Repository: "myrepo", Reference: "latest"}, + }, + { + "/v2/myorg/myapp/manifests/v1.0", + route.Operation{Kind: route.KindManifest, Repository: "myorg/myapp", Reference: "v1.0"}, + }, + { + "/v2/myrepo/blobs/" + validDigest, + route.Operation{Kind: route.KindBlob, Repository: "myrepo", Digest: validDigest}, + }, + {"/v2/myrepo/blobs/uploads/", route.Operation{Kind: route.KindUpload, Repository: "myrepo"}}, + { + "/v2/myrepo/blobs/uploads/" + validUUID, + route.Operation{Kind: route.KindUpload, Repository: "myrepo", UploadID: validUUID}, + }, + {"/v2/myrepo/tags/list", route.Operation{Kind: route.KindTagList, Repository: "myrepo"}}, + } + for _, tt := range tests { + t.Run(tt.path, func(t *testing.T) { + op, err := route.Parse(tt.path) + require.NoError(t, err) + assert.Equal(t, tt.want, op) + }) + } +} + +// TestParse_ResolvesRepositoryFromLastMarker proves a reserved-looking +// component earlier in the path cannot shorten the repository: the parser +// (and therefore authorization) always sees the repository the handler +// would serve. +func TestParse_ResolvesRepositoryFromLastMarker(t *testing.T) { + tests := []struct { + path string + wantRepo string + }{ + {"/v2/manifests/victim/manifests/latest", "manifests/victim"}, + {"/v2/allowed/tags/victim/manifests/latest", "allowed/tags/victim"}, + {"/v2/blobs/impostor/blobs/" + validDigest, "blobs/impostor"}, + } + for _, tt := range tests { + t.Run(tt.path, func(t *testing.T) { + op, err := route.Parse(tt.path) + require.NoError(t, err) + assert.Equal(t, tt.wantRepo, op.Repository) + }) + } +} + +func TestParse_Errors(t *testing.T) { + notFound := []string{ + "/healthz", + "/v2/myrepo", + "/v2/myrepo/manifests/", + "/v2/myrepo/manifests/a/b", + "/v2/myrepo/blobs/", + "/v2/myrepo/tags/list/", + "/v2//manifests/latest", + } + for _, path := range notFound { + t.Run("not found "+path, func(t *testing.T) { + _, err := route.Parse(path) + assert.ErrorIs(t, err, route.ErrNotFound) + }) + } + + cases := []struct { + path string + wantCode string + }{ + {"/v2/BadRepo/manifests/latest", route.CodeNameInvalid}, + {"/v2/myrepo/manifests/bad tag", route.CodeTagInvalid}, + {"/v2/myrepo/blobs/notadigest", route.CodeDigestInvalid}, + {"/v2/myrepo/blobs/uploads/not-a-uuid", route.CodeBlobUploadInvalid}, + } + for _, tc := range cases { + t.Run("invalid "+tc.path, func(t *testing.T) { + _, err := route.Parse(tc.path) + assert.ErrorIs(t, err, route.ErrInvalid) + var parseErr *route.ParseError + require.ErrorAs(t, err, &parseErr) + assert.Equal(t, tc.wantCode, parseErr.Code) + }) + } +} + +func TestRequiresRepositoryAuth_ProtocolRootsOnly(t *testing.T) { + base, err := route.Parse("/v2/") + require.NoError(t, err) + assert.False(t, base.RequiresRepositoryAuth()) + + catalog, err := route.Parse("/v2/_catalog") + require.NoError(t, err) + assert.False(t, catalog.RequiresRepositoryAuth()) + + manifest, err := route.Parse("/v2/myrepo/manifests/latest") + require.NoError(t, err) + assert.True(t, manifest.RequiresRepositoryAuth()) +} diff --git a/internal/adapters/in/traffic/l4_roundtrip_test.go b/internal/adapters/in/traffic/l4_roundtrip_test.go new file mode 100644 index 000000000..8a68e1daf --- /dev/null +++ b/internal/adapters/in/traffic/l4_roundtrip_test.go @@ -0,0 +1,63 @@ +package traffic + +import ( + "context" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +// TestAppL4TCPUdpRoundTrip proves one applied graph relays TCP and UDP +// simultaneously through distinct entrypoints to distinct loopback +// backends: the app-plane shape (service:--:-) +// round-trips end to end on both protocols. +func TestAppL4TCPUdpRoundTrip(t *testing.T) { + tcpBackend := startTCPEchoServer(t, 0) + udpBackend := startUDPEchoServer(t) + + tcpAddr, err := backendFromAddress("game--server", tcpBackend.address) + require.NoError(t, err) + udpAddr, err := backendFromAddress("game--server", udpBackend.address) + require.NoError(t, err) + udpAddr.Protocol = domain.NetworkProtocolUDP + + graph := domain.TrafficGraph{ + Options: domain.TrafficOptions{ + TCP: domain.TCPOptions{DialTimeout: time.Second, IdleTimeout: time.Minute, DrainTimeout: 50 * time.Millisecond}, + UDP: domain.UDPOptions{IdleTimeout: time.Minute, DrainTimeout: 50 * time.Millisecond}, + }, + EntryPoints: []domain.EntryPoint{ + {Name: "tcp", Address: freeTCPAddress(t), Protocol: domain.EntryPointProtocolTCP}, + {Name: "udp", Address: freeUDPAddress(t), Protocol: domain.EntryPointProtocolUDP}, + }, + Routers: []domain.TrafficRouter{ + {Name: "app-game--server--tcp-9000", EntryPoint: "tcp", Protocol: domain.RouterProtocolTCP, Service: "service:game--server:tcp-9000"}, + {Name: "app-game--server--udp-9000", EntryPoint: "udp", Protocol: domain.RouterProtocolUDP, Service: "service:game--server:udp-9000"}, + }, + Services: []domain.TrafficService{ + {Name: "service:game--server:tcp-9000", Backends: []domain.TrafficBackend{tcpAddr}}, + {Name: "service:game--server:udp-9000", Backends: []domain.TrafficBackend{udpAddr}}, + }, + } + require.NoError(t, graph.Validate()) + + manager := NewManager() + require.NoError(t, manager.Apply(context.Background(), &graph)) + defer shutdownManager(t, manager) + + tcpConn := dialTCP(t, graph.EntryPoints[0].Address) + defer tcpConn.Close() + assertRoundTrip(t, tcpConn, "hello tcp") + + udpConn := dialUDP(t, graph.EntryPoints[1].Address) + defer udpConn.Close() + assertUDPRoundTrip(t, udpConn, "hello udp") + + status := manager.Status() + assert.Equal(t, "ok", status.LastReloadStatus) + assert.Len(t, status.Routers, 2) +} diff --git a/internal/adapters/in/traffic/manager.go b/internal/adapters/in/traffic/manager.go index 194fba123..95915a50f 100644 --- a/internal/adapters/in/traffic/manager.go +++ b/internal/adapters/in/traffic/manager.go @@ -3,6 +3,7 @@ package traffic import ( "context" + "errors" "fmt" "maps" "net" @@ -10,6 +11,7 @@ import ( "strings" "sync" "sync/atomic" + "syscall" "time" "github.com/bnema/zerowrap" @@ -337,6 +339,9 @@ func conflictingTCPRuntime(current map[string]*entryPointRuntime, entryPoint dom func (m *Manager) bindTCPEntryPoint(ctx context.Context, entryPoint domain.EntryPoint) (*entryPointRuntime, error) { listener, err := (&net.ListenConfig{}).Listen(ctx, "tcp", entryPoint.Address) if err != nil { + if errors.Is(err, syscall.EADDRINUSE) { + return nil, fmt.Errorf("bind tcp entrypoint %q on %s: inspect host-level TCP listeners, including other users or Gordon instances: %w", entryPoint.Name, entryPoint.Address, err) + } return nil, fmt.Errorf("bind tcp entrypoint %q on %s: %w", entryPoint.Name, entryPoint.Address, err) } trusted, err := parseTrustedCIDRs(entryPoint.TrustedCIDRs) @@ -405,6 +410,9 @@ func conflictingUDPRuntime(current map[string]*udpEntryPointRuntime, entryPoint func (m *Manager) bindUDPEntryPoint(ctx context.Context, entryPoint domain.EntryPoint) (*udpEntryPointRuntime, error) { packetConn, err := (&net.ListenConfig{}).ListenPacket(ctx, "udp", entryPoint.Address) if err != nil { + if errors.Is(err, syscall.EADDRINUSE) { + return nil, fmt.Errorf("bind udp entrypoint %q on %s: inspect host-level UDP listeners, including other users or Gordon instances: %w", entryPoint.Name, entryPoint.Address, err) + } return nil, fmt.Errorf("bind udp entrypoint %q on %s: %w", entryPoint.Name, entryPoint.Address, err) } trusted, err := parseTrustedCIDRs(entryPoint.TrustedCIDRs) diff --git a/internal/adapters/in/traffic/udp.go b/internal/adapters/in/traffic/udp.go index 1b95ab44d..9af810105 100644 --- a/internal/adapters/in/traffic/udp.go +++ b/internal/adapters/in/traffic/udp.go @@ -4,6 +4,7 @@ import ( "context" "errors" "net" + "net/netip" "strconv" "sync" "sync/atomic" @@ -28,28 +29,42 @@ type udpEntryPointRuntime struct { started atomic.Bool closed atomic.Bool - ctx context.Context - cancel context.CancelFunc - done chan struct{} - doneOnce sync.Once + ctx context.Context + cancel context.CancelFunc + // runWG tracks every goroutine owned by this runtime (read loop, datagram + // workers, expiry loop and per-session backend loops). Counts are only added + // while holding mu and before the owning goroutine starts; stop() flips + // accepting under the same lock before it ever waits, so no Add can race a + // Wait. + runWG sync.WaitGroup datagrams chan udpDatagram - mu sync.Mutex - sessions map[string]*udpSession + // dialContext dials the upstream backend. It is a private instance field so + // tests can drive in-flight dial interleavings deterministically; production + // code always uses the standard UDP dialer set at construction and never + // mutates it after the runtime starts. + dialContext func(ctx context.Context, network, address string) (net.Conn, error) + + mu sync.Mutex + accepting bool + sessions map[netip.AddrPort]*udpSession } type udpDatagram struct { - clientAddr net.Addr + clientAddr netip.AddrPort packet []byte } type udpSession struct { - clientAddr net.Addr + clientAddr netip.AddrPort + // remoteAddr is the immutable packet destination precomputed once at + // publication, so backendLoop never has to allocate or format an address + // per packet. It is shared read-only across the session's goroutines. + remoteAddr *net.UDPAddr backend net.Conn backendRef domain.TrafficBackend lastSeen atomic.Int64 - done chan struct{} once sync.Once } @@ -62,9 +77,12 @@ func newUDPEntryPointRuntime(parentCtx context.Context, manager *Manager, entryP trusted: trusted, ctx: ctx, cancel: cancel, - done: make(chan struct{}), + accepting: true, datagrams: make(chan udpDatagram, udpDatagramBacklog), - sessions: map[string]*udpSession{}, + sessions: map[netip.AddrPort]*udpSession{}, + dialContext: func(ctx context.Context, network, address string) (net.Conn, error) { + return (&net.Dialer{}).DialContext(ctx, network, address) + }, } } @@ -72,6 +90,13 @@ func (r *udpEntryPointRuntime) start() { if !r.started.CompareAndSwap(false, true) { return } + r.mu.Lock() + if !r.accepting { + r.mu.Unlock() + return + } + r.runWG.Add(udpWorkerCount + 2) // read loop, expiry loop and datagram workers + r.mu.Unlock() entryPoint := r.entryPointSnapshot() trafficInfo(r.ctx).Str("entrypoint", entryPoint.Name).Str("address", entryPoint.Address).Str("protocol", string(entryPoint.Protocol)).Msg("started udp traffic entrypoint") for range udpWorkerCount { @@ -82,7 +107,7 @@ func (r *udpEntryPointRuntime) start() { } func (r *udpEntryPointRuntime) readLoop() { - defer r.closeDone() + defer r.runWG.Done() buf := make([]byte, udpBufferSize) for { n, clientAddr, err := r.packetConn.ReadFrom(buf) @@ -93,7 +118,12 @@ func (r *udpEntryPointRuntime) readLoop() { r.counters.totalErrors.Add(1) continue } - if !trustedRemoteAddr(r.trustedSnapshot(), clientAddr) { + key, ok := udpAddrPort(clientAddr) + if !ok { + r.counters.totalRefused.Add(1) + continue + } + if !trustedAddrPort(r.trustedSnapshot(), key) { r.counters.totalRefused.Add(1) continue } @@ -108,7 +138,7 @@ func (r *udpEntryPointRuntime) readLoop() { } packet := append([]byte(nil), buf[:n]...) select { - case r.datagrams <- udpDatagram{clientAddr: clientAddr, packet: packet}: + case r.datagrams <- udpDatagram{clientAddr: key, packet: packet}: case <-r.ctx.Done(): return default: @@ -118,6 +148,7 @@ func (r *udpEntryPointRuntime) readLoop() { } func (r *udpEntryPointRuntime) datagramWorker() { + defer r.runWG.Done() for { select { case <-r.ctx.Done(): @@ -133,9 +164,9 @@ func (r *udpEntryPointRuntime) datagramWorker() { } } -func (r *udpEntryPointRuntime) handleDatagram(clientAddr net.Addr, packet []byte) { +func (r *udpEntryPointRuntime) handleDatagram(clientAddr netip.AddrPort, packet []byte) { options := effectiveUDPOptions(snapshotUDPOptions(r.manager.snapshot.Load())) - session, ok := r.session(clientAddr.String(), clientAddr, options) + session, ok := r.session(clientAddr, options) if !ok { return } @@ -143,16 +174,20 @@ func (r *udpEntryPointRuntime) handleDatagram(clientAddr net.Addr, packet []byte r.counters.bytesIn.Add(int64(n)) if err != nil { r.counters.totalErrors.Add(1) - r.removeSession(clientAddr.String()) + r.removeSession(session) } } -func (r *udpEntryPointRuntime) session(key string, clientAddr net.Addr, options domain.UDPOptions) (*udpSession, bool) { +func (r *udpEntryPointRuntime) session(clientAddr netip.AddrPort, options domain.UDPOptions) (*udpSession, bool) { r.mu.Lock() - if session := r.sessions[key]; session != nil { - session.touch() + if existing := r.sessions[clientAddr]; existing != nil { + existing.touch() + r.mu.Unlock() + return existing, true + } + if !r.accepting { r.mu.Unlock() - return session, true + return nil, false } r.mu.Unlock() @@ -170,33 +205,60 @@ func (r *udpEntryPointRuntime) session(key string, clientAddr net.Addr, options } r.mu.Unlock() dialCtx, cancel := context.WithTimeout(r.ctx, udpDialTimeout(options)) - backendConn, err := (&net.Dialer{}).DialContext(dialCtx, "udp", net.JoinHostPort(backend.Host, strconv.Itoa(backend.Port))) + backendConn, err := r.dialContext(dialCtx, "udp", net.JoinHostPort(backend.Host, strconv.Itoa(backend.Port))) cancel() if err != nil { r.counters.totalErrors.Add(1) return nil, false } - session := &udpSession{clientAddr: clientAddr, backend: backendConn, backendRef: backend, done: make(chan struct{})} + session := &udpSession{ + clientAddr: clientAddr, + remoteAddr: net.UDPAddrFromAddrPort(clientAddr), + backend: backendConn, + backendRef: backend, + } session.touch() + return r.publishSession(session, options) +} +// publishSession runs the locked post-dial admission, re-dedup and publication +// step. It must only be called after the backend dial completes. It returns the +// session callers should use: +// - the newly dialed session once it is published, +// - an existing session that won a concurrent creation race, with the newly +// dialed backend closed and the existing session touched, +// - nil when admission has closed or the session limit was reached, with the +// newly dialed backend closed. +// +// The runWG count for a published session is added while holding mu and before +// its goroutine starts, so it can never race stop()'s Wait (see the runWG +// field comment). The backendLoop goroutine starts after the lock is released. +func (r *udpEntryPointRuntime) publishSession(session *udpSession, options domain.UDPOptions) (*udpSession, bool) { r.mu.Lock() - if existing := r.sessions[key]; existing != nil { + if !r.accepting { + r.mu.Unlock() + _ = session.backend.Close() + return nil, false + } + if existing := r.sessions[session.clientAddr]; existing != nil { r.mu.Unlock() - _ = backendConn.Close() + _ = session.backend.Close() + existing.touch() return existing, true } if options.MaxSessions > 0 && len(r.sessions) >= options.MaxSessions { r.mu.Unlock() - _ = backendConn.Close() + _ = session.backend.Close() r.counters.totalRefused.Add(1) return nil, false } - r.sessions[key] = session - r.mu.Unlock() - + r.sessions[session.clientAddr] = session + r.runWG.Add(1) r.counters.activeUDPSessions.Add(1) r.counters.totalAccepted.Add(1) - go r.backendLoop(key, session) + r.mu.Unlock() + + go r.backendLoop(session) return session, true } @@ -211,8 +273,9 @@ func udpDialTimeout(options domain.UDPOptions) time.Duration { return 5 * time.Second } -func (r *udpEntryPointRuntime) backendLoop(key string, session *udpSession) { - defer r.removeSession(key) +func (r *udpEntryPointRuntime) backendLoop(session *udpSession) { + defer r.runWG.Done() + defer r.removeSession(session) buf := make([]byte, udpBufferSize) for { n, err := session.backend.Read(buf) @@ -222,7 +285,7 @@ func (r *udpEntryPointRuntime) backendLoop(key string, session *udpSession) { } return } - written, err := r.packetConn.WriteTo(buf[:n], session.clientAddr) + written, err := r.packetConn.WriteTo(buf[:n], session.remoteAddr) r.counters.bytesOut.Add(int64(written)) if err != nil { if !r.isClosed() { @@ -234,6 +297,7 @@ func (r *udpEntryPointRuntime) backendLoop(key string, session *udpSession) { } func (r *udpEntryPointRuntime) expireLoop() { + defer r.runWG.Done() ticker := time.NewTicker(25 * time.Millisecond) defer ticker.Stop() for { @@ -266,14 +330,22 @@ func (r *udpEntryPointRuntime) expireIdleSessions(idleTimeout time.Duration) { } } -func (r *udpEntryPointRuntime) removeSession(key string) { +// removeSession drops session from the session map and closes its backend only +// when it is still the session currently registered for its client address. A +// stale remover (expiry, backend read/write failure, reload or CIDR cleanup) can +// therefore never evict or close a replacement published under the same key. +// The backend socket is closed outside the lock. +func (r *udpEntryPointRuntime) removeSession(session *udpSession) { + if session == nil { + return + } r.mu.Lock() - session := r.sessions[key] - if session != nil { - delete(r.sessions, key) + current := r.sessions[session.clientAddr] + if current == session { + delete(r.sessions, session.clientAddr) } r.mu.Unlock() - if session == nil { + if current != session { return } session.close() @@ -310,29 +382,26 @@ func (r *udpEntryPointRuntime) stop(ctx context.Context, drainTimeout time.Durat if r.closed.CompareAndSwap(false, true) { entryPoint := r.entryPointSnapshot() trafficInfo(ctx).Str("entrypoint", entryPoint.Name).Str("address", entryPoint.Address).Msg("stopping udp traffic entrypoint") + // Closing admission under mu serializes with start() and session(), + // which are the only places that add to runWG. Once this critical + // section completes no further Add can occur, so waitRuntime may Wait + // safely. + r.mu.Lock() + r.accepting = false + r.mu.Unlock() r.cancel() _ = r.packetConn.Close() - if !r.started.Load() { - r.closeDone() - } } - select { - case <-r.done: - case <-ctx.Done(): - r.closeSessions() - return - } - if r.waitSessions(ctx, drainTimeout) { + if r.waitRuntime(ctx, drainTimeout) { trafficInfo(ctx).Str("entrypoint", r.entryPointSnapshot().Name).Msg("stopped udp traffic entrypoint") return } trafficDebug(ctx).Str("entrypoint", r.entryPointSnapshot().Name).Dur("drain_timeout", drainTimeout).Msg("forcing udp traffic entrypoint drain") r.closeSessions() - select { - case <-r.sessionsDone(): - trafficInfo(ctx).Str("entrypoint", r.entryPointSnapshot().Name).Msg("stopped udp traffic entrypoint") - case <-ctx.Done(): + if !r.waitRuntime(ctx, drainTimeout) { + r.runWG.Wait() } + trafficInfo(ctx).Str("entrypoint", r.entryPointSnapshot().Name).Msg("stopped udp traffic entrypoint") } func (r *udpEntryPointRuntime) drainSessionsAfter(drainTimeout time.Duration) { @@ -358,56 +427,40 @@ func (r *udpEntryPointRuntime) drainSessionsMatchingAfter(match func(*udpSession }() } -func (r *udpEntryPointRuntime) waitSessions(ctx context.Context, drainTimeout time.Duration) bool { +func (r *udpEntryPointRuntime) waitRuntime(ctx context.Context, drainTimeout time.Duration) bool { if drainTimeout <= 0 { drainTimeout = defaultUDPOptions().DrainTimeout } drainCtx, cancel := context.WithTimeout(ctx, drainTimeout) defer cancel() + done := make(chan struct{}) + go func() { + r.runWG.Wait() + close(done) + }() select { - case <-r.sessionsDone(): + case <-done: return true case <-drainCtx.Done(): return false } } -func (r *udpEntryPointRuntime) closeDone() { - r.doneOnce.Do(func() { close(r.done) }) -} - -func (r *udpEntryPointRuntime) sessionsDone() <-chan struct{} { - done := make(chan struct{}) - go func() { - for { - r.mu.Lock() - count := len(r.sessions) - r.mu.Unlock() - if count == 0 { - close(done) - return - } - time.Sleep(10 * time.Millisecond) - } - }() - return done -} - func (r *udpEntryPointRuntime) closeSessions() { r.closeSessionsMatching(func(*udpSession) bool { return true }) } func (r *udpEntryPointRuntime) closeSessionsMatching(match func(*udpSession) bool) { r.mu.Lock() - keys := make([]string, 0, len(r.sessions)) - for key, session := range r.sessions { + sessions := make([]*udpSession, 0, len(r.sessions)) + for _, session := range r.sessions { if match(session) { - keys = append(keys, key) + sessions = append(sessions, session) } } r.mu.Unlock() - for _, key := range keys { - r.removeSession(key) + for _, session := range sessions { + r.removeSession(session) } } @@ -424,15 +477,15 @@ func (r *udpEntryPointRuntime) updateEntryPoint(entryPoint domain.EntryPoint, tr r.mu.Lock() r.entryPoint = entryPoint r.trusted = trusted - stale := make([]string, 0) - for key, session := range r.sessions { - if !trustedRemoteAddr(trusted, session.clientAddr) { - stale = append(stale, key) + stale := make([]*udpSession, 0) + for _, session := range r.sessions { + if !trustedAddrPort(trusted, session.clientAddr) { + stale = append(stale, session) } } r.mu.Unlock() - for _, key := range stale { - r.removeSession(key) + for _, session := range stale { + r.removeSession(session) } } @@ -450,11 +503,59 @@ func (r *udpEntryPointRuntime) trustedSnapshot() []*net.IPNet { func (r *udpEntryPointRuntime) isClosed() bool { return r.closed.Load() } +// udpAddrPort converts a packet source address into a typed, comparable session +// key. IPv4-mapped addresses are normalized to their IPv4 form so that both +// representations share a single session; IPv6 zones are preserved. Invalid +// addresses are rejected before any trusted-list handling so they can never be +// admitted as a zero-valued session key. +func udpAddrPort(addr net.Addr) (netip.AddrPort, bool) { + if udpAddr, ok := addr.(*net.UDPAddr); ok { + key := normalizeUDPAddrPort(udpAddr.AddrPort()) + return key, key.IsValid() + } + parsed, err := netip.ParseAddrPort(addr.String()) + if err != nil { + return netip.AddrPort{}, false + } + key := normalizeUDPAddrPort(parsed) + return key, key.IsValid() +} + +func normalizeUDPAddrPort(addrPort netip.AddrPort) netip.AddrPort { + if !addrPort.IsValid() { + return addrPort + } + return netip.AddrPortFrom(addrPort.Addr().Unmap(), addrPort.Port()) +} + +// trustedAddrPort reports whether addr is within the trusted CIDRs. An empty +// trusted list admits every client. +// +// Matching is intentionally by IP only: the IPv6 zone identifies the local +// interface a packet arrived on, not the peer, so a zoned link-local client +// matches the same CIDR as its zoneless form. Zones remain part of the session +// key (see udpAddrPort) and are dropped only for trust evaluation. The address +// is unmapped before matching so a v4-mapped client matches an IPv4 CIDR. +func trustedAddrPort(trusted []*net.IPNet, addr netip.AddrPort) bool { + if !addr.IsValid() { + return false + } + if len(trusted) == 0 { + return true + } + ip := addr.Addr().Unmap().AsSlice() + for _, network := range trusted { + if network.Contains(ip) { + return true + } + } + return false +} + func (s *udpSession) touch() { s.lastSeen.Store(time.Now().UnixNano()) } func (s *udpSession) close() { s.once.Do(func() { _ = s.backend.Close() - close(s.done) }) } diff --git a/internal/adapters/in/traffic/udp_test.go b/internal/adapters/in/traffic/udp_test.go index 9923b6c04..721572250 100644 --- a/internal/adapters/in/traffic/udp_test.go +++ b/internal/adapters/in/traffic/udp_test.go @@ -3,6 +3,8 @@ package traffic import ( "context" "net" + "net/netip" + "sync" "testing" "time" @@ -106,6 +108,391 @@ func TestUDPReloadReplacesSessionWhenBackendChanges(t *testing.T) { assertUDPRoundTripWant(t, newConn, "third", "b:third") } +func TestUDPStaleRemovalKeepsReplacementSession(t *testing.T) { + clientAddr := netip.MustParseAddrPort("127.0.0.1:4100") + stale := newTestUDPSession(t, clientAddr) + replacement := newTestUDPSession(t, clientAddr) + runtime, _ := newTestUDPRuntime(t, freeUDPAddress(t)) + runtime.mu.Lock() + runtime.sessions[clientAddr] = replacement + runtime.counters.activeUDPSessions.Store(1) + runtime.mu.Unlock() + + // A stale remover carries the pointer of the session that used to own the + // client key. Expiry, backend read/write failures, reload and CIDR cleanup + // all funnel through removeSession, so none of them may evict the session + // that replaced it. + runtime.removeSession(stale) + + runtime.mu.Lock() + current, ok := runtime.sessions[clientAddr] + runtime.mu.Unlock() + require.True(t, ok, "replacement session must stay registered") + require.Same(t, replacement, current) + require.Equal(t, int64(1), runtime.counters.activeUDPSessions.Load(), "stale removal must not decrement the active counter") + require.False(t, testUDPSessionClosed(replacement), "replacement backend must stay open") + + runtime.removeSession(replacement) + + runtime.mu.Lock() + _, ok = runtime.sessions[clientAddr] + runtime.mu.Unlock() + require.False(t, ok, "removing the current session must unregister it") + require.Equal(t, int64(0), runtime.counters.activeUDPSessions.Load(), "removing the current session must decrement once") + require.True(t, testUDPSessionClosed(replacement)) + + // Removing again must be idempotent: no double decrement and no panic. + runtime.removeSession(replacement) + require.Equal(t, int64(0), runtime.counters.activeUDPSessions.Load()) +} + +func TestUDPExpiryClosesOnlyRegisteredIdleSessionOnce(t *testing.T) { + clientAddr := netip.MustParseAddrPort("127.0.0.1:4200") + // stale was replaced under the same key and is no longer registered. + stale := newTestUDPSession(t, clientAddr) + stale.lastSeen.Store(time.Now().Add(-time.Hour).UnixNano()) + current := newTestUDPSession(t, clientAddr) + runtime, _ := newTestUDPRuntime(t, freeUDPAddress(t)) + runtime.mu.Lock() + runtime.sessions[clientAddr] = current + runtime.counters.activeUDPSessions.Store(1) + runtime.mu.Unlock() + + runtime.expireIdleSessions(time.Minute) + require.False(t, testUDPSessionClosed(current), "fresh registered session must not expire") + require.False(t, testUDPSessionClosed(stale), "an unregistered session must never be closed by expiry") + require.Equal(t, int64(1), runtime.counters.activeUDPSessions.Load()) + + current.lastSeen.Store(time.Now().Add(-time.Hour).UnixNano()) + runtime.expireIdleSessions(time.Minute) + + runtime.mu.Lock() + _, ok := runtime.sessions[clientAddr] + runtime.mu.Unlock() + require.False(t, ok) + require.True(t, testUDPSessionClosed(current)) + require.Equal(t, int64(0), runtime.counters.activeUDPSessions.Load(), "expiry must decrement exactly once") + + // The backend loop of the expired session also removes it on return; the + // identity check keeps the counter coherent and must not double decrement. + runtime.removeSession(current) + require.Equal(t, int64(0), runtime.counters.activeUDPSessions.Load()) +} + +func TestUDPAdmissionGateBlocksPublicationAfterClosingBegins(t *testing.T) { + backend := startUDPEchoServer(t) + runtime, _ := newTestUDPRuntime(t, backend.address) + // Model the instant stop() closes admission but before it cancels the + // runtime context: a datagram worker that reaches session() now must not + // publish a new session. + runtime.mu.Lock() + runtime.accepting = false + runtime.mu.Unlock() + runtime.closed.Store(true) + + session, ok := runtime.session(netip.MustParseAddrPort("127.0.0.1:4300"), effectiveUDPOptions(domain.UDPOptions{})) + require.False(t, ok) + require.Nil(t, session) + + runtime.mu.Lock() + require.Empty(t, runtime.sessions) + runtime.mu.Unlock() + require.Equal(t, int64(0), runtime.counters.activeUDPSessions.Load()) + requireUDPRuntimeGoroutinesDone(t, runtime) +} + +func TestUDPPublishSessionRejectsCompletedDialAfterAdmissionClosure(t *testing.T) { + runtime, _ := newTestUDPRuntime(t, freeUDPAddress(t)) + clientAddr := netip.MustParseAddrPort("127.0.0.1:4800") + dialed := newTestUDPSession(t, clientAddr) + runtime.mu.Lock() + runtime.accepting = false + runtime.mu.Unlock() + runtime.closed.Store(true) + + session, ok := runtime.publishSession(dialed, effectiveUDPOptions(domain.UDPOptions{})) + require.False(t, ok) + require.Nil(t, session) + require.True(t, testUDPSessionClosed(dialed), "a rejected dial must have its backend closed") + runtime.mu.Lock() + require.Empty(t, runtime.sessions) + runtime.mu.Unlock() + require.Equal(t, int64(0), runtime.counters.activeUDPSessions.Load()) + require.Equal(t, int64(0), runtime.counters.totalAccepted.Load()) + requireUDPRuntimeGoroutinesDone(t, runtime) +} + +func TestUDPPublishSessionReusesExistingAndTouchesIt(t *testing.T) { + runtime, _ := newTestUDPRuntime(t, freeUDPAddress(t)) + clientAddr := netip.MustParseAddrPort("127.0.0.1:4900") + existing := newTestUDPSession(t, clientAddr) + existing.lastSeen.Store(time.Now().Add(-time.Hour).UnixNano()) + dialed := newTestUDPSession(t, clientAddr) + runtime.mu.Lock() + runtime.sessions[clientAddr] = existing + runtime.counters.activeUDPSessions.Store(1) + runtime.mu.Unlock() + + session, ok := runtime.publishSession(dialed, effectiveUDPOptions(domain.UDPOptions{})) + require.True(t, ok) + require.Same(t, existing, session) + require.True(t, testUDPSessionClosed(dialed), "the losing dial must have its backend closed") + require.False(t, testUDPSessionClosed(existing)) + require.Greater(t, existing.lastSeen.Load(), time.Now().Add(-time.Minute).UnixNano(), "reuse must touch the existing session") + runtime.mu.Lock() + current := runtime.sessions[clientAddr] + runtime.mu.Unlock() + require.Same(t, existing, current) + require.Equal(t, int64(1), runtime.counters.activeUDPSessions.Load(), "reuse must not add a session") + require.Equal(t, int64(0), runtime.counters.totalAccepted.Load(), "reuse must not count as a new accept") + requireUDPRuntimeGoroutinesDone(t, runtime) +} + +func TestUDPRemoveSessionNilIsSafe(t *testing.T) { + runtime, _ := newTestUDPRuntime(t, freeUDPAddress(t)) + require.NotPanics(t, func() { runtime.removeSession(nil) }) + runtime.mu.Lock() + require.Empty(t, runtime.sessions) + runtime.mu.Unlock() + require.Equal(t, int64(0), runtime.counters.activeUDPSessions.Load()) +} + +func TestUDPReadLoopRejectsInvalidSourceAddress(t *testing.T) { + backend := startUDPEchoServer(t) + packetConn := newScriptedPacketConn() + runtime, _ := newTestUDPRuntimeWithPacketConn(t, backend.address, packetConn) + runtime.start() + + packetConn.packets <- scriptedUDPPacket{payload: []byte("zero"), addr: &net.UDPAddr{}} + packetConn.packets <- scriptedUDPPacket{payload: []byte("junk"), addr: testUDPNetAddr{network: "udp", address: "not-an-address"}} + require.Eventually(t, func() bool { return runtime.counters.totalRefused.Load() == 2 }, time.Second, 5*time.Millisecond) + runtime.mu.Lock() + require.Empty(t, runtime.sessions, "invalid source addresses must never open a session") + runtime.mu.Unlock() + require.Equal(t, int64(0), runtime.counters.activeUDPSessions.Load()) + + // A valid datagram after the invalid ones proves the read loop keeps serving. + packetConn.packets <- scriptedUDPPacket{payload: []byte("ok"), addr: &net.UDPAddr{IP: net.ParseIP("127.0.0.1"), Port: 43210}} + require.Eventually(t, func() bool { return runtime.counters.activeUDPSessions.Load() == 1 }, time.Second, 5*time.Millisecond) + require.Equal(t, int64(1), runtime.counters.totalAccepted.Load()) +} + +func TestUDPStopBeforeStartLeavesNoRuntimeGoroutines(t *testing.T) { + backend := startUDPEchoServer(t) + runtime, _ := newTestUDPRuntime(t, backend.address) + ctx, cancel := context.WithTimeout(context.Background(), time.Second) + defer cancel() + + runtime.stop(ctx, 50*time.Millisecond) + // Starting a runtime whose admission already closed must not launch workers. + runtime.start() + + runtime.mu.Lock() + require.Empty(t, runtime.sessions) + runtime.mu.Unlock() + requireUDPRuntimeGoroutinesDone(t, runtime) + require.Equal(t, int64(0), runtime.counters.activeUDPSessions.Load()) +} + +func TestUDPShutdownLeavesNoSessionsOrRuntimeGoroutines(t *testing.T) { + backend := startUDPEchoServer(t) + graph := udpGraph(t, freeUDPAddress(t), backend.address) + graph.Options.UDP.DrainTimeout = 50 * time.Millisecond + manager := NewManager() + require.NoError(t, manager.Apply(context.Background(), &graph)) + + conn := dialUDP(t, graph.EntryPoints[0].Address) + defer conn.Close() + assertUDPRoundTrip(t, conn, "held") + require.Eventually(t, func() bool { return manager.Status().Counters.ActiveUDPSessions == 1 }, time.Second, 10*time.Millisecond) + + manager.mu.Lock() + runtime := manager.udpListeners["udp"] + manager.mu.Unlock() + require.NotNil(t, runtime) + + ctx, cancel := context.WithTimeout(context.Background(), time.Second) + defer cancel() + require.NoError(t, manager.Shutdown(ctx)) + + runtime.mu.Lock() + require.Empty(t, runtime.sessions) + runtime.mu.Unlock() + require.Equal(t, int64(0), runtime.counters.activeUDPSessions.Load()) + requireUDPRuntimeGoroutinesDone(t, runtime) +} + +func TestUDPConcurrentCreationDuringShutdownLeavesNoSessions(t *testing.T) { + for range 5 { + backend := startUDPEchoServer(t) + runtime, _ := newTestUDPRuntime(t, backend.address) + runtime.start() + + var creators sync.WaitGroup + for worker := range 8 { + creators.Add(1) + go func(worker int) { + defer creators.Done() + for attempt := range 25 { + key := netip.AddrPortFrom(netip.AddrFrom4([4]byte{127, 0, 0, 1}), uint16(5000+worker*100+attempt)) + runtime.session(key, effectiveUDPOptions(domain.UDPOptions{})) + } + }(worker) + } + + ctx, cancel := context.WithTimeout(context.Background(), time.Second) + runtime.stop(ctx, 50*time.Millisecond) + cancel() + creators.Wait() + + runtime.mu.Lock() + require.Empty(t, runtime.sessions, "no session may be published after closing began") + runtime.mu.Unlock() + require.Equal(t, int64(0), runtime.counters.activeUDPSessions.Load()) + requireUDPRuntimeGoroutinesDone(t, runtime) + } +} + +// TestUDPInFlightDialDoesNotPublishAfterAdmissionClosure blocks the backend dial +// on the runtime's private dial seam, closes admission exactly as stop() does, +// then lets the dial complete: session() must reject the completed dial without +// publishing a session or a backend goroutine. The seam is instance-local, so no +// package-global state is touched. +func TestUDPInFlightDialDoesNotPublishAfterAdmissionClosure(t *testing.T) { + runtime, _ := newTestUDPRuntime(t, "127.0.0.1:15353") + release := make(chan struct{}) + dialStarted := make(chan struct{}, 1) + liveConn := newLiveUDPConn(t) + runtime.dialContext = func(ctx context.Context, _, _ string) (net.Conn, error) { + select { + case dialStarted <- struct{}{}: + default: + } + select { + case <-release: + return liveConn, nil + case <-ctx.Done(): + return nil, ctx.Err() + } + } + + type result struct { + session *udpSession + ok bool + } + results := make(chan result, 1) + go func() { + session, ok := runtime.session(netip.MustParseAddrPort("127.0.0.1:4700"), effectiveUDPOptions(domain.UDPOptions{})) + results <- result{session: session, ok: ok} + }() + + select { + case <-dialStarted: + case <-time.After(5 * time.Second): + t.Fatal("backend dial never reached the dial seam") + } + + runtime.mu.Lock() + runtime.accepting = false + runtime.mu.Unlock() + runtime.closed.Store(true) + close(release) + + select { + case got := <-results: + require.False(t, got.ok, "a dial completing after admission closure must not open a session") + require.Nil(t, got.session) + case <-time.After(5 * time.Second): + t.Fatal("in-flight session creation did not finish") + } + + runtime.mu.Lock() + require.Empty(t, runtime.sessions) + runtime.mu.Unlock() + require.Equal(t, int64(0), runtime.counters.activeUDPSessions.Load()) + require.Equal(t, int64(0), runtime.counters.totalAccepted.Load()) + requireUDPRuntimeGoroutinesDone(t, runtime) +} + +func TestUDPAddrPortNormalizesIPv4MappedAndPreservesZones(t *testing.T) { + plainV4, ok := udpAddrPort(&net.UDPAddr{IP: net.ParseIP("127.0.0.1"), Port: 9000}) + require.True(t, ok) + mappedV4, ok := udpAddrPort(&net.UDPAddr{IP: net.ParseIP("::ffff:127.0.0.1"), Port: 9000}) + require.True(t, ok) + wantV4 := netip.MustParseAddrPort("127.0.0.1:9000") + require.Equal(t, wantV4, plainV4) + require.Equal(t, wantV4, mappedV4, "IPv4-mapped addresses must share the IPv4 session key") + + zoned, ok := udpAddrPort(&net.UDPAddr{IP: net.ParseIP("fe80::1"), Zone: "eth0", Port: 9001}) + require.True(t, ok) + require.Equal(t, netip.MustParseAddrPort("[fe80::1%eth0]:9001"), zoned) + otherZone, ok := udpAddrPort(&net.UDPAddr{IP: net.ParseIP("fe80::1"), Zone: "eth1", Port: 9001}) + require.True(t, ok) + require.NotEqual(t, zoned, otherZone, "IPv6 zones must remain part of the key") + + parsed, ok := udpAddrPort(testUDPNetAddr{network: "udp", address: "[2001:db8::1]:9002"}) + require.True(t, ok) + require.Equal(t, netip.MustParseAddrPort("[2001:db8::1]:9002"), parsed) + + _, ok = udpAddrPort(testUDPNetAddr{network: "udp", address: "not-an-address"}) + require.False(t, ok) +} + +func TestUDPAddrPortRejectsInvalidAddresses(t *testing.T) { + _, ok := udpAddrPort(&net.UDPAddr{}) + require.False(t, ok, "zero UDPAddr must be rejected") + _, ok = udpAddrPort(&net.UDPAddr{IP: nil, Port: 53}) + require.False(t, ok, "UDPAddr without an IP must be rejected") + _, ok = udpAddrPort(testUDPNetAddr{network: "udp", address: "not-an-address"}) + require.False(t, ok) + + // The trusted-list path must not admit an invalid address either, even when + // the list is empty and would otherwise trust every client. + require.False(t, trustedAddrPort(nil, netip.AddrPort{})) +} + +func TestUDPSessionPrecomputesImmutableRemoteAddr(t *testing.T) { + backend := startUDPEchoServer(t) + runtime, _ := newTestUDPRuntime(t, backend.address) + options := effectiveUDPOptions(domain.UDPOptions{}) + + v4, ok := runtime.session(netip.MustParseAddrPort("127.0.0.1:4600"), options) + require.True(t, ok) + require.NotNil(t, v4.remoteAddr) + require.Equal(t, "127.0.0.1:4600", v4.remoteAddr.String()) + runtime.removeSession(v4) + + zoned, ok := runtime.session(netip.MustParseAddrPort("[fe80::1%eth0]:4601"), options) + require.True(t, ok) + require.NotNil(t, zoned.remoteAddr) + require.Equal(t, netip.MustParseAddrPort("[fe80::1%eth0]:4601"), zoned.remoteAddr.AddrPort(), "precomputed destination must preserve the IPv6 zone") + runtime.removeSession(zoned) +} + +func TestUDPTrustedAddrPortMatchesCIDRs(t *testing.T) { + ipv4Trusted, err := parseTrustedCIDRs([]string{"127.0.0.0/8"}) + require.NoError(t, err) + require.True(t, trustedAddrPort(ipv4Trusted, netip.MustParseAddrPort("127.0.0.1:1"))) + require.True(t, trustedAddrPort(ipv4Trusted, netip.MustParseAddrPort("[::ffff:127.0.0.1]:1")), "v4-mapped client must match IPv4 CIDR") + require.False(t, trustedAddrPort(ipv4Trusted, netip.MustParseAddrPort("192.0.2.1:1"))) + require.False(t, trustedAddrPort(ipv4Trusted, netip.MustParseAddrPort("[2001:db8::1]:1")), "IPv6 client must not match IPv4 CIDR") + + ipv6Trusted, err := parseTrustedCIDRs([]string{"2001:db8::/32"}) + require.NoError(t, err) + require.True(t, trustedAddrPort(ipv6Trusted, netip.MustParseAddrPort("[2001:db8::1]:1"))) + // Intentional UDP semantic: CIDR trust is evaluated on the IP only. The + // zone names the local interface a packet arrived on, not the peer, so a + // zoned and zoneless client of the same address are equally trusted. Zones + // stay part of the session key; do not "fix" this to compare zones. + require.True(t, trustedAddrPort(ipv6Trusted, netip.MustParseAddrPort("[2001:db8::1%eth0]:1")), "zone must not affect CIDR matching") + require.True(t, trustedAddrPort(ipv6Trusted, netip.MustParseAddrPort("[2001:db8::1%eth1]:1")), "any zone on a trusted address is trusted") + require.False(t, trustedAddrPort(ipv6Trusted, netip.MustParseAddrPort("[2001:dead::2%eth0]:1")), "an untrusted address is untrusted regardless of zone") + require.False(t, trustedAddrPort(ipv6Trusted, netip.MustParseAddrPort("127.0.0.1:1"))) + + require.True(t, trustedAddrPort(nil, netip.MustParseAddrPort("203.0.113.9:1")), "empty trusted list admits every client") + require.False(t, trustedAddrPort(ipv4Trusted, netip.AddrPort{}), "invalid client address is not trusted") +} + func assertUDPRejected(t *testing.T, manager *Manager, address string, refused int64) { t.Helper() conn := dialUDP(t, address) @@ -330,6 +717,130 @@ func assertUDPRoundTripWant(t *testing.T, conn *net.UDPConn, message string, wan assert.Equal(t, want, string(buf[:n])) } +// newTestUDPRuntime builds a UDP runtime backed by a valid graph snapshot +// without binding through Manager.Apply, so tests can drive publication, +// expiry and shutdown interleavings directly. +func newTestUDPRuntime(t *testing.T, backendAddress string) (*udpEntryPointRuntime, *Manager) { + t.Helper() + packetConn, err := net.ListenPacket("udp", "127.0.0.1:0") + require.NoError(t, err) + return newTestUDPRuntimeWithPacketConn(t, backendAddress, packetConn) +} + +// newTestUDPRuntimeWithPacketConn builds a runtime over a caller-supplied +// PacketConn so the read loop can be driven deterministically. +func newTestUDPRuntimeWithPacketConn(t *testing.T, backendAddress string, packetConn net.PacketConn) (*udpEntryPointRuntime, *Manager) { + t.Helper() + graph := udpGraph(t, freeUDPAddress(t), backendAddress) + manager := NewManager() + manager.snapshot.Store(&graph) + runtime := newUDPEntryPointRuntime(context.Background(), manager, graph.EntryPoints[0], packetConn, nil) + t.Cleanup(func() { + ctx, cancel := context.WithTimeout(context.Background(), time.Second) + defer cancel() + runtime.stop(ctx, 50*time.Millisecond) + _ = packetConn.Close() + }) + return runtime, manager +} + +// scriptedPacketConn feeds the read loop a fixed sequence of packets, including +// source addresses that cannot be converted to a session key. +type scriptedUDPPacket struct { + payload []byte + addr net.Addr +} + +type scriptedPacketConn struct { + packets chan scriptedUDPPacket + closed chan struct{} + once sync.Once +} + +func newScriptedPacketConn() *scriptedPacketConn { + return &scriptedPacketConn{packets: make(chan scriptedUDPPacket, 16), closed: make(chan struct{})} +} + +func (c *scriptedPacketConn) ReadFrom(p []byte) (int, net.Addr, error) { + select { + case packet := <-c.packets: + return copy(p, packet.payload), packet.addr, nil + case <-c.closed: + return 0, nil, net.ErrClosed + } +} + +func (c *scriptedPacketConn) WriteTo(p []byte, _ net.Addr) (int, error) { return len(p), nil } + +func (c *scriptedPacketConn) Close() error { + c.once.Do(func() { close(c.closed) }) + return nil +} + +func (c *scriptedPacketConn) LocalAddr() net.Addr { return &net.UDPAddr{IP: net.IPv4zero} } +func (c *scriptedPacketConn) SetDeadline(time.Time) error { return nil } +func (c *scriptedPacketConn) SetReadDeadline(time.Time) error { return nil } +func (c *scriptedPacketConn) SetWriteDeadline(time.Time) error { return nil } + +func newTestUDPSession(t *testing.T, clientAddr netip.AddrPort) *udpSession { + t.Helper() + session := &udpSession{clientAddr: clientAddr, remoteAddr: net.UDPAddrFromAddrPort(clientAddr), backend: newLiveUDPConn(t)} + session.touch() + return session +} + +// newLiveUDPConn dials a live loopback UDP listener so probe writes never fail +// with ICMP port-unreachable errors, and tears both down with the test. +func newLiveUDPConn(t *testing.T) net.Conn { + t.Helper() + listener, err := net.ListenPacket("udp", "127.0.0.1:0") + require.NoError(t, err) + t.Cleanup(func() { _ = listener.Close() }) + go func() { + buf := make([]byte, 64) + for { + if _, _, err := listener.ReadFrom(buf); err != nil { + return + } + } + }() + conn, err := net.Dial("udp", listener.LocalAddr().String()) + require.NoError(t, err) + t.Cleanup(func() { _ = conn.Close() }) + return conn +} + +// testUDPSessionClosed probes the backend with a write. Synthetic sessions are +// connected to a live loopback listener (see newLiveUDPConn), so an open socket +// always accepts the probe and a closed one always errors, without the ICMP +// port-unreachable flakiness of writing to an unbound port. +func testUDPSessionClosed(session *udpSession) bool { + _, err := session.backend.Write([]byte("probe")) + return err != nil +} + +func requireUDPRuntimeGoroutinesDone(t *testing.T, runtime *udpEntryPointRuntime) { + t.Helper() + done := make(chan struct{}) + go func() { + runtime.runWG.Wait() + close(done) + }() + select { + case <-done: + case <-time.After(2 * time.Second): + t.Fatal("udp runtime goroutines did not complete") + } +} + +type testUDPNetAddr struct { + network string + address string +} + +func (a testUDPNetAddr) Network() string { return a.network } +func (a testUDPNetAddr) String() string { return a.address } + type udpEchoServer struct{ address string } func startUDPEchoServer(t *testing.T) udpEchoServer { diff --git a/internal/adapters/localadmin/localadmin.go b/internal/adapters/localadmin/localadmin.go new file mode 100644 index 000000000..655b3ca72 --- /dev/null +++ b/internal/adapters/localadmin/localadmin.go @@ -0,0 +1,412 @@ +// Package localadmin owns the owner-only Unix socket policy shared by the +// Gordon daemon and the local CLI: runtime directory discovery, strict +// filesystem validation, and safe listener lifecycle. +// +// It deliberately depends only on the standard library so the composition +// root (internal/app) and the incoming CLI adapter can share it without +// creating an adapter-to-application dependency. +package localadmin + +import ( + "errors" + "fmt" + "io/fs" + "net" + "os" + "path/filepath" + "syscall" + "time" +) + +// SocketName is the fixed filename of the owner-only admin socket. +const SocketName = "admin.sock" + +// DirPermissions is the required mode of the runtime directory. +const DirPermissions fs.FileMode = 0o700 + +// SocketPermissions is the required mode of the admin socket. +const SocketPermissions fs.FileMode = 0o600 + +// OtherPermissions is the union of permission bits that must never be set. +const OtherPermissions fs.FileMode = 0o077 + +// staleDialTimeout bounds the liveness probe performed before removing a +// stale socket path. +const staleDialTimeout = 500 * time.Millisecond + +// ErrUnsafePath marks a path that failed ownership, type, or mode validation. +var ErrUnsafePath = errors.New("unsafe local admin path") + +// ErrActiveListener marks a socket that already has a live listener. +var ErrActiveListener = errors.New("local admin socket already has an active listener") + +// SocketPath joins dir with the fixed socket filename. +func SocketPath(dir string) string { + return filepath.Join(dir, SocketName) +} + +// EnsureRuntimeDir returns, creates, and validates the daemon runtime +// directory: $XDG_RUNTIME_DIR/gordon when set, otherwise $HOME/.gordon/run. +// A set XDG_RUNTIME_DIR must be absolute; any failure to ensure the preferred +// directory fails closed and never falls back to $HOME, since a silent +// fallback would change daemon identity. +func EnsureRuntimeDir() (string, error) { + if xdg := os.Getenv("XDG_RUNTIME_DIR"); xdg != "" { + if !filepath.IsAbs(xdg) { + return "", fmt.Errorf("%w: XDG_RUNTIME_DIR is not absolute: %q", ErrUnsafePath, xdg) + } + dir := filepath.Join(xdg, "gordon") + if err := EnsureDir(dir); err != nil { + return "", err + } + return dir, nil + } + + home, err := os.UserHomeDir() + if err != nil { + return "", fmt.Errorf("resolve home directory: %w", err) + } + dir := filepath.Join(home, ".gordon", "run") + if err := EnsureDir(dir); err != nil { + return "", err + } + return dir, nil +} + +// ClientDirCandidates lists CLI discovery candidates in daemon priority +// order: $XDG_RUNTIME_DIR/gordon, /run/user//gordon, then +// $HOME/.gordon/run. Directories are not created. +func ClientDirCandidates() []string { + var candidates []string + + if xdg := os.Getenv("XDG_RUNTIME_DIR"); xdg != "" { + candidates = append(candidates, filepath.Join(xdg, "gordon")) + } + + systemDir := filepath.Join("/run/user", fmt.Sprintf("%d", os.Getuid()), "gordon") + if len(candidates) == 0 || candidates[0] != systemDir { + candidates = append(candidates, systemDir) + } + + if home, err := os.UserHomeDir(); err == nil { + candidates = append(candidates, filepath.Join(home, ".gordon", "run")) + } + + return candidates +} + +// EnsureDir creates dir with owner-only permissions when missing, tightens an +// existing directory, and then validates it as an owner-owned real directory +// without group/other permission bits. +func EnsureDir(dir string) error { + // Validate the existing ancestor chain before creating anything so + // MkdirAll never creates directories through a symlinked, foreign, or + // otherwise unsafe ancestor. + if err := validateExistingAncestorChain(dir); err != nil { + return err + } + if err := os.MkdirAll(dir, DirPermissions); err != nil { + return fmt.Errorf("create local admin runtime directory: %w", err) + } + + fi, err := os.Lstat(dir) + if err != nil { + return fmt.Errorf("inspect local admin runtime directory: %w", err) + } + // Refuse symlinks, non-directories, and foreign owners before any chmod so + // a hostile path never has its target modified through a link. + if err := validateDirIdentity(fi, dir); err != nil { + return err + } + if fi.Mode().Perm()&OtherPermissions != 0 { + if err := os.Chmod(dir, DirPermissions); err != nil { + return fmt.Errorf("restrict local admin runtime directory: %w", err) + } + } + return ValidateDir(dir) +} + +// ValidateDir reports whether dir is an owner-owned real directory without +// group/other permission bits. +func ValidateDir(dir string) error { + if err := validateParentChain(dir); err != nil { + return err + } + fi, err := os.Lstat(dir) + if err != nil { + return fmt.Errorf("inspect local admin runtime directory: %w", err) + } + if err := validateDirIdentity(fi, dir); err != nil { + return err + } + if fi.Mode().Perm()&OtherPermissions != 0 { + return fmt.Errorf("%w: runtime directory %s has group/other permissions %04o", ErrUnsafePath, dir, fi.Mode().Perm()) + } + return nil +} + +// validateExistingAncestorChain validates every existing ancestor of dir +// before MkdirAll runs, so missing directories are never created through a +// symlinked, non-directory, or foreign-writable ancestor. Missing ancestors +// are skipped: they will be created by MkdirAll once the existing chain is +// known safe. +func validateExistingAncestorChain(dir string) error { + for current := filepath.Dir(filepath.Clean(dir)); ; current = filepath.Dir(current) { + fi, err := os.Lstat(current) + if err != nil { + if errors.Is(err, fs.ErrNotExist) { + parent := filepath.Dir(current) + if parent == current { + return nil + } + continue + } + return fmt.Errorf("inspect local admin path component %s: %w", current, err) + } + if err := validatePathComponent(fi, current); err != nil { + return err + } + parent := filepath.Dir(current) + if parent == current { + return nil + } + } +} + +func validatePathComponent(fi os.FileInfo, path string) error { + return validatePathComponentForUID(fi, path, uint32(os.Geteuid())) // #nosec G115 -- Unix effective UIDs fit uid_t. +} + +func validatePathComponentForUID(fi os.FileInfo, path string, euid uint32) error { + if fi.Mode()&os.ModeSymlink != 0 || !fi.IsDir() { + return fmt.Errorf("%w: path component %s is not a real directory", ErrUnsafePath, path) + } + st, ok := fi.Sys().(*syscall.Stat_t) + if !ok { + return fmt.Errorf("%w: cannot determine owner of %s", ErrUnsafePath, path) + } + if st.Uid != 0 && st.Uid != euid { + return fmt.Errorf("%w: path component %s is owned by uid %d, not root or %d", ErrUnsafePath, path, st.Uid, euid) + } + writable := fi.Mode().Perm()&0o022 != 0 + stickyRoot := st.Uid == 0 && fi.Mode()&os.ModeSticky != 0 + if writable && !stickyRoot { + return fmt.Errorf("%w: path component %s is replaceable", ErrUnsafePath, path) + } + return nil +} + +// validateDirIdentity checks the properties that no chmod can repair: a symlink +// target, a non-directory, or another user's path. +func validateParentChain(dir string) error { + for current := filepath.Clean(dir); ; current = filepath.Dir(current) { + fi, err := os.Lstat(current) + if err != nil { + return fmt.Errorf("inspect local admin path component %s: %w", current, err) + } + if err := validatePathComponent(fi, current); err != nil { + return err + } + parent := filepath.Dir(current) + if parent == current { + return nil + } + } +} + +func validateDirIdentity(fi os.FileInfo, dir string) error { + if fi.Mode()&os.ModeSymlink != 0 { + return fmt.Errorf("%w: runtime directory %s is a symlink", ErrUnsafePath, dir) + } + if !fi.IsDir() { + return fmt.Errorf("%w: runtime directory %s is not a directory", ErrUnsafePath, dir) + } + if err := validateOwner(fi); err != nil { + return fmt.Errorf("runtime directory %s: %w", dir, err) + } + return nil +} + +// ValidateSocket reports whether path is an owner-owned Unix socket without +// group/other permission bits and without a symlink at the final component. +func ValidateSocket(path string) error { + if err := validateParentChain(filepath.Dir(path)); err != nil { + return err + } + fi, err := os.Lstat(path) + if err != nil { + return err + } + return validateSocketFile(fi, path) +} + +// Listen binds the owner-only admin socket inside dir and returns the +// listener plus the identity of the file it created. +// +// It fails closed: the directory must be owner-owned, a real directory, and +// owner-only; an existing path must already be an owner-owned socket; and an +// existing socket is removed only after a bounded dial proves no listener is +// active. Any other existing path is never replaced. +func Listen(dir string) (*net.UnixListener, os.FileInfo, error) { + if err := EnsureDir(dir); err != nil { + return nil, nil, err + } + + path := SocketPath(dir) + if err := clearStale(path); err != nil { + return nil, nil, err + } + + ln, err := net.ListenUnix("unix", &net.UnixAddr{Name: path, Net: "unix"}) + if err != nil { + return nil, nil, fmt.Errorf("bind local admin socket: %w", err) + } + // Keep unlink-on-close under our control so shutdown removes the socket + // only after verifying it is still the file we bound. + ln.SetUnlinkOnClose(false) + + bound, err := os.Lstat(path) + if err != nil { + _ = ln.Close() + return nil, nil, fmt.Errorf("inspect newly bound local admin socket: %w", err) + } + if err := os.Chmod(path, SocketPermissions); err != nil { + return nil, nil, cleanupListen(ln, path, bound, fmt.Errorf("restrict local admin socket: %w", err)) + } + + fi, err := os.Lstat(path) + if err != nil { + return nil, nil, cleanupListen(ln, path, bound, fmt.Errorf("inspect local admin socket: %w", err)) + } + if !os.SameFile(bound, fi) { + return nil, nil, cleanupListen(ln, path, bound, fmt.Errorf("%w: local admin socket was replaced during setup", ErrUnsafePath)) + } + if err := validateSocketFile(fi, path); err != nil { + return nil, nil, cleanupListen(ln, path, bound, err) + } + + return ln, fi, nil +} + +// RemoveOwnedSocket removes path only when it is still the same file that was +// bound. A replaced path is left untouched and reported as unsafe. +func RemoveOwnedSocket(path string, bound os.FileInfo) error { + if bound == nil { + return nil + } + current, err := os.Lstat(path) + if errors.Is(err, fs.ErrNotExist) { + return nil + } + if err != nil { + return fmt.Errorf("inspect local admin socket for removal: %w", err) + } + if !os.SameFile(bound, current) { + return fmt.Errorf("%w: local admin socket %s was replaced; not removing it", ErrUnsafePath, path) + } + if err := os.Remove(path); err != nil && !errors.Is(err, fs.ErrNotExist) { + return fmt.Errorf("remove local admin socket: %w", err) + } + return nil +} + +func cleanupListen(ln *net.UnixListener, path string, bound os.FileInfo, cause error) error { + _ = ln.Close() + _ = RemoveOwnedSocket(path, bound) + return cause +} + +func validateSocketFile(fi os.FileInfo, path string) error { + if err := validateSocketIdentity(fi, path); err != nil { + return err + } + if fi.Mode().Perm()&OtherPermissions != 0 { + return fmt.Errorf("%w: %s has group/other permissions %04o", ErrUnsafePath, path, fi.Mode().Perm()) + } + return nil +} + +// validateSocketIdentity checks only the properties that must hold before a +// stale path may be removed: a non-symlink Unix socket owned by this euid. +// Callers that consume the socket additionally require mode 0600. +func validateSocketIdentity(fi os.FileInfo, path string) error { + if fi.Mode()&os.ModeSymlink != 0 { + return fmt.Errorf("%w: %s is a symlink", ErrUnsafePath, path) + } + if fi.Mode()&os.ModeSocket == 0 { + return fmt.Errorf("%w: %s is not a Unix socket", ErrUnsafePath, path) + } + if err := validateOwner(fi); err != nil { + return fmt.Errorf("%s: %w", path, err) + } + return nil +} + +func validateOwner(fi os.FileInfo) error { + st, ok := fi.Sys().(*syscall.Stat_t) + if !ok { + return fmt.Errorf("%w: cannot determine file owner", ErrUnsafePath) + } + if int64(st.Uid) != int64(os.Geteuid()) { + return fmt.Errorf("%w: owned by uid %d, not %d", ErrUnsafePath, st.Uid, os.Geteuid()) + } + return nil +} + +// clearStale removes an existing socket path only when it is an owner-owned +// socket with no active listener. Everything else fails closed. +func clearStale(path string) error { + fi, err := os.Lstat(path) + if errors.Is(err, fs.ErrNotExist) { + return nil + } + if err != nil { + return fmt.Errorf("inspect local admin socket: %w", err) + } + if err := validateSocketIdentity(fi, path); err != nil { + return fmt.Errorf("refusing to replace local admin path: %w", err) + } + + live, err := listening(path) + if err != nil { + return err + } + if live { + return fmt.Errorf("%w at %s", ErrActiveListener, path) + } + + current, err := os.Lstat(path) + if errors.Is(err, fs.ErrNotExist) { + return nil + } + if err != nil { + return fmt.Errorf("reinspect stale local admin socket: %w", err) + } + if !os.SameFile(fi, current) { + return fmt.Errorf("%w: local admin socket changed during stale check", ErrUnsafePath) + } + if err := os.Remove(path); err != nil && !errors.Is(err, fs.ErrNotExist) { + return fmt.Errorf("remove stale local admin socket: %w", err) + } + return nil +} + +// ProbeSocket reports whether a validated socket has a live listener. +func ProbeSocket(path string) (bool, error) { + return listening(path) +} + +// listening dials path with a bounded timeout. A live listener reports true; +// an explicit no-listener error reports false. Any other dial failure is +// returned so an unexpected condition never leads to removing a path. +func listening(path string) (bool, error) { + conn, err := net.DialTimeout("unix", path, staleDialTimeout) + if err == nil { + _ = conn.Close() + return true, nil + } + if errors.Is(err, syscall.ECONNREFUSED) || errors.Is(err, syscall.ENOENT) || errors.Is(err, fs.ErrNotExist) { + return false, nil + } + return false, fmt.Errorf("probe local admin socket %s: %w", path, err) +} diff --git a/internal/adapters/localadmin/localadmin_test.go b/internal/adapters/localadmin/localadmin_test.go new file mode 100644 index 000000000..3e99431d7 --- /dev/null +++ b/internal/adapters/localadmin/localadmin_test.go @@ -0,0 +1,397 @@ +package localadmin + +import ( + "context" + "io/fs" + "net" + "os" + "path/filepath" + "strconv" + "syscall" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func TestEnsureRuntimeDirPrefersXDG(t *testing.T) { + xdg := t.TempDir() + t.Setenv("XDG_RUNTIME_DIR", xdg) + + dir, err := EnsureRuntimeDir() + require.NoError(t, err) + assert.Equal(t, filepath.Join(xdg, "gordon"), dir) + + fi, err := os.Lstat(dir) + require.NoError(t, err) + assert.Equal(t, fs.FileMode(0o700), fi.Mode().Perm()) +} + +func TestEnsureRuntimeDirFallsBackToHome(t *testing.T) { + home := t.TempDir() + t.Setenv("XDG_RUNTIME_DIR", "") + t.Setenv("HOME", home) + + dir, err := EnsureRuntimeDir() + require.NoError(t, err) + assert.Equal(t, filepath.Join(home, ".gordon", "run"), dir) +} + +func TestEnsureRuntimeDirRejectsUnsafeXDG(t *testing.T) { + xdg := t.TempDir() + // Make the selected XDG gordon path a symlink. Falling back would silently + // change daemon identity, so an unsafe selected path fails closed. + require.NoError(t, os.Symlink(t.TempDir(), filepath.Join(xdg, "gordon"))) + t.Setenv("XDG_RUNTIME_DIR", xdg) + t.Setenv("HOME", t.TempDir()) + + _, err := EnsureRuntimeDir() + require.Error(t, err) + assert.ErrorIs(t, err, ErrUnsafePath) +} + +func TestEnsureRuntimeDirRejectsRelativeXDG(t *testing.T) { + home := t.TempDir() + t.Setenv("XDG_RUNTIME_DIR", filepath.Join("relative", "runtime")) + t.Setenv("HOME", home) + + _, err := EnsureRuntimeDir() + require.Error(t, err) + assert.ErrorIs(t, err, ErrUnsafePath) + // Fail closed: no silent fallback to $HOME. + assert.NoDirExists(t, filepath.Join(home, ".gordon", "run")) +} + +func TestEnsureRuntimeDirNoFallbackOnCreateError(t *testing.T) { + // An overlong XDG_RUNTIME_DIR component makes ancestor inspection fail + // with ENAMETOOLONG, which is not ErrUnsafePath. The preferred candidate + // must fail immediately rather than silently falling back to $HOME. + base := t.TempDir() + longName := make([]byte, 5000) + for i := range longName { + longName[i] = 'x' + } + home := t.TempDir() + t.Setenv("XDG_RUNTIME_DIR", filepath.Join(base, string(longName))) + t.Setenv("HOME", home) + + _, err := EnsureRuntimeDir() + require.Error(t, err) + assert.NotErrorIs(t, err, ErrUnsafePath) + assert.NoDirExists(t, filepath.Join(home, ".gordon", "run")) +} + +func TestEnsureDirRefusesSymlinkAncestor(t *testing.T) { + target := t.TempDir() + link := filepath.Join(t.TempDir(), "link") + require.NoError(t, os.Symlink(target, link)) + + dir := filepath.Join(link, "gordon") + err := EnsureDir(dir) + require.Error(t, err) + assert.ErrorIs(t, err, ErrUnsafePath) + // Nothing must be created through the symlinked ancestor. + assert.NoDirExists(t, filepath.Join(target, "gordon")) + assert.NoFileExists(t, dir) +} + +func TestEnsureDirTightensPermissions(t *testing.T) { + dir := filepath.Join(t.TempDir(), "gordon") + require.NoError(t, os.Mkdir(dir, 0o755)) + + require.NoError(t, EnsureDir(dir)) + + fi, err := os.Lstat(dir) + require.NoError(t, err) + assert.Equal(t, fs.FileMode(0o700), fi.Mode().Perm()) +} + +func TestEnsureDirRejectsOwnerOwnedWritableAncestor(t *testing.T) { + ancestor := t.TempDir() + require.NoError(t, os.Chmod(ancestor, 0o770)) + + dir := filepath.Join(ancestor, "nested", "gordon") + err := EnsureDir(dir) + require.Error(t, err) + assert.ErrorIs(t, err, ErrUnsafePath) + assert.NoDirExists(t, filepath.Join(ancestor, "nested")) +} + +type fileInfoWithOwner struct { + os.FileInfo + uid uint32 +} + +func (fi fileInfoWithOwner) Sys() any { + return &syscall.Stat_t{Uid: fi.uid} +} + +func TestValidatePathComponentRejectsForeignOwnedAncestors(t *testing.T) { + base, err := os.Lstat(t.TempDir()) + require.NoError(t, err) + + for _, mode := range []fs.FileMode{0o755, 0o700} { + t.Run(mode.String(), func(t *testing.T) { + fi := fileInfoWithOwner{FileInfo: fileInfoWithMode{FileInfo: base, mode: os.ModeDir | mode}, uid: 2000} + err := validatePathComponentForUID(fi, "/runtime", 1000) + require.Error(t, err) + assert.ErrorIs(t, err, ErrUnsafePath) + assert.Contains(t, err.Error(), "owned by uid 2000") + }) + } +} + +type fileInfoWithMode struct { + os.FileInfo + mode fs.FileMode +} + +func (fi fileInfoWithMode) Mode() fs.FileMode { return fi.mode } + +func TestValidatePathComponentAllowsRootOwnedStickyAncestor(t *testing.T) { + base, err := os.Lstat(t.TempDir()) + require.NoError(t, err) + fi := fileInfoWithOwner{ + FileInfo: fileInfoWithMode{FileInfo: base, mode: os.ModeDir | os.ModeSticky | 0o777}, + uid: 0, + } + + assert.NoError(t, validatePathComponentForUID(fi, "/tmp", 1000)) +} + +func TestValidateDirAllowsStickyRootAncestor(t *testing.T) { + root := os.TempDir() + fi, err := os.Lstat(root) + require.NoError(t, err) + st, ok := fi.Sys().(*syscall.Stat_t) + if !ok || st.Uid != 0 || fi.Mode()&os.ModeSticky == 0 || fi.Mode().Perm()&0o022 == 0 { + t.Skip("system temporary directory is not writable sticky-root") + } + + dir := filepath.Join(t.TempDir(), "gordon") + require.NoError(t, os.Mkdir(dir, DirPermissions)) + assert.NoError(t, ValidateDir(dir)) +} + +func TestEnsureDirRejectsSymlinkDirectory(t *testing.T) { + target := t.TempDir() + link := filepath.Join(t.TempDir(), "gordon") + require.NoError(t, os.Symlink(target, link)) + + err := EnsureDir(link) + require.Error(t, err) + assert.ErrorIs(t, err, ErrUnsafePath) +} + +func TestEnsureDirRejectsRegularFile(t *testing.T) { + path := filepath.Join(t.TempDir(), "gordon") + require.NoError(t, os.WriteFile(path, []byte("x"), 0o600)) + + err := EnsureDir(path) + require.Error(t, err) +} + +func TestEnsureDirRejectsForeignOwner(t *testing.T) { + if os.Geteuid() != 0 { + t.Skip("requires root to change directory ownership") + } + dir := filepath.Join(t.TempDir(), "gordon") + require.NoError(t, os.Mkdir(dir, 0o700)) + require.NoError(t, os.Chown(dir, 65534, 65534)) + + err := EnsureDir(dir) + require.Error(t, err) + assert.ErrorIs(t, err, ErrUnsafePath) +} + +func TestListenCreatesOwnerOnlySocket(t *testing.T) { + dir := filepath.Join(t.TempDir(), "gordon") + + ln, info, err := Listen(dir) + require.NoError(t, err) + t.Cleanup(func() { _ = ln.Close() }) + + path := SocketPath(dir) + fi, err := os.Lstat(path) + require.NoError(t, err) + assert.NotZero(t, fi.Mode()&os.ModeSocket) + assert.Equal(t, fs.FileMode(0o600), fi.Mode().Perm()) + assert.True(t, os.SameFile(info, fi)) + require.NoError(t, ValidateSocket(path)) +} + +func TestListenRefusesActiveListener(t *testing.T) { + dir := filepath.Join(t.TempDir(), "gordon") + + ln, _, err := Listen(dir) + require.NoError(t, err) + t.Cleanup(func() { _ = ln.Close() }) + + go func() { + for { + conn, acceptErr := ln.Accept() + if acceptErr != nil { + return + } + _ = conn.Close() + } + }() + + _, _, err = Listen(dir) + require.Error(t, err) + assert.ErrorIs(t, err, ErrActiveListener) + assert.FileExists(t, SocketPath(dir)) +} + +func TestListenRemovesStaleOwnedSocket(t *testing.T) { + dir := filepath.Join(t.TempDir(), "gordon") + require.NoError(t, EnsureDir(dir)) + + stale, err := net.ListenUnix("unix", &net.UnixAddr{Name: SocketPath(dir), Net: "unix"}) + require.NoError(t, err) + stale.SetUnlinkOnClose(false) + require.NoError(t, stale.Close()) + require.FileExists(t, SocketPath(dir)) + + ln, _, err := Listen(dir) + require.NoError(t, err) + t.Cleanup(func() { _ = ln.Close() }) + assert.FileExists(t, SocketPath(dir)) +} + +func TestListenRefusesRegularFileCollision(t *testing.T) { + dir := filepath.Join(t.TempDir(), "gordon") + require.NoError(t, EnsureDir(dir)) + path := SocketPath(dir) + require.NoError(t, os.WriteFile(path, []byte("not a socket"), 0o600)) + + _, _, err := Listen(dir) + require.Error(t, err) + assert.ErrorIs(t, err, ErrUnsafePath) + // The unknown file is never deleted or replaced. + assert.FileExists(t, path) +} + +func TestListenRefusesSymlinkSocket(t *testing.T) { + dir := filepath.Join(t.TempDir(), "gordon") + require.NoError(t, EnsureDir(dir)) + realDir := t.TempDir() + real, err := net.ListenUnix("unix", &net.UnixAddr{Name: filepath.Join(realDir, "real.sock"), Net: "unix"}) + require.NoError(t, err) + t.Cleanup(func() { _ = real.Close() }) + + require.NoError(t, os.Symlink(filepath.Join(realDir, "real.sock"), SocketPath(dir))) + + _, _, err = Listen(dir) + require.Error(t, err) + assert.ErrorIs(t, err, ErrUnsafePath) +} + +func TestValidateSocketRejectsGroupAccessibleSocket(t *testing.T) { + dir := t.TempDir() + ln, _, err := Listen(dir) + require.NoError(t, err) + t.Cleanup(func() { _ = ln.Close() }) + + require.NoError(t, os.Chmod(SocketPath(dir), 0o660)) + err = ValidateSocket(SocketPath(dir)) + require.Error(t, err) + assert.ErrorIs(t, err, ErrUnsafePath) +} + +func TestRemoveOwnedSocketRemovesBoundSocket(t *testing.T) { + dir := filepath.Join(t.TempDir(), "gordon") + ln, info, err := Listen(dir) + require.NoError(t, err) + t.Cleanup(func() { _ = ln.Close() }) + + require.NoError(t, RemoveOwnedSocket(SocketPath(dir), info)) + assert.NoFileExists(t, SocketPath(dir)) +} + +func TestRemoveOwnedSocketLeavesReplacement(t *testing.T) { + dir := filepath.Join(t.TempDir(), "gordon") + ln, info, err := Listen(dir) + require.NoError(t, err) + t.Cleanup(func() { _ = ln.Close() }) + + path := SocketPath(dir) + require.NoError(t, os.Remove(path)) + require.NoError(t, os.WriteFile(path, []byte("replacement"), 0o600)) + + err = RemoveOwnedSocket(path, info) + require.Error(t, err) + assert.ErrorIs(t, err, ErrUnsafePath) + assert.FileExists(t, path) +} + +func TestRemoveOwnedSocketIgnoresMissingPath(t *testing.T) { + assert.NoError(t, RemoveOwnedSocket(filepath.Join(t.TempDir(), "absent.sock"), nil)) +} + +func TestClientDirCandidatesOrderAndDedupe(t *testing.T) { + home := t.TempDir() + t.Setenv("HOME", home) + + uid := strconv.Itoa(os.Getuid()) + systemDir := filepath.Join("/run/user", uid, "gordon") + + t.Run("xdg coincides with the systemd default", func(t *testing.T) { + t.Setenv("XDG_RUNTIME_DIR", filepath.Join("/run/user", uid)) + assert.Equal(t, []string{systemDir, filepath.Join(home, ".gordon", "run")}, ClientDirCandidates()) + }) + + t.Run("xdg differs from the systemd default", func(t *testing.T) { + xdg := t.TempDir() + t.Setenv("XDG_RUNTIME_DIR", xdg) + assert.Equal(t, []string{ + filepath.Join(xdg, "gordon"), + systemDir, + filepath.Join(home, ".gordon", "run"), + }, ClientDirCandidates()) + }) +} + +func TestDialContextRejectsPeerUIDMismatchBeforeUse(t *testing.T) { + dir := filepath.Join(t.TempDir(), "gordon") + ln, _, err := Listen(dir) + require.NoError(t, err) + t.Cleanup(func() { _ = ln.Close() }) + + accepted := make(chan net.Conn, 1) + go func() { + conn, acceptErr := ln.Accept() + if acceptErr == nil { + accepted <- conn + } + }() + + _, err = dialContextForUID(context.Background(), SocketPath(dir), uint32(os.Geteuid()+1)) + require.Error(t, err) + assert.ErrorIs(t, err, ErrPeerUIDMismatch) + + select { + case conn := <-accepted: + defer conn.Close() + require.NoError(t, conn.SetReadDeadline(time.Now().Add(time.Second))) + buf := make([]byte, 1) + n, readErr := conn.Read(buf) + assert.Zero(t, n, "no request bytes may be sent before peer authentication") + assert.Error(t, readErr) + case <-time.After(time.Second): + t.Fatal("server did not accept authenticated connection attempt") + } +} + +func TestListenRefusesForeignOwnedSocketDirectory(t *testing.T) { + if os.Geteuid() != 0 { + t.Skip("requires root to change directory ownership") + } + dir := filepath.Join(t.TempDir(), "gordon") + require.NoError(t, os.Mkdir(dir, 0o700)) + require.NoError(t, os.Chown(dir, 65534, 65534)) + + _, _, err := Listen(dir) + require.Error(t, err) +} diff --git a/internal/adapters/localadmin/peer_linux.go b/internal/adapters/localadmin/peer_linux.go new file mode 100644 index 000000000..b31e1df82 --- /dev/null +++ b/internal/adapters/localadmin/peer_linux.go @@ -0,0 +1,80 @@ +//go:build linux + +package localadmin + +import ( + "context" + "errors" + "fmt" + "net" + "os" + "path/filepath" + "time" + + "golang.org/x/sys/unix" +) + +// ErrPeerUIDMismatch marks a Unix socket whose listening process does not run +// as the current effective user. +var ErrPeerUIDMismatch = errors.New("local admin peer uid mismatch") + +// DialContext connects to path and authenticates the listening process before +// returning the connection to an HTTP transport. +func DialContext(ctx context.Context, path string) (net.Conn, error) { + euid := os.Geteuid() + if euid < 0 { + return nil, fmt.Errorf("authenticate local admin peer: invalid effective uid %d", euid) + } + return dialContextForUID(ctx, path, uint32(euid)) // #nosec G115 -- euid is non-negative and Linux uid_t is uint32. +} + +func dialContextForUID(ctx context.Context, path string, expectedUID uint32) (net.Conn, error) { + if err := ValidateDir(filepath.Dir(path)); err != nil { + return nil, err + } + if err := ValidateSocket(path); err != nil { + return nil, err + } + + dialer := &net.Dialer{Timeout: 10 * time.Second, KeepAlive: 30 * time.Second} + conn, err := dialer.DialContext(ctx, "unix", path) + if err != nil { + return nil, err + } + unixConn, ok := conn.(*net.UnixConn) + if !ok { + _ = conn.Close() + return nil, fmt.Errorf("authenticate local admin peer: unexpected connection type %T", conn) + } + + uid, err := peerUID(unixConn) + if err != nil { + _ = conn.Close() + return nil, fmt.Errorf("authenticate local admin peer: %w", err) + } + if uid != expectedUID { + _ = conn.Close() + return nil, fmt.Errorf("%w: got %d, want %d", ErrPeerUIDMismatch, uid, expectedUID) + } + return conn, nil +} + +func peerUID(conn *net.UnixConn) (uint32, error) { + raw, err := conn.SyscallConn() + if err != nil { + return 0, err + } + var ( + cred *unix.Ucred + sockErr error + ) + if err := raw.Control(func(fd uintptr) { + cred, sockErr = unix.GetsockoptUcred(int(fd), unix.SOL_SOCKET, unix.SO_PEERCRED) + }); err != nil { + return 0, err + } + if sockErr != nil { + return 0, sockErr + } + return cred.Uid, nil +} diff --git a/internal/adapters/localadmin/peer_other.go b/internal/adapters/localadmin/peer_other.go new file mode 100644 index 000000000..d62ef8f1a --- /dev/null +++ b/internal/adapters/localadmin/peer_other.go @@ -0,0 +1,20 @@ +//go:build !linux + +package localadmin + +import ( + "context" + "errors" + "net" +) + +// ErrPeerUIDMismatch marks a Unix socket whose listening process does not run +// as the current effective user. +var ErrPeerUIDMismatch = errors.New("local admin peer uid mismatch") + +// DialContext fails closed on platforms where Gordon cannot authenticate the +// peer credentials of a Unix-domain socket. Remote administration remains +// available on these platforms. +func DialContext(context.Context, string) (net.Conn, error) { + return nil, errors.New("authenticate local admin peer: unsupported platform") +} diff --git a/internal/adapters/out/appsecrets/store.go b/internal/adapters/out/appsecrets/store.go new file mode 100644 index 000000000..18f209b35 --- /dev/null +++ b/internal/adapters/out/appsecrets/store.go @@ -0,0 +1,81 @@ +// Package appsecrets implements out.SecretWriter on pass for v3 app +// secrets at gordon/apps///. Values never pass +// through app state, diffs, logs, or backup metadata — only this +// explicit write path and the deployment-time read path carry them. +package appsecrets + +import ( + "context" + "fmt" + "os/exec" + "strings" + "time" + + "github.com/bnema/zerowrap" + + "github.com/bnema/gordon/internal/domain" +) + +// Store writes app secret values through the pass CLI. +type Store struct { + timeout time.Duration + log zerowrap.Logger +} + +// NewStore creates a pass-backed app secret writer. +func NewStore(log zerowrap.Logger) *Store { + return &Store{timeout: 30 * time.Second, log: log} +} + +// SetSecret implements out.SecretWriter. +func (s *Store) SetSecret(ctx context.Context, path, value string) error { + if err := validateAppSecretPath(path); err != nil { + return err + } + if strings.Contains(value, "\n") { + return fmt.Errorf("appsecrets: secret value must not contain newlines: %w", domain.ErrInvalidAppSpec) + } + if len(value) == 0 || len(value) > domain.MaxAppEnvValueLen { + return fmt.Errorf("appsecrets: secret value must be 1-%d bytes: %w", domain.MaxAppEnvValueLen, domain.ErrInvalidAppSpec) + } + timeoutCtx, cancel := context.WithTimeout(ctx, s.timeout) + defer cancel() + cmd := exec.CommandContext(timeoutCtx, "pass", "insert", "-m", "-f", path) //nolint:gosec // binary is constant ("pass"); path validated below + cmd.Stdin = strings.NewReader(value) + output, err := cmd.CombinedOutput() + if err != nil { + return fmt.Errorf("appsecrets: pass insert failed: %s: %w", strings.TrimSpace(string(output)), err) + } + s.log.Info().Str("path", path).Msg("appsecrets: secret written") + return nil +} + +// DeleteSecret implements out.SecretWriter. +func (s *Store) DeleteSecret(ctx context.Context, path string) error { + if err := validateAppSecretPath(path); err != nil { + return err + } + timeoutCtx, cancel := context.WithTimeout(ctx, s.timeout) + defer cancel() + cmd := exec.CommandContext(timeoutCtx, "pass", "rm", "-f", path) //nolint:gosec // binary is constant ("pass"); path validated below + output, err := cmd.CombinedOutput() + if err != nil { + return fmt.Errorf("appsecrets: pass rm failed: %s: %w", strings.TrimSpace(string(output)), err) + } + s.log.Info().Str("path", path).Msg("appsecrets: secret deleted") + return nil +} + +// validateAppSecretPath constrains writes to the app secret namespace. +func validateAppSecretPath(path string) error { + rest, ok := strings.CutPrefix(path, "gordon/apps/") + if !ok || rest == "" { + return fmt.Errorf("appsecrets: path %q must live under gordon/apps/: %w", path, domain.ErrInvalidAppSpec) + } + for _, part := range strings.Split(rest, "/") { + if part == "" || part == "." || part == ".." || strings.ContainsAny(part, " \t\n\r\\\"'$`*?[]{}()|&;<>!") { + return fmt.Errorf("appsecrets: path %q has invalid segment: %w", path, domain.ErrInvalidAppSpec) + } + } + return nil +} diff --git a/internal/adapters/out/appsecrets/store_test.go b/internal/adapters/out/appsecrets/store_test.go new file mode 100644 index 000000000..ce4f4fe78 --- /dev/null +++ b/internal/adapters/out/appsecrets/store_test.go @@ -0,0 +1,23 @@ +package appsecrets + +import ( + "context" + "testing" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +func TestValidateAppSecretPath(t *testing.T) { + store := NewStore(zerowrap.Default()) + ctx := context.Background() + + // Path validation fires before any pass invocation. + require.ErrorIs(t, func() error { return store.SetSecret(ctx, "elsewhere/x", "v") }(), domain.ErrInvalidAppSpec) + require.ErrorIs(t, func() error { return store.SetSecret(ctx, "gordon/apps/a/../b", "v") }(), domain.ErrInvalidAppSpec) + require.ErrorIs(t, func() error { return store.SetSecret(ctx, "gordon/apps/blog/web/db", "a\nb") }(), domain.ErrInvalidAppSpec) + assert.NotNil(t, store) +} diff --git a/internal/adapters/out/appstate/export_test.go b/internal/adapters/out/appstate/export_test.go new file mode 100644 index 000000000..2b6257daa --- /dev/null +++ b/internal/adapters/out/appstate/export_test.go @@ -0,0 +1,76 @@ +package appstate + +// Test-only hooks live here instead of the production API: they rewrite +// durable records to reproduce corrupt or version-skewed states that +// production code can never create. + +import ( + "errors" + "fmt" + + bolt "go.etcd.io/bbolt" + bolterrors "go.etcd.io/bbolt/errors" + + "github.com/bnema/gordon/internal/domain" +) + +// CorruptCheckpointForTest rewrites the checkpoint record. Tests use it +// to simulate corrupt or version-skewed state.db payloads. +func CorruptCheckpointForTest(s *Store, rewrite func([]byte) []byte) error { + return s.db.Update(func(tx *bolt.Tx) error { + bucket := tx.Bucket(bucketCheckpoint) + if bucket == nil { + return nil + } + raw := append([]byte(nil), bucket.Get(keyStoreRecord)...) + return bucket.Put(keyStoreRecord, rewrite(raw)) + }) +} + +// SeedDesiredRecordForTest writes an exact desired-record payload, creating +// the app bucket when absent. Tests use it to plant records written by an +// older Gordon, whose JSON omits fields the current schema has. +func SeedDesiredRecordForTest(s *Store, app string, raw []byte) error { + return s.db.Update(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, app, true) + if err != nil { + return err + } + return bucket.Put(keyDesired, raw) + }) +} + +// CorruptDesiredForTest rewrites one app's desired record. Tests use it +// to simulate a corrupt desired payload with other apps unaffected. +func CorruptDesiredForTest(s *Store, app string, rewrite func([]byte) []byte) error { + return s.db.Update(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, app, false) + if err != nil { + return err + } + if bucket == nil { + return nil + } + raw := append([]byte(nil), bucket.Get(keyDesired)...) + return bucket.Put(keyDesired, rewrite(raw)) + }) +} + +// ResetOwnershipHistoryForTest drops the incarnation-indexed ownership +// history and its migration marker. Tests use it to reproduce a +// pre-migration database that already holds current-shaped records. +func ResetOwnershipHistoryForTest(s *Store) error { + return s.db.Update(func(tx *bolt.Tx) error { + if err := tx.DeleteBucket(bucketOwnershipHistory); err != nil && !errors.Is(err, bolterrors.ErrBucketNotFound) { + return fmt.Errorf("appstate: reset ownership history: %w", domain.ErrAppStateIO) + } + meta := tx.Bucket(bucketMeta) + if meta == nil { + return nil + } + if err := meta.Delete(keyOwnershipMigration); err != nil { + return fmt.Errorf("appstate: reset ownership migration: %w", domain.ErrAppStateIO) + } + return nil + }) +} diff --git a/internal/adapters/out/appstate/protection.go b/internal/adapters/out/appstate/protection.go new file mode 100644 index 000000000..e3dafb78e --- /dev/null +++ b/internal/adapters/out/appstate/protection.go @@ -0,0 +1,429 @@ +package appstate + +import ( + "context" + "fmt" + + bolt "go.etcd.io/bbolt" + + "github.com/bnema/gordon/internal/domain" +) + +// ProtectionSnapshot implements out.PruneProtectionStore: one bbolt +// View transaction reads every durable fact prune plans against, so no +// candidate is ever judged against facts from a different moment. +// +// Anything that cannot be read completely is reported as an inventory +// gap instead of being silently omitted: an apparently unprotected +// snapshot must never be an artifact of a corrupt record. +func (s *Store) ProtectionSnapshot(ctx context.Context) (*domain.PruneProtectionSnapshot, error) { + if err := checkCtx(ctx); err != nil { + return nil, err + } + + snapshot := &domain.PruneProtectionSnapshot{} + if err := s.db.View(func(tx *bolt.Tx) error { + return readProtectionLocked(tx, snapshot) + }); err != nil { + return nil, err + } + + snapshot.Roots = dedupeRoots(snapshot.Roots) + snapshot.VolumeClaims = dedupeClaims(snapshot.VolumeClaims) + return snapshot, nil +} + +// readProtectionLocked fills snapshot from one consistent read view. +func readProtectionLocked(tx *bolt.Tx, snapshot *domain.PruneProtectionSnapshot) error { + if err := readOwnershipHistoryLocked(tx, snapshot); err != nil { + return err + } + // A missing history bucket is a real gap: ownership history is the + // only positive proof that a volume was ever Gordon's. + if tx.Bucket(bucketOwnershipHistory) == nil { + snapshot.Gaps = append(snapshot.Gaps, domain.InventoryGap{ + Source: domain.InventorySourceOwnership, + Reason: domain.PruneReasonUnknownInventory, + Detail: "ownership history bucket is missing", + }) + } + + apps := tx.Bucket(bucketApps) + if apps == nil { + return nil + } + return apps.ForEach(func(name, _ []byte) error { + bucket := apps.Bucket(name) + if bucket == nil { + return nil + } + // A bucket we cannot decode is reported, never skipped: prune + // must not treat an unreadable app as a removed app. + readAppProtectionLocked(bucket, string(name), snapshot) + return nil + }) +} + +// readAppProtectionLocked collects desired, active, apply, operation, +// inhibition, and ownership facts of one app. +func readAppProtectionLocked(bucket *bolt.Bucket, app string, snapshot *domain.PruneProtectionSnapshot) { + record, err := ensureAppRecordLocked(bucket, app, false) + if err != nil { + snapshot.Gaps = append(snapshot.Gaps, domain.InventoryGap{ + Source: domain.InventorySourceAppState, + Reason: domain.PruneReasonUnknownInventory, + Detail: fmt.Sprintf("app %s: app record unreadable", app), + }) + // The same failure hides this app's ownership claims, so volume + // planning must fail closed too. + snapshot.Gaps = append(snapshot.Gaps, domain.InventoryGap{ + Source: domain.InventorySourceOwnership, + Reason: domain.PruneReasonUnknownInventory, + Detail: fmt.Sprintf("app %s: ownership unreadable with its app record", app), + }) + return + } + appID := record.ID + + readDesiredProtectionLocked(bucket, app, snapshot) + readActiveProtectionLocked(bucket, app, snapshot) + readApplyIntentProtectionLocked(bucket, app, snapshot) + readOperationProtectionLocked(bucket, app, snapshot) + readInhibitionProtectionLocked(bucket, app, snapshot) + readAppOwnershipProtectionLocked(bucket, app, appID, snapshot) +} + +func readDesiredProtectionLocked(bucket *bolt.Bucket, app string, snapshot *domain.PruneProtectionSnapshot) { + var desired domain.AppDesiredRevision + found, err := getRecord(bucket, keyDesired, &desired, "desired") + if err != nil { + snapshot.Gaps = append(snapshot.Gaps, domain.InventoryGap{ + Source: domain.InventorySourceAppState, + Reason: domain.PruneReasonUnknownInventory, + Detail: fmt.Sprintf("app %s: desired record unreadable", app), + }) + return + } + if !found { + return + } + owner := app + "@desired" + for _, service := range desired.Spec.Services { + snapshot.Roots = append(snapshot.Roots, + domain.ImageRefRoots(domain.ProtectionDesiredRevision, service.Image, owner+"."+service.Name)...) + } +} + +func readActiveProtectionLocked(bucket *bolt.Bucket, app string, snapshot *domain.PruneProtectionSnapshot) { + var active domain.AppActive + found, err := getRecord(bucket, keyActive, &active, "active") + if err != nil { + snapshot.Gaps = append(snapshot.Gaps, domain.InventoryGap{ + Source: domain.InventorySourceAppState, + Reason: domain.PruneReasonUnknownInventory, + Detail: fmt.Sprintf("app %s: active record unreadable", app), + }) + return + } + if !found { + return + } + // ACTIVE covers running, stopped, and partially converged services + // alike: every entry names an image that must survive. + for serviceName, service := range active.Services { + owner := app + "@active." + serviceName + snapshot.Roots = append(snapshot.Roots, domain.ImageRefRoots(domain.ProtectionActiveService, service.Image, owner)...) + if service.Digest != "" { + snapshot.Roots = append(snapshot.Roots, domain.ProtectionRoot{ + Kind: domain.ProtectionActiveService, Ref: service.Digest, + Repository: domain.ImageRefRepository(service.Image), Owner: owner, + }) + } + if service.Container != "" { + snapshot.Roots = append(snapshot.Roots, domain.ProtectionRoot{ + Kind: domain.ProtectionContainerUse, Ref: service.Container, Owner: owner, + }) + } + } +} + +func readApplyIntentProtectionLocked(bucket *bolt.Bucket, app string, snapshot *domain.PruneProtectionSnapshot) { + keys, err := listKeys(bucket, subIntents, false) + if err != nil { + snapshot.Gaps = append(snapshot.Gaps, domain.InventoryGap{ + Source: domain.InventorySourceAppState, + Reason: domain.PruneReasonUnknownInventory, + Detail: fmt.Sprintf("app %s: apply intents unreadable", app), + }) + return + } + for _, intentID := range keys { + intent, err := loadApplyIntentLocked(bucket, app, intentID) + if err != nil { + snapshot.Gaps = append(snapshot.Gaps, domain.InventoryGap{ + Source: domain.InventorySourceAppState, + Reason: domain.PruneReasonUnknownInventory, + Detail: fmt.Sprintf("app %s: apply intent %s unreadable", app, intentID), + }) + continue + } + // Only staged and committed intents can still materialize; an + // applied intent is already covered by ACTIVE. + if intent.State != domain.AppIntentStaged && intent.State != domain.AppIntentCommitted { + continue + } + owner := app + "@intent." + intentID + for _, service := range intent.Spec.Services { + snapshot.Roots = append(snapshot.Roots, + domain.ImageRefRoots(domain.ProtectionApplyIntent, service.Image, owner+"."+service.Name)...) + } + } +} + +func readOperationProtectionLocked(bucket *bolt.Bucket, app string, snapshot *domain.PruneProtectionSnapshot) { + keys, err := listKeys(bucket, subOps, false) + if err != nil { + snapshot.Gaps = append(snapshot.Gaps, domain.InventoryGap{ + Source: domain.InventorySourceOperation, + Reason: domain.PruneReasonUnknownInventory, + Detail: fmt.Sprintf("app %s: operation journal unreadable", app), + }) + return + } + ops, err := subBucket(bucket, subOps, false) + if err != nil { + return + } + if ops == nil { + return + } + for _, opID := range keys { + var op domain.AppOperation + found, err := getRecord(ops, []byte(opID), &op, "operation") + if err != nil || !found { + snapshot.Gaps = append(snapshot.Gaps, domain.InventoryGap{ + Source: domain.InventorySourceOperation, + Reason: domain.PruneReasonUnknownInventory, + Detail: fmt.Sprintf("app %s: operation %s unreadable", app, opID), + }) + continue + } + // An operation with no terminal outcome may still be running: + // every image it named is protected. + if op.Outcome != "" { + continue + } + owner := app + "@op." + opID + for _, step := range op.Steps { + if step.Digest != "" { + snapshot.Roots = append(snapshot.Roots, domain.ProtectionRoot{ + Kind: domain.ProtectionOperation, Ref: step.Digest, + Repository: domain.ImageRefRepository(step.Image), Owner: owner, + }) + } + snapshot.Roots = append(snapshot.Roots, + domain.ImageRefRoots(domain.ProtectionOperation, step.Image, owner)...) + } + } +} + +func readInhibitionProtectionLocked(bucket *bolt.Bucket, app string, snapshot *domain.PruneProtectionSnapshot) { + var inhibitions []domain.AppRecoveryInhibition + found, err := getRecord(bucket, keyInhibitions, &inhibitions, "recovery inhibitions") + if err != nil { + snapshot.Gaps = append(snapshot.Gaps, domain.InventoryGap{ + Source: domain.InventorySourceAppState, + Reason: domain.PruneReasonUnknownInventory, + Detail: fmt.Sprintf("app %s: recovery inhibitions unreadable", app), + }) + return + } + if !found { + return + } + for _, inhibition := range inhibitions { + if inhibition.ContainerID == "" { + continue + } + snapshot.Roots = append(snapshot.Roots, domain.ProtectionRoot{ + Kind: domain.ProtectionRecoveryInhibition, + Ref: inhibition.ContainerID, + Owner: app + "@" + inhibition.Service, + }) + } +} + +// readAppOwnershipProtectionLocked records the current incarnation's +// claims. It is the per-app record, which always reflects the live +// incarnation; historical claims come from the ownership-history +// bucket. +func readAppOwnershipProtectionLocked(bucket *bolt.Bucket, app, appID string, snapshot *domain.PruneProtectionSnapshot) { + var ownership domain.AppOwnership + found, err := getRecord(bucket, keyOwnership, &ownership, "ownership") + if err != nil { + snapshot.Gaps = append(snapshot.Gaps, domain.InventoryGap{ + Source: domain.InventorySourceOwnership, + Reason: domain.PruneReasonUnknownInventory, + Detail: fmt.Sprintf("app %s: ownership record unreadable", app), + }) + return + } + if !found { + return + } + // The ownership record's own ID is authoritative: the app record's + // UUID can lag behind it after a re-deploy. + claimAppID := ownership.ID + if claimAppID == "" { + claimAppID = appID + } + claims, gap := claimsForOwnership(app, claimAppID, ownership) + if gap != nil { + snapshot.Gaps = append(snapshot.Gaps, *gap) + } + snapshot.VolumeClaims = append(snapshot.VolumeClaims, claims...) + snapshot.ImageClaims = append(snapshot.ImageClaims, imageClaimsForOwnership(app, claimAppID, ownership)...) +} + +// readOwnershipHistoryLocked records every historical incarnation's +// claims. A reused app name never overwrites them, so a volume retained +// by a removed app stays protected after the name is taken again. +func readOwnershipHistoryLocked(tx *bolt.Tx, snapshot *domain.PruneProtectionSnapshot) error { + history := tx.Bucket(bucketOwnershipHistory) + if history == nil { + return nil + } + return history.ForEach(func(incarnation, value []byte) error { + if value == nil { + return nil + } + var ownership domain.AppOwnership + if err := unmarshal(value, &ownership, "ownership history"); err != nil { + snapshot.Gaps = append(snapshot.Gaps, domain.InventoryGap{ + Source: domain.InventorySourceOwnership, + Reason: domain.PruneReasonUnknownInventory, + Detail: fmt.Sprintf("ownership history %s unreadable", incarnation), + }) + return nil + } + claimApp := ownership.App + if claimApp == "" { + claimApp = string(incarnation) + } + claims, gap := claimsForOwnership(claimApp, ownership.ID, ownership) + if gap != nil { + snapshot.Gaps = append(snapshot.Gaps, *gap) + return nil + } + snapshot.VolumeClaims = append(snapshot.VolumeClaims, claims...) + snapshot.ImageClaims = append(snapshot.ImageClaims, imageClaimsForOwnership(claimApp, ownership.ID, ownership)...) + return nil + }) +} + +// imageClaimsForOwnership converts one ownership record into image claims. +// An absent or unrecognized lifecycle state maps conservatively to +// retained: only an explicit released record can ever be eligible. +func imageClaimsForOwnership(app, appID string, ownership domain.AppOwnership) []domain.ImageClaim { + claims := make([]domain.ImageClaim, 0, len(ownership.Images)) + for _, image := range ownership.Images { + if image.Reference == "" && image.Digest == "" { + continue + } + claims = append(claims, domain.ImageClaim{ + Reference: image.Reference, + Digest: image.Digest, + App: app, + AppID: appID, + Service: image.Service, + State: mapOwnedImageState(image.State), + }) + } + return claims +} + +// mapOwnedImageState maps a durable owned-image state to a claim state. +// Unknown values protect. +func mapOwnedImageState(state string) domain.VolumeClaimState { + switch state { + case domain.AppResourceReleased: + return domain.VolumeClaimReleased + case domain.AppResourceAttached: + return domain.VolumeClaimAttached + default: + return domain.VolumeClaimRetained + } +} + +// claimsForOwnership converts one ownership record into volume claims. +// An absent or unrecognized lifecycle state maps conservatively to +// retained: only an explicit released record can ever be eligible. +func claimsForOwnership(app, appID string, ownership domain.AppOwnership) ([]domain.VolumeClaim, *domain.InventoryGap) { + claims := make([]domain.VolumeClaim, 0, len(ownership.Volumes)) + for _, volume := range ownership.Volumes { + name := volume.RuntimeName + if name == "" { + name = volume.Name + } + if name == "" { + continue + } + claims = append(claims, domain.VolumeClaim{ + Name: name, + App: app, + AppID: appID, + Service: volume.Service, + State: mapOwnedVolumeState(volume.State), + }) + } + return claims, nil +} + +// mapOwnedVolumeState maps a durable owned-volume state to a claim +// state. Unknown values protect. +func mapOwnedVolumeState(state string) domain.VolumeClaimState { + switch state { + case domain.AppResourceReleased: + return domain.VolumeClaimReleased + case domain.AppResourceAttached: + return domain.VolumeClaimAttached + default: + return domain.VolumeClaimRetained + } +} + +// dedupeRoots drops duplicate roots, preserving first-seen order. +func dedupeRoots(roots []domain.ProtectionRoot) []domain.ProtectionRoot { + seen := make(map[string]struct{}, len(roots)) + out := make([]domain.ProtectionRoot, 0, len(roots)) + for _, root := range roots { + if !root.Valid() { + continue + } + key := string(root.Kind) + "\x00" + root.Ref + "\x00" + root.Repository + if _, dup := seen[key]; dup { + continue + } + seen[key] = struct{}{} + out = append(out, root) + } + return out +} + +// dedupeClaims drops identical claims, preserving first-seen order. The +// lifecycle state is part of the identity: a released claim and an +// attached claim for the same runtime volume are different facts, and +// collapsing them would lose the one that protects the volume. +func dedupeClaims(claims []domain.VolumeClaim) []domain.VolumeClaim { + seen := make(map[string]struct{}, len(claims)) + out := make([]domain.VolumeClaim, 0, len(claims)) + for _, claim := range claims { + key := claim.Name + "\x00" + claim.AppID + "\x00" + claim.App + "\x00" + claim.Service + "\x00" + string(claim.State) + if _, dup := seen[key]; dup { + continue + } + seen[key] = struct{}{} + out = append(out, claim) + } + return out +} diff --git a/internal/adapters/out/appstate/protection_test.go b/internal/adapters/out/appstate/protection_test.go new file mode 100644 index 000000000..2dda30bac --- /dev/null +++ b/internal/adapters/out/appstate/protection_test.go @@ -0,0 +1,378 @@ +package appstate_test + +import ( + "context" + "strings" + "sync" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/out/appstate" + "github.com/bnema/gordon/internal/domain" +) + +func hasRoot(t *testing.T, snapshot *domain.PruneProtectionSnapshot, kind domain.ProtectionRootKind, ref string) bool { + t.Helper() + for _, root := range snapshot.Roots { + if root.Kind == kind && root.Ref == ref { + return true + } + } + return false +} + +func tagRefKey(t *testing.T, repository, tag string) string { + t.Helper() + ref, err := domain.NewRegistryTagRef(repository, tag) + require.NoError(t, err) + return ref.Key() +} + +// TestStore_ProtectionSnapshotCoversEveryRoot proves one snapshot +// transaction collects desired, active, inhibition, apply, operation, +// and ownership facts together. +func TestStore_ProtectionSnapshotCoversEveryRoot(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + desired := testIntent("shop", "rev-1") + desired.Spec.Services[0].Image = "registry.example/shop:1" + require.NoError(t, store.StageApply(ctx, desired)) + require.NoError(t, store.CommitApply(ctx, "shop", desired.Intent)) + require.NoError(t, store.MaterializeApply(ctx, "shop", desired.Intent)) + + require.NoError(t, store.SaveActive(ctx, domain.AppActive{ + App: "shop", + Services: map[string]domain.AppEffectiveService{ + "web": { + EffectiveRevision: "rev-1", + Image: "registry.example/shop:1", + Digest: "sha256:" + strings.Repeat("ab", 32), + Container: "c-live", + }, + }, + })) + require.NoError(t, store.SaveRecoveryInhibition(ctx, domain.AppRecoveryInhibition{ + App: "shop", Service: "web", ContainerID: "c-inhibited", + Reason: domain.AppInhibitReplacementPending, + })) + + // A staged intent may still materialize. + staged := testIntent("shop", "rev-2") + staged.Intent = "apply-staged" + staged.Spec.Services[0].Image = "registry.example/shop:2" + require.NoError(t, store.StageApply(ctx, staged)) + + // An unfinished operation pins its own image. + require.NoError(t, store.SaveOperation(ctx, domain.AppOperation{ + Op: "op-open", Kind: "deploy", App: "shop", + Steps: []domain.AppOperationStep{ + {ID: "preflight", State: domain.AppStepSucceeded}, + {ID: "service.web.replace", State: domain.AppStepPending, + Digest: "sha256:" + strings.Repeat("cd", 32), + Image: "registry.example/shop:3"}, + }, + })) + + // A finished operation pins nothing. + require.NoError(t, store.SaveOperation(ctx, domain.AppOperation{ + Op: "op-done", Kind: "deploy", App: "shop", Outcome: domain.AppOutcomeSuccess, + Steps: []domain.AppOperationStep{ + {ID: "service.web.replace", State: domain.AppStepSucceeded, + Digest: "sha256:" + strings.Repeat("ef", 32), + Image: "registry.example/shop:4"}, + }, + })) + + require.NoError(t, store.SaveOwnership(ctx, domain.AppOwnership{ + App: "shop", ID: "app-uuid-shop", + Volumes: []domain.AppOwnedVolume{ + {Name: "data", Service: "web", RuntimeName: "gordon-shop--web--vol--data", State: domain.AppResourceAttached}, + }, + })) + + snapshot, err := store.ProtectionSnapshot(ctx) + require.NoError(t, err) + require.True(t, snapshot.Complete(), "snapshot gaps: %+v", snapshot.Gaps) + + assert.True(t, hasRoot(t, snapshot, domain.ProtectionDesiredRevision, tagRefKey(t, "registry.example/shop", "1")), + "desired revision root missing") + assert.True(t, hasRoot(t, snapshot, domain.ProtectionActiveService, tagRefKey(t, "registry.example/shop", "1")), + "active service root missing") + assert.True(t, hasRoot(t, snapshot, domain.ProtectionActiveService, "sha256:"+strings.Repeat("ab", 32)), + "active digest root missing") + assert.True(t, hasRoot(t, snapshot, domain.ProtectionContainerUse, "c-live"), + "active container root missing") + assert.True(t, hasRoot(t, snapshot, domain.ProtectionRecoveryInhibition, "c-inhibited"), + "recovery inhibition root missing") + assert.True(t, hasRoot(t, snapshot, domain.ProtectionApplyIntent, tagRefKey(t, "registry.example/shop", "2")), + "staged apply intent root missing") + assert.True(t, hasRoot(t, snapshot, domain.ProtectionOperation, "sha256:"+strings.Repeat("cd", 32)), + "unfinished operation digest root missing") + assert.True(t, hasRoot(t, snapshot, domain.ProtectionOperation, tagRefKey(t, "registry.example/shop", "3")), + "unfinished operation image root missing") + assert.False(t, hasRoot(t, snapshot, domain.ProtectionOperation, "sha256:"+strings.Repeat("ef", 32)), + "finished operation must not pin content") + assert.False(t, hasRoot(t, snapshot, domain.ProtectionOperation, tagRefKey(t, "registry.example/shop", "4")), + "finished operation must not pin content") + + claim, ok := snapshot.VolumeClaimFor("gordon-shop--web--vol--data") + require.True(t, ok) + assert.Equal(t, domain.VolumeClaimAttached, claim.State) + assert.Equal(t, "shop", claim.App) + assert.True(t, snapshot.VolumeProtects("gordon-shop--web--vol--data")) +} + +// TestStore_ProtectionSnapshotStoppedAppKeepsActive proves a stopped app +// with no running container still protects its image and container. +func TestStore_ProtectionSnapshotStoppedAppKeepsActive(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + require.NoError(t, store.SaveActive(ctx, domain.AppActive{ + App: "blog", + StopIntent: true, + Services: map[string]domain.AppEffectiveService{ + "web": {EffectiveRevision: "rev-1", Image: "registry.example/blog:7", Container: "c-stopped"}, + }, + })) + require.NoError(t, store.SaveIntent(ctx, domain.AppStopIntent{App: "blog", Stopped: true, UpdatedBy: "test"})) + + snapshot, err := store.ProtectionSnapshot(ctx) + require.NoError(t, err) + assert.True(t, hasRoot(t, snapshot, domain.ProtectionActiveService, tagRefKey(t, "registry.example/blog", "7"))) + assert.True(t, hasRoot(t, snapshot, domain.ProtectionContainerUse, "c-stopped")) +} + +// TestStore_ProtectionSnapshotKeepsHistoricalIncarnations proves a +// reused app name never loses the earlier incarnation's retained +// resources. +func TestStore_ProtectionSnapshotKeepsHistoricalIncarnations(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + const volume = "gordon-blog--db--vol--d" + require.NoError(t, store.SaveOwnership(ctx, domain.AppOwnership{ + App: "blog", ID: "app-uuid-1", + Volumes: []domain.AppOwnedVolume{ + {Name: "d", Service: "db", RuntimeName: volume, State: domain.AppResourceRetained}, + }, + })) + // Remove frees the name; a new incarnation takes it and owns + // nothing yet. + require.NoError(t, store.SaveOwnership(ctx, domain.AppOwnership{App: "blog", ID: "app-uuid-2"})) + + snapshot, err := store.ProtectionSnapshot(ctx) + require.NoError(t, err) + require.True(t, snapshot.VolumeProtects(volume), "historical retained claim was lost") + claims := snapshot.VolumeClaimsFor(volume) + require.Len(t, claims, 1) + assert.Equal(t, domain.VolumeClaimRetained, claims[0].State) + assert.Equal(t, "app-uuid-1", claims[0].AppID) +} + +// TestStore_ProtectionSnapshotKeepsDifferentClaimStates proves two +// incarnations' claims on one volume are distinct facts: a released +// claim must never overwrite or hide another incarnation's attached +// claim for the same runtime volume. +func TestStore_ProtectionSnapshotKeepsDifferentClaimStates(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + const volume = "gordon-app--web--vol--data" + // The old incarnation released the volume. + require.NoError(t, store.SaveOwnership(ctx, domain.AppOwnership{ + App: "app", ID: "app-uuid-old", + Volumes: []domain.AppOwnedVolume{ + {Name: "data", Service: "web", RuntimeName: volume, State: domain.AppResourceReleased}, + }, + })) + // A new incarnation attached it again. + require.NoError(t, store.SaveOwnership(ctx, domain.AppOwnership{ + App: "app", ID: "app-uuid-new", + Volumes: []domain.AppOwnedVolume{ + {Name: "data", Service: "web", RuntimeName: volume, State: domain.AppResourceAttached}, + }, + })) + + snapshot, err := store.ProtectionSnapshot(ctx) + require.NoError(t, err) + claims := snapshot.VolumeClaimsFor(volume) + require.Len(t, claims, 2, "both incarnations' claims must survive") + assert.True(t, snapshot.VolumeProtects(volume), "the attached claim must still protect the volume") +} + +// TestStore_ProtectionSnapshotReleasedIsTheOnlyDeletableState proves +// only an explicit released record can ever make a volume eligible. +func TestStore_ProtectionSnapshotReleasedIsTheOnlyDeletableState(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + const volume = "gordon-shop--web--vol--data" + require.NoError(t, store.SaveOwnership(ctx, domain.AppOwnership{ + App: "shop", ID: "app-uuid-shop", + Volumes: []domain.AppOwnedVolume{ + {Name: "data", Service: "web", RuntimeName: volume, State: domain.AppResourceReleased}, + }, + })) + + snapshot, err := store.ProtectionSnapshot(ctx) + require.NoError(t, err) + assert.False(t, snapshot.VolumeProtects(volume)) + claim, ok := snapshot.VolumeClaimFor(volume) + require.True(t, ok) + assert.Equal(t, domain.VolumeClaimReleased, claim.State) +} + +// TestStore_ProtectionSnapshotUnknownStateIsRetained proves an +// unrecognized lifecycle state never becomes deletable. +func TestStore_ProtectionSnapshotUnknownStateIsRetained(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + const volume = "gordon-shop--web--vol--data" + require.NoError(t, store.SaveOwnership(ctx, domain.AppOwnership{ + App: "shop", ID: "app-uuid-shop", + Volumes: []domain.AppOwnedVolume{ + {Name: "data", Service: "web", RuntimeName: volume}, + }, + })) + + snapshot, err := store.ProtectionSnapshot(ctx) + require.NoError(t, err) + assert.True(t, snapshot.VolumeProtects(volume), "empty state must map to retained") +} + +// TestStore_ProtectionSnapshotMigrationBackfillsHistory proves the +// one-time migration reconstructs history for records written before +// the history bucket existed, and that reopening is idempotent. +func TestStore_ProtectionSnapshotMigrationBackfillsHistory(t *testing.T) { + ctx := context.Background() + dir := t.TempDir() + store, err := appstate.NewStore(dir, testLogger()) + require.NoError(t, err) + + const volume = "gordon-blog--db--vol--d" + require.NoError(t, store.SaveOwnership(ctx, domain.AppOwnership{ + App: "blog", ID: "app-uuid-1", + Volumes: []domain.AppOwnedVolume{ + {Name: "d", Service: "db", RuntimeName: volume, State: domain.AppResourceRetained}, + }, + })) + + // Reproduce a pre-migration database: per-app records exist, the + // incarnation index does not. + require.NoError(t, appstate.ResetOwnershipHistoryForTest(store)) + snapshot, err := store.ProtectionSnapshot(ctx) + require.NoError(t, err) + assert.False(t, snapshot.Complete(), "missing history bucket must be reported") + assert.True(t, snapshot.VolumeProtects(volume), "per-app record still protects") + require.NoError(t, store.Close()) + + // Reopening runs the migration once and reuses the existing index + // on every later open. + reopened, err := appstate.NewStore(dir, testLogger()) + require.NoError(t, err) + t.Cleanup(func() { assert.NoError(t, reopened.Close()) }) + snapshot, err = reopened.ProtectionSnapshot(ctx) + require.NoError(t, err) + assert.True(t, snapshot.Complete(), "snapshot gaps after migration: %+v", snapshot.Gaps) + assert.True(t, snapshot.VolumeProtects(volume)) + + require.NoError(t, reopened.Close()) + third, err := appstate.NewStore(dir, testLogger()) + require.NoError(t, err) + t.Cleanup(func() { assert.NoError(t, third.Close()) }) + ownership, err := third.LoadOwnership(ctx, "blog") + require.NoError(t, err) + assert.Equal(t, "app-uuid-1", ownership.ID) + snapshot, err = third.ProtectionSnapshot(ctx) + require.NoError(t, err) + assert.True(t, snapshot.VolumeProtects(volume)) +} + +// TestStore_ProtectionSnapshotReportsCorruptRecords proves unreadable +// state becomes a gap instead of an apparently unprotected snapshot. +func TestStore_ProtectionSnapshotReportsCorruptRecords(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + desired := testIntent("shop", "rev-1") + require.NoError(t, store.StageApply(ctx, desired)) + require.NoError(t, store.CommitApply(ctx, "shop", desired.Intent)) + require.NoError(t, store.MaterializeApply(ctx, "shop", desired.Intent)) + require.NoError(t, appstate.CorruptDesiredForTest(store, "shop", func([]byte) []byte { + return []byte("{not json") + })) + + snapshot, err := store.ProtectionSnapshot(ctx) + require.NoError(t, err) + assert.False(t, snapshot.Complete()) + require.NotEmpty(t, snapshot.Gaps) + assert.Equal(t, domain.InventorySourceAppState, snapshot.Gaps[0].Source) +} + +// TestStore_ProtectionSnapshotConcurrentWrites proves a snapshot is +// atomic against concurrent state writes. +func TestStore_ProtectionSnapshotConcurrentWrites(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + const rounds = 40 + var wg sync.WaitGroup + for index := 0; index < rounds; index++ { + wg.Add(1) + go func(index int) { + defer wg.Done() + app := "app-" + itoa2(index%4) + active := domain.AppActive{ + App: app, + Services: map[string]domain.AppEffectiveService{ + "web": {EffectiveRevision: "rev-" + itoa2(index), Image: "registry.example/" + app + ":1"}, + }, + } + assert.NoError(t, store.SaveActive(ctx, active)) + assert.NoError(t, store.SaveOwnership(ctx, domain.AppOwnership{ + App: app, ID: "app-uuid-" + itoa2(index%4), + Volumes: []domain.AppOwnedVolume{ + {Name: "data", Service: "web", RuntimeName: "gordon-" + itoa2(index%4) + "--web--vol--data"}, + }, + })) + snapshot, err := store.ProtectionSnapshot(ctx) + assert.NoError(t, err) + assert.NoError(t, snapshotValidate(snapshot)) + }(index) + } + wg.Wait() +} + +// snapshotValidate checks the snapshot invariant callers rely on: every +// root and claim is well formed. +func snapshotValidate(snapshot *domain.PruneProtectionSnapshot) error { + for _, root := range snapshot.Roots { + if !root.Valid() { + return assert.AnError + } + } + for _, gap := range snapshot.Gaps { + if !gap.Valid() { + return assert.AnError + } + } + return nil +} + +// TestStore_ProtectionSnapshotCancellation proves a canceled context +// never opens a read transaction. +func TestStore_ProtectionSnapshotCancellation(t *testing.T) { + store := newTestStore(t) + ctx, cancel := context.WithCancel(context.Background()) + cancel() + _, err := store.ProtectionSnapshot(ctx) + assert.ErrorIs(t, err, context.Canceled) +} + +var _ = time.Second diff --git a/internal/adapters/out/appstate/store.go b/internal/adapters/out/appstate/store.go new file mode 100644 index 000000000..bfc5eb1de --- /dev/null +++ b/internal/adapters/out/appstate/store.go @@ -0,0 +1,1662 @@ +// Package appstate implements the out.AppState boundary on bbolt: +// one state.db file with per-app buckets, single-writer transactions, +// and crash-safe commit semantics from bbolt itself. It replaces the +// prepared multi-file JSON adapter (no compatibility layer). +// It never reads secret values and never touches OCI manifest storage. +package appstate + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "os" + "path/filepath" + "sort" + "time" + + "github.com/bnema/zerowrap" + "github.com/google/uuid" + bolt "go.etcd.io/bbolt" + bolterrors "go.etcd.io/bbolt/errors" + + "github.com/bnema/gordon/internal/domain" +) + +// Bucket layout inside state.db. +var ( + bucketMeta = []byte("meta") + bucketCheckpoint = []byte("checkpoint") + bucketApps = []byte("apps") + bucketOwnershipHistory = []byte("ownership_history") + keyStoreRecord = []byte("store") + keyAppRecord = []byte("app") + keyDesired = []byte("desired") + keyActive = []byte("active") + keyIntent = []byte("intent") + keyOwnership = []byte("ownership") + keyInhibitions = []byte("recovery_inhibitions") + keyOwnershipMigration = []byte("ownership_history_v1") + subRevisions = []byte("revisions") + subIntents = []byte("intents") + subOps = []byte("ops") +) + +// Store layout constants. +const ( + appsDir = "apps" + stateFile = "state.db" + dirPerm = 0o750 +) + +// appRecord tracks the stable internal UUID for one public app name. +type appRecord struct { + ID string `json:"id"` + Name string `json:"name"` +} + +// Store is a bbolt-backed out.AppState implementation. +type Store struct { + db *bolt.DB + log zerowrap.Logger + // retention bounds unreferenced-revision GC per app. Zero means + // the compiled default (AppDefaultRevisionRetention). + retention int +} + +// WithRevisionRetention overrides the unreferenced-revision GC bound. +// Values <= 0 restore the compiled default. +func (s *Store) WithRevisionRetention(n int) *Store { + if n > 0 { + s.retention = n + } else { + s.retention = 0 + } + return s +} + +// Retention returns the effective unreferenced-revision GC bound. +func (s *Store) Retention() int { + if s.retention > 0 { + return s.retention + } + return domain.AppDefaultRevisionRetention +} + +// NewStore creates the store rooted at dataDir (…/apps/state.db). +func NewStore(dataDir string, log zerowrap.Logger) (*Store, error) { + root := filepath.Join(dataDir, appsDir) + if err := os.MkdirAll(root, dirPerm); err != nil { + return nil, fmt.Errorf("appstate: create store root: %w", domain.ErrAppStateIO) + } + db, err := bolt.Open(filepath.Join(root, stateFile), 0o600, &bolt.Options{Timeout: 5 * time.Second}) + if err != nil { + return nil, fmt.Errorf("appstate: open state db: %w", domain.ErrAppStateIO) + } + store := &Store{db: db, log: log} + if err := store.init(); err != nil { + _ = db.Close() + return nil, err + } + return store, nil +} + +// Close releases the underlying database handle. +func (s *Store) Close() error { + if s == nil || s.db == nil { + return nil + } + if err := s.db.Close(); err != nil { + return fmt.Errorf("appstate: close state db: %w", domain.ErrAppStateIO) + } + return nil +} + +// init ensures top-level buckets and the default checkpoint exist. +func (s *Store) init() error { + return s.db.Update(func(tx *bolt.Tx) error { + if _, err := tx.CreateBucketIfNotExists(bucketMeta); err != nil { + return fmt.Errorf("appstate: create meta bucket: %w", domain.ErrAppStateIO) + } + checkpointBucket, err := tx.CreateBucketIfNotExists(bucketCheckpoint) + if err != nil { + return fmt.Errorf("appstate: create checkpoint bucket: %w", domain.ErrAppStateIO) + } + if checkpointBucket.Get(keyStoreRecord) == nil { + raw, err := marshal(domain.AppStoreCheckpoint{Version: domain.AppStoreVersion}) + if err != nil { + return err + } + if err := checkpointBucket.Put(keyStoreRecord, raw); err != nil { + return fmt.Errorf("appstate: init checkpoint: %w", domain.ErrAppStateIO) + } + } + if _, err := tx.CreateBucketIfNotExists(bucketApps); err != nil { + return fmt.Errorf("appstate: create apps bucket: %w", domain.ErrAppStateIO) + } + return migrateOwnershipHistoryLocked(tx) + }) +} + +// migrateOwnershipHistoryLocked populates the incarnation-indexed +// ownership history once. A reused app name never overwrites the +// history of an earlier incarnation, so retained resources stay +// protected after a remove/recreate. Existing records are copied +// verbatim: claimsForOwnership maps every unrecognized lifecycle state +// conservatively to retained. +func migrateOwnershipHistoryLocked(tx *bolt.Tx) error { + meta := tx.Bucket(bucketMeta) + if meta == nil { + return fmt.Errorf("appstate: missing meta bucket: %w", domain.ErrAppStateIO) + } + history, err := tx.CreateBucketIfNotExists(bucketOwnershipHistory) + if err != nil { + return fmt.Errorf("appstate: create ownership history bucket: %w", domain.ErrAppStateIO) + } + if meta.Get(keyOwnershipMigration) != nil { + return nil + } + + apps := tx.Bucket(bucketApps) + if apps != nil { + if err := apps.ForEach(func(name, _ []byte) error { + bucket := apps.Bucket(name) + if bucket == nil { + return nil + } + var ownership domain.AppOwnership + found, err := getRecord(bucket, keyOwnership, &ownership, "ownership") + if err != nil { + // A corrupt record cannot be migrated. It stays unwritten + // and surfaces as an inventory gap on every snapshot. + return nil + } + if !found { + return nil + } + app := string(name) + if ownership.App == "" { + ownership.App = app + } + record, err := ensureAppRecordLocked(bucket, app, true) + if err != nil { + return err + } + if ownership.ID == "" { + ownership.ID = record.ID + } + return putOwnershipHistoryLocked(history, ownership) + }); err != nil { + return err + } + } + + if err := meta.Put(keyOwnershipMigration, []byte("1")); err != nil { + return fmt.Errorf("appstate: record ownership migration: %w", domain.ErrAppStateIO) + } + return nil +} + +// putOwnershipHistoryLocked writes one incarnation's ownership record +// into the history bucket. Records without an incarnation ID have no +// stable key and are skipped. +func putOwnershipHistoryLocked(history *bolt.Bucket, ownership domain.AppOwnership) error { + if history == nil || ownership.ID == "" { + return nil + } + raw, err := marshal(ownership) + if err != nil { + return err + } + if err := history.Put([]byte(ownership.ID), raw); err != nil { + return fmt.Errorf("appstate: record ownership history: %w", domain.ErrAppStateIO) + } + return nil +} + +// marshal encodes a record as canonical JSON. +func marshal(v any) ([]byte, error) { + raw, err := json.Marshal(v) + if err != nil { + return nil, fmt.Errorf("appstate: encode: %w", domain.ErrAppStateIO) + } + return raw, nil +} + +// unmarshal decodes JSON; broken payloads map to ErrAppStateCorrupt. +func unmarshal(raw []byte, v any, what string) error { + if err := json.Unmarshal(raw, v); err != nil { + return fmt.Errorf("appstate: corrupt %s: %w", what, domain.ErrAppStateCorrupt) + } + return nil +} + +// appBucket returns the per-app bucket, creating it on writes. +func appBucket(tx *bolt.Tx, app string, create bool) (*bolt.Bucket, error) { + apps := tx.Bucket(bucketApps) + if apps == nil { + return nil, fmt.Errorf("appstate: missing apps bucket: %w", domain.ErrAppStateIO) + } + if create { + bucket, err := apps.CreateBucketIfNotExists([]byte(app)) + if err != nil { + return nil, fmt.Errorf("appstate: create app bucket: %w", domain.ErrAppStateIO) + } + return bucket, nil + } + return apps.Bucket([]byte(app)), nil +} + +// subBucket returns a named child bucket, creating it on writes. +func subBucket(parent *bolt.Bucket, name []byte, create bool) (*bolt.Bucket, error) { + if parent == nil { + return nil, nil + } + if create { + bucket, err := parent.CreateBucketIfNotExists(name) + if err != nil { + return nil, fmt.Errorf("appstate: create bucket %s: %w", name, domain.ErrAppStateIO) + } + return bucket, nil + } + return parent.Bucket(name), nil +} + +// ensureAppRecordLocked returns the stable UUID for an app, assigning one +// on first write. Reads leave the database untouched. +func ensureAppRecordLocked(bucket *bolt.Bucket, app string, create bool) (appRecord, error) { + var record appRecord + raw := bucket.Get(keyAppRecord) + if raw == nil { + if !create { + return appRecord{Name: app}, nil + } + record = appRecord{ID: newAppID(), Name: app} + encoded, err := marshal(record) + if err != nil { + return appRecord{}, err + } + if err := bucket.Put(keyAppRecord, encoded); err != nil { + return appRecord{}, fmt.Errorf("appstate: record app id: %w", domain.ErrAppStateIO) + } + return record, nil + } + if err := unmarshal(raw, &record, "app"); err != nil { + return appRecord{}, err + } + if record.Name == "" { + record.Name = app + } + if record.ID == "" && create { + record.ID = newAppID() + encoded, err := marshal(record) + if err != nil { + return appRecord{}, err + } + if err := bucket.Put(keyAppRecord, encoded); err != nil { + return appRecord{}, fmt.Errorf("appstate: record app id: %w", domain.ErrAppStateIO) + } + } + return record, nil +} + +// newAppID allocates a stable internal app UUID. +func newAppID() string { + return "app-" + uuid.NewString() +} + +// checkCtx aborts before opening a transaction when canceled. +func checkCtx(ctx context.Context) error { + if err := ctx.Err(); err != nil { + return err + } + return nil +} + +// getRecord reads one JSON record from a bucket. +func getRecord(bucket *bolt.Bucket, key []byte, v any, what string) (bool, error) { + if bucket == nil { + return false, nil + } + raw := bucket.Get(key) + if raw == nil { + return false, nil + } + if err := unmarshal(raw, v, what); err != nil { + return false, err + } + return true, nil +} + +// putRecord writes one JSON record into a bucket. +func putRecord(bucket *bolt.Bucket, key []byte, v any) error { + raw, err := marshal(v) + if err != nil { + return err + } + if err := bucket.Put(key, raw); err != nil { + return fmt.Errorf("appstate: write record: %w", domain.ErrAppStateIO) + } + return nil +} + +// listKeys returns sorted string keys of a child bucket. +func listKeys(parent *bolt.Bucket, name []byte, reverse bool) ([]string, error) { + child, err := subBucket(parent, name, false) + if err != nil { + return nil, err + } + if child == nil { + return nil, nil + } + var keys []string + if err := child.ForEach(func(k, _ []byte) error { + keys = append(keys, string(k)) + return nil + }); err != nil { + return nil, fmt.Errorf("appstate: list records: %w", domain.ErrAppStateIO) + } + if reverse { + sort.Sort(sort.Reverse(sort.StringSlice(keys))) + } else { + sort.Strings(keys) + } + return keys, nil +} + +// Recover implements out.AppState. +func (s *Store) Recover(ctx context.Context) error { + if err := checkCtx(ctx); err != nil { + return err + } + return s.db.Update(func(tx *bolt.Tx) error { + appsBucket := tx.Bucket(bucketApps) + if appsBucket == nil { + return nil + } + return appsBucket.ForEach(func(app, _ []byte) error { + // Skip nested-bucket cursor noise: app entries are buckets. + bucket := appsBucket.Bucket(app) + if bucket == nil { + return nil + } + return materializeCommittedLocked(tx, bucket) + }) + }) +} + +// LoadCheckpoint implements out.AppState. +func (s *Store) LoadCheckpoint(ctx context.Context) (domain.AppStoreCheckpoint, error) { + if err := checkCtx(ctx); err != nil { + return domain.AppStoreCheckpoint{}, err + } + var checkpoint domain.AppStoreCheckpoint + if err := s.db.View(func(tx *bolt.Tx) error { + bucket := tx.Bucket(bucketCheckpoint) + if bucket == nil { + checkpoint = domain.AppStoreCheckpoint{Version: domain.AppStoreVersion} + return nil + } + raw := bucket.Get(keyStoreRecord) + if raw == nil { + checkpoint = domain.AppStoreCheckpoint{Version: domain.AppStoreVersion} + return nil + } + return unmarshal(raw, &checkpoint, "store") + }); err != nil { + return domain.AppStoreCheckpoint{}, err + } + if checkpoint.Version > domain.AppStoreVersion { + return domain.AppStoreCheckpoint{}, fmt.Errorf("appstate: unsupported store version %d: %w", checkpoint.Version, domain.ErrAppStateIncompatible) + } + return checkpoint, nil +} + +// ListApps implements out.AppState. +func (s *Store) ListApps(ctx context.Context) ([]string, error) { + if err := checkCtx(ctx); err != nil { + return nil, err + } + var apps []string + if err := s.db.View(func(tx *bolt.Tx) error { + appsBucket := tx.Bucket(bucketApps) + if appsBucket == nil { + return nil + } + return appsBucket.ForEach(func(app, _ []byte) error { + if appsBucket.Bucket(app) != nil { + apps = append(apps, string(app)) + } + return nil + }) + }); err != nil { + return nil, err + } + sort.Strings(apps) + return apps, nil +} + +// LoadDesired implements out.AppState. +func (s *Store) LoadDesired(ctx context.Context, app string) (domain.AppDesiredRevision, bool, error) { + if err := checkCtx(ctx); err != nil { + return domain.AppDesiredRevision{}, false, err + } + var desired domain.AppDesiredRevision + var ok bool + if err := s.db.View(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, app, false) + if err != nil { + return err + } + found, err := getRecord(bucket, keyDesired, &desired, "desired") + if err != nil { + return err + } + ok = found + return nil + }); err != nil { + return domain.AppDesiredRevision{}, false, err + } + return desired, ok, nil +} + +// LoadRevision implements out.AppState. +func (s *Store) LoadRevision(ctx context.Context, app, revision string) (domain.AppDesiredRevision, error) { + if err := checkCtx(ctx); err != nil { + return domain.AppDesiredRevision{}, err + } + var rev domain.AppDesiredRevision + if err := s.db.View(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, app, false) + if err != nil { + return err + } + revisions, err := subBucket(bucket, subRevisions, false) + if err != nil { + return err + } + if revisions == nil { + return fmt.Errorf("appstate: revision %s not found: %w", revision, domain.ErrAppRevisionNotFound) + } + raw := revisions.Get([]byte(revision)) + if raw == nil { + return fmt.Errorf("appstate: revision %s not found: %w", revision, domain.ErrAppRevisionNotFound) + } + return unmarshal(raw, &rev, "revision") + }); err != nil { + return domain.AppDesiredRevision{}, err + } + return rev, nil +} + +// ListRevisions returns revision ids ordered newest first. +func (s *Store) ListRevisions(ctx context.Context, app string) ([]string, error) { + if err := checkCtx(ctx); err != nil { + return nil, err + } + var revs []string + if err := s.db.View(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, app, false) + if err != nil { + return err + } + keys, err := listKeys(bucket, subRevisions, true) + if err != nil { + return err + } + revs = keys + return nil + }); err != nil { + return nil, err + } + return revs, nil +} + +// LoadActive implements out.AppState. +func (s *Store) LoadActive(ctx context.Context, app string) (domain.AppActive, bool, error) { + if err := checkCtx(ctx); err != nil { + return domain.AppActive{}, false, err + } + var active domain.AppActive + var ok bool + if err := s.db.View(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, app, false) + if err != nil { + return err + } + found, err := getRecord(bucket, keyActive, &active, "active") + if err != nil { + return err + } + ok = found + return nil + }); err != nil { + return domain.AppActive{}, false, err + } + return active, ok, nil +} + +// LoadIntent implements out.AppState. Absent intent means running. +func (s *Store) LoadIntent(ctx context.Context, app string) (domain.AppStopIntent, error) { + if err := checkCtx(ctx); err != nil { + return domain.AppStopIntent{}, err + } + var intent domain.AppStopIntent + if err := s.db.View(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, app, false) + if err != nil { + return err + } + found, err := getRecord(bucket, keyIntent, &intent, "intent") + if err != nil { + return err + } + if !found { + intent = domain.AppStopIntent{App: app} + } + return nil + }); err != nil { + return domain.AppStopIntent{}, err + } + return intent, nil +} + +// SaveIntent implements out.AppState. +func (s *Store) SaveIntent(ctx context.Context, intent domain.AppStopIntent) error { + if err := checkCtx(ctx); err != nil { + return err + } + return s.db.Update(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, intent.App, true) + if err != nil { + return err + } + if _, err := ensureAppRecordLocked(bucket, intent.App, true); err != nil { + return err + } + return putRecord(bucket, keyIntent, intent) + }) +} + +// LoadOwnership implements out.AppState. +func (s *Store) LoadOwnership(ctx context.Context, app string) (domain.AppOwnership, error) { + if err := checkCtx(ctx); err != nil { + return domain.AppOwnership{}, err + } + var ownership domain.AppOwnership + if err := s.db.View(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, app, false) + if err != nil { + return err + } + if bucket == nil { + ownership = domain.AppOwnership{App: app} + return nil + } + found, err := getRecord(bucket, keyOwnership, &ownership, "ownership") + if err != nil { + return err + } + if !found { + record, err := ensureAppRecordLocked(bucket, app, false) + if err != nil { + return err + } + ownership = domain.AppOwnership{App: app, ID: record.ID} + } + return nil + }); err != nil { + return domain.AppOwnership{}, err + } + return ownership, nil +} + +// LoadRecoveryInhibitions implements out.AppState. +func (s *Store) LoadRecoveryInhibitions(ctx context.Context, app string) ([]domain.AppRecoveryInhibition, error) { + if err := checkCtx(ctx); err != nil { + return nil, err + } + var inhibitions []domain.AppRecoveryInhibition + if err := s.db.View(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, app, false) + if err != nil { + return err + } + _, err = getRecord(bucket, keyInhibitions, &inhibitions, "recovery inhibitions") + return err + }); err != nil { + return nil, err + } + return inhibitions, nil +} + +// SaveRecoveryInhibition implements out.AppState. +func (s *Store) SaveRecoveryInhibition(ctx context.Context, inhibition domain.AppRecoveryInhibition) error { + if err := checkCtx(ctx); err != nil { + return err + } + return s.db.Update(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, inhibition.App, true) + if err != nil { + return err + } + record, err := ensureAppRecordLocked(bucket, inhibition.App, true) + if err != nil { + return err + } + if inhibition.AppID == "" { + inhibition.AppID = record.ID + } + var inhibitions []domain.AppRecoveryInhibition + if found, err := getRecord(bucket, keyInhibitions, &inhibitions, "recovery inhibitions"); err != nil { + return err + } else if !found { + inhibitions = nil + } + replaced := false + for i, existing := range inhibitions { + if existing.Service == inhibition.Service && existing.ContainerID == inhibition.ContainerID { + inhibitions[i] = inhibition + replaced = true + break + } + } + if !replaced { + inhibitions = append(inhibitions, inhibition) + } + return putRecord(bucket, keyInhibitions, inhibitions) + }) +} + +// ClearRecoveryInhibition implements out.AppState. +func (s *Store) ClearRecoveryInhibition(ctx context.Context, app, service, containerID string) error { + if err := checkCtx(ctx); err != nil { + return err + } + return s.db.Update(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, app, false) + if err != nil || bucket == nil { + return err + } + var inhibitions []domain.AppRecoveryInhibition + found, err := getRecord(bucket, keyInhibitions, &inhibitions, "recovery inhibitions") + if err != nil || !found { + return err + } + kept := make([]domain.AppRecoveryInhibition, 0, len(inhibitions)) + for _, existing := range inhibitions { + if existing.Service == service && existing.ContainerID == containerID { + continue + } + kept = append(kept, existing) + } + if len(kept) == len(inhibitions) { + return nil + } + return putRecord(bucket, keyInhibitions, kept) + }) +} + +// SaveOwnership implements out.AppState. +func (s *Store) SaveOwnership(ctx context.Context, ownership domain.AppOwnership) error { + if err := checkCtx(ctx); err != nil { + return err + } + return s.db.Update(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, ownership.App, true) + if err != nil { + return err + } + record, err := ensureAppRecordLocked(bucket, ownership.App, true) + if err != nil { + return err + } + if ownership.ID == "" { + ownership.ID = record.ID + } + // A new UUID means a new incarnation: never carry the old + // incarnation's resources implicitly. The old incarnation's + // history record stays in the ownership-history bucket. + if ownership.ID != record.ID && len(ownership.Volumes) == 0 && len(ownership.Secrets) == 0 && len(ownership.Networks) == 0 { + record = appRecord{ID: ownership.ID, Name: ownership.App} + encoded, err := marshal(record) + if err != nil { + return err + } + if err := bucket.Put(keyAppRecord, encoded); err != nil { + return fmt.Errorf("appstate: record app id: %w", domain.ErrAppStateIO) + } + } + if err := putRecord(bucket, keyOwnership, ownership); err != nil { + return err + } + return putOwnershipHistoryLocked(tx.Bucket(bucketOwnershipHistory), ownership) + }) +} + +// AcceptApply implements out.AppState: stage, commit, materialize, then +// best-effort garbage collection. Each step is its own transaction so a +// crash leaves a state Recover or GC can finish. +func (s *Store) AcceptApply(ctx context.Context, intent domain.AppApplyIntent) error { + if err := s.StageApply(ctx, intent); err != nil { + return fmt.Errorf("appstate: stage apply: %w", err) + } + if err := s.CommitApply(ctx, intent.App, intent.Intent); err != nil { + return fmt.Errorf("appstate: commit apply: %w", err) + } + if err := s.MaterializeApply(ctx, intent.App, intent.Intent); err != nil { + return fmt.Errorf("appstate: materialize apply (re-query by intent): %w", err) + } + if err := s.CollectGarbage(ctx, intent.App, []string{intent.Intent}); err != nil { + s.log.Warn().Err(err).Str("app", intent.App).Msg("appstate: garbage collection failed, continuing") + } + return nil +} + +// StageApply writes a staged intent. No reservation or revision is +// visible until CommitApply. It is one step of AcceptApply, exposed for +// crash-recovery tests. +func (s *Store) StageApply(ctx context.Context, intent domain.AppApplyIntent) error { + if err := checkCtx(ctx); err != nil { + return err + } + intent.State = domain.AppIntentStaged + return s.db.Update(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, intent.App, true) + if err != nil { + return err + } + if _, err := ensureAppRecordLocked(bucket, intent.App, true); err != nil { + return err + } + intents, err := subBucket(bucket, subIntents, true) + if err != nil { + return err + } + return putRecord(intents, []byte(intent.Intent), intent) + }) +} + +// CommitApply is the single atomic commit point: staged→committed. It is +// one step of AcceptApply, exposed for crash-recovery tests. +func (s *Store) CommitApply(ctx context.Context, app, intentID string) error { + if err := checkCtx(ctx); err != nil { + return err + } + return s.db.Update(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, app, true) + if err != nil { + return err + } + intent, err := loadApplyIntentLocked(bucket, app, intentID) + if err != nil { + return err + } + if intent.State != domain.AppIntentStaged { + return fmt.Errorf("appstate: intent %s is %s, not staged: %w", intentID, intent.State, domain.ErrAppStateConflict) + } + if err := verifySupersedesLocked(bucket, intent); err != nil { + return err + } + intent.State = domain.AppIntentCommitted + intents, err := subBucket(bucket, subIntents, true) + if err != nil { + return err + } + return putRecord(intents, []byte(intentID), intent) + }) +} + +// MaterializeApply finishes a committed intent. Idempotent: safe to +// replay. It is one step of AcceptApply, exposed for crash-recovery tests. +func (s *Store) MaterializeApply(ctx context.Context, app, intentID string) error { + if err := checkCtx(ctx); err != nil { + return err + } + return s.db.Update(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, app, true) + if err != nil { + return err + } + intent, err := loadApplyIntentLocked(bucket, app, intentID) + if err != nil { + return err + } + if intent.State == domain.AppIntentStaged { + return fmt.Errorf("appstate: intent %s is staged, commit first: %w", intentID, domain.ErrAppStateConflict) + } + if intent.State == domain.AppIntentApplied { + return nil + } + if err := materializeIntentLocked(tx, bucket, intent); err != nil { + return err + } + intent.State = domain.AppIntentApplied + intents, err := subBucket(bucket, subIntents, true) + if err != nil { + return err + } + return putRecord(intents, []byte(intentID), intent) + }) +} + +// materializeIntentLocked writes revision + desired + checkpoint deltas. +func materializeIntentLocked(tx *bolt.Tx, bucket *bolt.Bucket, intent domain.AppApplyIntent) error { + if err := verifySupersedesLocked(bucket, intent); err != nil { + return err + } + rev := domain.AppDesiredRevision{ + Revision: intent.Revision, + App: intent.App, + Supersedes: intent.Supersedes, + AcceptedAt: intent.CreatedAt, + SourceSHA256: intent.SourceSHA256, + Spec: intent.Spec, + Reservations: intent.Reservations, + SecretsRequired: secretsRequired(intent.Spec), + SecretsEnv: secretsEnv(intent.Spec), + Status: domain.AppStepPending, + } + revisions, err := subBucket(bucket, subRevisions, true) + if err != nil { + return err + } + if err := putRecord(revisions, []byte(intent.Revision), rev); err != nil { + return err + } + if err := putRecord(bucket, keyDesired, rev); err != nil { + return err + } + return foldReservationsLocked(tx, intent) +} + +// foldReservationsLocked folds one committed intent's deltas into the checkpoint. +func foldReservationsLocked(tx *bolt.Tx, intent domain.AppApplyIntent) error { + bucket := tx.Bucket(bucketCheckpoint) + if bucket == nil { + return fmt.Errorf("appstate: missing checkpoint bucket: %w", domain.ErrAppStateIO) + } + var checkpoint domain.AppStoreCheckpoint + raw := bucket.Get(keyStoreRecord) + if raw == nil { + checkpoint = domain.AppStoreCheckpoint{Version: domain.AppStoreVersion} + } else if err := unmarshal(raw, &checkpoint, "store"); err != nil { + return err + } + kept := checkpoint.Reservations[:0] + for _, res := range checkpoint.Reservations { + if res.App != intent.App { + kept = append(kept, res) + } + } + kept = append(kept, intent.Reservations...) + checkpoint.Version = domain.AppStoreVersion + checkpoint.Reservations = kept + encoded, err := marshal(checkpoint) + if err != nil { + return err + } + if err := bucket.Put(keyStoreRecord, encoded); err != nil { + return fmt.Errorf("appstate: write checkpoint: %w", domain.ErrAppStateIO) + } + return nil +} + +// materializeCommittedLocked finishes all committed intents for one app. +// The caller holds the write tx, so revision, desired, checkpoint, and +// intent updates commit atomically. +func materializeCommittedLocked(tx *bolt.Tx, bucket *bolt.Bucket) error { + intents, err := subBucket(bucket, subIntents, false) + if err != nil { + return err + } + if intents == nil { + return nil + } + type pending struct { + id string + intent domain.AppApplyIntent + } + var committed []pending + if err := intents.ForEach(func(k, v []byte) error { + var intent domain.AppApplyIntent + if err := unmarshal(v, &intent, "intent"); err != nil { + return err + } + if intent.State == domain.AppIntentCommitted { + committed = append(committed, pending{id: string(k), intent: intent}) + } + return nil + }); err != nil { + return err + } + for _, item := range committed { + if err := materializeIntentLocked(tx, bucket, item.intent); err != nil { + if errors.Is(err, domain.ErrAppStateConflict) { + // A newer apply superseded this committed intent while the + // daemon was away; it must never overwrite newer desired + // state and can be dropped. + if delErr := intents.Delete([]byte(item.id)); delErr != nil { + return fmt.Errorf("appstate: drop superseded intent: %w", domain.ErrAppStateIO) + } + continue + } + return err + } + item.intent.State = domain.AppIntentApplied + if err := putRecord(intents, []byte(item.id), item.intent); err != nil { + return err + } + } + return nil +} + +// verifySupersedesLocked rejects an intent whose expected former revision +// no longer matches the current desired revision: a concurrent apply that +// committed first wins, and the stale intent can never overwrite it. +func verifySupersedesLocked(bucket *bolt.Bucket, intent domain.AppApplyIntent) error { + var desired domain.AppDesiredRevision + found, err := getRecord(bucket, keyDesired, &desired, "desired") + if err != nil { + return err + } + current := "" + if found { + current = desired.Revision + } + if current != intent.Supersedes { + return fmt.Errorf("appstate: intent %s expected desired %q but found %q: %w", + intent.Intent, intent.Supersedes, current, domain.ErrAppStateConflict) + } + return nil +} + +// LoadApplyIntent returns one apply intent by id. +func (s *Store) LoadApplyIntent(ctx context.Context, app, intentID string) (domain.AppApplyIntent, error) { + if err := checkCtx(ctx); err != nil { + return domain.AppApplyIntent{}, err + } + var intent domain.AppApplyIntent + if err := s.db.View(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, app, false) + if err != nil { + return err + } + loaded, err := loadApplyIntentLocked(bucket, app, intentID) + if err != nil { + return err + } + intent = loaded + return nil + }); err != nil { + return domain.AppApplyIntent{}, err + } + return intent, nil +} + +func loadApplyIntentLocked(bucket *bolt.Bucket, app, intentID string) (domain.AppApplyIntent, error) { + var intent domain.AppApplyIntent + if bucket == nil { + return domain.AppApplyIntent{}, fmt.Errorf("appstate: intent %s not found: %w", intentID, domain.ErrAppIntentNotFound) + } + intents, err := subBucket(bucket, subIntents, false) + if err != nil { + return domain.AppApplyIntent{}, err + } + if intents == nil { + return domain.AppApplyIntent{}, fmt.Errorf("appstate: intent %s not found: %w", intentID, domain.ErrAppIntentNotFound) + } + raw := intents.Get([]byte(intentID)) + if raw == nil { + return domain.AppApplyIntent{}, fmt.Errorf("appstate: intent %s not found: %w", intentID, domain.ErrAppIntentNotFound) + } + if err := unmarshal(raw, &intent, "intent"); err != nil { + return domain.AppApplyIntent{}, err + } + _ = app + return intent, nil +} + +// ListIntents returns apply intent ids for recovery scans. +func (s *Store) ListIntents(ctx context.Context, app string) ([]string, error) { + if err := checkCtx(ctx); err != nil { + return nil, err + } + var ids []string + if err := s.db.View(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, app, false) + if err != nil { + return err + } + keys, err := listKeys(bucket, subIntents, false) + if err != nil { + return err + } + ids = keys + return nil + }); err != nil { + return nil, err + } + return ids, nil +} + +// CollectGarbage removes staged orphans, applied intents, and +// unreferenced revisions beyond retention. Referenced revisions +// (desired/active/in-flight) are never evicted. +func (s *Store) CollectGarbage(ctx context.Context, app string, inFlight []string) error { + if err := checkCtx(ctx); err != nil { + return err + } + return s.db.Update(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, app, true) + if err != nil { + return err + } + protected, err := protectedRevisionsLocked(bucket, app, inFlight) + if err != nil { + return err + } + if err := sweepIntentsLocked(bucket, inFlight); err != nil { + return err + } + return sweepRevisionsLocked(bucket, protected, s.Retention()) + }) +} + +// protectedRevisionsLocked returns revisions that GC must never evict. +func protectedRevisionsLocked(bucket *bolt.Bucket, app string, inFlight []string) (map[string]struct{}, error) { + protected := map[string]struct{}{} + for _, rev := range inFlight { + protected[rev] = struct{}{} + } + var desired domain.AppDesiredRevision + if found, err := getRecord(bucket, keyDesired, &desired, "desired"); err != nil { + return nil, err + } else if found { + protected[desired.Revision] = struct{}{} + } + var active domain.AppActive + if found, err := getRecord(bucket, keyActive, &active, "active"); err != nil { + return nil, err + } else if found { + for _, svc := range active.Services { + protected[svc.EffectiveRevision] = struct{}{} + } + } + intents, err := subBucket(bucket, subIntents, false) + if err != nil { + return nil, err + } + if intents != nil { + if err := intents.ForEach(func(_, v []byte) error { + var intent domain.AppApplyIntent + if err := unmarshal(v, &intent, "intent"); err != nil { + return err + } + // Only unfinished work pins its revision: staged orphans + // and committed-but-unmaterialized intents. Applied intents + // are history; their revisions age out by retention. + if intent.State == domain.AppIntentStaged || intent.State == domain.AppIntentCommitted { + protected[intent.Revision] = struct{}{} + } + return nil + }); err != nil { + return nil, err + } + } + _ = app + return protected, nil +} + +// sweepIntentsLocked removes applied intents and orphaned staged intents. +// Committed intents are never swept here: they carry unmaterialized +// deltas that recovery must complete. +func sweepIntentsLocked(bucket *bolt.Bucket, inFlight []string) error { + intents, err := subBucket(bucket, subIntents, false) + if err != nil { + return err + } + if intents == nil { + return nil + } + // A live apply's intent must never be swept as an orphan: the caller + // names its own in-flight intent so concurrent work is preserved. + protected := make(map[string]struct{}, len(inFlight)) + for _, id := range inFlight { + protected[id] = struct{}{} + } + var drop [][]byte + if err := intents.ForEach(func(k, v []byte) error { + if _, ok := protected[string(k)]; ok { + return nil + } + var intent domain.AppApplyIntent + if err := unmarshal(v, &intent, "intent"); err != nil { + return err + } + if intent.State == domain.AppIntentApplied || intent.State == domain.AppIntentStaged { + key := append([]byte(nil), k...) + drop = append(drop, key) + } + return nil + }); err != nil { + return err + } + for _, key := range drop { + if err := intents.Delete(key); err != nil { + return fmt.Errorf("appstate: remove intent: %w", domain.ErrAppStateIO) + } + } + return nil +} + +// sweepRevisionsLocked removes unreferenced revisions beyond retention. +func sweepRevisionsLocked(bucket *bolt.Bucket, protected map[string]struct{}, retention int) error { + revisions, err := subBucket(bucket, subRevisions, false) + if err != nil { + return err + } + if revisions == nil { + return nil + } + var keys []string + if err := revisions.ForEach(func(k, _ []byte) error { + keys = append(keys, string(k)) + return nil + }); err != nil { + return fmt.Errorf("appstate: list revisions: %w", domain.ErrAppStateIO) + } + sort.Sort(sort.Reverse(sort.StringSlice(keys))) + kept := 0 + for _, rev := range keys { + if _, ok := protected[rev]; ok { + continue + } + if kept >= retention { + if err := revisions.Delete([]byte(rev)); err != nil { + return fmt.Errorf("appstate: remove revision: %w", domain.ErrAppStateIO) + } + continue + } + kept++ + } + return nil +} + +// SaveOperation implements out.AppState. +func (s *Store) SaveOperation(ctx context.Context, op domain.AppOperation) error { + if err := checkCtx(ctx); err != nil { + return err + } + return s.db.Update(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, op.App, true) + if err != nil { + return err + } + if _, err := ensureAppRecordLocked(bucket, op.App, true); err != nil { + return err + } + ops, err := subBucket(bucket, subOps, true) + if err != nil { + return err + } + return putRecord(ops, []byte(op.Op), op) + }) +} + +// AppExists implements out.AppState: a read-only live-identity check +// that never creates an app bucket. +func (s *Store) AppExists(ctx context.Context, app string) (bool, error) { + if err := checkCtx(ctx); err != nil { + return false, err + } + exists := false + err := s.db.View(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, app, false) + if err != nil { + return err + } + if bucket == nil { + return nil + } + known, err := appHasLiveIdentityLocked(bucket) + if err != nil { + return err + } + exists = known + return nil + }) + if err != nil { + return false, err + } + return exists, nil +} + +// ClaimOperation implements out.AppState: one atomic check-and-write for +// a mutation request key. +func (s *Store) ClaimOperation(ctx context.Context, candidate domain.AppOperation) (domain.AppOperation, bool, error) { + if err := checkCtx(ctx); err != nil { + return domain.AppOperation{}, false, err + } + if candidate.Op == "" { + return domain.AppOperation{}, false, fmt.Errorf("appstate: claim operation without a key: %w", domain.ErrInvalidAppSpec) + } + var existing domain.AppOperation + claimed := false + err := s.db.Update(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, candidate.App, false) + if err != nil { + return err + } + if stored, ok, err := claimedOperation(bucket, candidate); err != nil { + return err + } else if ok { + existing = stored + return nil + } + if bucket == nil { + return fmt.Errorf("appstate: app %q has no state: %w", candidate.App, domain.ErrAppNotFound) + } + known, err := appHasLiveIdentityLocked(bucket) + if err != nil { + return err + } + if !known { + return fmt.Errorf("appstate: app %q has no live identity: %w", candidate.App, domain.ErrAppNotFound) + } + writeBucket, err := appBucket(tx, candidate.App, true) + if err != nil { + return err + } + if _, err := ensureAppRecordLocked(writeBucket, candidate.App, true); err != nil { + return err + } + ops, err := subBucket(writeBucket, subOps, true) + if err != nil { + return err + } + if err := putRecord(ops, []byte(candidate.Op), candidate); err != nil { + return err + } + existing = candidate + claimed = true + return nil + }) + if err != nil { + return domain.AppOperation{}, false, err + } + return existing, claimed, nil +} + +// claimedOperation returns the journal already stored under the candidate +// key when its request identity matches. A key answering a different +// request identity is a conflict, never a replay. +func claimedOperation(bucket *bolt.Bucket, candidate domain.AppOperation) (domain.AppOperation, bool, error) { + if bucket == nil { + return domain.AppOperation{}, false, nil + } + ops, err := subBucket(bucket, subOps, false) + if err != nil || ops == nil { + return domain.AppOperation{}, false, err + } + raw := ops.Get([]byte(candidate.Op)) + if raw == nil { + return domain.AppOperation{}, false, nil + } + var stored domain.AppOperation + if err := unmarshal(raw, &stored, "operation"); err != nil { + return domain.AppOperation{}, false, err + } + if stored.Request != candidate.Request { + return domain.AppOperation{}, false, fmt.Errorf( + "appstate: operation key %s already answered a different request: %w", + candidate.Op, domain.ErrAppStateConflict, + ) + } + return stored, true, nil +} + +// appHasLiveIdentityLocked reports whether an existing app bucket holds a +// live identity: desired state, active state, or an ownership +// incarnation. Operation journals alone are not identity — a retired app +// keeps them so its own key still replays, while a new key must never +// recreate state for the freed name. +func appHasLiveIdentityLocked(bucket *bolt.Bucket) (bool, error) { + if bucket.Get(keyDesired) != nil || bucket.Get(keyActive) != nil { + return true, nil + } + var ownership domain.AppOwnership + found, err := getRecord(bucket, keyOwnership, &ownership, "ownership") + if err != nil { + return false, err + } + return found && ownership.ID != "", nil +} + +// LoadOperation implements out.AppState. +func (s *Store) LoadOperation(ctx context.Context, app, opID string) (domain.AppOperation, error) { + if err := checkCtx(ctx); err != nil { + return domain.AppOperation{}, err + } + var op domain.AppOperation + if err := s.db.View(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, app, false) + if err != nil { + return err + } + ops, err := subBucket(bucket, subOps, false) + if err != nil { + return err + } + if ops == nil { + return fmt.Errorf("appstate: operation %s not found: %w", opID, domain.ErrAppOperationNotFound) + } + raw := ops.Get([]byte(opID)) + if raw == nil { + return fmt.Errorf("appstate: operation %s not found: %w", opID, domain.ErrAppOperationNotFound) + } + return unmarshal(raw, &op, "operation") + }); err != nil { + return domain.AppOperation{}, err + } + return op, nil +} + +// LoadLatestOperation implements out.AppState. +func (s *Store) LoadLatestOperation(ctx context.Context, app string) (domain.AppOperation, bool, error) { + if err := checkCtx(ctx); err != nil { + return domain.AppOperation{}, false, err + } + var latest domain.AppOperation + found := false + err := s.db.View(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, app, false) + if err != nil { + return err + } + ops, err := subBucket(bucket, subOps, false) + if err != nil || ops == nil { + return err + } + return ops.ForEach(func(_, raw []byte) error { + var op domain.AppOperation + if err := unmarshal(raw, &op, "operation"); err != nil { + return err + } + if !found || op.StartedAt.After(latest.StartedAt) { + latest, found = op, true + } + return nil + }) + }) + if err != nil { + return domain.AppOperation{}, false, err + } + return latest, found, nil +} + +// RetireApp implements out.AppState. In one transaction it archives the +// live incarnation's ownership record (with every resource marked +// retained), drops the live ownership record, resets the app's UUID so a +// later reapply allocates a NEW incarnation, and clears desired, active, +// intent, staged intents, and revisions. The archived record stays the +// positive proof that retained volumes and secrets exist, so prune keeps +// protecting them while a name reuse can never adopt them. +func (s *Store) RetireApp(ctx context.Context, app string) error { + if err := checkCtx(ctx); err != nil { + return err + } + return s.db.Update(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, app, true) + if err != nil { + return err + } + if err := archiveRetainedOwnershipLocked(tx, bucket); err != nil { + return err + } + return clearIncarnationStateLocked(bucket, app) + }) +} + +// archiveRetainedOwnershipLocked archives the live ownership record with +// every resource marked retained, so prune keeps protecting them. A +// missing record is not an error. +func archiveRetainedOwnershipLocked(tx *bolt.Tx, bucket *bolt.Bucket) error { + var ownership domain.AppOwnership + found, err := getRecord(bucket, keyOwnership, &ownership, "ownership") + if err != nil { + return err + } + if !found || ownership.ID == "" { + return nil + } + for i := range ownership.Volumes { + ownership.Volumes[i].State = domain.AppResourceRetained + } + for i := range ownership.Secrets { + ownership.Secrets[i].State = domain.AppResourceRetained + } + for i := range ownership.Images { + ownership.Images[i].State = domain.AppResourceReleased + } + return putOwnershipHistoryLocked(tx.Bucket(bucketOwnershipHistory), ownership) +} + +// clearIncarnationStateLocked drops the live ownership record and all +// operational state, then resets the app record so the next write assigns +// a fresh incarnation UUID. +func clearIncarnationStateLocked(bucket *bolt.Bucket, app string) error { + for _, key := range [][]byte{keyOwnership, keyDesired, keyActive, keyIntent, keyInhibitions} { + if err := bucket.Delete(key); err != nil { + return fmt.Errorf("appstate: retire %s: %w", app, domain.ErrAppStateIO) + } + } + for _, sub := range [][]byte{subRevisions, subIntents} { + if err := bucket.DeleteBucket(sub); err != nil && !errors.Is(err, bolterrors.ErrBucketNotFound) { + return fmt.Errorf("appstate: retire %s: %w", app, domain.ErrAppStateIO) + } + } + encoded, err := marshal(appRecord{Name: app}) + if err != nil { + return err + } + if err := bucket.Put(keyAppRecord, encoded); err != nil { + return fmt.Errorf("appstate: reset app record: %w", domain.ErrAppStateIO) + } + return nil +} + +// SaveActive implements out.AppState. +func (s *Store) SaveActive(ctx context.Context, active domain.AppActive) error { + if err := checkCtx(ctx); err != nil { + return err + } + return s.db.Update(func(tx *bolt.Tx) error { + bucket, err := appBucket(tx, active.App, true) + if err != nil { + return err + } + if _, err := ensureAppRecordLocked(bucket, active.App, true); err != nil { + return err + } + return putRecord(bucket, keyActive, active) + }) +} + +// RegisterBackendBinds implements out.AppState. Backend claims join the +// global checkpoint in the same write tx that checks them, so two +// concurrent deploys cannot both claim one bind. +func (s *Store) RegisterBackendBinds(ctx context.Context, binds []domain.AppListenerReservation) error { + if err := checkCtx(ctx); err != nil { + return err + } + if err := validateBackendClaims(binds); err != nil { + return err + } + return s.db.Update(func(tx *bolt.Tx) error { + checkpoint, bucket, err := loadCheckpointLocked(tx) + if err != nil { + return err + } + if err := checkBackendConflicts(checkpoint.Reservations, binds); err != nil { + return err + } + checkpoint.Reservations = replaceContainerClaims(checkpoint.Reservations, binds) + checkpoint.Version = domain.AppStoreVersion + return putCheckpointLocked(bucket, checkpoint) + }) +} + +// validateBackendClaims rejects non-backend or malformed claims before +// any storage mutation. +func validateBackendClaims(binds []domain.AppListenerReservation) error { + for _, bind := range binds { + if bind.Owner != domain.OwnerGordonBackend { + return fmt.Errorf("appstate: backend claim owner must be %q: %w", domain.OwnerGordonBackend, domain.ErrAppReservationConflict) + } + if bind.Proto != "tcp" && bind.Proto != "udp" { + return fmt.Errorf("appstate: backend claim proto must be tcp or udp: %w", domain.ErrAppReservationConflict) + } + if bind.IP != "127.0.0.1" || bind.Port <= 0 || bind.ContainerID == "" { + return fmt.Errorf("appstate: malformed backend claim: %w", domain.ErrAppReservationConflict) + } + } + return nil +} + +// loadCheckpointLocked reads the global checkpoint in a write tx. +func loadCheckpointLocked(tx *bolt.Tx) (domain.AppStoreCheckpoint, *bolt.Bucket, error) { + bucket := tx.Bucket(bucketCheckpoint) + if bucket == nil { + return domain.AppStoreCheckpoint{}, nil, fmt.Errorf("appstate: missing checkpoint bucket: %w", domain.ErrAppStateIO) + } + var checkpoint domain.AppStoreCheckpoint + raw := bucket.Get(keyStoreRecord) + if raw == nil { + return domain.AppStoreCheckpoint{Version: domain.AppStoreVersion}, bucket, nil + } + if err := unmarshal(raw, &checkpoint, "store"); err != nil { + return domain.AppStoreCheckpoint{}, nil, err + } + return checkpoint, bucket, nil +} + +// putCheckpointLocked persists the global checkpoint in a write tx. +func putCheckpointLocked(bucket *bolt.Bucket, checkpoint domain.AppStoreCheckpoint) error { + encoded, err := marshal(checkpoint) + if err != nil { + return err + } + if err := bucket.Put(keyStoreRecord, encoded); err != nil { + return fmt.Errorf("appstate: write checkpoint: %w", domain.ErrAppStateIO) + } + return nil +} + +// checkBackendConflicts fails closed on another app's identical loopback +// claim. Same-app claims never self-conflict. +func checkBackendConflicts(existing []domain.AppListenerReservation, binds []domain.AppListenerReservation) error { + for _, bind := range binds { + for _, other := range existing { + if other.App == bind.App { + continue + } + if backendOverlaps(other, bind) { + return fmt.Errorf("appstate: backend bind %s conflicts with %s/%s: %w", + backendDescribe(bind), other.App, other.Service, domain.ErrAppReservationConflict) + } + } + } + return nil +} + +// replaceContainerClaims swaps one container's backend claims for the +// new set, preserving every other claim (old and replacement coexist +// until the old container's verified withdrawal). +func replaceContainerClaims(existing []domain.AppListenerReservation, binds []domain.AppListenerReservation) []domain.AppListenerReservation { + if len(binds) == 0 { + return existing + } + kept := existing[:0] + for _, res := range existing { + if res.Owner == domain.OwnerGordonBackend && res.App == binds[0].App && res.ContainerID == binds[0].ContainerID { + continue + } + kept = append(kept, res) + } + return append(kept, binds...) +} + +// ReleaseBackendBinds implements out.AppState. +func (s *Store) ReleaseBackendBinds(ctx context.Context, app, containerID string) error { + if err := checkCtx(ctx); err != nil { + return err + } + return s.db.Update(func(tx *bolt.Tx) error { + bucket := tx.Bucket(bucketCheckpoint) + if bucket == nil { + return fmt.Errorf("appstate: missing checkpoint bucket: %w", domain.ErrAppStateIO) + } + var checkpoint domain.AppStoreCheckpoint + raw := bucket.Get(keyStoreRecord) + if raw == nil { + return nil + } + if err := unmarshal(raw, &checkpoint, "store"); err != nil { + return err + } + kept := checkpoint.Reservations[:0] + for _, existing := range checkpoint.Reservations { + if existing.Owner == domain.OwnerGordonBackend && existing.App == app && existing.ContainerID == containerID { + continue + } + kept = append(kept, existing) + } + checkpoint.Reservations = kept + encoded, err := marshal(checkpoint) + if err != nil { + return err + } + if err := bucket.Put(keyStoreRecord, encoded); err != nil { + return fmt.Errorf("appstate: write checkpoint: %w", domain.ErrAppStateIO) + } + return nil + }) +} + +// backendOverlaps reports exact loopback bind collisions per protocol. +// A TCP and a UDP claim on the same host port never collide: they are +// distinct sockets. Claims are exact 127.0.0.1:port allocations +// (kernel-arbitrated), so only the same IP+port collides — never user +// wildcard publishes on other IPs. +func backendOverlaps(a, b domain.AppListenerReservation) bool { + if a.Proto != b.Proto || (a.Proto != "tcp" && a.Proto != "udp") { + return false + } + return a.IP == "127.0.0.1" && b.IP == "127.0.0.1" && a.Port == b.Port +} + +func backendDescribe(bind domain.AppListenerReservation) string { + return fmt.Sprintf("%s://%s:%d", bind.Proto, bind.IP, bind.Port) +} +func secretsRequired(spec domain.AppSpec) []string { + var paths []string + for _, svc := range spec.Services { + for _, name := range svc.Secrets { + paths = append(paths, domain.AppSecretPath(spec.Name, svc.Name, name)) + } + } + sort.Strings(paths) + return paths +} + +// secretsEnv maps pass paths back to env keys. +func secretsEnv(spec domain.AppSpec) map[string]string { + env := map[string]string{} + for _, svc := range spec.Services { + for envKey, name := range svc.Secrets { + env[domain.AppSecretPath(spec.Name, svc.Name, name)] = envKey + } + } + return env +} diff --git a/internal/adapters/out/appstate/store_claim_test.go b/internal/adapters/out/appstate/store_claim_test.go new file mode 100644 index 000000000..9538276c4 --- /dev/null +++ b/internal/adapters/out/appstate/store_claim_test.go @@ -0,0 +1,211 @@ +package appstate_test + +import ( + "context" + "sync" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/out/appstate" + "github.com/bnema/gordon/internal/domain" +) + +// knownApp gives the app a live identity so keyed mutations are allowed. +func knownApp(t *testing.T, store interface { + SaveOwnership(context.Context, domain.AppOwnership) error +}, app string) { + t.Helper() + require.NoError(t, store.SaveOwnership(context.Background(), domain.AppOwnership{App: app})) +} + +func deployClaim(app, key, revision string) domain.AppOperation { + return domain.AppOperation{ + Op: key, + Kind: "deploy", + App: app, + StartedAt: time.Now().UTC(), + Request: domain.AppOperationRequestFor("deploy", app, revision, ""), + } +} + +// TestStore_ClaimOperationCreatesInFlightJournal proves an absent key is +// persisted as an in-flight claim so no effect can run twice. +func TestStore_ClaimOperationCreatesInFlightJournal(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + knownApp(t, store, "blog") + + candidate := deployClaim("blog", "key-1", "rev-1") + existing, claimed, err := store.ClaimOperation(ctx, candidate) + require.NoError(t, err) + require.True(t, claimed) + assert.Equal(t, "key-1", existing.Op) + assert.False(t, existing.Terminal(), "a fresh claim is not terminal") + + stored, err := store.LoadOperation(ctx, "blog", "key-1") + require.NoError(t, err) + assert.Equal(t, candidate.Request, stored.Request) + assert.False(t, stored.Terminal()) +} + +// TestStore_ClaimOperationReplaysSameRequest proves a repeat of one +// request returns the stored journal instead of claiming again. +func TestStore_ClaimOperationReplaysSameRequest(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + knownApp(t, store, "blog") + + require.NoError(t, store.SaveOperation(ctx, domain.AppOperation{ + Op: "key-1", Kind: "deploy", App: "blog", + Request: domain.AppOperationRequestFor("deploy", "blog", "rev-1", ""), + Outcome: domain.AppOutcomeSuccess, + StartedAt: time.Now().UTC(), + })) + + existing, claimed, err := store.ClaimOperation(ctx, deployClaim("blog", "key-1", "rev-1")) + require.NoError(t, err) + assert.False(t, claimed) + assert.Equal(t, domain.AppOutcomeSuccess, existing.Outcome) +} + +// TestStore_ClaimOperationConflictsOnDifferentRequest proves one key can +// never answer two different requests. +func TestStore_ClaimOperationConflictsOnDifferentRequest(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + knownApp(t, store, "blog") + + _, claimed, err := store.ClaimOperation(ctx, deployClaim("blog", "key-1", "rev-1")) + require.NoError(t, err) + require.True(t, claimed) + + _, _, err = store.ClaimOperation(ctx, deployClaim("blog", "key-1", "rev-2")) + require.ErrorIs(t, err, domain.ErrAppStateConflict) + + // A different kind with the same key is a different request too. + stop := domain.AppOperation{ + Op: "key-1", Kind: "stop", App: "blog", StartedAt: time.Now().UTC(), + Request: domain.AppOperationRequestFor("stop", "blog", "", ""), + } + _, _, err = store.ClaimOperation(ctx, stop) + require.ErrorIs(t, err, domain.ErrAppStateConflict) +} + +// TestStore_ClaimOperationRefusesUnknownAppWithoutState proves a keyed +// mutation of an unknown name creates no durable state at all. +func TestStore_ClaimOperationRefusesUnknownAppWithoutState(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + _, _, err := store.ClaimOperation(ctx, deployClaim("ghost", "key-1", "rev-1")) + require.ErrorIs(t, err, domain.ErrAppNotFound) + + apps, err := store.ListApps(ctx) + require.NoError(t, err) + assert.NotContains(t, apps, "ghost", "a refused claim must not create an app entry") + + _, err = store.LoadOperation(ctx, "ghost", "key-1") + require.ErrorIs(t, err, domain.ErrAppOperationNotFound) +} + +// TestStore_ClaimOperationIsAtomicUnderConcurrency proves two concurrent +// claims of one key produce exactly one winner. +func TestStore_ClaimOperationIsAtomicUnderConcurrency(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + knownApp(t, store, "blog") + + const racers = 8 + var ( + wg sync.WaitGroup + mu sync.Mutex + winners int + ) + start := make(chan struct{}) + for range racers { + wg.Add(1) + go func() { + defer wg.Done() + <-start + _, claimed, err := store.ClaimOperation(ctx, deployClaim("blog", "key-1", "rev-1")) + mu.Lock() + defer mu.Unlock() + if err != nil { + t.Errorf("claim: %v", err) + return + } + if claimed { + winners++ + } + }() + } + close(start) + wg.Wait() + assert.Equal(t, 1, winners, "exactly one claim may win") +} + +// TestStore_ClaimOperationReplaysAfterReopen proves a claimed operation +// survives a daemon restart: the reopened store replays the stored journal +// instead of claiming the key again. +func TestStore_ClaimOperationReplaysAfterReopen(t *testing.T) { + ctx := context.Background() + dir := t.TempDir() + store, err := appstate.NewStore(dir, testLogger()) + require.NoError(t, err) + require.NoError(t, store.SaveOwnership(ctx, domain.AppOwnership{App: "blog"})) + + _, claimed, err := store.ClaimOperation(ctx, deployClaim("blog", "key-1", "rev-1")) + require.NoError(t, err) + require.True(t, claimed) + op := deployClaim("blog", "key-1", "rev-1") + op.Outcome = domain.AppOutcomeSuccess + require.NoError(t, store.SaveOperation(ctx, op)) + require.NoError(t, store.Close()) + + reopened, err := appstate.NewStore(dir, testLogger()) + require.NoError(t, err) + t.Cleanup(func() { assert.NoError(t, reopened.Close()) }) + + existing, claimed, err := reopened.ClaimOperation(ctx, deployClaim("blog", "key-1", "rev-1")) + require.NoError(t, err) + assert.False(t, claimed, "a reopened store replays the persisted claim") + assert.Equal(t, domain.AppOutcomeSuccess, existing.Outcome) +} + +// TestStore_ClaimOperationSurvivesRetire proves an operation journal +// stays reachable after removal while a new key against the freed name +// is refused. +func TestStore_ClaimOperationSurvivesRetire(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + knownApp(t, store, "blog") + + remove := domain.AppOperation{ + Op: "remove-1", Kind: "remove", App: "blog", StartedAt: time.Now().UTC(), + Request: domain.AppOperationRequestFor("remove", "blog", "", ""), + } + _, claimed, err := store.ClaimOperation(ctx, remove) + require.NoError(t, err) + require.True(t, claimed) + remove.Outcome = domain.AppOutcomeSuccess + remove.Steps = []domain.AppOperationStep{{ID: "service.web.remove", State: domain.AppStepSucceeded}} + require.NoError(t, store.SaveOperation(ctx, remove)) + require.NoError(t, store.RetireApp(ctx, "blog")) + + replayed, claimed, err := store.ClaimOperation(ctx, domain.AppOperation{ + Op: "remove-1", Kind: "remove", App: "blog", + Request: domain.AppOperationRequestFor("remove", "blog", "", ""), + }) + require.NoError(t, err) + assert.False(t, claimed, "a repeated remove replays its stored result") + assert.Equal(t, domain.AppOutcomeSuccess, replayed.Outcome) + + _, _, err = store.ClaimOperation(ctx, domain.AppOperation{ + Op: "remove-2", Kind: "remove", App: "blog", + Request: domain.AppOperationRequestFor("remove", "blog", "", ""), + }) + require.ErrorIs(t, err, domain.ErrAppNotFound, "a new key must not resurrect a retired name") +} diff --git a/internal/adapters/out/appstate/store_coordination_test.go b/internal/adapters/out/appstate/store_coordination_test.go new file mode 100644 index 000000000..ed215e3e7 --- /dev/null +++ b/internal/adapters/out/appstate/store_coordination_test.go @@ -0,0 +1,74 @@ +package appstate_test + +import ( + "context" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +// TestStore_CommitApplyRejectsSupersededIntent proves an apply that stages +// against an outdated desired revision cannot commit once a newer apply has +// published: concurrent applies can never overwrite each other silently. +func TestStore_CommitApplyRejectsSupersededIntent(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + require.NoError(t, store.StageApply(ctx, testIntent("blog", "rev-1"))) + require.NoError(t, store.CommitApply(ctx, "blog", "apply-test-rev-1")) + require.NoError(t, store.MaterializeApply(ctx, "blog", "apply-test-rev-1")) + + // Staged while no desired state existed; by commit time rev-1 is live. + require.NoError(t, store.StageApply(ctx, testIntent("blog", "rev-2"))) + require.ErrorIs(t, store.CommitApply(ctx, "blog", "apply-test-rev-2"), domain.ErrAppStateConflict) +} + +// TestStore_RecoverDropsSupersededCommittedIntent proves recovery never +// overwrites newer desired state with a committed intent that lost the race. +func TestStore_RecoverDropsSupersededCommittedIntent(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + require.NoError(t, store.StageApply(ctx, testIntent("blog", "rev-1"))) + require.NoError(t, store.CommitApply(ctx, "blog", "apply-test-rev-1")) + require.NoError(t, store.MaterializeApply(ctx, "blog", "apply-test-rev-1")) + + // A committed intent that never materialized. + require.NoError(t, store.StageApply(ctx, testIntentAfter("blog", "rev-2", "rev-1"))) + require.NoError(t, store.CommitApply(ctx, "blog", "apply-test-rev-2")) + + // A newer apply wins and publishes rev-3. + require.NoError(t, store.StageApply(ctx, testIntentAfter("blog", "rev-3", "rev-1"))) + require.NoError(t, store.CommitApply(ctx, "blog", "apply-test-rev-3")) + require.NoError(t, store.MaterializeApply(ctx, "blog", "apply-test-rev-3")) + + require.NoError(t, store.Recover(ctx)) + + desired, ok, err := store.LoadDesired(ctx, "blog") + require.NoError(t, err) + require.True(t, ok) + assert.Equal(t, "rev-3", desired.Revision, "the superseded intent must never overwrite newer state") + _, err = store.LoadApplyIntent(ctx, "blog", "apply-test-rev-2") + require.ErrorIs(t, err, domain.ErrAppIntentNotFound, "the stale intent is dropped") +} + +// TestStore_CollectGarbagePreservesInFlightIntent proves a live apply's +// staged intent is never swept as an orphan. +func TestStore_CollectGarbagePreservesInFlightIntent(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + require.NoError(t, store.StageApply(ctx, testIntent("blog", "rev-inflight"))) + require.NoError(t, store.CollectGarbage(ctx, "blog", []string{"apply-test-rev-inflight"})) + + _, err := store.LoadApplyIntent(ctx, "blog", "apply-test-rev-inflight") + require.NoError(t, err, "an in-flight intent must survive garbage collection") + + // Without the in-flight marker it is an orphan and is swept. + require.NoError(t, store.CollectGarbage(ctx, "blog", nil)) + _, err = store.LoadApplyIntent(ctx, "blog", "apply-test-rev-inflight") + require.ErrorIs(t, err, domain.ErrAppIntentNotFound) +} diff --git a/internal/adapters/out/appstate/store_retire_test.go b/internal/adapters/out/appstate/store_retire_test.go new file mode 100644 index 000000000..18c936a85 --- /dev/null +++ b/internal/adapters/out/appstate/store_retire_test.go @@ -0,0 +1,134 @@ +package appstate_test + +import ( + "context" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +// TestStore_RetireAppEndsIncarnation proves removing an app archives its +// ownership, clears its operational state, and detaches the name so a +// reapply allocates a new incarnation that cannot inherit the old +// secrets or volumes. +func TestStore_RetireAppEndsIncarnation(t *testing.T) { + store := newTestStore(t) + ctx := context.Background() + + require.NoError(t, store.SaveOwnership(ctx, domain.AppOwnership{ + App: "blog", ID: "app-old", + Volumes: []domain.AppOwnedVolume{{ + Name: "data", Service: "web", + RuntimeName: "gordon-blog--web--vol--data", + State: domain.AppResourceAttached, + }}, + Secrets: []domain.AppOwnedSecret{{ + Service: "web", Env: "DATABASE_URL", Name: "database-url", + Path: "gordon/apps/app-old/blog/web/database-url", State: domain.AppResourceAttached, + }}, + Images: []domain.AppOwnedImage{{ + Service: "web", Reference: "registry.example.com/blog/web:1.4.2", + Digest: "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + State: domain.AppResourceAttached, + }}, + })) + require.NoError(t, store.SaveActive(ctx, domain.AppActive{ + App: "blog", + Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-old", EffectiveRevision: "rev-old"}, + }, + })) + require.NoError(t, store.SaveIntent(ctx, domain.AppStopIntent{App: "blog", Stopped: true})) + + require.NoError(t, store.RetireApp(ctx, "blog")) + + ownership, err := store.LoadOwnership(ctx, "blog") + require.NoError(t, err) + assert.Empty(t, ownership.ID, "the name must not retain the old incarnation UUID") + assert.Empty(t, ownership.Volumes) + assert.Empty(t, ownership.Secrets) + + _, ok, err := store.LoadActive(ctx, "blog") + require.NoError(t, err) + assert.False(t, ok, "active state must be cleared") + + _, ok, err = store.LoadDesired(ctx, "blog") + require.NoError(t, err) + assert.False(t, ok, "desired state must be cleared") + + intent, err := store.LoadIntent(ctx, "blog") + require.NoError(t, err) + assert.False(t, intent.Stopped, "intent must be cleared") + + // Reapply: the first write after retirement allocates a new incarnation. + require.NoError(t, store.SaveOwnership(ctx, domain.AppOwnership{App: "blog"})) + reused, err := store.LoadOwnership(ctx, "blog") + require.NoError(t, err) + require.NotEmpty(t, reused.ID) + assert.NotEqual(t, "app-old", reused.ID, "a reused name must get a new incarnation") + + // The retired incarnation's resources remain protected (prune reads the + // archived record), so they are retained rather than adopted or deleted. + snapshot, err := store.ProtectionSnapshot(ctx) + require.NoError(t, err) + assert.True(t, containsVolumeClaim(snapshot.VolumeClaims, "gordon-blog--web--vol--data", domain.VolumeClaimRetained), + "retained volumes must stay protected after retirement") + claim, ok := snapshot.ImageClaimFor("registry.example.com/blog/web:1.4.2") + if !ok { + t.Fatalf("retired image claim missing from %+v", snapshot.ImageClaims) + } + assert.Equal(t, domain.VolumeClaimReleased, claim.State, "a removed app releases its pinned images") +} + +func TestStore_RetireAppIsAtomicOnEmptyApp(t *testing.T) { + store := newTestStore(t) + require.NoError(t, store.RetireApp(context.Background(), "never-deployed")) + + ownership, err := store.LoadOwnership(context.Background(), "never-deployed") + require.NoError(t, err) + assert.Empty(t, ownership.ID) +} + +// TestProtectionSnapshot_OperationDigestRootCarriesRepository proves an +// unfinished operation's digest root is repository-qualified so prune can +// walk that manifest's closure. +func TestProtectionSnapshot_OperationDigestRootCarriesRepository(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + const manifestDigest = "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + + require.NoError(t, store.SaveOperation(ctx, domain.AppOperation{ + Op: "op-open", Kind: "deploy", App: "blog", + Steps: []domain.AppOperationStep{{ + ID: "service.web.replace", State: domain.AppStepPending, + Digest: manifestDigest, + Image: "registry.example.com/blog/web:1.4.2", + }}, + })) + + snapshot, err := store.ProtectionSnapshot(ctx) + require.NoError(t, err) + + var digestRoot *domain.ProtectionRoot + for i := range snapshot.Roots { + if snapshot.Roots[i].Ref == manifestDigest { + digestRoot = &snapshot.Roots[i] + } + } + if digestRoot == nil { + t.Fatalf("operation digest root missing from %+v", snapshot.Roots) + } + assert.Equal(t, "blog/web", digestRoot.Repository) +} + +func containsVolumeClaim(claims []domain.VolumeClaim, name string, state domain.VolumeClaimState) bool { + for _, claim := range claims { + if claim.Name == name && claim.State == state { + return true + } + } + return false +} diff --git a/internal/adapters/out/appstate/store_test.go b/internal/adapters/out/appstate/store_test.go new file mode 100644 index 000000000..4a4a62b52 --- /dev/null +++ b/internal/adapters/out/appstate/store_test.go @@ -0,0 +1,617 @@ +package appstate_test + +import ( + "context" + "sync" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/out/appstate" + "github.com/bnema/gordon/internal/domain" +) + +func testLogger() zerowrap.Logger { + return zerowrap.Default() +} + +func newTestStore(t *testing.T) *appstate.Store { + t.Helper() + dir := t.TempDir() + store, err := appstate.NewStore(dir, testLogger()) + require.NoError(t, err) + t.Cleanup(func() { + assert.NoError(t, store.Close()) + }) + return store +} + +func testIntent(app, rev string) domain.AppApplyIntent { + return testIntentAfter(app, rev, "") +} + +// testIntentAfter builds an intent that expects a specific former desired +// revision, as the apply use case does. +func testIntentAfter(app, rev, supersedes string) domain.AppApplyIntent { + intent := domain.AppApplyIntent{ + Intent: "apply-test-" + rev, + App: app, + Revision: rev, + Spec: domain.AppSpec{ + Name: app, + Services: []domain.AppService{ + {Name: "web", Image: "img:1", StopGrace: 10 * time.Second}, + }, + }, + Reservations: []domain.AppListenerReservation{ + {Proto: "http", Host: app + ".example.com", Service: "web", App: app}, + }, + CreatedAt: time.Now().UTC(), + } + intent.Supersedes = supersedes + return intent +} + +func backendClaim(app, service, container string, port int) domain.AppListenerReservation { + return domain.AppListenerReservation{ + Proto: "tcp", IP: "127.0.0.1", Port: port, + Service: service, App: app, + Owner: domain.OwnerGordonBackend, ContainerID: container, + } +} + +// TestStore_BackendBindsRegisterRelease proves the gordon-backend claim +// lifecycle: registration records distinguishable loopback claims, +// same-app replacement claims coexist, release drops exactly the retired +// container's claims. +func TestStore_BackendBindsRegisterRelease(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + require.NoError(t, store.RegisterBackendBinds(ctx, []domain.AppListenerReservation{ + backendClaim("blog", "web", "c-old", 18080), + })) + // Replacement coexists (different container, different port). + require.NoError(t, store.RegisterBackendBinds(ctx, []domain.AppListenerReservation{ + backendClaim("blog", "web", "c-new", 18081), + })) + + checkpoint, err := store.LoadCheckpoint(ctx) + require.NoError(t, err) + backends := 0 + for _, res := range checkpoint.Reservations { + if res.Owner == domain.OwnerGordonBackend { + backends++ + assert.Equal(t, "127.0.0.1", res.IP) + } + } + assert.Equal(t, 2, backends) + + // Release drops exactly the retired container's claim. + require.NoError(t, store.ReleaseBackendBinds(ctx, "blog", "c-old")) + checkpoint, err = store.LoadCheckpoint(ctx) + require.NoError(t, err) + backends = 0 + for _, res := range checkpoint.Reservations { + if res.Owner == domain.OwnerGordonBackend { + backends++ + assert.Equal(t, "c-new", res.ContainerID) + } + } + assert.Equal(t, 1, backends) + + // Unknown release is ignored. + require.NoError(t, store.ReleaseBackendBinds(ctx, "blog", "c-gone")) +} + +// TestStore_BackendBindsConflictCrossApp proves fail-closed registration: +// another app's identical loopback claim conflicts; same-app claims never +// self-conflict. +func TestStore_BackendBindsConflictCrossApp(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + require.NoError(t, store.RegisterBackendBinds(ctx, []domain.AppListenerReservation{ + backendClaim("blog", "web", "c-blog", 18080), + })) + err := store.RegisterBackendBinds(ctx, []domain.AppListenerReservation{ + backendClaim("shop", "web", "c-shop", 18080), + }) + require.ErrorIs(t, err, domain.ErrAppReservationConflict) + + // Same app re-registering (restart re-inspection) succeeds. + require.NoError(t, store.RegisterBackendBinds(ctx, []domain.AppListenerReservation{ + backendClaim("blog", "web", "c-blog", 18080), + })) +} + +// TestStore_BackendBindsUDPProtocol proves UDP claims register alongside +// TCP claims on the same host port without conflict (distinct sockets), +// while a second UDP claim on that port from another app conflicts. +func TestStore_BackendBindsUDPProtocol(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + tcp := backendClaim("blog", "web", "c-blog", 19000) + udp := backendClaim("blog", "web", "c-blog", 19000) + udp.Proto = "udp" + require.NoError(t, store.RegisterBackendBinds(ctx, []domain.AppListenerReservation{tcp, udp})) + + otherUDP := backendClaim("shop", "web", "c-shop", 19000) + otherUDP.Proto = "udp" + err := store.RegisterBackendBinds(ctx, []domain.AppListenerReservation{otherUDP}) + require.ErrorIs(t, err, domain.ErrAppReservationConflict) + + // Non-backend protocols are rejected. + bad := backendClaim("blog", "web", "c-blog", 19001) + bad.Proto = "sctp" + require.ErrorIs(t, store.RegisterBackendBinds(ctx, []domain.AppListenerReservation{bad}), domain.ErrAppReservationConflict) +} + +func TestStore_StageCommitMaterialize(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + require.NoError(t, store.StageApply(ctx, testIntent("blog", "rev-1"))) + // Not visible before commit. + _, ok, err := store.LoadDesired(ctx, "blog") + require.NoError(t, err) + assert.False(t, ok) + + require.NoError(t, store.CommitApply(ctx, "blog", "apply-test-rev-1")) + require.NoError(t, store.MaterializeApply(ctx, "blog", "apply-test-rev-1")) + + desired, ok, err := store.LoadDesired(ctx, "blog") + require.NoError(t, err) + require.True(t, ok) + assert.Equal(t, "rev-1", desired.Revision) + assert.Equal(t, "pending", desired.Status) + + // Idempotent replay. + require.NoError(t, store.MaterializeApply(ctx, "blog", "apply-test-rev-1")) + + // Checkpoint folded. + checkpoint, err := store.LoadCheckpoint(ctx) + require.NoError(t, err) + require.Len(t, checkpoint.Reservations, 1) + assert.Equal(t, "blog.example.com", checkpoint.Reservations[0].Host) + + // Commit of a committed intent fails. + require.ErrorIs(t, store.CommitApply(ctx, "blog", "apply-test-rev-1"), domain.ErrAppStateConflict) + // Materialize of unknown intent fails. + require.ErrorIs(t, store.MaterializeApply(ctx, "blog", "apply-nope"), domain.ErrAppIntentNotFound) +} + +func TestStore_RecoverCompletesCommitted(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + require.NoError(t, store.StageApply(ctx, testIntent("blog", "rev-1"))) + require.NoError(t, store.CommitApply(ctx, "blog", "apply-test-rev-1")) + // Simulate crash: never materialized. Recover finishes it. + require.NoError(t, store.Recover(ctx)) + + desired, ok, err := store.LoadDesired(ctx, "blog") + require.NoError(t, err) + require.True(t, ok) + assert.Equal(t, "rev-1", desired.Revision) +} + +func TestStore_ConcurrentSamePortCrossApp(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + // Two apps claiming the same host: store accepts both intents + // (conflict detection lives in the use case); the checkpoint folds + // per-app, and ListApps stays consistent under concurrency. + var wg sync.WaitGroup + errs := make([]error, 4) + for i := range 4 { + wg.Add(1) + go func(i int) { + defer wg.Done() + app := []string{"a", "b", "c", "d"}[i] + intent := testIntent(app, "rev-1") + if err := store.StageApply(ctx, intent); err != nil { + errs[i] = err + return + } + if err := store.CommitApply(ctx, app, intent.Intent); err != nil { + errs[i] = err + return + } + errs[i] = store.MaterializeApply(ctx, app, intent.Intent) + }(i) + } + wg.Wait() + for _, err := range errs { + assert.NoError(t, err) + } + apps, err := store.ListApps(ctx) + require.NoError(t, err) + assert.Equal(t, []string{"a", "b", "c", "d"}, apps) +} + +func TestStore_GarbageProtectsReferenced(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + require.NoError(t, store.StageApply(ctx, testIntent("blog", "rev-1"))) + require.NoError(t, store.CommitApply(ctx, "blog", "apply-test-rev-1")) + require.NoError(t, store.MaterializeApply(ctx, "blog", "apply-test-rev-1")) + require.NoError(t, store.StageApply(ctx, testIntentAfter("blog", "rev-2", "rev-1"))) + require.NoError(t, store.CommitApply(ctx, "blog", "apply-test-rev-2")) + require.NoError(t, store.MaterializeApply(ctx, "blog", "apply-test-rev-2")) + + // rev-1 is superseded but still listed; GC with no in-flight keeps + // retention count (2 < 8) — both survive. + require.NoError(t, store.CollectGarbage(ctx, "blog", nil)) + revs, err := store.ListRevisions(ctx, "blog") + require.NoError(t, err) + assert.Contains(t, revs, "rev-1") + assert.Contains(t, revs, "rev-2") + + // Mark rev-2 active; GC must never evict it even with in-flight noise. + active := domain.AppActive{ + App: "blog", ConvergedRevision: "rev-2", Converged: true, + Services: map[string]domain.AppEffectiveService{ + "web": {EffectiveRevision: "rev-2", Image: "img:1"}, + }, + } + require.NoError(t, store.SaveActive(ctx, active)) + require.NoError(t, store.CollectGarbage(ctx, "blog", []string{"rev-99"})) + _, err = store.LoadRevision(ctx, "blog", "rev-2") + require.NoError(t, err) +} + +func TestStore_CorruptAndIncompatible(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + // Corrupt desired record → scoped corrupt error, other apps unaffected. + require.NoError(t, store.StageApply(ctx, testIntent("blog", "rev-1"))) + require.NoError(t, store.CommitApply(ctx, "blog", "apply-test-rev-1")) + require.NoError(t, store.MaterializeApply(ctx, "blog", "apply-test-rev-1")) + require.NoError(t, appstate.CorruptDesiredForTest(store, "blog", func([]byte) []byte { return []byte("{nope") })) + _, _, err := store.LoadDesired(ctx, "blog") + require.ErrorIs(t, err, domain.ErrAppStateCorrupt) + + require.NoError(t, store.StageApply(ctx, testIntent("other", "rev-1"))) + require.NoError(t, store.CommitApply(ctx, "other", "apply-test-rev-1")) + require.NoError(t, store.MaterializeApply(ctx, "other", "apply-test-rev-1")) + _, ok, err := store.LoadDesired(ctx, "other") + require.NoError(t, err) + assert.True(t, ok) + + // Incompatible version → checkpoint load refuses the database. + require.NoError(t, appstate.CorruptCheckpointForTest(store, func([]byte) []byte { return []byte(`{"version":99}`) })) + _, err = store.LoadCheckpoint(ctx) + require.ErrorIs(t, err, domain.ErrAppStateIncompatible) +} + +func TestStore_AcceptApplyPublishesDesiredRevision(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + require.NoError(t, store.AcceptApply(ctx, testIntent("blog", "rev-1"))) + + desired, ok, err := store.LoadDesired(ctx, "blog") + require.NoError(t, err) + require.True(t, ok) + assert.Equal(t, "rev-1", desired.Revision) + checkpoint, err := store.LoadCheckpoint(ctx) + require.NoError(t, err) + assert.NotEmpty(t, checkpoint.Reservations) + intent, err := store.LoadApplyIntent(ctx, "blog", "apply-test-rev-1") + require.NoError(t, err) + assert.Equal(t, domain.AppIntentApplied, intent.State) + + // A stale supersedes pointer is refused at commit and publishes nothing. + err = store.AcceptApply(ctx, testIntentAfter("blog", "rev-2", "rev-stale")) + require.Error(t, err) + desired, _, err = store.LoadDesired(ctx, "blog") + require.NoError(t, err) + assert.Equal(t, "rev-1", desired.Revision) +} + +func TestStore_GarbageSweepsStagedOrphan(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + // Staged but never committed = orphan. GC removes it; committed stays. + require.NoError(t, store.StageApply(ctx, testIntent("blog", "rev-orphan"))) + require.NoError(t, store.StageApply(ctx, testIntent("blog", "rev-keep"))) + require.NoError(t, store.CommitApply(ctx, "blog", "apply-test-rev-keep")) + require.NoError(t, store.CollectGarbage(ctx, "blog", nil)) + + _, err := store.LoadApplyIntent(ctx, "blog", "apply-test-rev-orphan") + require.ErrorIs(t, err, domain.ErrAppIntentNotFound) + _, err = store.LoadApplyIntent(ctx, "blog", "apply-test-rev-keep") + require.NoError(t, err) +} + +func TestStore_OperationJournal(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + op := domain.AppOperation{ + Op: "op-1", Kind: "deploy", App: "blog", InputRevision: "rev-1", + StartedAt: time.Now().UTC(), + Steps: []domain.AppOperationStep{{ID: "preflight", State: "succeeded"}}, + } + require.NoError(t, store.SaveOperation(ctx, op)) + loaded, err := store.LoadOperation(ctx, "blog", "op-1") + require.NoError(t, err) + assert.Equal(t, "rev-1", loaded.InputRevision) + require.Len(t, loaded.Steps, 1) + + _, err = store.LoadOperation(ctx, "blog", "op-nope") + require.ErrorIs(t, err, domain.ErrAppOperationNotFound) +} + +func TestStore_IntentAndOwnership(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + intent, err := store.LoadIntent(ctx, "fresh") + require.NoError(t, err) + assert.False(t, intent.Stopped) + + require.NoError(t, store.SaveIntent(ctx, domain.AppStopIntent{App: "fresh", Stopped: true, UpdatedBy: "op-1", UpdatedAt: time.Now().UTC()})) + intent, err = store.LoadIntent(ctx, "fresh") + require.NoError(t, err) + assert.True(t, intent.Stopped) + + ownership, err := store.LoadOwnership(ctx, "fresh") + require.NoError(t, err) + assert.Equal(t, "fresh", ownership.App) + ownership.ID = "app-uuid-1" + ownership.Volumes = []domain.AppOwnedVolume{{Name: "d", Service: "db", RuntimeName: "gordon-fresh--db--vol--d", State: "attached"}} + require.NoError(t, store.SaveOwnership(ctx, ownership)) + ownership, err = store.LoadOwnership(ctx, "fresh") + require.NoError(t, err) + require.Len(t, ownership.Volumes, 1) + assert.Equal(t, "app-uuid-1", ownership.ID) +} + +func TestStore_RetentionEvictsUnreferencedBeyondDefault(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + // Apply 10 revisions; only desired stays protected. Default keeps 8 + // unreferenced + every protected reference. + previous := "" + for i := 1; i <= 10; i++ { + rev := "rev-" + itoa2(i) + require.NoError(t, store.StageApply(ctx, testIntentAfter("blog", rev, previous))) + require.NoError(t, store.CommitApply(ctx, "blog", "apply-test-"+rev)) + require.NoError(t, store.MaterializeApply(ctx, "blog", "apply-test-"+rev)) + previous = rev + } + require.NoError(t, store.CollectGarbage(ctx, "blog", nil)) + revs, err := store.ListRevisions(ctx, "blog") + require.NoError(t, err) + // Desired (rev-10) protected + 8 newest unreferenced kept. + assert.Len(t, revs, domain.AppDefaultRevisionRetention+1) + assert.Contains(t, revs, "rev-10") + assert.NotContains(t, revs, "rev-01") +} + +func TestStore_ConfiguredRetentionOverridesDefault(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + store.WithRevisionRetention(2) + assert.Equal(t, 2, store.Retention()) + + previous := "" + for i := 1; i <= 5; i++ { + rev := "rev-" + itoa2(i) + require.NoError(t, store.StageApply(ctx, testIntentAfter("blog", rev, previous))) + require.NoError(t, store.CommitApply(ctx, "blog", "apply-test-"+rev)) + require.NoError(t, store.MaterializeApply(ctx, "blog", "apply-test-"+rev)) + previous = rev + } + require.NoError(t, store.CollectGarbage(ctx, "blog", nil)) + revs, err := store.ListRevisions(ctx, "blog") + require.NoError(t, err) + // Desired (rev-05) protected + 2 newest unreferenced kept. + assert.Len(t, revs, 3) + assert.Contains(t, revs, "rev-05") + assert.NotContains(t, revs, "rev-01") + + store.WithRevisionRetention(0) + assert.Equal(t, domain.AppDefaultRevisionRetention, store.Retention()) +} + +func itoa2(n int) string { + if n < 10 { + return "0" + string(rune('0'+n)) + } + return string(rune('0'+n/10)) + string(rune('0'+n%10)) +} + +func TestStore_ReopenPersistsAcrossClose(t *testing.T) { + dir := t.TempDir() + store, err := appstate.NewStore(dir, testLogger()) + require.NoError(t, err) + ctx := context.Background() + require.NoError(t, store.StageApply(ctx, testIntent("blog", "rev-1"))) + require.NoError(t, store.CommitApply(ctx, "blog", "apply-test-rev-1")) + require.NoError(t, store.MaterializeApply(ctx, "blog", "apply-test-rev-1")) + require.NoError(t, store.Close()) + + reopened, err := appstate.NewStore(dir, testLogger()) + require.NoError(t, err) + t.Cleanup(func() { assert.NoError(t, reopened.Close()) }) + desired, ok, err := reopened.LoadDesired(ctx, "blog") + require.NoError(t, err) + require.True(t, ok) + assert.Equal(t, "rev-1", desired.Revision) +} + +func TestStore_OwnershipUUIDSurvivesNameReuse(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + // First incarnation owns a retained volume under its UUID. + require.NoError(t, store.SaveOwnership(ctx, domain.AppOwnership{ + App: "blog", + ID: "app-uuid-1", + Volumes: []domain.AppOwnedVolume{ + {Name: "d", Service: "db", RuntimeName: "gordon-blog--db--vol--d", State: "retained"}, + }, + })) + + // Simulate remove freeing the name, then a new app reusing it. + // The old record stays under the old UUID; the new app gets a new + // UUID and must not implicitly adopt the retained volume. + require.NoError(t, store.SaveOwnership(ctx, domain.AppOwnership{App: "blog", ID: "app-uuid-2"})) + ownership, err := store.LoadOwnership(ctx, "blog") + require.NoError(t, err) + assert.Equal(t, "app-uuid-2", ownership.ID) + assert.Empty(t, ownership.Volumes) +} + +func TestStore_RecoveryInhibitionsAreDurableAndGenerationScoped(t *testing.T) { + ctx := context.Background() + dir := t.TempDir() + store, err := appstate.NewStore(dir, testLogger()) + require.NoError(t, err) + + require.NoError(t, store.SaveRecoveryInhibition(ctx, domain.AppRecoveryInhibition{ + App: "blog", Service: "web", ContainerID: "c-old", + Reason: domain.AppInhibitReplacementPending, Operation: "op-1", + })) + require.NoError(t, store.SaveRecoveryInhibition(ctx, domain.AppRecoveryInhibition{ + App: "blog", Service: "worker", ContainerID: "c-worker", + Reason: domain.AppInhibitReplacementPending, Operation: "op-2", + })) + require.NoError(t, store.SaveRecoveryInhibition(ctx, domain.AppRecoveryInhibition{ + App: "wiki", Service: "web", ContainerID: "c-wiki", + Reason: domain.AppInhibitReplacementPending, Operation: "op-3", + })) + + // Re-saving the same generation replaces it instead of duplicating. + require.NoError(t, store.SaveRecoveryInhibition(ctx, domain.AppRecoveryInhibition{ + App: "blog", Service: "web", ContainerID: "c-old", + Reason: "removed", Operation: "op-4", + })) + blog, err := store.LoadRecoveryInhibitions(ctx, "blog") + require.NoError(t, err) + require.Len(t, blog, 2) + byService := map[string]domain.AppRecoveryInhibition{} + for _, inhibition := range blog { + byService[inhibition.Service] = inhibition + } + assert.Equal(t, "removed", byService["web"].Reason) + assert.Equal(t, "op-4", byService["web"].Operation) + assert.Equal(t, "c-worker", byService["worker"].ContainerID) + + wiki, err := store.LoadRecoveryInhibitions(ctx, "wiki") + require.NoError(t, err) + require.Len(t, wiki, 1) + + // Clearing one generation never touches another. + require.NoError(t, store.ClearRecoveryInhibition(ctx, "blog", "web", "c-old")) + blog, err = store.LoadRecoveryInhibitions(ctx, "blog") + require.NoError(t, err) + require.Len(t, blog, 1) + assert.Equal(t, "worker", blog[0].Service) + require.NoError(t, store.ClearRecoveryInhibition(ctx, "blog", "web", "c-old"), + "clearing an absent record is not an error") + + unknown, err := store.LoadRecoveryInhibitions(ctx, "absent") + require.NoError(t, err) + assert.Empty(t, unknown) + + require.NoError(t, store.Close()) + + // Reboot: the markers outlive the process. + reopened, err := appstate.NewStore(dir, testLogger()) + require.NoError(t, err) + t.Cleanup(func() { assert.NoError(t, reopened.Close()) }) + blog, err = reopened.LoadRecoveryInhibitions(ctx, "blog") + require.NoError(t, err) + require.Len(t, blog, 1) + assert.Equal(t, "c-worker", blog[0].ContainerID) + + cancelled, cancel := context.WithCancel(ctx) + cancel() + if _, err := reopened.LoadRecoveryInhibitions(cancelled, "blog"); err == nil { + t.Fatal("a cancelled context must fail the read") + } + require.Error(t, reopened.SaveRecoveryInhibition(cancelled, domain.AppRecoveryInhibition{App: "blog"})) + require.Error(t, reopened.ClearRecoveryInhibition(cancelled, "blog", "worker", "c-worker")) +} + +func TestStore_DesiredDevicesRoundTrip(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + intent := testIntent("blog", "rev-1") + intent.Spec.Services[0].Devices = []string{"test_gpu"} + require.NoError(t, store.StageApply(ctx, intent)) + require.NoError(t, store.CommitApply(ctx, "blog", "apply-test-rev-1")) + require.NoError(t, store.MaterializeApply(ctx, "blog", "apply-test-rev-1")) + + desired, ok, err := store.LoadDesired(ctx, "blog") + require.NoError(t, err) + require.True(t, ok) + require.Len(t, desired.Spec.Services, 1) + assert.Equal(t, []string{"test_gpu"}, desired.Spec.Services[0].Devices, "logical names persist in revisions") +} + +// TestStore_DesiredWithoutDevicesLoads proves a desired record written by a +// Gordon release that predates service devices still loads: the absent JSON +// key decodes to a nil slice, never an error or a phantom device. +func TestStore_DesiredWithoutDevicesLoads(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + + require.NoError(t, appstate.SeedDesiredRecordForTest(store, "blog", []byte(legacyDesiredWithoutDevices))) + + desired, ok, err := store.LoadDesired(ctx, "blog") + require.NoError(t, err) + require.True(t, ok) + require.Len(t, desired.Spec.Services, 1) + assert.Nil(t, desired.Spec.Services[0].Devices, "a missing devices key must not become a phantom device") +} + +// legacyDesiredWithoutDevices is a verbatim state.db desired record from a +// release predating service devices. Unlike current serializer output, its +// service object carries no "Devices" key at all rather than a null one. +const legacyDesiredWithoutDevices = `{ + "revision": "rev-1", + "app": "blog", + "accepted_at": "2026-01-01T00:00:00Z", + "source_sha256": "", + "spec": { + "Name": "blog", + "Env": null, + "Services": [ + { + "Name": "web", + "Image": "img:1", + "Command": null, + "StopGrace": 10000000000, + "Readiness": {"Type": "", "Path": "", "Contains": "", "Port": 0, "Timeout": 0}, + "HTTP": null, + "TCP": null, + "UDP": null, + "Secrets": null, + "Volumes": null, + "Binds": null, + "Databases": null, + "Backup": {"Postgres": null, "Volume": null} + } + ], + "Networks": null + }, + "reservations": [{"proto": "http", "host": "blog.example.com", "service": "web", "app": "blog"}], + "secrets_required": null, + "secrets_env": {}, + "status": "pending" +}` diff --git a/internal/adapters/out/docker/live_network_probe_test.go b/internal/adapters/out/docker/live_network_probe_test.go new file mode 100644 index 000000000..095b5e28f --- /dev/null +++ b/internal/adapters/out/docker/live_network_probe_test.go @@ -0,0 +1,255 @@ +//go:build live + +// Live bounded-network-readiness verification against a real Docker +// daemon. These tests are opt-in: build with `-tags live` and run with the +// daemon reachable. They create and remove their own networks and +// containers, and assert no helper is ever left behind. +// +// go test -tags live ./internal/adapters/out/docker/ -run TestLiveNetworkProbe -count=1 -v + +package docker + +import ( + "context" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +// liveProbeNetwork creates an owned private network for one probe scenario. +func liveProbeNetwork(t *testing.T, runtime *Runtime, ctx context.Context) string { + t.Helper() + name := liveName("gordon-live-probe") + require.NoError(t, runtime.CreateNetwork(ctx, name, domain.NetworkConfig{ + Driver: "bridge", Labels: domain.AppPrivateNetworkLabels("live", name), + })) + t.Cleanup(func() { + cleanup, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + _ = runtime.RemoveNetwork(cleanup, name) + }) + return name +} + +// liveProbeHelpers counts leftover bounded-readiness helpers. +func liveProbeHelpers(t *testing.T, runtime *Runtime, ctx context.Context) int { + t.Helper() + containers, err := runtime.ListContainers(ctx, true) + require.NoError(t, err) + count := 0 + for _, c := range containers { + if c != nil && c.Labels[domain.LabelPurpose] == domain.PurposeNetworkProbe { + count++ + } + } + return count +} + +// liveProbeTarget starts one listener container on the network. +func liveProbeTarget(t *testing.T, runtime *Runtime, ctx context.Context, network, listenScript string) *domain.Container { + t.Helper() + return liveCreate(t, runtime, ctx, &domain.ContainerConfig{ + Image: liveImage, + Name: liveName("live-probe-target"), + NetworkMode: network, + Entrypoint: []string{"sh", "-c", listenScript}, + AutoRemove: false, + }) +} + +// TestLiveNetworkProbe_PeerFacingReadyAndCleanedUp proves a published-free +// internal listener answers a bounded probe, the helper is removed, and the +// target exposes no host port. +func TestLiveNetworkProbe_PeerFacingReadyAndCleanedUp(t *testing.T) { + runtime, ctx := liveRuntime(t) + network := liveProbeNetwork(t, runtime, ctx) + target := liveProbeTarget(t, runtime, ctx, network, + `while true; do printf "HTTP/1.0 200 OK\r\nContent-Length: 2\r\n\r\nok" | nc -l -p 8080; done`) + + before := liveProbeHelpers(t, runtime, ctx) + inspect, err := runtime.InspectContainer(ctx, target.ID) + require.NoError(t, err) + require.False(t, inspect.StartedAt.IsZero(), "the target must report an execution start") + assert.Empty(t, inspect.Ports, "an internal listener must not publish a host port") + + result, err := runtime.ProbeContainerNetwork(ctx, domain.ContainerNetworkProbeRequest{ + TargetContainerID: target.ID, + ExpectedStartedAt: inspect.StartedAt, + Network: network, + Protocol: domain.ProbeProtocolHTTP, + Port: 8080, + Path: "/healthz", + Timeout: 10 * time.Second, + }) + require.NoError(t, err) + assert.True(t, result.Ready) + assert.Equal(t, 200, result.Status) + + assert.Equal(t, before, liveProbeHelpers(t, runtime, ctx), "the helper must always be removed") +} + +// TestLiveNetworkProbe_LoopbackOnlyListenerFails proves a target listening +// only on its own loopback never reports ready from the network. +func TestLiveNetworkProbe_LoopbackOnlyListenerFails(t *testing.T) { + runtime, ctx := liveRuntime(t) + network := liveProbeNetwork(t, runtime, ctx) + target := liveProbeTarget(t, runtime, ctx, network, + `while true; do printf "HTTP/1.0 200 OK\r\nContent-Length: 2\r\n\r\nok" | nc -l -p 8081 -s 127.0.0.1; done`) + + inspect, err := runtime.InspectContainer(ctx, target.ID) + require.NoError(t, err) + + result, err := runtime.ProbeContainerNetwork(ctx, domain.ContainerNetworkProbeRequest{ + TargetContainerID: target.ID, + ExpectedStartedAt: inspect.StartedAt, + Network: network, + Protocol: domain.ProbeProtocolHTTP, + Port: 8081, + Path: "/", + Timeout: 5 * time.Second, + }) + require.NoError(t, err) + assert.False(t, result.Ready, "a loopback-only listener must never be reachable over the private network") + assert.NotEmpty(t, result.Diagnostic) +} + +// TestLiveNetworkProbe_TCPReady proves a TCP readiness session accepts an +// open port and rejects a closed one. +func TestLiveNetworkProbe_TCPReady(t *testing.T) { + runtime, ctx := liveRuntime(t) + network := liveProbeNetwork(t, runtime, ctx) + target := liveProbeTarget(t, runtime, ctx, network, + `while true; do nc -l -p 5432 >/dev/null 2>&1; done`) + + inspect, err := runtime.InspectContainer(ctx, target.ID) + require.NoError(t, err) + + ready, err := runtime.ProbeContainerNetwork(ctx, domain.ContainerNetworkProbeRequest{ + TargetContainerID: target.ID, + ExpectedStartedAt: inspect.StartedAt, + Network: network, + Protocol: domain.ProbeProtocolTCP, + Port: 5432, + Timeout: 5 * time.Second, + }) + require.NoError(t, err) + assert.True(t, ready.Ready) + + closed, err := runtime.ProbeContainerNetwork(ctx, domain.ContainerNetworkProbeRequest{ + TargetContainerID: target.ID, + ExpectedStartedAt: inspect.StartedAt, + Network: network, + Protocol: domain.ProbeProtocolTCP, + Port: 5499, + Timeout: 2 * time.Second, + }) + require.NoError(t, err) + assert.False(t, closed.Ready) +} + +// TestLiveNetworkProbe_RejectsStaleGeneration proves a probe whose expected +// execution start no longer matches the target fails instead of reporting +// readiness for a generation that restarted. +func TestLiveNetworkProbe_RejectsStaleGeneration(t *testing.T) { + runtime, ctx := liveRuntime(t) + network := liveProbeNetwork(t, runtime, ctx) + target := liveProbeTarget(t, runtime, ctx, network, + `while true; do printf "HTTP/1.0 200 OK\r\nContent-Length: 2\r\n\r\nok" | nc -l -p 8080; done`) + + _, err := runtime.ProbeContainerNetwork(ctx, domain.ContainerNetworkProbeRequest{ + TargetContainerID: target.ID, + ExpectedStartedAt: time.Now().Add(-time.Hour).UTC(), + Network: network, + Protocol: domain.ProbeProtocolHTTP, + Port: 8080, + Path: "/", + Timeout: 5 * time.Second, + }) + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrAppStateConflict) +} + +// TestLiveNetworkProbe_NonHTTPGreetingIsNotReady proves a listener whose +// first line merely starts with 2 or 3 (a non-HTTP service on a +// misdirected port) is never reported ready. +func TestLiveNetworkProbe_NonHTTPGreetingIsNotReady(t *testing.T) { + runtime, ctx := liveRuntime(t) + network := liveProbeNetwork(t, runtime, ctx) + target := liveProbeTarget(t, runtime, ctx, network, + `while true; do printf '300 ready\r\n' | nc -l -p 8083; done`) + + inspect, err := runtime.InspectContainer(ctx, target.ID) + require.NoError(t, err) + + result, err := runtime.ProbeContainerNetwork(ctx, domain.ContainerNetworkProbeRequest{ + TargetContainerID: target.ID, + ExpectedStartedAt: inspect.StartedAt, + Network: network, + Protocol: domain.ProbeProtocolHTTP, + Port: 8083, + Path: "/", + Timeout: 5 * time.Second, + }) + require.NoError(t, err) + assert.False(t, result.Ready, "a non-HTTP greeting must never satisfy HTTP readiness") +} + +// TestLiveNetworkProbe_HungTargetIsUnhealthyNotInfrastructure proves a +// target that accepts the connection but never answers yields an unhealthy +// result rather than an infrastructure error: the retry loop must keep +// polling instead of aborting the wait. +func TestLiveNetworkProbe_HungTargetIsUnhealthyNotInfrastructure(t *testing.T) { + runtime, ctx := liveRuntime(t) + network := liveProbeNetwork(t, runtime, ctx) + target := liveProbeTarget(t, runtime, ctx, network, + `while true; do sleep 3600 | nc -l -p 8084; done`) + + inspect, err := runtime.InspectContainer(ctx, target.ID) + require.NoError(t, err) + + result, err := runtime.ProbeContainerNetwork(ctx, domain.ContainerNetworkProbeRequest{ + TargetContainerID: target.ID, + ExpectedStartedAt: inspect.StartedAt, + Network: network, + Protocol: domain.ProbeProtocolHTTP, + Port: 8084, + Path: "/", + Timeout: 4 * time.Second, + }) + require.NoError(t, err, "a hanging target is a readiness failure, not an infrastructure error") + assert.False(t, result.Ready) + assert.NotEmpty(t, result.Diagnostic) +} + +// TestLiveNetworkProbe_RejectsForeignNetwork proves the helper is bound to +// exactly one network: a target attached elsewhere is never probed across +// networks. +func TestLiveNetworkProbe_RejectsForeignNetwork(t *testing.T) { + runtime, ctx := liveRuntime(t) + ownerNetwork := liveProbeNetwork(t, runtime, ctx) + foreignNetwork := liveProbeNetwork(t, runtime, ctx) + target := liveProbeTarget(t, runtime, ctx, ownerNetwork, + `while true; do printf "HTTP/1.0 200 OK\r\nContent-Length: 2\r\n\r\nok" | nc -l -p 8080; done`) + + before := liveProbeHelpers(t, runtime, ctx) + inspect, err := runtime.InspectContainer(ctx, target.ID) + require.NoError(t, err) + + _, err = runtime.ProbeContainerNetwork(ctx, domain.ContainerNetworkProbeRequest{ + TargetContainerID: target.ID, + ExpectedStartedAt: inspect.StartedAt, + Network: foreignNetwork, + Protocol: domain.ProbeProtocolHTTP, + Port: 8080, + Path: "/", + Timeout: 5 * time.Second, + }) + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrAppStateConflict) + + assert.Equal(t, before, liveProbeHelpers(t, runtime, ctx), "no helper may survive a rejected probe") +} diff --git a/internal/adapters/out/docker/live_runtime_test.go b/internal/adapters/out/docker/live_runtime_test.go new file mode 100644 index 000000000..f30730cfe --- /dev/null +++ b/internal/adapters/out/docker/live_runtime_test.go @@ -0,0 +1,208 @@ +//go:build live + +// Live runtime verification against a real Docker daemon. These tests are +// opt-in: build with `-tags live` and run with the daemon reachable. They +// create and remove their own networks, volumes, and containers. +// +// go test -tags live ./internal/adapters/out/docker/ -run TestLive -count=1 -v + +package docker + +import ( + "context" + "fmt" + "net" + "strings" + "testing" + "time" + + "github.com/moby/moby/client" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +const liveImage = "golang:1.27-alpine" + +func liveRuntime(t *testing.T) (*Runtime, context.Context) { + t.Helper() + cli, err := client.New( + client.FromEnv, + client.WithAPIVersionNegotiation(), + ) + require.NoError(t, err) + runtime := NewRuntimeWithClient(cli) + + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute) + t.Cleanup(cancel) + require.NoError(t, runtime.Ping(ctx), "a live Docker daemon is required") + return runtime, ctx +} + +func liveName(prefix string) string { + return fmt.Sprintf("%s-%d", prefix, time.Now().UnixNano()) +} + +func liveCreate(t *testing.T, runtime *Runtime, ctx context.Context, config *domain.ContainerConfig) *domain.Container { + t.Helper() + created, err := runtime.CreateContainer(ctx, config) + require.NoError(t, err) + t.Cleanup(func() { + cleanup, cancel := context.WithTimeout(context.Background(), 60*time.Second) + defer cancel() + // Force-remove in one step. A separate StopContainer would wait the + // full stop grace (30s) and then consume the cleanup budget, leaving + // the container behind and exhausting the daemon's address pool. + _ = runtime.RemoveContainer(cleanup, created.ID, true) + }) + require.NoError(t, runtime.StartContainer(ctx, created.ID)) + return created +} + +// TestLiveRuntime_PrivateNetworksIsolateApps proves containers on two +// app-owned networks cannot reach each other, while containers sharing one +// network can. +func TestLiveRuntime_PrivateNetworksIsolateApps(t *testing.T) { + runtime, ctx := liveRuntime(t) + netA := liveName("gordon-live-a") + netB := liveName("gordon-live-b") + for _, name := range []string{netA, netB} { + require.NoError(t, runtime.CreateNetwork(ctx, name, domain.NetworkConfig{ + Driver: "bridge", Labels: domain.AppPrivateNetworkLabels("live", name), + })) + networkName := name + t.Cleanup(func() { + cleanup, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + _ = runtime.RemoveNetwork(cleanup, networkName) + }) + } + + listenerName := liveName("live-a") + peerName := liveName("live-b") + foreignName := liveName("live-c") + listenCmd := []string{"sh", "-c", "while true; do nc -l -p 1234 >/dev/null 2>&1; done"} + probeCmd := []string{"sh", "-c", "nc -z -w 2 " + listenerName + " 1234"} + + liveCreate(t, runtime, ctx, &domain.ContainerConfig{ + Image: liveImage, Name: listenerName, NetworkMode: netA, + Entrypoint: listenCmd, AutoRemove: false, + }) + liveCreate(t, runtime, ctx, &domain.ContainerConfig{ + Image: liveImage, Name: peerName, NetworkMode: netA, + Entrypoint: []string{"sleep", "120"}, AutoRemove: false, + }) + liveCreate(t, runtime, ctx, &domain.ContainerConfig{ + Image: liveImage, Name: foreignName, NetworkMode: netB, + Entrypoint: []string{"sleep", "120"}, AutoRemove: false, + }) + + // Give the listener a moment, then prove same-network reachability. + require.Eventually(t, func() bool { + result, err := runtime.ExecInContainer(ctx, peerName, probeCmd) + return err == nil && result.ExitCode == 0 + }, 30*time.Second, time.Second, "containers on the same app network must reach each other") + + // A container on another app network must not reach the listener: the + // name does not resolve and the network is not routable. + result, err := runtime.ExecInContainer(ctx, foreignName, probeCmd) + if err == nil { + assert.NotZero(t, result.ExitCode, "cross-app network traffic must be impossible") + } +} + +// TestLiveRuntime_LoopbackPublishAndObservedBind proves a backend port is +// published only on loopback and that the adapter reports exactly that +// binding. +func TestLiveRuntime_LoopbackPublishAndObservedBind(t *testing.T) { + runtime, ctx := liveRuntime(t) + network := liveName("gordon-live-l4") + require.NoError(t, runtime.CreateNetwork(ctx, network, domain.NetworkConfig{ + Driver: "bridge", Labels: domain.AppPrivateNetworkLabels("live", network), + })) + t.Cleanup(func() { + cleanup, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + _ = runtime.RemoveNetwork(cleanup, network) + }) + + created := liveCreate(t, runtime, ctx, &domain.ContainerConfig{ + Image: liveImage, Name: liveName("live-l4"), NetworkMode: network, + Entrypoint: []string{"sh", "-c", "while true; do nc -l -p 8080 >/dev/null 2>&1; done"}, + PortPublishes: []domain.ContainerPortPublish{{ + HostIP: "127.0.0.1", HostPort: 0, ContainerPort: 8080, Protocol: domain.NetworkProtocolTCP, + }}, + AutoRemove: false, + }) + + binds, err := runtime.GetContainerBackendBinds(ctx, created.ID, []domain.ContainerBackendPort{ + {ContainerPort: 8080, Protocol: domain.NetworkProtocolTCP}, + }) + require.NoError(t, err) + require.Len(t, binds, 1) + hostPort := binds[0].HostPort + require.Positive(t, hostPort) + + // The loopback bind accepts a connection from the host. A wildcard or + // container-address bind would have been rejected by + // GetContainerBackendBinds above, which requires exactly one 127.0.0.1 + // mapping. + require.Eventually(t, func() bool { + conn, dialErr := net.DialTimeout("tcp", fmt.Sprintf("127.0.0.1:%d", hostPort), time.Second) + if dialErr != nil { + return false + } + _ = conn.Close() + return true + }, 30*time.Second, time.Second, "the loopback bind must accept connections") +} + +// TestLiveRuntime_ReadOnlyVolumeIsMountedReadOnly proves a volume declared +// read-only rejects writes while a writable volume accepts them. +func TestLiveRuntime_ReadOnlyVolumeIsMountedReadOnly(t *testing.T) { + runtime, ctx := liveRuntime(t) + network := liveName("gordon-live-ro") + require.NoError(t, runtime.CreateNetwork(ctx, network, domain.NetworkConfig{ + Driver: "bridge", Labels: domain.AppPrivateNetworkLabels("live", network), + })) + t.Cleanup(func() { + cleanup, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + _ = runtime.RemoveNetwork(cleanup, network) + }) + + roVolume := liveName("live-ro-vol") + rwVolume := liveName("live-rw-vol") + for _, name := range []string{roVolume, rwVolume} { + require.NoError(t, runtime.CreateVolume(ctx, name, map[string]string{domain.LabelManaged: "true"})) + volumeName := name + t.Cleanup(func() { + cleanup, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + _ = runtime.RemoveVolume(cleanup, volumeName, true) + }) + } + + created := liveCreate(t, runtime, ctx, &domain.ContainerConfig{ + Image: liveImage, Name: liveName("live-ro"), NetworkMode: network, + Entrypoint: []string{"sleep", "120"}, + Volumes: map[string]string{"/rw": rwVolume}, + ReadOnlyVolumes: map[string]string{ + "/ro": roVolume, + }, + AutoRemove: false, + }) + + writable, err := runtime.ExecInContainer(ctx, created.ID, []string{"sh", "-c", "echo data > /rw/file && cat /rw/file"}) + require.NoError(t, err) + assert.Zero(t, writable.ExitCode, "a writable volume must accept writes") + + readOnly, err := runtime.ExecInContainer(ctx, created.ID, []string{"sh", "-c", "echo data > /ro/file"}) + require.NoError(t, err) + assert.NotZero(t, readOnly.ExitCode, "a read-only volume must reject writes") + assert.True(t, strings.Contains(string(readOnly.Stderr), "Read-only") || + strings.Contains(string(readOnly.Stderr), "read-only") || + strings.Contains(string(readOnly.Stderr), "Permission denied"), + "unexpected read-only failure output: %s", string(readOnly.Stderr)) +} diff --git a/internal/adapters/out/docker/live_shared_network_test.go b/internal/adapters/out/docker/live_shared_network_test.go new file mode 100644 index 000000000..3be88040a --- /dev/null +++ b/internal/adapters/out/docker/live_shared_network_test.go @@ -0,0 +1,65 @@ +//go:build live + +package docker + +import ( + "context" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +// TestLiveRuntime_SharedMembershipAndDetachment proves cross-app traffic is +// possible only for services explicitly enrolled in the same shared +// network, and that unrelated services stay detached. +func TestLiveRuntime_SharedMembershipAndDetachment(t *testing.T) { + runtime, ctx := liveRuntime(t) + netA := liveProbeNetwork(t, runtime, ctx) + netB := liveProbeNetwork(t, runtime, ctx) + + sharedName := liveName("gordon-live-shared") + require.NoError(t, runtime.CreateNetwork(ctx, sharedName, domain.NetworkConfig{ + Driver: "bridge", Labels: domain.AppSharedNetworkLabels("live-shared"), + })) + t.Cleanup(func() { + cleanup, cancel := context.WithTimeout(context.Background(), 30*time.Second) + defer cancel() + _ = runtime.RemoveNetwork(cleanup, sharedName) + }) + + listenerName := liveName("live-shared-listener") + listener := liveCreate(t, runtime, ctx, &domain.ContainerConfig{ + Image: liveImage, Name: listenerName, NetworkMode: netA, + Entrypoint: []string{"sh", "-c", "while true; do nc -l -p 9100 >/dev/null 2>&1; done"}, + AutoRemove: false, + }) + require.NoError(t, runtime.ConnectContainerToNetwork(ctx, listener.ID, sharedName)) + + // Enrolled on the shared network from another app network. + enrolled := liveCreate(t, runtime, ctx, &domain.ContainerConfig{ + Image: liveImage, Name: liveName("live-shared-enrolled"), NetworkMode: netB, + Entrypoint: []string{"sleep", "180"}, AutoRemove: false, + }) + require.NoError(t, runtime.ConnectContainerToNetwork(ctx, enrolled.ID, sharedName)) + + // An unrelated service on the same app network but NOT enrolled. + outsider := liveCreate(t, runtime, ctx, &domain.ContainerConfig{ + Image: liveImage, Name: liveName("live-shared-outsider"), NetworkMode: netB, + Entrypoint: []string{"sleep", "180"}, AutoRemove: false, + }) + + probe := []string{"sh", "-c", "nc -z -w 3 " + listenerName + " 9100"} + + require.Eventually(t, func() bool { + result, err := runtime.ExecInContainer(ctx, enrolled.ID, probe) + return err == nil && result.ExitCode == 0 + }, 60*time.Second, 2*time.Second, "an enrolled service must reach the peer through the shared network") + + result, err := runtime.ExecInContainer(ctx, outsider.ID, probe) + require.NoError(t, err, "the exec itself must run; a transport error is not proof of isolation") + assert.NotZero(t, result.ExitCode, "an unrelated service must not reach a peer it is not enrolled with") +} diff --git a/internal/adapters/out/docker/log_stream.go b/internal/adapters/out/docker/log_stream.go new file mode 100644 index 000000000..e371e050d --- /dev/null +++ b/internal/adapters/out/docker/log_stream.go @@ -0,0 +1,133 @@ +package docker + +import ( + "bufio" + "context" + "fmt" + "io" + "strings" + "time" + + cerrdefs "github.com/containerd/errdefs" + "github.com/moby/moby/api/pkg/stdcopy" + "github.com/moby/moby/client" + + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/domain" +) + +// maxLogLineBytes bounds one exported line; longer lines are truncated +// so one runaway writer cannot exhaust memory. +const maxLogLineBytes = 64 * 1024 + +var _ out.ContainerLogStreamer = (*Runtime)(nil) + +// StreamContainerLogs implements out.ContainerLogStreamer. It follows +// stdout and stderr with runtime timestamps, demultiplexes them, and +// emits one line at a time until the stream ends or ctx is canceled. +func (r *Runtime) StreamContainerLogs(ctx context.Context, containerID string, since time.Time, emit func(domain.ContainerLogLine)) error { + options := client.ContainerLogsOptions{ + ShowStdout: true, + ShowStderr: true, + Follow: true, + Timestamps: true, + } + if !since.IsZero() { + options.Since = since.UTC().Format(time.RFC3339Nano) + } + logs, err := r.client.ContainerLogs(ctx, containerID, options) + if err != nil { + if cerrdefs.IsNotFound(err) { + return fmt.Errorf("%w: %s", domain.ErrContainerNotFound, containerID) + } + return fmt.Errorf("follow container logs: %w", err) + } + defer logs.Close() + // Closing the stream unblocks the demux copy on cancellation. + stop := context.AfterFunc(ctx, func() { _ = logs.Close() }) + defer stop() + + stdoutR, stdoutW := io.Pipe() + stderrR, stderrW := io.Pipe() + done := make(chan struct{}, 2) + lines := make(chan domain.ContainerLogLine) + go scanLogLines(stdoutR, domain.LogStreamStdout, lines, done) + go scanLogLines(stderrR, domain.LogStreamStderr, lines, done) + + copyErr := make(chan error, 1) + go func() { + _, err := stdcopy.StdCopy(stdoutW, stderrW, logs) + _ = stdoutW.Close() + _ = stderrW.Close() + copyErr <- err + }() + + for pending := 2; pending > 0; { + select { + case line := <-lines: + emit(line) + case <-done: + pending-- + } + } + err = <-copyErr + if ctx.Err() != nil { + return ctx.Err() + } + if err != nil { + return fmt.Errorf("demultiplex container logs: %w", err) + } + return nil +} + +// scanLogLines splits one demultiplexed stream into timestamped lines. +// The reader is always drained so the demux copy never blocks. +func scanLogLines(r *io.PipeReader, stream string, lines chan<- domain.ContainerLogLine, done chan<- struct{}) { + defer func() { done <- struct{}{} }() + reader := bufio.NewReaderSize(r, 4096) + for { + raw, err := readBoundedLine(reader) + if raw != "" { + lines <- parseTimestampedLine(raw, stream) + } + if err != nil { + _ = r.CloseWithError(err) + return + } + } +} + +// readBoundedLine returns one line without its newline, truncated to +// maxLogLineBytes; the rest of an oversized line is discarded. +func readBoundedLine(reader *bufio.Reader) (string, error) { + var b strings.Builder + for { + chunk, err := reader.ReadSlice('\n') + if room := maxLogLineBytes - b.Len(); room > 0 { + if len(chunk) > room { + chunk = chunk[:room] + } + b.Write(chunk) + } + if err == bufio.ErrBufferFull { + continue + } + return strings.TrimRight(b.String(), "\r\n"), err + } +} + +// parseTimestampedLine splits the RFC3339Nano prefix Docker adds when +// Timestamps is set. Lines without a valid prefix keep the receive time. +func parseTimestampedLine(raw, stream string) domain.ContainerLogLine { + raw = strings.ToValidUTF8(raw, "\uFFFD") + line := domain.ContainerLogLine{Time: time.Now().UTC(), Stream: stream, Body: raw} + prefix, body, ok := strings.Cut(raw, " ") + if !ok { + return line + } + if ts, err := time.Parse(time.RFC3339Nano, prefix); err == nil { + line.Time = ts + line.Body = body + } + return line +} diff --git a/internal/adapters/out/docker/log_stream_test.go b/internal/adapters/out/docker/log_stream_test.go new file mode 100644 index 000000000..c89e416f9 --- /dev/null +++ b/internal/adapters/out/docker/log_stream_test.go @@ -0,0 +1,94 @@ +package docker + +import ( + "bufio" + "context" + "encoding/binary" + "net/http" + "net/http/httptest" + "strings" + "testing" + "time" + + "github.com/moby/moby/api/pkg/stdcopy" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +func TestRuntime_StreamContainerLogs_DemuxesTimestampedLines(t *testing.T) { + var gotFollow string + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + gotFollow = r.URL.Query().Get("follow") + w.Header().Set("Content-Type", "application/vnd.docker.multiplexed-stream") + _, _ = w.Write(muxFrame(stdcopy.Stdout, "2026-09-10T12:00:01.5Z hello\n")) + _, _ = w.Write(muxFrame(stdcopy.Stderr, "2026-09-10T12:00:02Z boom\n")) + })) + defer server.Close() + + runtime := newRuntimeForHTTPServer(t, server) + var lines []domain.ContainerLogLine + err := runtime.StreamContainerLogs(context.Background(), "c1", time.Time{}, func(l domain.ContainerLogLine) { + lines = append(lines, l) + }) + + require.NoError(t, err) + assert.Equal(t, "1", gotFollow) + require.Len(t, lines, 2) + byStream := map[string]domain.ContainerLogLine{} + for _, l := range lines { + byStream[l.Stream] = l + } + assert.Equal(t, "hello", byStream[domain.LogStreamStdout].Body) + assert.Equal(t, time.Date(2026, 9, 10, 12, 0, 1, 500000000, time.UTC), byStream[domain.LogStreamStdout].Time) + assert.Equal(t, "boom", byStream[domain.LogStreamStderr].Body) +} + +// muxFrame encodes one Docker multiplexed-stream frame. +func muxFrame(stream stdcopy.StdType, payload string) []byte { + frame := make([]byte, 8, 8+len(payload)) + frame[0] = byte(stream) + binary.BigEndian.PutUint32(frame[4:], uint32(len(payload))) + return append(frame, payload...) +} + +func TestRuntime_StreamContainerLogs_NotFound(t *testing.T) { + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(http.StatusNotFound) + _, _ = w.Write([]byte(`{"message":"No such container: gone"}`)) + })) + defer server.Close() + + runtime := newRuntimeForHTTPServer(t, server) + err := runtime.StreamContainerLogs(context.Background(), "gone", time.Time{}, func(domain.ContainerLogLine) {}) + + assert.ErrorIs(t, err, domain.ErrContainerNotFound) +} + +func TestReadBoundedLine_TruncatesOversizedLines(t *testing.T) { + long := strings.Repeat("x", maxLogLineBytes+100) + reader := bufio.NewReaderSize(strings.NewReader(long+"\nnext\n"), 16) + + first, err := readBoundedLine(reader) + require.NoError(t, err) + second, err := readBoundedLine(reader) + require.NoError(t, err) + + assert.Len(t, first, maxLogLineBytes) + assert.Equal(t, "next", second) +} + +func TestParseTimestampedLine_KeepsBodyWithoutTimestamp(t *testing.T) { + line := parseTimestampedLine("plain text line", domain.LogStreamStdout) + assert.Equal(t, "plain text line", line.Body) +} + +func TestParseTimestampedLine_SanitizesInvalidUTF8(t *testing.T) { + line := parseTimestampedLine("2026-09-10T12:00:01Z bad\xffbytes", domain.LogStreamStdout) + assert.Equal(t, "bad\uFFFDbytes", line.Body) + + plain := parseTimestampedLine("plain\xff", domain.LogStreamStdout) + assert.Equal(t, "plain\uFFFD", plain.Body) +} diff --git a/internal/adapters/out/docker/network_probe.go b/internal/adapters/out/docker/network_probe.go new file mode 100644 index 000000000..db38e51f1 --- /dev/null +++ b/internal/adapters/out/docker/network_probe.go @@ -0,0 +1,439 @@ +package docker + +import ( + "bytes" + "context" + "errors" + "fmt" + "io" + "math" + "strconv" + "strings" + "time" + + "github.com/bnema/zerowrap" + cerrdefs "github.com/containerd/errdefs" + "github.com/moby/moby/api/pkg/stdcopy" + "github.com/moby/moby/api/types/container" + "github.com/moby/moby/client" + + "github.com/bnema/gordon/internal/domain" +) + +// DefaultNetworkProbeImage is the digest-pinned, multi-arch helper image +// one bounded internal readiness session runs. It provides only a POSIX +// shell with busybox nc/printf. Pin by digest, never by tag: the helper +// must be byte-identical across hosts and engines. The image must be +// pre-provisioned and available offline on the target host, because a +// probe never pulls. +const DefaultNetworkProbeImage = "alpine@sha256:d9e853e87e55526f6b2917df91a2115c36dd7c696a35be12163d44e6e2a4b6bc" + +// Bounded resources and windows for the probe helper. They exist only to +// bound a misbehaving helper, not to size a workload. +const ( + probeHelperMemory = 32 << 20 + probeHelperNanoCPUs = 500_000_000 + probeHelperPIDs = 32 + // probeHelperCleanup bounds a force-remove of one helper. + probeHelperCleanup = 30 * time.Second + probeMaxSeconds = 30 + // probeHelperOrphanAge is how old a leftover probe helper must be + // before a later probe reclaims it. Live helpers last seconds, so this + // never touches an in-flight session, including a concurrent one. + probeHelperOrphanAge = 10 * time.Minute +) + +// networkProbeArgv is the helper argv: a shell script plus positional +// arguments, so the target address never appears in a shell string. +type networkProbeArgv struct { + script string + args []string +} + +// httpProbeScript sends one HTTP/1.0 request over a raw connection and +// accepts only a 2xx/3xx status line. It never follows redirects and never +// consults a proxy. The first line must be an HTTP status line: a +// non-HTTP greeting that merely starts with 2 or 3 is never readiness. It +// prints the observed status (000 when none) for diagnostics. +const httpProbeScript = `path="$1"; tmo="$2"; ip="$3"; port="$4" +line=$(printf 'GET %s HTTP/1.0\r\nHost: readiness\r\nConnection: close\r\n\r\n' "$path" | nc -w "$tmo" "$ip" "$port" | head -c 128 | head -n1) +code=$(printf '%s\n' "$line" | awk '$1 ~ /^HTTP\/[0-9]+\.[0-9]+$/ && $2 ~ /^[0-9][0-9][0-9]$/ { print $2 }') +case "$code" in + 2??|3??) printf '%s\n' "$code"; exit 0;; + [0-9][0-9][0-9]) printf '%s\n' "$code"; exit 1;; + *) printf '%s\n' 000; exit 1;; +esac` + +// tcpProbeScript accepts only an accepted connection. +const tcpProbeScript = `nc -z -w "$1" "$2" "$3"` + +// ProbeContainerNetwork runs one bounded readiness session against one +// declared internal container port. The helper joins only the exact +// network in the request, targets the exact inspected endpoint IP (never +// a service alias), and is always force-removed under an independent +// cleanup context. An infrastructure failure returns an error and must +// never be retried as an unhealthy attempt; an unhealthy target returns a +// result with Ready=false. +func (r *Runtime) ProbeContainerNetwork(ctx context.Context, request domain.ContainerNetworkProbeRequest) (result domain.ContainerNetworkProbeResult, retErr error) { + ctx = zerowrap.CtxWithFields(ctx, map[string]any{ + zerowrap.FieldLayer: "adapter", + zerowrap.FieldAdapter: "docker", + zerowrap.FieldAction: "ProbeContainerNetwork", + "network": request.Network, + }) + log := zerowrap.FromCtx(ctx) + + if err := request.Validate(); err != nil { + return domain.ContainerNetworkProbeResult{}, err + } + sessionCtx, sessionCancel := context.WithTimeout(ctx, request.Timeout) + defer sessionCancel() + targetIP, err := r.inspectProbeTarget(sessionCtx, request) + if err != nil { + return domain.ContainerNetworkProbeResult{}, err + } + + helperID, cleanup, err := r.createProbeHelper(sessionCtx, request, targetIP) + if err != nil { + return domain.ContainerNetworkProbeResult{}, err + } + defer func() { + result, retErr = probeCleanupResult(log, cleanup, result, retErr) + }() + + if _, err := r.client.ContainerStart(sessionCtx, helperID, client.ContainerStartOptions{}); err != nil { + return domain.ContainerNetworkProbeResult{}, log.WrapErr(err, "failed to start network probe helper") + } + + timedOut, err := r.waitForProbeHelper(sessionCtx, helperID) + if err != nil { + return domain.ContainerNetworkProbeResult{}, err + } + // A bound expiry is a readiness failure, not an infrastructure + // failure. Return it before any post-session step, and never classify + // it through a context that just expired. Caller cancellation stays an + // error so the deployment stops immediately. + if timedOut { + return probeWaitTimeoutResult(ctx) + } + + exitCode, stdout, err := r.probeHelperOutcome(sessionCtx, helperID) + if result, handled := classifyProbeStepError(sessionCtx, ctx, err); handled { + return result, err + } + // Revalidate identity before trusting the outcome: a target that + // restarted during the session must not report readiness for a + // generation that no longer exists. The caller may retry this as a + // candidate that is simply not ready yet. + finalIP, err := r.inspectProbeTarget(sessionCtx, request) + if result, handled := classifyProbeStepError(sessionCtx, ctx, err); handled { + return result, err + } + if finalIP != targetIP { + return domain.ContainerNetworkProbeResult{}, fmt.Errorf("deployment: probe target endpoint changed during readiness: %w", domain.ErrAppStateConflict) + } + if err := ctx.Err(); err != nil { + return domain.ContainerNetworkProbeResult{}, err + } + + result = domain.ContainerNetworkProbeResult{Ready: exitCode == 0} + if request.Protocol == domain.ProbeProtocolHTTP { + result.Status = parseProbeStatus(stdout) + result.Ready = result.Ready && result.Status >= 200 && result.Status < 400 + } + if !result.Ready { + result.Diagnostic = probeDiagnostic(request, result.Status) + } + return result, nil +} + +func probeWaitTimeoutResult(parentCtx context.Context) (domain.ContainerNetworkProbeResult, error) { + if err := parentCtx.Err(); err != nil { + return domain.ContainerNetworkProbeResult{}, err + } + return probeTimeoutFailure(), nil +} + +func classifyProbeStepError(sessionCtx, parentCtx context.Context, err error) (domain.ContainerNetworkProbeResult, bool) { + if err == nil { + return domain.ContainerNetworkProbeResult{}, false + } + if probeBoundExpired(sessionCtx, parentCtx) { + return probeTimeoutFailure(), true + } + return domain.ContainerNetworkProbeResult{}, true +} + +// probeTimeoutFailure is the bounded readiness failure of a session whose +// own bound expired. It is not an infrastructure error: the target may +// simply be slow, so the caller may retry it as an unhealthy attempt. +func probeTimeoutFailure() domain.ContainerNetworkProbeResult { + return domain.ContainerNetworkProbeResult{Diagnostic: "probe did not complete within its bound"} +} + +// probeBoundExpired reports whether a post-wait probe step failed because +// the session's own bound expired. A caller cancellation is never a bound +// expiry: it must stay an error so the deployment stops immediately. +func probeBoundExpired(sessionCtx, parentCtx context.Context) bool { + return parentCtx.Err() == nil && errors.Is(sessionCtx.Err(), context.DeadlineExceeded) +} + +func probeCleanupResult(log zerowrap.Logger, cleanup func() error, result domain.ContainerNetworkProbeResult, retErr error) (domain.ContainerNetworkProbeResult, error) { + cleanErr := cleanup() + if cleanErr == nil { + return result, retErr + } + log.Warn().Err(cleanErr).Msg("failed to remove network probe helper") + if retErr != nil { + return domain.ContainerNetworkProbeResult{}, fmt.Errorf("%w: %v; prior probe error: %w", domain.ErrNetworkProbeCleanup, cleanErr, retErr) + } + return domain.ContainerNetworkProbeResult{}, fmt.Errorf("%w: %v", domain.ErrNetworkProbeCleanup, cleanErr) +} + +// inspectProbeTarget resolves the exact endpoint IP of the target on the +// requested network and enforces the expected execution start. It fails +// closed, wrapping domain.ErrContainerNotFound or domain.ErrAppStateConflict +// when the target is gone, not running, not attached to that network, or +// has restarted, so the caller can tell "not ready yet" from a broken +// helper. +func (r *Runtime) inspectProbeTarget(ctx context.Context, request domain.ContainerNetworkProbeRequest) (string, error) { + inspectResult, err := r.client.ContainerInspect(ctx, request.TargetContainerID, client.ContainerInspectOptions{}) + if err != nil { + if cerrdefs.IsNotFound(err) { + return "", fmt.Errorf("%w: %s", domain.ErrContainerNotFound, request.TargetContainerID) + } + return "", fmt.Errorf("deployment: inspect probe target: %w", err) + } + observed := inspectResult.Container + if observed.State == nil || observed.State.Status != container.StateRunning { + return "", fmt.Errorf("deployment: probe target %s is not running: %w", request.TargetContainerID, domain.ErrAppStateConflict) + } + startedAt, _ := time.Parse(time.RFC3339Nano, observed.State.StartedAt) + if startedAt.Year() <= 1 { + startedAt = time.Time{} + } + if !startedAt.Equal(request.ExpectedStartedAt) { + return "", fmt.Errorf("deployment: probe target %s execution changed during readiness: %w", request.TargetContainerID, domain.ErrAppStateConflict) + } + if observed.NetworkSettings == nil { + return "", fmt.Errorf("deployment: probe target %s has no network settings: %w", request.TargetContainerID, domain.ErrAppStateConflict) + } + endpoint, ok := observed.NetworkSettings.Networks[request.Network] + if !ok || endpoint == nil || !endpoint.IPAddress.IsValid() { + return "", fmt.Errorf("deployment: probe target %s is not attached to network %q: %w", request.TargetContainerID, request.Network, domain.ErrAppStateConflict) + } + return endpoint.IPAddress.String(), nil +} + +// createProbeHelper creates the hardened helper container on only the +// requested network. It has no host ports, mounts, volumes, environment, +// or shared networks, runs non-root on a read-only rootfs with all +// capabilities dropped, and never gains new privileges. +func (r *Runtime) createProbeHelper(ctx context.Context, request domain.ContainerNetworkProbeRequest, targetIP string) (string, func() error, error) { + log := zerowrap.FromCtx(ctx) + // Reclaim helpers orphaned by a previous crash before adding another: + // in-process cleanup cannot run after SIGKILL, so a later probe is the + // only owner that can bound that accumulation. + r.sweepOrphanedProbeHelpers(ctx) + + command := buildNetworkProbeCommand(request, targetIP) + entrypoint := append([]string{"sh", "-c", command.script, "sh"}, command.args...) + + created, err := r.client.ContainerCreate(ctx, client.ContainerCreateOptions{ + Config: &container.Config{ + Image: DefaultNetworkProbeImage, + Entrypoint: entrypoint, + User: "65534:65534", + AttachStdout: true, + AttachStderr: true, + Labels: map[string]string{ + domain.LabelPurpose: domain.PurposeNetworkProbe, + }, + }, + HostConfig: probeHelperHostConfig(request.Network), + Name: fmt.Sprintf("gordon-network-probe-%d", time.Now().UTC().UnixNano()), + }) + if err != nil { + if cerrdefs.IsNotFound(err) { + return "", nil, log.WrapErr(fmt.Errorf( + "network probe helper image %s is not available on this host: pre-provision the pinned image, the probe never pulls: %w", + DefaultNetworkProbeImage, err), "failed to create network probe helper") + } + return "", nil, log.WrapErr(err, "failed to create network probe helper") + } + + var cleanupErr error + var cleaned bool + cleanup := func() error { + if cleaned { + return cleanupErr + } + cleaned = true + // Independent bounded context: the session context may already be + // expired, but the helper must still be force-removed. + removeCtx, cancel := context.WithTimeout(context.Background(), probeHelperCleanup) + defer cancel() + cleanupErr = r.RemoveContainer(removeCtx, created.ID, true) + return cleanupErr + } + return created.ID, cleanup, nil +} + +// sweepOrphanedProbeHelpers best-effort removes probe helpers older than +// probeHelperOrphanAge. It is bounded to the probe purpose label, so it +// can never touch a workload, a volume archive helper, or an in-flight +// session. A sweep failure never fails the probe. +func (r *Runtime) sweepOrphanedProbeHelpers(ctx context.Context) { + log := zerowrap.FromCtx(ctx) + list, err := r.client.ContainerList(ctx, client.ContainerListOptions{ + All: true, + Filters: client.Filters{}.Add("label", domain.LabelPurpose+"="+domain.PurposeNetworkProbe), + }) + if err != nil { + log.Debug().Err(err).Msg("network probe orphan sweep: list failed") + return + } + cutoff := time.Now().Add(-probeHelperOrphanAge) + for _, orphan := range list.Items { + if time.Unix(orphan.Created, 0).After(cutoff) { + continue // live session, leave it alone + } + if err := ctx.Err(); err != nil { + return + } + removeCtx, cancel := context.WithTimeout(ctx, probeHelperCleanup) + removeErr := r.RemoveContainer(removeCtx, orphan.ID, true) + cancel() + if removeErr != nil { + log.Debug().Err(removeErr).Msg("network probe orphan sweep: remove failed") + continue + } + log.Info().Str("container", orphan.ID).Msg("reclaimed orphaned network probe helper") + } +} + +// probeHelperHostConfig is the smallest host configuration able to reach +// the target's private network and run a shell probe. Every field that +// would widen the helper's reach is left at its zero value. +func probeHelperHostConfig(network string) *container.HostConfig { + pids := int64(probeHelperPIDs) + return &container.HostConfig{ + AutoRemove: false, + NetworkMode: container.NetworkMode(network), + ReadonlyRootfs: true, + SecurityOpt: []string{"no-new-privileges:true"}, + CapDrop: []string{"ALL"}, + CapAdd: []string{}, + Resources: container.Resources{ + Memory: probeHelperMemory, + NanoCPUs: probeHelperNanoCPUs, + PidsLimit: &pids, + }, + } +} + +// waitForProbeHelper blocks until the helper exits or the bound expires. +// A bound expiry is reported as a non-fatal timeout: the target may be +// hanging, which is a readiness failure, not an infrastructure failure. +func (r *Runtime) waitForProbeHelper(ctx context.Context, helperID string) (bool, error) { + waitResult := r.client.ContainerWait(ctx, helperID, client.ContainerWaitOptions{Condition: container.WaitConditionNotRunning}) + select { + case err := <-waitResult.Error: + if err != nil { + if ctx.Err() != nil { + return true, nil + } + return false, fmt.Errorf("deployment: wait for network probe helper: %w", err) + } + return false, nil + case <-waitResult.Result: + return false, nil + case <-ctx.Done(): + return true, nil + } +} + +// probeHelperOutcome reads the helper's exit code and bounded stdout. +func (r *Runtime) probeHelperOutcome(ctx context.Context, helperID string) (int, string, error) { + log := zerowrap.FromCtx(ctx) + logs, err := r.client.ContainerLogs(ctx, helperID, client.ContainerLogsOptions{ShowStdout: true, ShowStderr: true}) + if err != nil { + return 0, "", log.WrapErr(err, "failed to read network probe helper logs") + } + defer func() { _ = logs.Close() }() + + var stdout cappedProbeBuffer + if _, err := stdcopy.StdCopy(&stdout, io.Discard, logs); err != nil { + return 0, "", log.WrapErr(err, "failed to demux network probe helper logs") + } + + inspectResult, err := r.client.ContainerInspect(ctx, helperID, client.ContainerInspectOptions{}) + if err != nil { + return 0, "", fmt.Errorf("deployment: inspect network probe helper: %w", err) + } + if inspectResult.Container.State == nil { + return 0, "", fmt.Errorf("deployment: network probe helper has no state: %w", domain.ErrAppStateConflict) + } + return inspectResult.Container.State.ExitCode, stdout.String(), nil +} + +type cappedProbeBuffer struct { + bytes.Buffer +} + +func (b *cappedProbeBuffer) Write(p []byte) (int, error) { + const limit = 64 + remaining := limit - b.Len() + if remaining > 0 { + _, _ = b.Buffer.Write(p[:min(len(p), remaining)]) + } + return len(p), nil +} + +// buildNetworkProbeCommand builds the helper argv for the requested +// protocol. The nc bound is kept strictly shorter than the attempt bound, +// so the helper exits while the caller context is still alive; it is +// rounded to whole seconds because nc -w takes seconds, floored at one, +// and clamped so one attempt can never outlive its bound. +func buildNetworkProbeCommand(request domain.ContainerNetworkProbeRequest, targetIP string) networkProbeArgv { + seconds := int(math.Ceil(request.Timeout.Seconds())) - 1 + if seconds < 1 { + seconds = 1 + } + if seconds > probeMaxSeconds { + seconds = probeMaxSeconds + } + port := strconv.Itoa(request.Port) + if request.Protocol == domain.ProbeProtocolHTTP { + return networkProbeArgv{ + script: httpProbeScript, + args: []string{request.Path, strconv.Itoa(seconds), targetIP, port}, + } + } + return networkProbeArgv{ + script: tcpProbeScript, + args: []string{strconv.Itoa(seconds), targetIP, port}, + } +} + +// parseProbeStatus extracts the HTTP status the helper observed; 0 when +// none was read. +func parseProbeStatus(stdout string) int { + status, err := strconv.Atoi(strings.TrimSpace(stdout)) + if err != nil || status < 0 || status > 999 { + return 0 + } + return status +} + +// probeDiagnostic describes an unhealthy attempt without leaking addresses +// or command output. +func probeDiagnostic(request domain.ContainerNetworkProbeRequest, status int) string { + if request.Protocol == domain.ProbeProtocolHTTP { + if status > 0 { + return fmt.Sprintf("http status %d", status) + } + return "no http response" + } + return "tcp connection refused or timed out" +} diff --git a/internal/adapters/out/docker/network_probe_test.go b/internal/adapters/out/docker/network_probe_test.go new file mode 100644 index 000000000..dfc9be80a --- /dev/null +++ b/internal/adapters/out/docker/network_probe_test.go @@ -0,0 +1,174 @@ +package docker + +import ( + "context" + "errors" + "fmt" + "strconv" + "strings" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/moby/moby/api/types/container" + + "github.com/bnema/gordon/internal/domain" +) + +// TestBuildNetworkProbeCommand proves the helper argv carries the exact +// target and clamps the timeout to whole seconds within its bound. +func TestBuildNetworkProbeCommand(t *testing.T) { + t.Run("http carries path, whole-second timeout, and exact endpoint", func(t *testing.T) { + command := buildNetworkProbeCommand(domain.ContainerNetworkProbeRequest{ + Protocol: domain.ProbeProtocolHTTP, + Path: "/healthz", + Port: 8080, + Timeout: 1500 * time.Millisecond, + }, "172.18.0.5") + assert.Equal(t, httpProbeScript, command.script) + assert.Equal(t, []string{"/healthz", "1", "172.18.0.5", "8080"}, command.args) + }) + + t.Run("tcp carries no path", func(t *testing.T) { + command := buildNetworkProbeCommand(domain.ContainerNetworkProbeRequest{ + Protocol: domain.ProbeProtocolTCP, + Port: 5432, + Timeout: time.Second, + }, "172.18.0.6") + assert.Equal(t, tcpProbeScript, command.script) + assert.Equal(t, []string{"1", "172.18.0.6", "5432"}, command.args) + }) + + t.Run("nc bound stays strictly under the attempt bound", func(t *testing.T) { + for _, timeout := range []time.Duration{1500 * time.Millisecond, 5 * time.Second, 30 * time.Second} { + command := buildNetworkProbeCommand(domain.ContainerNetworkProbeRequest{ + Protocol: domain.ProbeProtocolTCP, Port: 1, Timeout: timeout, + }, "10.0.0.1") + seconds, err := strconv.Atoi(command.args[0]) + require.NoError(t, err) + assert.Less(t, float64(seconds), timeout.Seconds(), + "the helper must exit before the caller context expires for %s", timeout) + } + }) + + t.Run("timeout is clamped into the bounded window", func(t *testing.T) { + tooSmall := buildNetworkProbeCommand(domain.ContainerNetworkProbeRequest{ + Protocol: domain.ProbeProtocolTCP, Port: 1, Timeout: time.Millisecond, + }, "10.0.0.1") + assert.Equal(t, "1", tooSmall.args[0]) + + tooLarge := buildNetworkProbeCommand(domain.ContainerNetworkProbeRequest{ + Protocol: domain.ProbeProtocolTCP, Port: 1, Timeout: time.Hour, + }, "10.0.0.1") + assert.Equal(t, "30", tooLarge.args[0]) + }) +} + +// TestProbeHelperHostConfig_IsHardened proves the helper can reach its +// network and nothing else: no ports, mounts, volumes, or extra capability. +func TestProbeHelperHostConfig_IsHardened(t *testing.T) { + config := probeHelperHostConfig("gordon--blog--net") + require.NotNil(t, config) + assert.Equal(t, "gordon--blog--net", string(config.NetworkMode)) + assert.True(t, config.ReadonlyRootfs) + assert.Equal(t, []string{"ALL"}, config.CapDrop) + assert.Empty(t, config.CapAdd) + assert.Equal(t, []string{"no-new-privileges:true"}, config.SecurityOpt) + assert.False(t, config.Privileged) + + assert.NotZero(t, config.Memory) + require.NotNil(t, config.PidsLimit) + assert.Positive(t, *config.PidsLimit) + assert.NotZero(t, config.NanoCPUs) + + assert.Empty(t, config.PortBindings) + assert.Empty(t, config.Binds) + assert.Empty(t, config.Mounts) + assert.Empty(t, config.VolumesFrom) + assert.Empty(t, config.Devices) + assert.NotEqual(t, container.NetworkMode("none"), config.NetworkMode, "the helper must reach its target network") +} + +// TestProbeHelperConfig_NoSharedNetworks proves the helper is created with +// exactly one network and no aliases. +func TestProbeHelperConfig_NoSharedNetworks(t *testing.T) { + config := probeHelperHostConfig("gordon--blog--net") + assert.Empty(t, config.ExtraHosts) + assert.Empty(t, config.DNS) + assert.Empty(t, config.Links) +} + +func TestParseProbeStatus(t *testing.T) { + assert.Equal(t, 200, parseProbeStatus("200\n")) + assert.Equal(t, 404, parseProbeStatus("404")) + assert.Equal(t, 0, parseProbeStatus("")) + assert.Equal(t, 0, parseProbeStatus("not-a-status")) + assert.Equal(t, 0, parseProbeStatus("1234")) +} + +func TestCappedProbeBuffer(t *testing.T) { + var buffer cappedProbeBuffer + input := strings.Repeat("x", 1024) + written, err := buffer.Write([]byte(input)) + require.NoError(t, err) + assert.Equal(t, len(input), written) + assert.Len(t, buffer.String(), 64) +} + +func TestHTTPProbeScript_RequiresThreeDigitStatus(t *testing.T) { + assert.Contains(t, httpProbeScript, "$2 ~ /^[0-9][0-9][0-9]$/") + assert.NotContains(t, httpProbeScript, "2*|3*") + assert.Contains(t, httpProbeScript, "head -c 128") +} + +func TestProbeCleanupResult_PriorErrorNeverHidesCleanupFailure(t *testing.T) { + cleanupErr := errors.New("remove failed") + priorErr := fmt.Errorf("stale: %w", domain.ErrAppStateConflict) + result, err := probeCleanupResult(zerowrap.Default(), func() error { return cleanupErr }, domain.ContainerNetworkProbeResult{Ready: true}, priorErr) + assert.False(t, result.Ready) + require.ErrorIs(t, err, domain.ErrNetworkProbeCleanup) + require.ErrorIs(t, err, priorErr) +} + +func TestProbeBoundExpired_OnlySessionDeadline(t *testing.T) { + parent, cancel := context.WithCancel(context.Background()) + session, sessionCancel := context.WithDeadline(parent, time.Now().Add(-time.Second)) + defer sessionCancel() + <-session.Done() + + assert.True(t, probeBoundExpired(session, parent), + "the session's own bound expiring is a readiness timeout") + + // Caller cancellation must never be reclassified as a readiness + // timeout: it has to stay an error so the deployment stops. + cancel() + <-session.Done() + assert.False(t, probeBoundExpired(session, parent), + "caller cancellation must stay an error") + + assert.False(t, probeBoundExpired(context.Background(), context.Background()), + "a live session has not expired") +} + +func TestProbeTimeoutFailure_IsUnhealthyNotError(t *testing.T) { + result := probeTimeoutFailure() + assert.False(t, result.Ready, "a bound expiry must not report readiness") + assert.NotEmpty(t, result.Diagnostic, "a bound expiry must carry a bounded diagnostic") +} + +func TestProbeDiagnostic(t *testing.T) { + http := domain.ContainerNetworkProbeRequest{Protocol: domain.ProbeProtocolHTTP} + assert.Equal(t, "http status 503", probeDiagnostic(http, 503)) + assert.Equal(t, "no http response", probeDiagnostic(http, 0)) + + tcp := domain.ContainerNetworkProbeRequest{Protocol: domain.ProbeProtocolTCP} + assert.Equal(t, "tcp connection refused or timed out", probeDiagnostic(tcp, 0)) +} + +func TestDefaultNetworkProbeImage_IsDigestPinned(t *testing.T) { + assert.Contains(t, DefaultNetworkProbeImage, "@sha256:", "the helper image must be pinned by digest") + assert.True(t, strings.HasPrefix(DefaultNetworkProbeImage, "alpine@"), "the helper needs only busybox shell tools") +} diff --git a/internal/adapters/out/docker/runtime.go b/internal/adapters/out/docker/runtime.go index 4ae66e982..099a05970 100644 --- a/internal/adapters/out/docker/runtime.go +++ b/internal/adapters/out/docker/runtime.go @@ -13,7 +13,10 @@ import ( "io" "maps" "math" + "net" + "net/http" "net/netip" + "net/url" "os" "path/filepath" "sort" @@ -39,7 +42,8 @@ import ( // Runtime implements the ContainerRuntime interface using Docker API. type Runtime struct { - client *client.Client + client *client.Client + runtimeName string } var _ out.ContainerRuntime = (*Runtime)(nil) @@ -114,7 +118,7 @@ func NewRuntimeWithSocket(socketPath string) (*Runtime, error) { return nil, fmt.Errorf("failed to create Docker client for %s: %w", socketPath, err) } - return &Runtime{client: cli}, nil + return &Runtime{client: cli, runtimeName: guessRuntimeName(socketPath)}, nil } // NewRuntimeWithClient creates a new Docker runtime instance with a custom client (for testing). @@ -140,6 +144,19 @@ func (r *Runtime) CreateContainer(ctx context.Context, config *domain.ContainerC return nil, err } binds := buildVolumeBinds(config, log) + mounts, err := buildBindMounts(config) + if err != nil { + return nil, err + } + // Device-bearing creates require a CDI-capable engine. The gate runs + // only when devices are requested: ordinary apps never pay for it + // and never fail it. + if len(config.CDIDevices) > 0 { + if err := r.requireCDISupport(ctx); err != nil { + return nil, err + } + } + deviceRequests := buildDeviceRequests(config) // Create container configuration containerConfig := &container.Config{ @@ -149,6 +166,7 @@ func (r *Runtime) CreateContainer(ctx context.Context, config *domain.ContainerC ExposedPorts: exposedPorts, WorkingDir: config.WorkingDir, Cmd: config.Cmd, + Entrypoint: config.Entrypoint, Labels: config.Labels, User: config.User, } @@ -159,6 +177,7 @@ func (r *Runtime) CreateContainer(ctx context.Context, config *domain.ContainerC Ulimits: []*units.Ulimit{ {Name: "nofile", Soft: 65536, Hard: 65536}, }, + DeviceRequests: deviceRequests, } if config.PidsLimit > 0 { resources.PidsLimit = &config.PidsLimit @@ -175,6 +194,7 @@ func (r *Runtime) CreateContainer(ctx context.Context, config *domain.ContainerC PortBindings: portBindings, AutoRemove: config.AutoRemove, Binds: binds, + Mounts: mounts, NetworkMode: container.NetworkMode(config.NetworkMode), Resources: resources, SecurityOpt: []string{"no-new-privileges:true"}, @@ -287,6 +307,199 @@ func buildVolumeBinds(config *domain.ContainerConfig, log zerowrap.Logger) []str return binds } +// buildBindMounts validates ephemeral host binds defensively and translates +// them into Docker bind mounts. The result is sorted by destination so the +// create request is deterministic. Host source paths are never logged or +// echoed in errors. +func buildBindMounts(config *domain.ContainerConfig) ([]mount.Mount, error) { + if len(config.Binds) == 0 { + return nil, nil + } + mounts := make([]mount.Mount, 0, len(config.Binds)) + seen := make(map[string]string, len(config.Binds)) + for _, bind := range config.Binds { + if bind.Source == "" || !filepath.IsAbs(bind.Source) || filepath.Clean(bind.Source) != bind.Source { + return nil, fmt.Errorf("invalid bind %q: source must be an absolute clean path", bind.Name) + } + if err := domain.ValidateBindDestination(bind.Destination); err != nil { + return nil, fmt.Errorf("invalid bind %q: %w", bind.Name, err) + } + if prev, ok := seen[bind.Destination]; ok { + return nil, fmt.Errorf("invalid bind %q: destination %q already used by bind %q", bind.Name, bind.Destination, prev) + } + seen[bind.Destination] = bind.Name + mounts = append(mounts, mount.Mount{ + Type: mount.TypeBind, + Source: bind.Source, + Target: bind.Destination, + ReadOnly: bind.ReadOnly, + }) + } + sort.SliceStable(mounts, func(i, j int) bool { + if mounts[i].Target != mounts[j].Target { + return mounts[i].Target < mounts[j].Target + } + return mounts[i].Source < mounts[j].Source + }) + return mounts, nil +} + +// requireCDISupport fails closed when the connected engine cannot serve +// native CDI device requests. Family and version come from the daemon's +// own /version response; Gordon requires Podman 5.4+ or Docker 28.3+ for +// device-bearing creates. The version is a capability gate, not hardware +// proof: CDI specs, toolkit, and node permissions must still be correct +// on the host. +func (r *Runtime) requireCDISupport(ctx context.Context) error { + version, err := r.client.ServerVersion(ctx, client.ServerVersionOptions{}) + if err != nil { + return engineProbeError(ctx, err) + } + if err := checkCDISupport(version, r.runtimeName); err != nil { + return err + } + return nil +} + +// engineProbeError maps a failed /version probe onto the CDI capability +// gate. Cancellation and deadlines survive so callers can still tell an +// aborted probe from an incapable engine; every other cause collapses to +// ErrRuntimeUnsupported with the cause text dropped, because transport +// errors embed the daemon endpoint and the operator's home path. +func engineProbeError(ctx context.Context, err error) error { + var ctxErr error + switch { + case errors.Is(err, context.Canceled), errors.Is(ctx.Err(), context.Canceled): + ctxErr = context.Canceled + case errors.Is(err, context.DeadlineExceeded), errors.Is(ctx.Err(), context.DeadlineExceeded): + ctxErr = context.DeadlineExceeded + } + if ctxErr != nil { + return fmt.Errorf("%w: cannot verify engine version: %w", domain.ErrRuntimeUnsupported, ctxErr) + } + return fmt.Errorf("%w: cannot verify engine version", domain.ErrRuntimeUnsupported) +} + +// SupportsCDIDevices implements out.ContainerRuntime. Deployment calls it +// in preflight for device-bearing revisions so an unsupported engine +// fails before any workload mutation; CreateContainer keeps the same gate +// as defense in depth. +func (r *Runtime) SupportsCDIDevices(ctx context.Context) error { + return r.requireCDISupport(ctx) +} + +// checkCDISupport verifies one daemon version response against the CDI +// support matrix. Family detection prefers the daemon's own components +// (Podman Engine vs Engine) and falls back to the socket-path hint only +// when the response names nothing recognizable. +func checkCDISupport(version client.ServerVersionResult, socketHint string) error { + family, daemonVersion := cdiEngineIdentity(version, socketHint) + var minimum string + switch family { + case "podman": + minimum = "5.4" + case "docker": + minimum = "28.3" + default: + return fmt.Errorf("%w: unrecognized engine %q version %q: device requests require Podman 5.4+ or Docker 28.3+ with native CDI configured", domain.ErrRuntimeUnsupported, family, daemonVersion) + } + if compareEngineVersion(daemonVersion, minimum) < 0 { + return fmt.Errorf("%w: %s %s below minimum %s: device requests require Podman 5.4+ or Docker 28.3+ with native CDI configured (check engine upgrade, CDI specs, toolkit, and device permissions)", domain.ErrRuntimeUnsupported, family, daemonVersion, minimum) + } + return nil +} + +// cdiEngineIdentity names the daemon family and its version from the +// /version response. Podman answers with a "Podman Engine" component; +// Docker answers with an "Engine" component and its version at top +// level. Unknown responses keep the raw version with an "unknown" +// family so the caller fails closed. +func cdiEngineIdentity(version client.ServerVersionResult, socketHint string) (family, daemonVersion string) { + for _, component := range version.Components { + switch component.Name { + case "Podman Engine": + return "podman", component.Version + case "Engine": + return "docker", component.Version + } + } + if strings.Contains(strings.ToLower(version.Platform.Name), "podman") { + return "podman", version.Version + } + switch socketHint { + case "podman", "docker": + return socketHint, version.Version + } + return "unknown", version.Version +} + +// compareEngineVersion compares dotted major.minor versions. A missing +// or unparsable version compares below any minimum: the gate fails +// closed rather than trusting an engine it cannot identify. +func compareEngineVersion(version, minimum string) int { + parse := func(value string) (int, int) { + parts := strings.SplitN(value, ".", 3) + if len(parts) < 2 { + return -1, -1 + } + major, ok := parseEngineVersionComponent(parts[0]) + if !ok { + return -1, -1 + } + minor, ok := parseEngineVersionComponent(parts[1]) + if !ok { + return -1, -1 + } + return major, minor + } + major, minor := parse(version) + minMajor, minMinor := parse(minimum) + switch { + case major != minMajor: + return major - minMajor + default: + return minor - minMinor + } +} + +// parseEngineVersionComponent parses one numeric major/minor component. +// Anything but digits is refused, so prerelease suffixes such as "-rc1" +// and malformed versions can never lift an engine above the gate. +func parseEngineVersionComponent(value string) (int, bool) { + value = strings.TrimSpace(value) + if value == "" { + return 0, false + } + for _, r := range value { + if r < '0' || r > '9' { + return 0, false + } + } + n, err := strconv.Atoi(value) + if err != nil { + return 0, false + } + return n, true +} + +// buildDeviceRequests encodes ephemeral CDI device IDs as one native CDI +// DeviceRequest. Capabilities and Options stay empty; Count stays 0 so the +// explicit DeviceIDs are the only grant. CDI IDs are never logged or echoed +// in errors: they are host inventory. +func buildDeviceRequests(config *domain.ContainerConfig) []container.DeviceRequest { + if len(config.CDIDevices) == 0 { + return nil + } + ids := append([]string(nil), config.CDIDevices...) + sort.Strings(ids) + return []container.DeviceRequest{ + { + Driver: "cdi", + DeviceIDs: ids, + }, + } +} + // StartContainer starts a container. func (r *Runtime) StartContainer(ctx context.Context, containerID string) error { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ @@ -299,6 +512,9 @@ func (r *Runtime) StartContainer(ctx context.Context, containerID string) error _, err := r.client.ContainerStart(ctx, containerID, client.ContainerStartOptions{}) if err != nil { + if cerrdefs.IsNotFound(err) { + return log.WrapErr(fmt.Errorf("%w: %s", domain.ErrContainerNotFound, containerID), "failed to start container") + } return log.WrapErr(err, "failed to start container") } @@ -330,8 +546,9 @@ func (r *Runtime) WaitForContainer(ctx context.Context, containerID string) erro return nil } -// StopContainer stops a container. -func (r *Runtime) StopContainer(ctx context.Context, containerID string) error { +// StopContainer stops a container, giving it grace to exit before the +// runtime kills it. A non-positive grace keeps the runtime default. +func (r *Runtime) StopContainer(ctx context.Context, containerID string, grace time.Duration) error { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ zerowrap.FieldLayer: "adapter", zerowrap.FieldAdapter: "docker", @@ -340,9 +557,11 @@ func (r *Runtime) StopContainer(ctx context.Context, containerID string) error { }) log := zerowrap.FromCtx(ctx) - timeout := 20 // 20 seconds before SIGKILL - _, err := r.client.ContainerStop(ctx, containerID, client.ContainerStopOptions{Timeout: &timeout}) + _, err := r.client.ContainerStop(ctx, containerID, client.ContainerStopOptions{Timeout: stopTimeout(grace)}) if err != nil { + if cerrdefs.IsNotFound(err) { + return log.WrapErr(fmt.Errorf("%w: %s", domain.ErrContainerNotFound, containerID), "failed to stop container") + } return log.WrapErr(err, "failed to stop container") } @@ -350,8 +569,9 @@ func (r *Runtime) StopContainer(ctx context.Context, containerID string) error { return nil } -// RestartContainer restarts a container. -func (r *Runtime) RestartContainer(ctx context.Context, containerID string) error { +// RestartContainer restarts a container, giving it grace to exit before +// the runtime kills it. A non-positive grace keeps the runtime default. +func (r *Runtime) RestartContainer(ctx context.Context, containerID string, grace time.Duration) error { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ zerowrap.FieldLayer: "adapter", zerowrap.FieldAdapter: "docker", @@ -360,9 +580,11 @@ func (r *Runtime) RestartContainer(ctx context.Context, containerID string) erro }) log := zerowrap.FromCtx(ctx) - timeout := 20 // 20 seconds before SIGKILL - _, err := r.client.ContainerRestart(ctx, containerID, client.ContainerRestartOptions{Timeout: &timeout}) + _, err := r.client.ContainerRestart(ctx, containerID, client.ContainerRestartOptions{Timeout: stopTimeout(grace)}) if err != nil { + if cerrdefs.IsNotFound(err) { + return log.WrapErr(fmt.Errorf("%w: %s", domain.ErrContainerNotFound, containerID), "failed to restart container") + } return log.WrapErr(err, "failed to restart container") } @@ -370,6 +592,21 @@ func (r *Runtime) RestartContainer(ctx context.Context, containerID string) erro return nil } +// stopTimeout converts a stop grace into the runtime API's whole-second +// timeout. A non-positive grace returns nil, which keeps the runtime's +// own default instead of an immediate kill. A fractional grace rounds up +// so the container never receives less time than the spec asked for. +func stopTimeout(grace time.Duration) *int { + if grace <= 0 { + return nil + } + seconds := int(math.Ceil(grace.Seconds())) + if seconds < 1 { + seconds = 1 + } + return &seconds +} + // RemoveContainer removes a container. func (r *Runtime) RemoveContainer(ctx context.Context, containerID string, force bool) error { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ @@ -383,6 +620,10 @@ func (r *Runtime) RemoveContainer(ctx context.Context, containerID string, force _, err := r.client.ContainerRemove(ctx, containerID, client.ContainerRemoveOptions{Force: force}) if err != nil { + if cerrdefs.IsNotFound(err) { + log.Debug().Msg("container not found, already removed") + return nil + } return log.WrapErr(err, "failed to remove container") } @@ -480,6 +721,9 @@ func (r *Runtime) InspectContainer(ctx context.Context, containerID string) (*do inspectResult, err := r.client.ContainerInspect(ctx, containerID, client.ContainerInspectOptions{}) if err != nil { + if cerrdefs.IsNotFound(err) { + return nil, log.WrapErr(fmt.Errorf("%w: %s", domain.ErrContainerNotFound, containerID), "failed to inspect container") + } return nil, log.WrapErr(err, "failed to inspect container") } resp := inspectResult.Container @@ -502,6 +746,14 @@ func (r *Runtime) InspectContainer(ctx context.Context, containerID string) (*do name := strings.TrimPrefix(resp.Name, "/") created, _ := time.Parse(time.RFC3339Nano, resp.Created) + // StartedAt scopes log readiness to the current execution: markers + // from a previous execution of the same container ID must not + // satisfy a probe. An unparsable value stays zero and is rejected by + // the readiness probe rather than silently widening the scope. + startedAt, _ := time.Parse(time.RFC3339Nano, resp.State.StartedAt) + if startedAt.Year() <= 1 { + startedAt = time.Time{} + } volumeMounts := make([]domain.ContainerVolumeMount, 0, len(resp.Mounts)) for _, m := range resp.Mounts { volumeMounts = append(volumeMounts, domain.ContainerVolumeMount{ @@ -524,9 +776,47 @@ func (r *Runtime) InspectContainer(ctx context.Context, containerID string) (*do Labels: resp.Config.Labels, VolumeMounts: volumeMounts, Created: created, + StartedAt: startedAt, + Env: resp.Config.Env, }, nil } +// GetContainerLogsSince gets container logs emitted at or after since. +// Readiness uses it with the observed execution start so markers from a +// previous execution of the same container ID cannot match. +func (r *Runtime) GetContainerLogsSince(ctx context.Context, containerID string, since time.Time, follow bool) (io.ReadCloser, error) { + ctx = zerowrap.CtxWithFields(ctx, map[string]any{ + zerowrap.FieldLayer: "adapter", + zerowrap.FieldAdapter: "docker", + zerowrap.FieldAction: "GetContainerLogsSince", + zerowrap.FieldEntityID: containerID, + }) + log := zerowrap.FromCtx(ctx) + + options := client.ContainerLogsOptions{ + ShowStdout: true, + ShowStderr: true, + Follow: follow, + Timestamps: true, + Tail: "10000", + } + if !since.IsZero() { + // RFC3339Nano keeps the sub-second part of the execution start: + // a marker emitted in the same second as a previous execution + // must not satisfy the probe. + options.Since = since.UTC().Format(time.RFC3339Nano) + } + logs, err := r.client.ContainerLogs(ctx, containerID, options) + if err != nil { + if cerrdefs.IsNotFound(err) { + return nil, log.WrapErr(fmt.Errorf("%w: %s", domain.ErrContainerNotFound, containerID), "failed to get container logs") + } + return nil, log.WrapErr(err, "failed to get container logs") + } + + return logs, nil +} + // GetContainerLogs gets container logs. func (r *Runtime) GetContainerLogs(ctx context.Context, containerID string, follow bool) (io.ReadCloser, error) { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ @@ -551,6 +841,70 @@ func (r *Runtime) GetContainerLogs(ctx context.Context, containerID string, foll return logs, nil } +func (r *Runtime) PullImageWithOptions(ctx context.Context, request domain.ImagePullRequest) error { + if request.Transport == domain.ImagePullTransportHTTP { + if r.runtimeName == "podman" || strings.Contains(strings.ToLower(r.client.DaemonHost()), "podman") { + return r.pullPodmanHTTP(ctx, request) + } + // Docker has no per-pull HTTP switch. Its daemon honors HTTP only for + // endpoints configured in insecure-registries; retain the compatible + // pull API and let a missing daemon policy fail explicitly. + if request.Username != "" || request.Password != "" { + return r.PullImageWithAuth(ctx, request.Reference, request.Username, request.Password) + } + return r.PullImage(ctx, request.Reference) + } + if request.Username != "" || request.Password != "" { + return r.PullImageWithAuth(ctx, request.Reference, request.Username, request.Password) + } + return r.PullImage(ctx, request.Reference) +} + +func (r *Runtime) pullPodmanHTTP(ctx context.Context, request domain.ImagePullRequest) error { + host := r.client.DaemonHost() + if !strings.HasPrefix(host, "unix://") { + return fmt.Errorf("unsupported Podman endpoint %q for HTTP pull", host) + } + httpClient := &http.Client{Transport: &http.Transport{DialContext: func(ctx context.Context, _, _ string) (net.Conn, error) { + return (&net.Dialer{}).DialContext(ctx, "unix", strings.TrimPrefix(host, "unix://")) + }}} + query := url.Values{"reference": {request.Reference}, "tlsVerify": {"false"}, "quiet": {"false"}} + req, err := http.NewRequestWithContext(ctx, http.MethodPost, "http://podman/v5.0.0/libpod/images/pull?"+query.Encode(), nil) + if err != nil { + return err + } + if request.Username != "" || request.Password != "" { + auth, err := json.Marshal(registry.AuthConfig{Username: request.Username, Password: request.Password}) + if err != nil { + return err + } + req.Header.Set(registry.AuthHeader, base64.StdEncoding.EncodeToString(auth)) + } + resp, err := httpClient.Do(req) + if err != nil { + return fmt.Errorf("podman HTTP image pull: %w", err) + } + defer resp.Body.Close() + if resp.StatusCode < 200 || resp.StatusCode >= 300 { + body, _ := io.ReadAll(io.LimitReader(resp.Body, 4096)) + return fmt.Errorf("podman HTTP image pull returned %s: %s: %w", resp.Status, strings.TrimSpace(string(body)), domain.ErrImagePullFailed) + } + decoder := json.NewDecoder(resp.Body) + for { + var event struct { + Error string `json:"error"` + } + if err := decoder.Decode(&event); errors.Is(err, io.EOF) { + break + } else if err != nil { + return fmt.Errorf("decode Podman pull response: %w", err) + } else if event.Error != "" { + return fmt.Errorf("podman image pull: %s: %w", event.Error, domain.ErrImagePullFailed) + } + } + return nil +} + // PullImage pulls an image. func (r *Runtime) PullImage(ctx context.Context, imageRef string) error { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ @@ -569,10 +923,8 @@ func (r *Runtime) PullImage(ctx context.Context, imageRef string) error { } defer reader.Close() - // Read the response to completion (this is required for the pull to complete) - _, err = io.Copy(io.Discard, reader) - if err != nil { - return log.WrapErr(err, "failed to read pull response") + if err := reader.Wait(ctx); err != nil { + return log.WrapErr(fmt.Errorf("%w: %w", domain.ErrImagePullFailed, err), "failed to complete image pull") } log.Info().Msg("image pulled successfully") @@ -629,10 +981,8 @@ func (r *Runtime) PullImageWithAuth(ctx context.Context, imageRef, username, pas } defer reader.Close() - // Read the response to completion (this is required for the pull to complete) - _, err = io.Copy(io.Discard, reader) - if err != nil { - return log.WrapErr(err, "failed to read pull response") + if err := reader.Wait(ctx); err != nil { + return log.WrapErr(fmt.Errorf("%w: %w", domain.ErrImagePullFailed, err), "failed to complete authenticated image pull") } log.Info().Msg("image pulled successfully with authentication") @@ -754,54 +1104,6 @@ func (r *Runtime) ListImagesDetailed(ctx context.Context) ([]runtimepkg.ImageDet return result, nil } -// PruneImages prunes unused images and reports reclaimed space. -func (r *Runtime) PruneImages(ctx context.Context, danglingOnly bool) (runtimepkg.PruneReport, error) { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "adapter", - zerowrap.FieldAdapter: "docker", - zerowrap.FieldAction: "PruneImages", - "dangling_only": danglingOnly, - }) - log := zerowrap.FromCtx(ctx) - - pruneFilters := make(client.Filters).Add("label", domain.LabelManaged+"=true") - if danglingOnly { - pruneFilters.Add("dangling", "true") - } - - pruneResult, err := r.client.ImagePrune(ctx, client.ImagePruneOptions{Filters: pruneFilters}) - if err != nil { - return runtimepkg.PruneReport{}, log.WrapErr(err, "failed to prune images") - } - - deletedIDs := make([]string, 0, len(pruneResult.Report.ImagesDeleted)) - for _, deleted := range pruneResult.Report.ImagesDeleted { - if deleted.Deleted != "" { - deletedIDs = append(deletedIDs, deleted.Deleted) - } - if deleted.Untagged != "" { - deletedIDs = append(deletedIDs, deleted.Untagged) - } - } - - spaceReclaimed := pruneResult.Report.SpaceReclaimed - if spaceReclaimed > math.MaxInt64 { - log.Warn(). - Uint64("space_reclaimed_bytes", spaceReclaimed). - Int64("space_reclaimed_capped_bytes", math.MaxInt64). - Msg("space reclaimed exceeds int64 max; capping value") - spaceReclaimed = uint64(math.MaxInt64) - } - - //nolint:gosec // spaceReclaimed is capped to MaxInt64 above - spaceReclaimedInt := int64(spaceReclaimed) - - return runtimepkg.PruneReport{ - DeletedIDs: deletedIDs, - SpaceReclaimed: spaceReclaimedInt, - }, nil -} - // Ping checks if Docker is responsive. func (r *Runtime) Ping(ctx context.Context) error { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ @@ -881,42 +1183,60 @@ func hasConfiguredHealthcheck(cfg *container.Config) bool { return !strings.EqualFold(strings.TrimSpace(cfg.Healthcheck.Test[0]), "NONE") } -// GetContainerPort gets the host port for a container's internal port. -func (r *Runtime) GetContainerPort(ctx context.Context, containerID string, internalPort int) (int, error) { +// GetContainerBackendBinds resolves protocol-specific container ports to +// their host binds in a single container inspection. +func (r *Runtime) GetContainerBackendBinds(ctx context.Context, containerID string, ports []domain.ContainerBackendPort) ([]domain.ContainerBackendBind, error) { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ zerowrap.FieldLayer: "adapter", zerowrap.FieldAdapter: "docker", - zerowrap.FieldAction: "GetContainerPort", + zerowrap.FieldAction: "GetContainerBackendBinds", zerowrap.FieldEntityID: containerID, - "internal_port": internalPort, + "ports": len(ports), }) log := zerowrap.FromCtx(ctx) inspectResult, err := r.client.ContainerInspect(ctx, containerID, client.ContainerInspectOptions{}) if err != nil { - return 0, log.WrapErr(err, "failed to inspect container") + return nil, log.WrapErr(err, "failed to inspect container") } resp := inspectResult.Container if resp.NetworkSettings == nil || resp.NetworkSettings.Ports == nil { - return 0, fmt.Errorf("no port mappings found for container %s", containerID) - } - - containerPort, err := network.ParsePort(fmt.Sprintf("%d/tcp", internalPort)) - if err != nil { - return 0, fmt.Errorf("invalid container port %d: %w", internalPort, err) - } - bindings, exists := resp.NetworkSettings.Ports[containerPort] - if !exists || len(bindings) == 0 { - return 0, fmt.Errorf("port %d not mapped for container %s", internalPort, containerID) + return nil, fmt.Errorf("no port mappings found for container %s", containerID) } - hostPort, err := strconv.Atoi(bindings[0].HostPort) - if err != nil { - return 0, fmt.Errorf("invalid host port for container %s: %w", containerID, err) + binds := make([]domain.ContainerBackendBind, 0, len(ports)) + for _, port := range ports { + containerPort, err := network.ParsePort(fmt.Sprintf("%d/%s", port.ContainerPort, port.Protocol)) + if err != nil { + return nil, fmt.Errorf("invalid container port %d: %w", port.ContainerPort, err) + } + bindings, exists := resp.NetworkSettings.Ports[containerPort] + if !exists || len(bindings) == 0 { + return nil, fmt.Errorf("port %d/%s not mapped for container %s", port.ContainerPort, port.Protocol, containerID) + } + // Exactly one loopback binding is required: a wildcard, multiple, + // or non-loopback mapping would expose the backend beyond the + // loopback contract the proxy and readiness rely on. + if len(bindings) != 1 { + return nil, fmt.Errorf("port %d/%s has %d host bindings for container %s; exactly one loopback binding is required", port.ContainerPort, port.Protocol, len(bindings), containerID) + } + binding := bindings[0] + if binding.HostIP != netip.MustParseAddr("127.0.0.1") { + return nil, fmt.Errorf("port %d/%s is published on host IP %q for container %s; loopback is required", port.ContainerPort, port.Protocol, binding.HostIP, containerID) + } + hostPort, err := strconv.Atoi(binding.HostPort) + if err != nil || hostPort < 1 || hostPort > 65535 { + return nil, fmt.Errorf("invalid host port %q for container %s", binding.HostPort, containerID) + } + binds = append(binds, domain.ContainerBackendBind{ + ContainerPort: port.ContainerPort, + HostPort: hostPort, + Protocol: port.Protocol, + }) } - return hostPort, nil + return binds, nil } // GetImageExposedPorts gets the exposed ports from an image. @@ -1177,7 +1497,7 @@ func (r *Runtime) VolumeExists(ctx context.Context, volumeName string) (bool, er } // CreateVolume creates a new Docker volume. -func (r *Runtime) CreateVolume(ctx context.Context, volumeName string) error { +func (r *Runtime) CreateVolume(ctx context.Context, volumeName string, labels map[string]string) error { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ zerowrap.FieldLayer: "adapter", zerowrap.FieldAdapter: "docker", @@ -1186,12 +1506,17 @@ func (r *Runtime) CreateVolume(ctx context.Context, volumeName string) error { }) log := zerowrap.FromCtx(ctx) + // The managed marker and creation time are adapter-owned; every + // caller-supplied label is preserved so ownership provenance is + // stamped at creation, not reconstructed later. + merged := make(map[string]string, len(labels)+2) + maps.Copy(merged, labels) + merged[domain.LabelManaged] = "true" + merged[domain.LabelCreated] = time.Now().UTC().Format(time.RFC3339) + _, err := r.client.VolumeCreate(ctx, client.VolumeCreateOptions{ - Name: volumeName, - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelCreated: time.Now().UTC().Format(time.RFC3339), - }, + Name: volumeName, + Labels: merged, }) if err != nil { return log.WrapErr(err, "failed to create volume") @@ -1532,6 +1857,19 @@ func (r *Runtime) GetImageLabels(ctx context.Context, imageRef string) (map[stri } // GetImageID returns the unique image ID (sha256 digest) for the given image reference. +func (r *Runtime) VerifyImageDigest(ctx context.Context, imageRef, digest string) error { + inspect, err := r.client.ImageInspect(ctx, imageRef) + if err != nil { + return fmt.Errorf("inspect image digest: %w", err) + } + for _, repoDigest := range inspect.RepoDigests { + if strings.HasSuffix(repoDigest, "@"+digest) { + return nil + } + } + return fmt.Errorf("expected digest %s is absent from local image metadata: %w", digest, domain.ErrImagePullFailed) +} + func (r *Runtime) GetImageID(ctx context.Context, imageRef string) (string, error) { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ zerowrap.FieldLayer: "adapter", diff --git a/internal/adapters/out/docker/runtime_backend_binds_test.go b/internal/adapters/out/docker/runtime_backend_binds_test.go new file mode 100644 index 000000000..ec7585917 --- /dev/null +++ b/internal/adapters/out/docker/runtime_backend_binds_test.go @@ -0,0 +1,128 @@ +package docker + +import ( + "context" + "net/http" + "net/http/httptest" + "strings" + "testing" + + "github.com/moby/moby/client" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +// backendBindsRuntime builds a docker runtime whose container inspection +// reports the given port mappings. +func backendBindsRuntime(t *testing.T, portsJSON string) *Runtime { + t.Helper() + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.Method == http.MethodGet && strings.HasSuffix(r.URL.Path, "/containers/abc123/json") { + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(`{"Id":"abc123","NetworkSettings":{"Ports":` + portsJSON + `}}`)) + return + } + t.Fatalf("unexpected request: %s %s", r.Method, r.URL.Path) + })) + t.Cleanup(server.Close) + + host := strings.TrimPrefix(server.URL, "http://") + cli, err := client.New(client.WithHost("tcp://"+host), client.WithAPIVersion("1.41"), client.WithHTTPClient(server.Client())) + require.NoError(t, err) + return NewRuntimeWithClient(cli) +} + +// TestRuntime_GetContainerBackendBindsRejectsMalformedMappings proves only +// an exact single loopback binding is accepted: a wildcard, non-loopback, +// empty, or duplicated mapping fails closed. +func TestRuntime_GetContainerBackendBindsRejectsMalformedMappings(t *testing.T) { + cases := []struct { + name string + ports string + }{ + {"wildcard", `{"9000/tcp":[{"HostIp":"0.0.0.0","HostPort":"32771"}]}`}, + {"non-loopback", `{"9000/tcp":[{"HostIp":"192.168.1.5","HostPort":"32771"}]}`}, + {"empty host ip", `{"9000/tcp":[{"HostIp":"","HostPort":"32771"}]}`}, + {"multiple bindings", `{"9000/tcp":[{"HostIp":"127.0.0.1","HostPort":"32771"},{"HostIp":"127.0.0.1","HostPort":"32772"}]}`}, + {"invalid host port", `{"9000/tcp":[{"HostIp":"127.0.0.1","HostPort":"0"}]}`}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + runtime := backendBindsRuntime(t, tc.ports) + _, err := runtime.GetContainerBackendBinds(context.Background(), "abc123", []domain.ContainerBackendPort{ + {ContainerPort: 9000, Protocol: domain.NetworkProtocolTCP}, + }) + require.Error(t, err) + }) + } +} + +// TestRuntime_GetContainerBackendBindsResolvesTCPAndUDPTogether proves one +// inspection resolves both a TCP and a UDP bind, including the same +// container port number on both protocols (distinct sockets). +func TestRuntime_GetContainerBackendBindsResolvesTCPAndUDPTogether(t *testing.T) { + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.Method == http.MethodGet && strings.HasSuffix(r.URL.Path, "/containers/abc123/json") { + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(`{ + "Id":"abc123", + "NetworkSettings":{"Ports":{ + "5432/tcp":[{"HostIp":"127.0.0.1","HostPort":"32770"}], + "9000/tcp":[{"HostIp":"127.0.0.1","HostPort":"32771"}], + "9000/udp":[{"HostIp":"127.0.0.1","HostPort":"32772"}] + }} + }`)) + return + } + t.Fatalf("unexpected request: %s %s", r.Method, r.URL.Path) + })) + defer server.Close() + + host := strings.TrimPrefix(server.URL, "http://") + cli, err := client.New(client.WithHost("tcp://"+host), client.WithAPIVersion("1.41"), client.WithHTTPClient(server.Client())) + require.NoError(t, err) + + runtime := NewRuntimeWithClient(cli) + binds, err := runtime.GetContainerBackendBinds(context.Background(), "abc123", []domain.ContainerBackendPort{ + {ContainerPort: 5432, Protocol: domain.NetworkProtocolTCP}, + {ContainerPort: 9000, Protocol: domain.NetworkProtocolTCP}, + {ContainerPort: 9000, Protocol: domain.NetworkProtocolUDP}, + }) + require.NoError(t, err) + require.Len(t, binds, 3) + assert.Equal(t, domain.ContainerBackendBind{ContainerPort: 5432, HostPort: 32770, Protocol: domain.NetworkProtocolTCP}, binds[0]) + assert.Equal(t, domain.ContainerBackendBind{ContainerPort: 9000, HostPort: 32771, Protocol: domain.NetworkProtocolTCP}, binds[1]) + assert.Equal(t, domain.ContainerBackendBind{ContainerPort: 9000, HostPort: 32772, Protocol: domain.NetworkProtocolUDP}, binds[2]) +} + +// TestRuntime_GetContainerBackendBindsFailsWhenProtocolMissing proves a +// missing protocol-specific mapping fails closed instead of falling +// back to the other protocol's bind. +func TestRuntime_GetContainerBackendBindsFailsWhenProtocolMissing(t *testing.T) { + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.Method == http.MethodGet && strings.HasSuffix(r.URL.Path, "/containers/abc123/json") { + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(`{ + "Id":"abc123", + "NetworkSettings":{"Ports":{ + "9000/tcp":[{"HostIp":"127.0.0.1","HostPort":"32771"}] + }} + }`)) + return + } + t.Fatalf("unexpected request: %s %s", r.Method, r.URL.Path) + })) + defer server.Close() + + host := strings.TrimPrefix(server.URL, "http://") + cli, err := client.New(client.WithHost("tcp://"+host), client.WithAPIVersion("1.41"), client.WithHTTPClient(server.Client())) + require.NoError(t, err) + + runtime := NewRuntimeWithClient(cli) + _, err = runtime.GetContainerBackendBinds(context.Background(), "abc123", []domain.ContainerBackendPort{ + {ContainerPort: 9000, Protocol: domain.NetworkProtocolUDP}, + }) + require.ErrorContains(t, err, "9000/udp") +} diff --git a/internal/adapters/out/docker/runtime_create_test.go b/internal/adapters/out/docker/runtime_create_test.go index 6a07bb44e..3ca13ae0f 100644 --- a/internal/adapters/out/docker/runtime_create_test.go +++ b/internal/adapters/out/docker/runtime_create_test.go @@ -123,3 +123,135 @@ func createTestContainer(t *testing.T, config *domain.ContainerConfig) map[strin return createBody } + +// TestRuntime_CreateContainerMountsDeclaredReadOnlyVolumes proves a mount +// declared read-only reaches the real Docker create request as read-only. +func TestRuntime_CreateContainerMountsDeclaredReadOnlyVolumes(t *testing.T) { + createBody := createTestContainer(t, &domain.ContainerConfig{ + Image: "nginx:latest", + Name: "gordon-app", + Volumes: map[string]string{"/data": "gordon-vol-data"}, + ReadOnlyVolumes: map[string]string{"/config": "gordon-vol-config"}, + }) + + hostConfig, ok := createBody["HostConfig"].(map[string]any) + require.True(t, ok) + binds, ok := hostConfig["Binds"].([]any) + require.True(t, ok) + assert.Contains(t, binds, "gordon-vol-data:/data") + assert.Contains(t, binds, "gordon-vol-config:/config:ro") +} + +// TestRuntime_CreateContainerAppliesNetworkAndResourceLimits proves the +// isolation contract reaches the real Docker create request: the container +// joins the named private network and carries the configured memory, CPU, +// and PID limits. +func TestRuntime_CreateContainerAppliesNetworkAndResourceLimits(t *testing.T) { + createBody := createTestContainer(t, &domain.ContainerConfig{ + Image: "nginx:latest", + Name: "gordon-app", + NetworkMode: "gordon-app-abc123", + Aliases: []string{"web"}, + Hostname: "web", + MemoryLimit: 512 << 20, + NanoCPUs: 1_500_000_000, + PidsLimit: 256, + }) + + hostConfig, ok := createBody["HostConfig"].(map[string]any) + require.True(t, ok) + assert.Equal(t, "gordon-app-abc123", hostConfig["NetworkMode"]) + assert.Equal(t, float64(512<<20), hostConfig["Memory"]) + assert.Equal(t, float64(1_500_000_000), hostConfig["NanoCpus"]) + assert.Equal(t, float64(256), hostConfig["PidsLimit"]) + + networking, ok := createBody["NetworkingConfig"].(map[string]any) + require.True(t, ok) + endpoints, ok := networking["EndpointsConfig"].(map[string]any) + require.True(t, ok) + assert.Contains(t, endpoints, "gordon-app-abc123") +} + +// TestRuntime_CreateContainerTranslatesBindMounts proves ephemeral host binds +// reach the Docker create request as TypeBind mounts in deterministic +// destination order, read-only preserved, alongside named-volume Binds. +func TestRuntime_CreateContainerTranslatesBindMounts(t *testing.T) { + createBody := createTestContainer(t, &domain.ContainerConfig{ + Image: "nginx:latest", + Name: "gordon-app", + Volumes: map[string]string{"/data": "gordon-vol-data"}, + Binds: []domain.ContainerBind{ + {Name: "rw", Source: "/srv/binds/rw", Destination: "/etc/app.conf"}, + {Name: "ro", Source: "/srv/binds/ro", Destination: "/var/lib/data", ReadOnly: true}, + }, + }) + + hostConfig, ok := createBody["HostConfig"].(map[string]any) + require.True(t, ok) + + // Existing named-volume Binds are preserved alongside explicit mounts. + binds, ok := hostConfig["Binds"].([]any) + require.True(t, ok) + assert.Contains(t, binds, "gordon-vol-data:/data") + + mounts, ok := hostConfig["Mounts"].([]any) + require.True(t, ok) + require.Len(t, mounts, 2) + + first, ok := mounts[0].(map[string]any) + require.True(t, ok) + assert.Equal(t, "bind", first["Type"]) + assert.Equal(t, "/srv/binds/rw", first["Source"]) + assert.Equal(t, "/etc/app.conf", first["Target"]) + assert.NotContains(t, first, "ReadOnly") + + second, ok := mounts[1].(map[string]any) + require.True(t, ok) + assert.Equal(t, "bind", second["Type"]) + assert.Equal(t, "/srv/binds/ro", second["Source"]) + assert.Equal(t, "/var/lib/data", second["Target"]) + assert.Equal(t, true, second["ReadOnly"]) +} + +func TestBuildBindMounts(t *testing.T) { + t.Run("sorts by destination", func(t *testing.T) { + mounts, err := buildBindMounts(&domain.ContainerConfig{Binds: []domain.ContainerBind{ + {Name: "z", Source: "/srv/binds/z", Destination: "/z"}, + {Name: "a", Source: "/srv/binds/a", Destination: "/a"}, + }}) + require.NoError(t, err) + require.Len(t, mounts, 2) + assert.Equal(t, "/a", mounts[0].Target) + assert.Equal(t, "/z", mounts[1].Target) + }) + + t.Run("invalid paths rejected without echoing source", func(t *testing.T) { + cases := []struct { + name string + bind domain.ContainerBind + }{ + {"relative source", domain.ContainerBind{Name: "b", Source: "srv/binds", Destination: "/etc/app.conf"}}, + {"empty source", domain.ContainerBind{Name: "b", Destination: "/etc/app.conf"}}, + {"unclean source", domain.ContainerBind{Name: "b", Source: "/srv/../binds", Destination: "/etc/app.conf"}}, + {"relative destination", domain.ContainerBind{Name: "b", Source: "/srv/binds", Destination: "etc/app.conf"}}, + {"sensitive destination", domain.ContainerBind{Name: "b", Source: "/srv/binds", Destination: "/dev"}}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + _, err := buildBindMounts(&domain.ContainerConfig{Binds: []domain.ContainerBind{tc.bind}}) + require.Error(t, err) + if tc.bind.Source != "" { + assert.NotContains(t, err.Error(), tc.bind.Source, "errors must never echo the host source") + } + }) + } + }) + + t.Run("duplicate destination rejected", func(t *testing.T) { + _, err := buildBindMounts(&domain.ContainerConfig{Binds: []domain.ContainerBind{ + {Name: "one", Source: "/srv/binds/one", Destination: "/etc/app.conf"}, + {Name: "two", Source: "/srv/binds/two", Destination: "/etc/app.conf"}, + }}) + require.Error(t, err) + }) +} diff --git a/internal/adapters/out/docker/runtime_devices_test.go b/internal/adapters/out/docker/runtime_devices_test.go new file mode 100644 index 000000000..70fbff633 --- /dev/null +++ b/internal/adapters/out/docker/runtime_devices_test.go @@ -0,0 +1,124 @@ +package docker + +import ( + "context" + "errors" + "fmt" + "testing" + + "github.com/moby/moby/api/types/container" + "github.com/moby/moby/api/types/system" + "github.com/moby/moby/client" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +func podmanVersion(version string) client.ServerVersionResult { + return client.ServerVersionResult{ + Version: version, + Components: []system.ComponentVersion{{Name: "Podman Engine", Version: version}}, + } +} + +func dockerVersion(version string) client.ServerVersionResult { + return client.ServerVersionResult{ + Version: version, + Components: []system.ComponentVersion{{Name: "Engine", Version: version}}, + } +} + +func TestBuildDeviceRequests_EmptyIsNil(t *testing.T) { + assert.Nil(t, buildDeviceRequests(&domain.ContainerConfig{})) + assert.Nil(t, buildDeviceRequests(&domain.ContainerConfig{CDIDevices: nil})) +} + +func TestBuildDeviceRequests_OneNativeCDIRequest(t *testing.T) { + requests := buildDeviceRequests(&domain.ContainerConfig{ + CDIDevices: []string{"example.com/gpu=GPU-b", "example.com/gpu=GPU-a"}, + }) + require.Len(t, requests, 1, "all CDI IDs travel in one native DeviceRequest") + request := requests[0] + assert.Equal(t, container.DeviceRequest{ + Driver: "cdi", + DeviceIDs: []string{"example.com/gpu=GPU-a", "example.com/gpu=GPU-b"}, + }, request, "driver is cdi, capabilities/options stay empty, count stays 0, IDs sorted") +} + +func TestBuildDeviceRequests_DoesNotMutateConfig(t *testing.T) { + config := &domain.ContainerConfig{CDIDevices: []string{"example.com/gpu=GPU-b", "example.com/gpu=GPU-a"}} + _ = buildDeviceRequests(config) + assert.Equal(t, []string{"example.com/gpu=GPU-b", "example.com/gpu=GPU-a"}, config.CDIDevices) +} + +func TestCheckCDISupportMatrix(t *testing.T) { + cases := []struct { + name string + version client.ServerVersionResult + hint string + wantError bool + }{ + {"podman 6.1.1", podmanVersion("6.1.1"), "", false}, + {"podman 5.4 exact", podmanVersion("5.4.0"), "", false}, + {"podman 4.9 refused", podmanVersion("4.9.5"), "", true}, + {"docker 28.3", dockerVersion("28.3.2"), "", false}, + {"docker 27 refused", dockerVersion("27.5.1"), "", true}, + {"unknown engine refused", client.ServerVersionResult{Version: "9.9.9"}, "unknown", true}, + {"socket hint podman accepted", client.ServerVersionResult{Version: "6.1.1"}, "podman", false}, + {"unparsable version refused", podmanVersion("not-a-version"), "", true}, + {"prerelease minor refused", podmanVersion("5.4-rc1"), "", true}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + err := checkCDISupport(tc.version, tc.hint) + if tc.wantError { + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrRuntimeUnsupported) + } else { + require.NoError(t, err) + } + }) + } +} + +func TestCompareEngineVersion(t *testing.T) { + assert.Greater(t, compareEngineVersion("6.1.1", "5.4"), 0) + assert.Less(t, compareEngineVersion("4.9.5", "5.4"), 0) + assert.Equal(t, 0, compareEngineVersion("5.4.0", "5.4")) + assert.Less(t, compareEngineVersion("not-a-version", "5.4"), 0) + assert.Equal(t, 0, compareEngineVersion("28.3.2", "28.3"), "patch releases do not affect the gate") + + // A prerelease suffix must never lift an engine above the gate: the + // minor component has to be numeric on its own. + assert.Less(t, compareEngineVersion("5.4-rc1", "5.4"), 0) + assert.Less(t, compareEngineVersion("28.3-rc1", "28.3"), 0) + assert.Less(t, compareEngineVersion("5.", "5.4"), 0) +} + +// TestEngineProbeError proves a failed /version probe keeps cancellation +// recognizable while every other cause collapses to the unsupported +// sentinel without the daemon endpoint in the text. +func TestEngineProbeError(t *testing.T) { + t.Run("canceled probe keeps the sentinel and drops the endpoint", func(t *testing.T) { + ctx, cancel := context.WithCancel(context.Background()) + cancel() + + err := engineProbeError(ctx, errors.New("dial unix /run/podman/podman.sock: i/o timeout")) + require.ErrorIs(t, err, context.Canceled) + require.ErrorIs(t, err, domain.ErrRuntimeUnsupported) + assert.NotContains(t, err.Error(), "podman.sock") + }) + + t.Run("canceled adapter cause is preserved", func(t *testing.T) { + err := engineProbeError(context.Background(), fmt.Errorf("probe: %w", context.Canceled)) + require.ErrorIs(t, err, context.Canceled) + }) + + t.Run("other probe failures stay redacted", func(t *testing.T) { + err := engineProbeError(context.Background(), errors.New("dial unix /run/podman/podman.sock: connection refused")) + require.ErrorIs(t, err, domain.ErrRuntimeUnsupported) + assert.False(t, errors.Is(err, context.Canceled)) + assert.NotContains(t, err.Error(), "podman.sock") + }) +} diff --git a/internal/adapters/out/docker/runtime_image_cleanup_test.go b/internal/adapters/out/docker/runtime_image_cleanup_test.go deleted file mode 100644 index a766c4434..000000000 --- a/internal/adapters/out/docker/runtime_image_cleanup_test.go +++ /dev/null @@ -1,190 +0,0 @@ -package docker - -import ( - "context" - "encoding/json" - "math" - "net/http" - "net/http/httptest" - "strings" - "testing" - "time" - - "github.com/moby/moby/client" - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" -) - -func TestRuntime_PruneImages_DanglingOnlyFilter(t *testing.T) { - server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - assert.Equal(t, http.MethodPost, r.Method) - assert.Equal(t, "/v1.41/images/prune", r.URL.Path) - - parsedFilters := client.Filters{} - require.NoError(t, json.Unmarshal([]byte(r.URL.Query().Get("filters")), &parsedFilters)) - assert.True(t, parsedFilters["dangling"]["true"]) - - w.Header().Set("Content-Type", "application/json") - _, _ = w.Write([]byte(`{"ImagesDeleted":[{"Deleted":"sha256:abc"},{"Untagged":"example:old"}],"SpaceReclaimed":1234}`)) - })) - defer server.Close() - - runtime := newRuntimeForHTTPServer(t, server) - report, err := runtime.PruneImages(context.Background(), true) - - require.NoError(t, err) - assert.Equal(t, []string{"sha256:abc", "example:old"}, report.DeletedIDs) - assert.EqualValues(t, 1234, report.SpaceReclaimed) -} - -func TestRuntime_PruneImages_FullUnusedHasNoDanglingFilter(t *testing.T) { - server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - assert.Equal(t, http.MethodPost, r.Method) - assert.Equal(t, "/v1.41/images/prune", r.URL.Path) - - parsedFilters := client.Filters{} - require.NoError(t, json.Unmarshal([]byte(r.URL.Query().Get("filters")), &parsedFilters)) - assert.Empty(t, parsedFilters["dangling"]) - - w.Header().Set("Content-Type", "application/json") - _, _ = w.Write([]byte(`{"ImagesDeleted":[{"Deleted":"sha256:def"}],"SpaceReclaimed":5678}`)) - })) - defer server.Close() - - runtime := newRuntimeForHTTPServer(t, server) - report, err := runtime.PruneImages(context.Background(), false) - - require.NoError(t, err) - assert.Equal(t, []string{"sha256:def"}, report.DeletedIDs) - assert.EqualValues(t, 5678, report.SpaceReclaimed) -} - -func TestRuntime_PruneImages_ReturnsErrorOnNon2xxResponse(t *testing.T) { - server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - assert.Equal(t, http.MethodPost, r.Method) - assert.Equal(t, "/v1.41/images/prune", r.URL.Path) - - w.WriteHeader(http.StatusInternalServerError) - _, _ = w.Write([]byte(`{"message":"internal error"}`)) - })) - defer server.Close() - - runtime := newRuntimeForHTTPServer(t, server) - _, err := runtime.PruneImages(context.Background(), true) - - require.Error(t, err) -} - -func TestRuntime_PruneImages_ReturnsErrorOnInvalidPayload(t *testing.T) { - server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - assert.Equal(t, http.MethodPost, r.Method) - assert.Equal(t, "/v1.41/images/prune", r.URL.Path) - - w.Header().Set("Content-Type", "application/json") - _, _ = w.Write([]byte(`not-json`)) - })) - defer server.Close() - - runtime := newRuntimeForHTTPServer(t, server) - _, err := runtime.PruneImages(context.Background(), true) - - require.Error(t, err) -} - -func TestRuntime_PruneImages_CapsSpaceReclaimedOverflow(t *testing.T) { - server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - assert.Equal(t, http.MethodPost, r.Method) - assert.Equal(t, "/v1.41/images/prune", r.URL.Path) - - w.Header().Set("Content-Type", "application/json") - _, _ = w.Write([]byte(`{"ImagesDeleted":[],"SpaceReclaimed":9223372036854775808}`)) - })) - defer server.Close() - - runtime := newRuntimeForHTTPServer(t, server) - report, err := runtime.PruneImages(context.Background(), true) - - require.NoError(t, err) - assert.EqualValues(t, math.MaxInt64, report.SpaceReclaimed) -} - -func TestRuntime_ListImagesDetailed_MapsImageSummary(t *testing.T) { - server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - assert.Equal(t, http.MethodGet, r.Method) - assert.Equal(t, "/v1.41/images/json", r.URL.Path) - assert.Equal(t, "1", r.URL.Query().Get("all")) - - w.Header().Set("Content-Type", "application/json") - _, _ = w.Write([]byte(`[{"Id":"sha256:img1","RepoTags":["alpine:latest"],"Size":321,"Created":1700000000}]`)) - })) - defer server.Close() - - runtime := newRuntimeForHTTPServer(t, server) - images, err := runtime.ListImagesDetailed(context.Background()) - - require.NoError(t, err) - require.Len(t, images, 1) - assert.Equal(t, "sha256:img1", images[0].ID) - assert.Equal(t, []string{"alpine:latest"}, images[0].RepoTags) - assert.EqualValues(t, 321, images[0].Size) - assert.Equal(t, time.Unix(1700000000, 0), images[0].Created) -} - -func TestRuntime_ListImagesDetailed_ReturnsErrorOnNon2xxResponse(t *testing.T) { - server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - assert.Equal(t, http.MethodGet, r.Method) - assert.Equal(t, "/v1.41/images/json", r.URL.Path) - - w.WriteHeader(http.StatusBadGateway) - _, _ = w.Write([]byte(`{"message":"upstream error"}`)) - })) - defer server.Close() - - runtime := newRuntimeForHTTPServer(t, server) - _, err := runtime.ListImagesDetailed(context.Background()) - - require.Error(t, err) -} - -func TestRuntime_ListImagesDetailed_ReturnsErrorOnInvalidPayload(t *testing.T) { - server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - assert.Equal(t, http.MethodGet, r.Method) - assert.Equal(t, "/v1.41/images/json", r.URL.Path) - - w.Header().Set("Content-Type", "application/json") - _, _ = w.Write([]byte(`{"not":`)) - })) - defer server.Close() - - runtime := newRuntimeForHTTPServer(t, server) - _, err := runtime.ListImagesDetailed(context.Background()) - - require.Error(t, err) -} - -func TestRuntime_PruneImages_UsesMaxInt64Boundary(t *testing.T) { - server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - assert.Equal(t, http.MethodPost, r.Method) - assert.Equal(t, "/v1.41/images/prune", r.URL.Path) - - w.Header().Set("Content-Type", "application/json") - _, _ = w.Write([]byte(`{"ImagesDeleted":[],"SpaceReclaimed":9223372036854775807}`)) - })) - defer server.Close() - - runtime := newRuntimeForHTTPServer(t, server) - report, err := runtime.PruneImages(context.Background(), true) - - require.NoError(t, err) - assert.EqualValues(t, math.MaxInt64, report.SpaceReclaimed) -} - -func newRuntimeForHTTPServer(t *testing.T, server *httptest.Server) *Runtime { - t.Helper() - - host := strings.TrimPrefix(server.URL, "http://") - cli, err := client.New(client.WithHost("tcp://"+host), client.WithAPIVersion("1.41"), client.WithHTTPClient(server.Client())) - require.NoError(t, err) - - return NewRuntimeWithClient(cli) -} diff --git a/internal/adapters/out/docker/runtime_prune.go b/internal/adapters/out/docker/runtime_prune.go new file mode 100644 index 000000000..4a712690d --- /dev/null +++ b/internal/adapters/out/docker/runtime_prune.go @@ -0,0 +1,202 @@ +package docker + +import ( + "context" + "fmt" + "sort" + "strings" + "time" + + "github.com/bnema/zerowrap" + "github.com/moby/moby/api/types/container" + "github.com/moby/moby/api/types/image" + "github.com/moby/moby/api/types/volume" + "github.com/moby/moby/client" + + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/domain" +) + +var _ out.PruneRuntime = (*Runtime)(nil) + +// InventoryRuntime implements out.PruneRuntime. +// +// It reads containers (running and stopped), images, and volumes in one +// pass. A failed list is recorded as an inventory gap, never as +// absence: the caller then fails closed per candidate instead of +// planning against an empty runtime. +func (r *Runtime) InventoryRuntime(ctx context.Context) (*domain.RuntimeInventory, error) { + ctx = zerowrap.CtxWithFields(ctx, map[string]any{ + zerowrap.FieldLayer: "adapter", + zerowrap.FieldAdapter: "docker", + zerowrap.FieldAction: "InventoryRuntime", + }) + log := zerowrap.FromCtx(ctx) + + inventory := &domain.RuntimeInventory{} + + containerList, containerErr := r.client.ContainerList(ctx, client.ContainerListOptions{All: true}) + if containerErr != nil { + log.Warn().Err(containerErr).Msg("failed to list containers for prune inventory") + inventory.Gaps = append(inventory.Gaps, domain.InventoryGap{ + Source: domain.InventorySourceRuntimeContainers, + Reason: domain.PruneReasonUnknownContainerUse, + Detail: "list containers failed", + }) + } else { + inventory.Containers = containerImageUses(containerList.Items) + } + + imageList, imageErr := r.client.ImageList(ctx, client.ImageListOptions{All: true}) + if imageErr != nil { + log.Warn().Err(imageErr).Msg("failed to list images for prune inventory") + inventory.Gaps = append(inventory.Gaps, domain.InventoryGap{ + Source: domain.InventorySourceRuntimeImages, + Reason: domain.PruneReasonUnknownInventory, + Detail: "list images failed", + }) + } else { + inventory.Images = runtimeImages(imageList.Items) + } + + volumeList, volumeErr := r.client.VolumeList(ctx, client.VolumeListOptions{}) + if volumeErr != nil { + log.Warn().Err(volumeErr).Msg("failed to list volumes for prune inventory") + inventory.Gaps = append(inventory.Gaps, domain.InventoryGap{ + Source: domain.InventorySourceRuntimeVolumes, + Reason: domain.PruneReasonUnknownInventory, + Detail: "list volumes failed", + }) + } else { + var mounts []volumeMount + if containerErr == nil { + mounts = volumeMounts(containerList.Items) + } + inventory.Volumes = volumeInfos(volumeList.Items, mounts) + } + + return inventory, nil +} + +// volumeMount is one volume mount observed on a container. +type volumeMount struct { + volume string + container string +} + +func volumeMounts(containers []container.Summary) []volumeMount { + var mounts []volumeMount + for _, container := range containers { + name := "" + if len(container.Names) > 0 { + name = strings.TrimPrefix(container.Names[0], "/") + } + if name == "" { + name = container.ID + } + for _, mount := range container.Mounts { + if mount.Type != "volume" || mount.Name == "" { + continue + } + mounts = append(mounts, volumeMount{volume: mount.Name, container: name}) + } + } + sort.Slice(mounts, func(i, j int) bool { + if mounts[i].volume != mounts[j].volume { + return mounts[i].volume < mounts[j].volume + } + return mounts[i].container < mounts[j].container + }) + return mounts +} + +// containerImageUses maps every container, running or not, to the image +// identity it was created from. A container whose image identity cannot +// be resolved marks ImageUnknown so no candidate can be proven unused. +func containerImageUses(containers []container.Summary) []domain.RuntimeContainerUse { + uses := make([]domain.RuntimeContainerUse, 0, len(containers)) + for _, container := range containers { + use := domain.RuntimeContainerUse{ + ContainerID: container.ID, + ImageID: container.ImageID, + ImageRef: container.Image, + Running: container.State == "running", + } + if use.ImageID == "" && use.ImageRef == "" { + use.ImageUnknown = true + } + uses = append(uses, use) + } + sort.Slice(uses, func(i, j int) bool { return uses[i].ContainerID < uses[j].ContainerID }) + return uses +} + +// runtimeImages maps the runtime image list to inventory entries. +func runtimeImages(images []image.Summary) []domain.RuntimeImage { + out := make([]domain.RuntimeImage, 0, len(images)) + for _, image := range images { + out = append(out, domain.RuntimeImage{ + ID: image.ID, + RepoTags: append([]string(nil), image.RepoTags...), + RepoDigests: append([]string(nil), image.RepoDigests...), + Labels: image.Labels, + Size: image.Size, + Created: time.Unix(image.Created, 0), + }) + } + sort.Slice(out, func(i, j int) bool { return out[i].ID < out[j].ID }) + return out +} + +// volumeInfos maps volumes plus observed container mounts to inventory +// entries. +func volumeInfos(volumes []volume.Volume, mounts []volumeMount) []*domain.VolumeInfo { + users := make(map[string][]string) + for _, mount := range mounts { + users[mount.volume] = append(users[mount.volume], mount.container) + } + + out := make([]*domain.VolumeInfo, 0, len(volumes)) + for _, volume := range volumes { + var size int64 + if volume.UsageData != nil { + size = volume.UsageData.Size + } + // A volume driver that cannot report creation time yields the + // zero time (for example some Podman responses). That is not a + // safety input: eligibility comes from ownership, not age. + created, _ := time.Parse(time.RFC3339, volume.CreatedAt) + containers := users[volume.Name] + out = append(out, &domain.VolumeInfo{ + Name: volume.Name, + Driver: volume.Driver, + MountPoint: volume.Mountpoint, + Size: size, + CreatedAt: created, + InUse: len(containers) > 0, + Containers: containers, + Labels: volume.Labels, + }) + } + sort.Slice(out, func(i, j int) bool { return out[i].Name < out[j].Name }) + return out +} + +// RemoveImageExact implements out.PruneRuntime. It removes one exact +// runtime image ID and never forces: an image that is still referenced +// fails loudly instead of being untagged behind the caller's back. +func (r *Runtime) RemoveImageExact(ctx context.Context, ref domain.RuntimeImageRef) error { + if !ref.Valid() { + return fmt.Errorf("docker: refusing to remove image with invalid identity %q", ref.ID) + } + return r.RemoveImage(ctx, ref.ID, false) +} + +// RemoveVolumeExact implements out.PruneRuntime. It removes one exact +// runtime volume name and never forces. +func (r *Runtime) RemoveVolumeExact(ctx context.Context, ref domain.RuntimeVolumeRef) error { + if !ref.Valid() { + return fmt.Errorf("docker: refusing to remove volume with invalid identity %q", ref.Name) + } + return r.RemoveVolume(ctx, ref.Name, false) +} diff --git a/internal/adapters/out/docker/runtime_prune_test.go b/internal/adapters/out/docker/runtime_prune_test.go new file mode 100644 index 000000000..1e0687f41 --- /dev/null +++ b/internal/adapters/out/docker/runtime_prune_test.go @@ -0,0 +1,214 @@ +package docker + +import ( + "context" + "net/http" + "net/http/httptest" + "strings" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +func TestRuntime_InventoryRuntime(t *testing.T) { + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + switch r.URL.Path { + case "/v1.41/containers/json": + assert.Equal(t, "1", r.URL.Query().Get("all")) + _, _ = w.Write([]byte(`[ + {"Id":"c-running","Names":["/shop-web-1"],"Image":"registry.example/shop:1","ImageID":"sha256:aaa","State":"running", + "Mounts":[{"Type":"volume","Name":"gordon-shop--web--vol--data"}]}, + {"Id":"c-stopped","Names":["/shop-web-2"],"Image":"registry.example/shop:2","ImageID":"sha256:bbb","State":"exited","Mounts":[]}, + {"Id":"c-opaque","Names":["/foreign"],"Image":"","ImageID":"","State":"exited","Mounts":[]} + ]`)) + case "/v1.41/images/json": + assert.Equal(t, "1", r.URL.Query().Get("all")) + _, _ = w.Write([]byte(`[ + {"Id":"sha256:aaa","RepoTags":["registry.example/shop:1","registry.example/shop:latest"], + "RepoDigests":["registry.example/shop@sha256:1111"],"Labels":{"gordon.app":"shop"},"Size":10,"Created":1700000000}, + {"Id":"sha256:bbb","RepoTags":[":"],"RepoDigests":[],"Labels":{},"Size":20,"Created":1700000000} + ]`)) + case "/v1.41/volumes": + _, _ = w.Write([]byte(`{"Volumes":[ + {"Name":"gordon-shop--web--vol--data","Driver":"local","Mountpoint":"/var/lib/docker/volumes/x/_data","CreatedAt":"2026-01-01T00:00:00Z","Labels":{"gordon.managed":"true","gordon.app":"shop"}}, + {"Name":"pgdata","Driver":"local","Mountpoint":"/var/lib/docker/volumes/pgdata/_data","CreatedAt":"not-a-time","Labels":{}} + ]}`)) + default: + t.Errorf("unexpected request %s %s", r.Method, r.URL.Path) + w.WriteHeader(http.StatusNotFound) + } + })) + defer server.Close() + + rt := newRuntimeForHTTPServer(t, server) + inventory, err := rt.InventoryRuntime(context.Background()) + require.NoError(t, err) + require.True(t, inventory.Complete(), "inventory gaps: %+v", inventory.Gaps) + + require.Len(t, inventory.Containers, 3) + byContainer := map[string]domain.RuntimeContainerUse{} + for _, use := range inventory.Containers { + byContainer[use.ContainerID] = use + } + assert.True(t, byContainer["c-running"].Running) + assert.False(t, byContainer["c-stopped"].Running) + assert.Equal(t, "sha256:bbb", byContainer["c-stopped"].ImageID) + assert.True(t, byContainer["c-opaque"].ImageUnknown, "unresolvable container image must be flagged") + + require.Len(t, inventory.Images, 2) + byImage := map[string]domain.RuntimeImage{} + for _, image := range inventory.Images { + byImage[image.ID] = image + } + // One image ID may carry several repo tags and repo digests. + assert.Equal(t, []string{"registry.example/shop:1", "registry.example/shop:latest"}, byImage["sha256:aaa"].RepoTags) + assert.Equal(t, []string{"registry.example/shop@sha256:1111"}, byImage["sha256:aaa"].RepoDigests) + assert.Equal(t, ":", byImage["sha256:bbb"].RepoTags[0]) + + require.Len(t, inventory.Volumes, 2) + byVolume := map[string]*domain.VolumeInfo{} + for _, volume := range inventory.Volumes { + byVolume[volume.Name] = volume + } + assert.True(t, byVolume["gordon-shop--web--vol--data"].InUse) + assert.Equal(t, []string{"shop-web-1"}, byVolume["gordon-shop--web--vol--data"].Containers) + assert.Equal(t, "shop", byVolume["gordon-shop--web--vol--data"].Labels[domain.LabelApp]) + assert.False(t, byVolume["pgdata"].InUse) + assert.Empty(t, byVolume["pgdata"].Labels) +} + +func TestRuntime_InventoryRuntimeReportsGapsNotAbsence(t *testing.T) { + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + if r.URL.Path == "/v1.41/images/json" { + w.WriteHeader(http.StatusInternalServerError) + _, _ = w.Write([]byte(`{"message":"boom"}`)) + return + } + switch r.URL.Path { + case "/v1.41/containers/json": + _, _ = w.Write([]byte(`[]`)) + case "/v1.41/volumes": + _, _ = w.Write([]byte(`{"Volumes":[]}`)) + } + })) + defer server.Close() + + rt := newRuntimeForHTTPServer(t, server) + inventory, err := rt.InventoryRuntime(context.Background()) + require.NoError(t, err) + require.False(t, inventory.Complete()) + require.Len(t, inventory.Gaps, 1) + assert.Equal(t, domain.InventorySourceRuntimeImages, inventory.Gaps[0].Source) + assert.Equal(t, domain.PruneReasonUnknownInventory, inventory.Gaps[0].Reason) +} + +func TestRuntime_InventoryRuntimeContainerListFailureProtectsImages(t *testing.T) { + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + switch r.URL.Path { + case "/v1.41/containers/json": + w.WriteHeader(http.StatusInternalServerError) + _, _ = w.Write([]byte(`{"message":"boom"}`)) + case "/v1.41/images/json": + _, _ = w.Write([]byte(`[]`)) + case "/v1.41/volumes": + _, _ = w.Write([]byte(`{"Volumes":[]}`)) + } + })) + defer server.Close() + + rt := newRuntimeForHTTPServer(t, server) + inventory, err := rt.InventoryRuntime(context.Background()) + require.NoError(t, err) + require.False(t, inventory.Complete()) + require.Len(t, inventory.Gaps, 1) + assert.Equal(t, domain.InventorySourceRuntimeContainers, inventory.Gaps[0].Source) + assert.Equal(t, domain.PruneReasonUnknownContainerUse, inventory.Gaps[0].Reason) +} + +func TestRuntime_RemoveImageExactUsesExactIdentityWithoutForce(t *testing.T) { + var method, path string + var force string + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + method = r.Method + path = r.URL.Path + force = r.URL.Query().Get("force") + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(`[]`)) + })) + defer server.Close() + + rt := newRuntimeForHTTPServer(t, server) + ref, err := domain.NewRuntimeImageRef("sha256:deadbeef") + require.NoError(t, err) + require.NoError(t, rt.RemoveImageExact(context.Background(), ref)) + + assert.Equal(t, http.MethodDelete, method) + assert.Equal(t, "/v1.41/images/sha256:deadbeef", path) + assert.NotEqual(t, "true", force) + assert.NotEqual(t, "1", force) +} + +func TestRuntime_RemoveVolumeExactUsesExactIdentityWithoutForce(t *testing.T) { + var method, path string + var force string + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + method = r.Method + path = r.URL.Path + force = r.URL.Query().Get("force") + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(`{}`)) + })) + defer server.Close() + + rt := newRuntimeForHTTPServer(t, server) + ref, err := domain.NewRuntimeVolumeRef("gordon-shop--web--vol--data") + require.NoError(t, err) + require.NoError(t, rt.RemoveVolumeExact(context.Background(), ref)) + + assert.Equal(t, http.MethodDelete, method) + assert.Equal(t, "/v1.41/volumes/gordon-shop--web--vol--data", path) + assert.NotEqual(t, "true", force) + assert.NotEqual(t, "1", force) +} + +func TestRuntime_ExactRemovalRejectsInvalidIdentity(t *testing.T) { + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + t.Errorf("invalid identity must not reach the runtime: %s %s", r.Method, r.URL.Path) + })) + defer server.Close() + + rt := newRuntimeForHTTPServer(t, server) + err := rt.RemoveImageExact(context.Background(), domain.RuntimeImageRef{}) + require.Error(t, err) + assert.Contains(t, err.Error(), "invalid identity") + + err = rt.RemoveVolumeExact(context.Background(), domain.RuntimeVolumeRef{}) + require.Error(t, err) + assert.Contains(t, err.Error(), "invalid identity") + + err = rt.RemoveImageExact(context.Background(), domain.RuntimeImageRef{ID: ""}) + require.Error(t, err) +} + +func TestRuntime_ExactRemovalPropagatesRuntimeErrors(t *testing.T) { + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(http.StatusConflict) + _, _ = w.Write([]byte(`{"message":"volume is in use"}`)) + })) + defer server.Close() + + rt := newRuntimeForHTTPServer(t, server) + ref, err := domain.NewRuntimeVolumeRef("in-use") + require.NoError(t, err) + err = rt.RemoveVolumeExact(context.Background(), ref) + require.Error(t, err) + assert.True(t, strings.Contains(err.Error(), "remove volume") || strings.Contains(err.Error(), "in use"), + "unexpected error: %v", err) +} diff --git a/internal/adapters/out/docker/runtime_recovery_test.go b/internal/adapters/out/docker/runtime_recovery_test.go new file mode 100644 index 000000000..e22868e57 --- /dev/null +++ b/internal/adapters/out/docker/runtime_recovery_test.go @@ -0,0 +1,148 @@ +package docker + +import ( + "context" + "errors" + "io" + "net/http" + "net/http/httptest" + "strconv" + "strings" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +// TestRuntime_InspectContainerReportsNotFoundSentinel proves a missing +// container is distinguishable from a transient runtime failure with +// errors.Is, so recovery never treats an API outage as a dead workload. +func TestRuntime_InspectContainerReportsNotFoundSentinel(t *testing.T) { + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(http.StatusNotFound) + _, _ = w.Write([]byte(`{"message":"No such container: gone"}`)) + })) + defer server.Close() + + runtime := newRuntimeForHTTPServer(t, server) + _, err := runtime.InspectContainer(context.Background(), "gone") + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrContainerNotFound) +} + +// TestRuntime_InspectContainerPreservesTransientErrors proves a non-404 +// failure is not reported as a missing container. +func TestRuntime_InspectContainerPreservesTransientErrors(t *testing.T) { + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(http.StatusInternalServerError) + _, _ = w.Write([]byte(`{"message":"server error"}`)) + })) + defer server.Close() + + runtime := newRuntimeForHTTPServer(t, server) + _, err := runtime.InspectContainer(context.Background(), "c-1") + require.Error(t, err) + assert.NotErrorIs(t, err, domain.ErrContainerNotFound) +} + +// TestRuntime_InspectContainerNormalizesStatusAndExecutionStart proves +// the state a recovery pass classifies on (running/exited/restarting/ +// paused) and the current execution start used to scope log readiness. +func TestRuntime_InspectContainerNormalizesStatusAndExecutionStart(t *testing.T) { + started := "2026-09-10T12:00:00.123456789Z" + statuses := []string{"running", "exited", "restarting", "paused", "created"} + for _, status := range statuses { + t.Run(status, func(t *testing.T) { + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(`{ + "Id":"c-1", + "Created":"2026-09-10T11:59:00Z", + "State":{"Status":"` + status + `","ExitCode":1,"StartedAt":"` + started + `"}, + "Config":{"Image":"img:1"} + }`)) + })) + defer server.Close() + + runtime := newRuntimeForHTTPServer(t, server) + container, err := runtime.InspectContainer(context.Background(), "c-1") + require.NoError(t, err) + assert.Equal(t, status, container.Status) + assert.Equal(t, 1, container.ExitCode, "a non-running container keeps its exit code") + assert.Equal(t, time.Date(2026, 9, 10, 12, 0, 0, 123456789, time.UTC), container.StartedAt.UTC()) + }) + } +} + +// TestRuntime_StartContainerReportsNotFoundSentinel proves the +// native-restart race can be told apart from a real start failure. +func TestRuntime_StartContainerReportsNotFoundSentinel(t *testing.T) { + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(http.StatusNotFound) + _, _ = w.Write([]byte(`{"message":"No such container: gone"}`)) + })) + defer server.Close() + + runtime := newRuntimeForHTTPServer(t, server) + err := runtime.StartContainer(context.Background(), "gone") + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrContainerNotFound) +} + +// TestRuntime_GetContainerLogsSinceScopesTheRequest proves readiness +// reads one exact container and passes the execution boundary to the +// runtime, so markers from a previous execution cannot match. +func TestRuntime_GetContainerLogsSinceScopesTheRequest(t *testing.T) { + // Sub-second boundary: a marker from the same second as a previous + // execution must not pass the filter. + since := time.Date(2026, 9, 10, 12, 0, 0, 123456789, time.UTC) + var gotPath, gotSince, gotTail string + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + gotPath = r.URL.Path + gotSince = r.URL.Query().Get("since") + gotTail = r.URL.Query().Get("tail") + w.Header().Set("Content-Type", "application/vnd.docker.raw-stream") + _, _ = w.Write([]byte("2026-09-10T12:00:01Z ready\n")) + })) + defer server.Close() + + runtime := newRuntimeForHTTPServer(t, server) + stream, err := runtime.GetContainerLogsSince(context.Background(), "c-exact", since, false) + require.NoError(t, err) + defer func() { _ = stream.Close() }() + body, err := io.ReadAll(stream) + require.NoError(t, err) + assert.Contains(t, string(body), "ready") + + assert.True(t, strings.HasSuffix(gotPath, "/containers/c-exact/logs"), "logs are read from the exact container ID") + // The Docker API takes the boundary as unix seconds with nanoseconds. + wireSince, parseErr := strconv.ParseFloat(gotSince, 64) + require.NoError(t, parseErr) + wantSince := float64(since.UnixNano()) / 1e9 + assert.InDelta(t, wantSince, wireSince, 1e-6, "the sub-second execution boundary reaches the runtime unchanged") + assert.Equal(t, "10000", gotTail) +} + +// TestRuntime_GetContainerLogsSinceReportsNotFoundSentinel proves a +// removed candidate fails readiness with an actionable error instead of +// hanging until the deadline. +func TestRuntime_GetContainerLogsSinceReportsNotFoundSentinel(t *testing.T) { + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(http.StatusNotFound) + _, _ = w.Write([]byte(`{"message":"No such container: gone"}`)) + })) + defer server.Close() + + runtime := newRuntimeForHTTPServer(t, server) + _, err := runtime.GetContainerLogsSince(context.Background(), "gone", time.Now(), false) + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrContainerNotFound) + assert.False(t, errors.Is(err, context.DeadlineExceeded)) +} diff --git a/internal/adapters/out/docker/stop_timeout_test.go b/internal/adapters/out/docker/stop_timeout_test.go new file mode 100644 index 000000000..02b74755b --- /dev/null +++ b/internal/adapters/out/docker/stop_timeout_test.go @@ -0,0 +1,42 @@ +package docker + +import ( + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// TestStopTimeout_ConvertsGraceToAPISeconds proves the runtime API never +// receives a truncated or negative grace: a non-positive grace keeps the +// runtime default (nil), and a fractional grace rounds up so the +// container always gets at least the declared time. +func TestStopTimeout_ConvertsGraceToAPISeconds(t *testing.T) { + cases := []struct { + name string + grace time.Duration + want *int + }{ + {name: "zero keeps the runtime default", grace: 0, want: nil}, + {name: "negative keeps the runtime default", grace: -time.Second, want: nil}, + {name: "default app grace", grace: 10 * time.Second, want: intPtr(10)}, + {name: "sub-second rounds up", grace: 500 * time.Millisecond, want: intPtr(1)}, + {name: "fractional rounds up", grace: 1500 * time.Millisecond, want: intPtr(2)}, + {name: "exact seconds are preserved", grace: 25 * time.Second, want: intPtr(25)}, + {name: "the app cap is preserved", grace: 5 * time.Minute, want: intPtr(300)}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + got := stopTimeout(tc.grace) + if tc.want == nil { + assert.Nil(t, got) + return + } + require.NotNil(t, got) + assert.Equal(t, *tc.want, *got) + }) + } +} + +func intPtr(v int) *int { return &v } diff --git a/internal/adapters/out/docker/testhelpers_test.go b/internal/adapters/out/docker/testhelpers_test.go new file mode 100644 index 000000000..d2799e74e --- /dev/null +++ b/internal/adapters/out/docker/testhelpers_test.go @@ -0,0 +1,23 @@ +package docker + +import ( + "net/http/httptest" + "strings" + "testing" + + "github.com/moby/moby/client" + "github.com/stretchr/testify/require" +) + +// newRuntimeForHTTPServer builds a Runtime whose Docker API endpoint is +// the given test server, so adapter behaviour can be asserted against +// the exact HTTP requests it issues. +func newRuntimeForHTTPServer(t *testing.T, server *httptest.Server) *Runtime { + t.Helper() + + host := strings.TrimPrefix(server.URL, "http://") + cli, err := client.New(client.WithHost("tcp://"+host), client.WithAPIVersion("1.41"), client.WithHTTPClient(server.Client())) + require.NoError(t, err) + + return NewRuntimeWithClient(cli) +} diff --git a/internal/adapters/out/domainsecrets/helpers_test.go b/internal/adapters/out/domainsecrets/helpers_test.go deleted file mode 100644 index dcc4027e5..000000000 --- a/internal/adapters/out/domainsecrets/helpers_test.go +++ /dev/null @@ -1,25 +0,0 @@ -package domainsecrets - -import ( - "fmt" - "testing" - - "github.com/bnema/zerowrap" -) - -func testLogger() zerowrap.Logger { - return zerowrap.Default() -} - -// CleanupPassAttachment removes attachment secrets from pass for testing. -// It's a test helper that should be called with defer to ensure cleanup. -// This function is exported to be shared across test packages. -func CleanupPassAttachment(_ *testing.T, containerName string, keys []string) { - for _, key := range keys { - path := fmt.Sprintf("%s/%s/%s", PassAttachmentPath, containerName, key) - _ = passCmd("rm", "-f", path) - } - - manifestPath := fmt.Sprintf("%s/%s/.keys", PassAttachmentPath, containerName) - _ = passCmd("rm", "-f", manifestPath) -} diff --git a/internal/adapters/out/domainsecrets/pass_store.go b/internal/adapters/out/domainsecrets/pass_store.go deleted file mode 100644 index fddfc4d1d..000000000 --- a/internal/adapters/out/domainsecrets/pass_store.go +++ /dev/null @@ -1,934 +0,0 @@ -// Package domainsecrets implements the DomainSecretStore adapter using pass. -package domainsecrets - -import ( - "context" - "fmt" - "os/exec" - "regexp" - "sort" - "strings" - "sync" - "time" - - "github.com/bnema/zerowrap" - - "github.com/bnema/gordon/internal/adapters/out/secrets" - "github.com/bnema/gordon/internal/boundaries/out" - "github.com/bnema/gordon/internal/domain" -) - -// ansiRegex matches ANSI escape sequences for stripping from pass output. -var ansiRegex = regexp.MustCompile(`\x1b\[[0-9;]*m`) - -const ( - PassDomainSecretsPath = "gordon/env" //nolint:gosec // Not a credential, this is a pass store path. - PassAttachmentPath = "gordon/env/attachments" //nolint:gosec // Not a credential, this is a pass store path. -) - -// PassStore implements the DomainSecretStore interface using the pass password manager. -type PassStore struct { - mu sync.Mutex - timeout time.Duration - log zerowrap.Logger -} - -// NewPassStore creates a new pass-based domain secret store. -func NewPassStore(log zerowrap.Logger) (*PassStore, error) { - // Use a timeout so a stalled GPG agent or keyring does not hang startup. - probeCtx, probeCancel := context.WithTimeout(context.Background(), 10*time.Second) - defer probeCancel() - if err := exec.CommandContext(probeCtx, "pass", "version").Run(); err != nil { //nolint:gosec // binary is a constant ("pass"), no user input - return nil, fmt.Errorf("pass is not available: %w", err) - } - - log.Debug(). - Str(zerowrap.FieldLayer, "adapter"). - Str(zerowrap.FieldAdapter, "domainsecrets"). - Str("provider", "pass"). - Msg("domain secret store initialized") - - return &PassStore{ - timeout: 10 * time.Second, - log: log, - }, nil -} - -// ListKeys returns the list of secret keys for a domain (not values). -func (s *PassStore) ListKeys(domainName string) ([]string, error) { - manifestPath, err := s.manifestPath(domainName) - if err != nil { - return nil, err - } - - ctx, cancel := context.WithTimeout(context.Background(), s.timeout) - defer cancel() - - content, exists, err := s.passShow(ctx, manifestPath) - if err != nil { - return nil, err - } - keys := []string{} - if exists { - for _, line := range strings.Split(content, "\n") { - key := strings.TrimSpace(line) - if key == "" { - continue - } - keys = append(keys, key) - } - } - sort.Strings(keys) - - // Recover keys that may exist in pass but are missing from the manifest. - discovered, err := s.listDomainKeys(domainName) - if err != nil { - return nil, err - } - - merged, changed := mergeUniqueKeys(keys, discovered) - if changed { - // Best-effort self-heal of stale manifest. - writeCtx, writeCancel := context.WithTimeout(context.Background(), s.timeout) - writeErr := s.passInsert(writeCtx, manifestPath, strings.Join(merged, "\n")) - writeCancel() - if writeErr != nil { - s.log.Warn(). - Str(zerowrap.FieldLayer, "adapter"). - Str(zerowrap.FieldAdapter, "domainsecrets"). - Str("domain", domainName). - Err(writeErr). - Msg("failed to self-heal pass manifest, continuing with recovered keys") - } - } - - return merged, nil -} - -// GetAll returns all secrets for a domain as a key-value map. -func (s *PassStore) GetAll(domainName string) (map[string]string, error) { - keys, err := s.ListKeys(domainName) - if err != nil { - return nil, err - } - - secretsMap := make(map[string]string) - for _, key := range keys { - keyPath, err := s.keyPath(domainName, key) - if err != nil { - return nil, err - } - - ctx, cancel := context.WithTimeout(context.Background(), s.timeout) - value, exists, err := s.passShow(ctx, keyPath) - cancel() - if err != nil { - return nil, err - } - if !exists { - s.log.Warn(). - Str(zerowrap.FieldLayer, "adapter"). - Str(zerowrap.FieldAdapter, "domainsecrets"). - Str("domain", domainName). - Str("key", key). - Msg("secret listed in manifest but missing in pass") - continue - } - secretsMap[key] = value - } - - return secretsMap, nil -} - -// SetIfEmpty sets multiple secrets for a domain only when no keys already exist. -// It performs the precondition check inside the store adapter so migration -// callers cannot accidentally split the check and write into separate steps. -func (s *PassStore) SetIfEmpty(domainName string, secretsMap map[string]string) ([]string, error) { - s.mu.Lock() - defer s.mu.Unlock() - - existingKeys, err := s.ListKeys(domainName) - if err != nil { - return nil, err - } - if len(existingKeys) > 0 { - return nil, fmt.Errorf("%w: pass secrets already exist for %s (found %d keys)", domain.ErrSecretsAlreadyExist, domainName, len(existingKeys)) - } - if err := s.Set(domainName, secretsMap); err != nil { - return nil, err - } - return sortedMapKeys(secretsMap), nil -} - -// Set sets or updates multiple secrets for a domain. -func (s *PassStore) Set(domainName string, secretsMap map[string]string) error { - if _, err := s.domainPath(domainName); err != nil { - return err - } - - existingKeys, err := s.ListKeys(domainName) - if err != nil { - return err - } - - existingSet := make(map[string]struct{}, len(existingKeys)) - for _, key := range existingKeys { - existingSet[key] = struct{}{} - } - - insertedNewPaths := make([]string, 0, len(secretsMap)) - for key, value := range secretsMap { - if err := domain.ValidateEnvKey(key); err != nil { - return err - } - - keyPath, err := s.keyPath(domainName, key) - if err != nil { - return err - } - - ctx, cancel := context.WithTimeout(context.Background(), s.timeout) - err = s.passInsert(ctx, keyPath, value) - cancel() - if err != nil { - s.cleanupInsertedPaths(insertedNewPaths) - return fmt.Errorf("failed to store secret %s: %w", key, err) - } - - if _, exists := existingSet[key]; !exists { - insertedNewPaths = append(insertedNewPaths, keyPath) - } - } - - keySet := make(map[string]struct{}, len(existingKeys)+len(secretsMap)) - for _, key := range existingKeys { - keySet[key] = struct{}{} - } - for key := range secretsMap { - keySet[key] = struct{}{} - } - - keys := make([]string, 0, len(keySet)) - for key := range keySet { - keys = append(keys, key) - } - sort.Strings(keys) - - manifestPath, err := s.manifestPath(domainName) - if err != nil { - return err - } - - ctx, cancel := context.WithTimeout(context.Background(), s.timeout) - if err := s.passInsert(ctx, manifestPath, strings.Join(keys, "\n")); err != nil { - cancel() - s.cleanupInsertedPaths(insertedNewPaths) - return fmt.Errorf("failed to update manifest: %w", err) - } - cancel() - - return nil -} - -// Delete removes a specific secret key from a domain. -func (s *PassStore) Delete(domainName, key string) error { - if _, err := s.domainPath(domainName); err != nil { - return err - } - - keyPath, err := s.keyPath(domainName, key) - if err != nil { - return err - } - - rmCtx, rmCancel := context.WithTimeout(context.Background(), s.timeout) - defer rmCancel() - if err := s.passRemove(rmCtx, keyPath); err != nil { - return err - } - - keys, err := s.ListKeys(domainName) - if err != nil { - return err - } - - updated := make([]string, 0, len(keys)) - for _, existingKey := range keys { - if existingKey == key { - continue - } - updated = append(updated, existingKey) - } - sort.Strings(updated) - - manifestPath, err := s.manifestPath(domainName) - if err != nil { - return err - } - - insertCtx, insertCancel := context.WithTimeout(context.Background(), s.timeout) - defer insertCancel() - if err := s.passInsert(insertCtx, manifestPath, strings.Join(updated, "\n")); err != nil { - return fmt.Errorf("failed to update manifest: %w", err) - } - - return nil -} - -// SetAttachmentIfEmpty sets attachment secrets only when no keys already exist. -// It performs the precondition check inside the store adapter so migration -// callers cannot accidentally split the check and write into separate steps. -func (s *PassStore) SetAttachmentIfEmpty(containerName string, secretsMap map[string]string) ([]string, error) { - s.mu.Lock() - defer s.mu.Unlock() - - existingKeys, err := s.listAttachmentKeysRecover(containerName) - if err != nil { - return nil, err - } - if len(existingKeys) > 0 { - return nil, fmt.Errorf("%w: pass secrets already exist for attachment %s (found %d keys)", domain.ErrSecretsAlreadyExist, containerName, len(existingKeys)) - } - if err := s.SetAttachment(containerName, secretsMap); err != nil { - return nil, err - } - return sortedMapKeys(secretsMap), nil -} - -// SetAttachment sets or updates multiple secrets for an attachment container. -func (s *PassStore) SetAttachment(containerName string, secretsMap map[string]string) error { - if _, err := s.attachmentPath(containerName); err != nil { - return err - } - - existingKeys, err := s.listAttachmentKeys(containerName) - if err != nil { - return err - } - - existingSet := make(map[string]struct{}, len(existingKeys)) - for _, key := range existingKeys { - existingSet[key] = struct{}{} - } - - insertedNewPaths := make([]string, 0, len(secretsMap)) - for key, value := range secretsMap { - if err := domain.ValidateEnvKey(key); err != nil { - return err - } - - keyPath, err := s.attachmentKeyPath(containerName, key) - if err != nil { - return err - } - - ctx, cancel := context.WithTimeout(context.Background(), s.timeout) - err = s.passInsert(ctx, keyPath, value) - cancel() - if err != nil { - s.cleanupInsertedPaths(insertedNewPaths) - return fmt.Errorf("failed to store attachment secret %s: %w", key, err) - } - - if _, exists := existingSet[key]; !exists { - insertedNewPaths = append(insertedNewPaths, keyPath) - } - } - - keySet := make(map[string]struct{}, len(existingKeys)+len(secretsMap)) - for _, key := range existingKeys { - keySet[key] = struct{}{} - } - for key := range secretsMap { - keySet[key] = struct{}{} - } - - keys := make([]string, 0, len(keySet)) - for key := range keySet { - keys = append(keys, key) - } - sort.Strings(keys) - - manifestPath, err := s.attachmentManifestPath(containerName) - if err != nil { - return err - } - - ctx, cancel := context.WithTimeout(context.Background(), s.timeout) - if err := s.passInsert(ctx, manifestPath, strings.Join(keys, "\n")); err != nil { - cancel() - s.cleanupInsertedPaths(insertedNewPaths) - return fmt.Errorf("failed to update attachment manifest: %w", err) - } - cancel() - - return nil -} - -// DeleteAttachment removes a specific secret key from an attachment container. -func (s *PassStore) DeleteAttachment(containerName, key string) error { - if _, err := s.attachmentPath(containerName); err != nil { - return err - } - - keyPath, err := s.attachmentKeyPath(containerName, key) - if err != nil { - return err - } - - rmCtx, rmCancel := context.WithTimeout(context.Background(), s.timeout) - defer rmCancel() - if err := s.passRemove(rmCtx, keyPath); err != nil { - return err - } - - keys, err := s.listAttachmentKeys(containerName) - if err != nil { - return err - } - - updated := make([]string, 0, len(keys)) - for _, existingKey := range keys { - if existingKey == key { - continue - } - updated = append(updated, existingKey) - } - sort.Strings(updated) - - manifestPath, err := s.attachmentManifestPath(containerName) - if err != nil { - return err - } - - insertCtx, insertCancel := context.WithTimeout(context.Background(), s.timeout) - defer insertCancel() - if err := s.passInsert(insertCtx, manifestPath, strings.Join(updated, "\n")); err != nil { - return fmt.Errorf("failed to update attachment manifest: %w", err) - } - - return nil -} - -// GetAllAttachment returns all secrets for an attachment container as a key-value map. -func (s *PassStore) GetAllAttachment(containerName string) (map[string]string, error) { - keys, err := s.listAttachmentKeys(containerName) - if err != nil { - return nil, err - } - - secretsMap := make(map[string]string) - for _, key := range keys { - keyPath, err := s.attachmentKeyPath(containerName, key) - if err != nil { - return nil, err - } - - ctx, cancel := context.WithTimeout(context.Background(), s.timeout) - value, exists, err := s.passShow(ctx, keyPath) - cancel() - if err != nil { - return nil, err - } - if !exists { - s.log.Warn(). - Str(zerowrap.FieldLayer, "adapter"). - Str(zerowrap.FieldAdapter, "domainsecrets"). - Str("container", containerName). - Str("key", key). - Msg("attachment secret listed in manifest but missing in pass") - continue - } - secretsMap[key] = value - } - - return secretsMap, nil -} - -// ListAttachmentKeys finds attachment secrets for a domain from pass. -// Supports both new (collision-resistant) and legacy container naming for backwards compatibility. -func (s *PassStore) ListAttachmentKeys(domainName string) ([]out.AttachmentSecrets, error) { - if _, err := s.domainPath(domainName); err != nil { - return nil, err - } - - ctx, cancel := context.WithTimeout(context.Background(), s.timeout) - defer cancel() - - containers, err := s.listTopLevelEntries(ctx, PassAttachmentPath) - if err != nil { - return nil, err - } - - // Try both new and legacy sanitization for backwards compatibility - sanitizedDomain := domain.SanitizeDomainForContainer(domainName) - sanitizedDomainLegacy := domain.SanitizeDomainForContainerLegacy(domainName) - prefixes := []string{ - "gordon-" + sanitizedDomain + "-", // New format (collision-resistant) - "gordon-" + sanitizedDomainLegacy + "-", // Old format (buggy but backwards compatible) - } - - seen := make(map[string]bool) - var results []out.AttachmentSecrets - for _, containerName := range containers { - matches := false - for _, prefix := range prefixes { - if strings.HasPrefix(containerName, prefix) { - matches = true - break - } - } - if !matches { - continue - } - - // Deduplicate results - if seen[containerName] { - continue - } - seen[containerName] = true - - manifestPath := fmt.Sprintf("%s/%s/.keys", PassAttachmentPath, containerName) - if err := secrets.ValidatePath(manifestPath); err != nil { - return nil, err - } - - showCtx, showCancel := context.WithTimeout(context.Background(), s.timeout) - content, exists, err := s.passShow(showCtx, manifestPath) - showCancel() - if err != nil { - return nil, err - } - if !exists { - continue - } - - keys := []string{} - for _, line := range strings.Split(content, "\n") { - key := strings.TrimSpace(line) - if key == "" { - continue - } - keys = append(keys, key) - } - - if len(keys) > 0 { - results = append(results, out.AttachmentSecrets{ - Service: containerName, - Keys: keys, - }) - } - } - - return results, nil -} - -func (s *PassStore) domainPath(domainName string) (string, error) { - safeDomain, err := s.sanitizeDomain(domainName) - if err != nil { - return "", err - } - path := fmt.Sprintf("%s/%s", PassDomainSecretsPath, safeDomain) - if err := secrets.ValidatePath(path); err != nil { - return "", err - } - return path, nil -} - -func (s *PassStore) keyPath(domainName, key string) (string, error) { - domainPath, err := s.domainPath(domainName) - if err != nil { - return "", err - } - if err := domain.ValidateEnvKey(key); err != nil { - return "", err - } - path := fmt.Sprintf("%s/%s", domainPath, key) - if err := secrets.ValidatePath(path); err != nil { - return "", err - } - return path, nil -} - -func (s *PassStore) manifestPath(domainName string) (string, error) { - domainPath, err := s.domainPath(domainName) - if err != nil { - return "", err - } - path := fmt.Sprintf("%s/.keys", domainPath) - if err := secrets.ValidatePath(path); err != nil { - return "", err - } - return path, nil -} - -func (s *PassStore) listDomainKeys(domainName string) ([]string, error) { - basePath, err := s.domainPath(domainName) - if err != nil { - return nil, err - } - - ctx, cancel := context.WithTimeout(context.Background(), s.timeout) - defer cancel() - - entries, err := s.listTopLevelEntries(ctx, basePath) - if err != nil { - return nil, err - } - - keys := make([]string, 0, len(entries)) - for _, entry := range entries { - if entry == ".keys" { - continue - } - if err := domain.ValidateEnvKey(entry); err != nil { - continue - } - keys = append(keys, entry) - } - - sort.Strings(keys) - return keys, nil -} - -func (s *PassStore) attachmentPath(containerName string) (string, error) { - if err := domain.ValidateContainerName(containerName); err != nil { - return "", err - } - path := fmt.Sprintf("%s/%s", PassAttachmentPath, containerName) - if err := secrets.ValidatePath(path); err != nil { - return "", err - } - return path, nil -} - -func (s *PassStore) attachmentKeyPath(containerName, key string) (string, error) { - attachmentPath, err := s.attachmentPath(containerName) - if err != nil { - return "", err - } - if err := domain.ValidateEnvKey(key); err != nil { - return "", err - } - path := fmt.Sprintf("%s/%s", attachmentPath, key) - if err := secrets.ValidatePath(path); err != nil { - return "", err - } - return path, nil -} - -func (s *PassStore) attachmentManifestPath(containerName string) (string, error) { - attachmentPath, err := s.attachmentPath(containerName) - if err != nil { - return "", err - } - path := fmt.Sprintf("%s/.keys", attachmentPath) - if err := secrets.ValidatePath(path); err != nil { - return "", err - } - return path, nil -} - -func (s *PassStore) listAttachmentKeys(containerName string) ([]string, error) { - manifestPath, err := s.attachmentManifestPath(containerName) - if err != nil { - return nil, err - } - - ctx, cancel := context.WithTimeout(context.Background(), s.timeout) - defer cancel() - - content, exists, err := s.passShow(ctx, manifestPath) - if err != nil { - return nil, err - } - if !exists { - return []string{}, nil - } - - keys := []string{} - for _, line := range strings.Split(content, "\n") { - key := strings.TrimSpace(line) - if key == "" { - continue - } - keys = append(keys, key) - } - - sort.Strings(keys) - return keys, nil -} - -func (s *PassStore) listAttachmentKeysRecover(containerName string) ([]string, error) { - keys, err := s.listAttachmentKeys(containerName) - if err != nil { - return nil, err - } - discovered, err := s.discoverAttachmentKeys(containerName) - if err != nil { - return nil, err - } - merged, changed := mergeUniqueKeys(keys, discovered) - if changed { - manifestPath, err := s.attachmentManifestPath(containerName) - if err != nil { - return nil, err - } - writeCtx, writeCancel := context.WithTimeout(context.Background(), s.timeout) - writeErr := s.passInsert(writeCtx, manifestPath, strings.Join(merged, "\n")) - writeCancel() - if writeErr != nil { - s.log.Warn(). - Str(zerowrap.FieldLayer, "adapter"). - Str(zerowrap.FieldAdapter, "domainsecrets"). - Str("container", containerName). - Err(writeErr). - Msg("failed to self-heal attachment pass manifest, continuing with recovered keys") - } - } - return merged, nil -} - -func (s *PassStore) discoverAttachmentKeys(containerName string) ([]string, error) { - basePath, err := s.attachmentPath(containerName) - if err != nil { - return nil, err - } - - ctx, cancel := context.WithTimeout(context.Background(), s.timeout) - defer cancel() - - entries, err := s.listTopLevelEntries(ctx, basePath) - if err != nil { - return nil, err - } - - keys := make([]string, 0, len(entries)) - for _, entry := range entries { - if entry == ".keys" { - continue - } - if err := domain.ValidateEnvKey(entry); err != nil { - continue - } - keys = append(keys, entry) - } - - sort.Strings(keys) - return keys, nil -} - -func (s *PassStore) sanitizeDomain(domainName string) (string, error) { - safeDomain, err := domain.SanitizeDomainForEnvFile(domainName) - if err != nil { - s.log.Warn(). - Str(zerowrap.FieldLayer, "adapter"). - Str(zerowrap.FieldAdapter, "domainsecrets"). - Str("domain", domainName). - Err(err). - Msg("rejected invalid domain") - return "", domain.ErrPathTraversal - } - return safeDomain, nil -} - -// ManifestExists checks if the manifest exists for a domain. -func (s *PassStore) ManifestExists(domainName string) (bool, error) { - manifestPath, err := s.manifestPath(domainName) - if err != nil { - return false, err - } - - ctx, cancel := context.WithTimeout(context.Background(), s.timeout) - defer cancel() - - _, exists, err := s.passShow(ctx, manifestPath) - if err != nil { - return false, err - } - - return exists, nil -} - -func (s *PassStore) passInsert(ctx context.Context, path, value string) error { - cmd := exec.CommandContext(ctx, "pass", "insert", "-m", "-f", path) //nolint:gosec // binary is constant ("pass"); path arguments validated by secrets path validator - cmd.Stdin = strings.NewReader(value) - output, err := cmd.CombinedOutput() - if err != nil { - return fmt.Errorf("pass insert failed: %s: %w", strings.TrimSpace(string(output)), err) - } - return nil -} - -func (s *PassStore) passRemove(ctx context.Context, path string) error { - cmd := exec.CommandContext(ctx, "pass", "rm", "-f", path) //nolint:gosec // binary is constant ("pass"); path arguments validated by secrets path validator - output, err := cmd.CombinedOutput() - if err != nil { - if passEntryMissing(string(output)) { - return nil - } - return fmt.Errorf("pass rm failed: %s: %w", strings.TrimSpace(string(output)), err) - } - return nil -} - -func (s *PassStore) passShow(ctx context.Context, path string) (string, bool, error) { - cmd := exec.CommandContext(ctx, "pass", "show", path) //nolint:gosec // binary is constant ("pass"); path arguments validated by secrets path validator - output, err := cmd.CombinedOutput() - if err != nil { - if passEntryMissing(string(output)) { - return "", false, nil - } - return "", false, fmt.Errorf("pass show failed: %s: %w", strings.TrimSpace(string(output)), err) - } - - clean := ansiRegex.ReplaceAllString(string(output), "") - clean = strings.TrimRight(clean, "\r\n") - return clean, true, nil -} - -func (s *PassStore) listTopLevelEntries(ctx context.Context, basePath string) ([]string, error) { - if err := secrets.ValidatePath(basePath); err != nil { - return nil, err - } - - cmd := exec.CommandContext(ctx, "pass", "ls", basePath) //nolint:gosec // binary is constant ("pass"); path arguments validated by secrets path validator - output, err := cmd.CombinedOutput() - if err != nil { - if passEntryMissing(string(output)) { - return []string{}, nil - } - return nil, fmt.Errorf("pass ls failed: %s: %w", strings.TrimSpace(string(output)), err) - } - - entries := []string{} - for _, entry := range parsePassListOutput(basePath, string(output)) { - if entry.depth == 1 { - entries = append(entries, entry.name) - } - } - - return entries, nil -} - -type passListEntry struct { - name string - depth int -} - -func parsePassListOutput(basePath, output string) []passListEntry { - lines := strings.Split(output, "\n") - entries := []passListEntry{} - - for _, line := range lines { - line = ansiRegex.ReplaceAllString(line, "") - if strings.TrimSpace(line) == "" { - continue - } - - depth := 0 - prefixLoop: - for { - switch { - case strings.HasPrefix(line, "│ "): - line = strings.TrimPrefix(line, "│ ") - depth++ - case strings.HasPrefix(line, "| "): - line = strings.TrimPrefix(line, "| ") - depth++ - case strings.HasPrefix(line, " "): - line = strings.TrimPrefix(line, " ") - depth++ - default: - break prefixLoop - } - } - - switch { - case strings.HasPrefix(line, "├── "): - line = strings.TrimPrefix(line, "├── ") - depth++ - case strings.HasPrefix(line, "└── "): - line = strings.TrimPrefix(line, "└── ") - depth++ - case strings.HasPrefix(line, "|-- "): - line = strings.TrimPrefix(line, "|-- ") - depth++ - case strings.HasPrefix(line, "`-- "): - line = strings.TrimPrefix(line, "`-- ") - depth++ - } - - name := strings.TrimSpace(line) - if name == "" || name == basePath { - continue - } - if depth == 0 { - continue - } - - entries = append(entries, passListEntry{name: name, depth: depth}) - } - - return entries -} - -func passEntryMissing(output string) bool { - clean := ansiRegex.ReplaceAllString(output, "") - lower := strings.ToLower(clean) - return strings.Contains(lower, "not in the password store") || - strings.Contains(lower, "password store is empty") -} - -func sortedMapKeys(values map[string]string) []string { - keys := make([]string, 0, len(values)) - for key := range values { - keys = append(keys, key) - } - sort.Strings(keys) - return keys -} - -func mergeUniqueKeys(primary, secondary []string) ([]string, bool) { - mergedSet := make(map[string]struct{}, len(primary)+len(secondary)) - for _, key := range primary { - mergedSet[key] = struct{}{} - } - for _, key := range secondary { - mergedSet[key] = struct{}{} - } - - merged := make([]string, 0, len(mergedSet)) - for key := range mergedSet { - merged = append(merged, key) - } - sort.Strings(merged) - - if len(merged) != len(primary) { - return merged, true - } - - for i := range primary { - if primary[i] != merged[i] { - return merged, true - } - } - - return merged, false -} - -func (s *PassStore) cleanupInsertedPaths(paths []string) { - if len(paths) == 0 { - return - } - - for _, path := range paths { - // Wrap in a closure so defer cancel() runs at the end of each iteration, - // not at the end of cleanupInsertedPaths (which would delay all cancels). - func(p string) { - ctx, cancel := context.WithTimeout(context.Background(), s.timeout) - defer cancel() - _ = s.passRemove(ctx, p) - }(path) - } -} diff --git a/internal/adapters/out/domainsecrets/pass_store_test.go b/internal/adapters/out/domainsecrets/pass_store_test.go deleted file mode 100644 index d5d9fd4ec..000000000 --- a/internal/adapters/out/domainsecrets/pass_store_test.go +++ /dev/null @@ -1,330 +0,0 @@ -package domainsecrets - -import ( - "context" - "fmt" - "os/exec" - "strings" - "testing" - "time" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" - - "github.com/bnema/gordon/internal/domain" -) - -func passCmd(args ...string) error { - ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second) - defer cancel() - return exec.CommandContext(ctx, "pass", args...).Run() -} - -func passInsertValue(path, value string) error { - ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second) - defer cancel() - cmd := exec.CommandContext(ctx, "pass", "insert", "-m", "-f", path) - cmd.Stdin = strings.NewReader(value) - _, err := cmd.CombinedOutput() - return err -} - -func passShow(path string) (string, error) { - ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second) - defer cancel() - out, err := exec.CommandContext(ctx, "pass", "show", path).CombinedOutput() - if err != nil { - return "", err - } - return strings.TrimSpace(string(out)), nil -} - -func requirePass(t *testing.T) { - if err := passCmd("version"); err != nil { - t.Skip("pass not available") - } - if err := passCmd("ls"); err != nil { - t.Skip("pass store not initialized") - } -} - -func cleanupPassDomain(_ *testing.T, domainName string, keys []string) { - safeDomain, err := domain.SanitizeDomainForEnvFile(domainName) - if err != nil { - return - } - - for _, key := range keys { - path := fmt.Sprintf("%s/%s/%s", PassDomainSecretsPath, safeDomain, key) - _ = passCmd("rm", "-f", path) - } - - manifestPath := fmt.Sprintf("%s/%s/.keys", PassDomainSecretsPath, safeDomain) - _ = passCmd("rm", "-f", manifestPath) -} - -func TestPassStore_SetGetDelete(t *testing.T) { - requirePass(t) - - store, err := NewPassStore(testLogger()) - require.NoError(t, err) - - domainName := fmt.Sprintf("test.%d.example.com", time.Now().UnixNano()) - keys := []string{"API_KEY", "DB_PASSWORD"} - defer cleanupPassDomain(t, domainName, keys) - - secretsMap := map[string]string{ - "API_KEY": "alpha", - "DB_PASSWORD": "bravo", - } - - err = store.Set(domainName, secretsMap) - require.NoError(t, err) - - keysList, err := store.ListKeys(domainName) - require.NoError(t, err) - assert.Len(t, keysList, 2) - assert.ElementsMatch(t, keys, keysList) - - values, err := store.GetAll(domainName) - require.NoError(t, err) - assert.Equal(t, "alpha", values["API_KEY"]) - assert.Equal(t, "bravo", values["DB_PASSWORD"]) - - err = store.Delete(domainName, "DB_PASSWORD") - require.NoError(t, err) - - keysList, err = store.ListKeys(domainName) - require.NoError(t, err) - assert.ElementsMatch(t, []string{"API_KEY"}, keysList) -} - -func TestPassStore_SetGetAttachment(t *testing.T) { - requirePass(t) - - store, err := NewPassStore(testLogger()) - require.NoError(t, err) - - containerName := fmt.Sprintf("gitea-postgres-%d", time.Now().UnixNano()) - keys := []string{"POSTGRES_USER", "POSTGRES_PASSWORD"} - defer CleanupPassAttachment(t, containerName, keys) - - secretsMap := map[string]string{ - "POSTGRES_USER": "gitea", - "POSTGRES_PASSWORD": "secret123", - } - - err = store.SetAttachment(containerName, secretsMap) - require.NoError(t, err) - - values, err := store.GetAllAttachment(containerName) - require.NoError(t, err) - assert.Equal(t, "gitea", values["POSTGRES_USER"]) - assert.Equal(t, "secret123", values["POSTGRES_PASSWORD"]) -} - -func TestPassStore_ListAttachmentKeys_AfterSet(t *testing.T) { - requirePass(t) - - store, err := NewPassStore(testLogger()) - require.NoError(t, err) - - containerName := fmt.Sprintf("redis-cache-%d", time.Now().UnixNano()) - keys := []string{"REDIS_PASSWORD"} - defer CleanupPassAttachment(t, containerName, keys) - - secretsMap := map[string]string{ - "REDIS_PASSWORD": "redis123", - } - - err = store.SetAttachment(containerName, secretsMap) - require.NoError(t, err) - - values, err := store.GetAllAttachment(containerName) - require.NoError(t, err) - assert.Len(t, values, 1) - assert.Equal(t, "redis123", values["REDIS_PASSWORD"]) -} - -func TestPassStore_DeleteAttachment(t *testing.T) { - requirePass(t) - - store, err := NewPassStore(testLogger()) - require.NoError(t, err) - - containerName := fmt.Sprintf("gitea-postgres-%d", time.Now().UnixNano()) - keys := []string{"POSTGRES_USER", "POSTGRES_PASSWORD"} - defer CleanupPassAttachment(t, containerName, keys) - - // Set 2 secrets - secretsMap := map[string]string{ - "POSTGRES_USER": "gitea", - "POSTGRES_PASSWORD": "secret123", - } - err = store.SetAttachment(containerName, secretsMap) - require.NoError(t, err) - - // Delete 1 - err = store.DeleteAttachment(containerName, "POSTGRES_PASSWORD") - require.NoError(t, err) - - // Verify remaining via GetAllAttachment - values, err := store.GetAllAttachment(containerName) - require.NoError(t, err) - assert.Len(t, values, 1) - assert.Equal(t, "gitea", values["POSTGRES_USER"]) - _, exists := values["POSTGRES_PASSWORD"] - assert.False(t, exists) -} - -func TestPassStore_Delete_Idempotent(t *testing.T) { - requirePass(t) - - store, err := NewPassStore(testLogger()) - require.NoError(t, err) - - domainName := fmt.Sprintf("idempotent.%d.example.com", time.Now().UnixNano()) - keys := []string{"API_KEY"} - defer cleanupPassDomain(t, domainName, keys) - - err = store.Set(domainName, map[string]string{"API_KEY": "alpha"}) - require.NoError(t, err) - - err = store.Delete(domainName, "API_KEY") - require.NoError(t, err) - - // Second delete of an already-removed key must be a no-op. - err = store.Delete(domainName, "API_KEY") - require.NoError(t, err) - - values, err := store.GetAll(domainName) - require.NoError(t, err) - assert.NotContains(t, values, "API_KEY") -} - -func TestPassStore_DeleteAttachment_Idempotent(t *testing.T) { - requirePass(t) - - store, err := NewPassStore(testLogger()) - require.NoError(t, err) - - containerName := fmt.Sprintf("idempotent-attachment-%d", time.Now().UnixNano()) - keys := []string{"POSTGRES_PASSWORD"} - defer CleanupPassAttachment(t, containerName, keys) - - err = store.SetAttachment(containerName, map[string]string{"POSTGRES_PASSWORD": "secret123"}) - require.NoError(t, err) - - err = store.DeleteAttachment(containerName, "POSTGRES_PASSWORD") - require.NoError(t, err) - - // Second delete of an already-removed key must be a no-op. - err = store.DeleteAttachment(containerName, "POSTGRES_PASSWORD") - require.NoError(t, err) - - values, err := store.GetAllAttachment(containerName) - require.NoError(t, err) - assert.NotContains(t, values, "POSTGRES_PASSWORD") -} - -func TestPassStore_SetAttachment_OverwriteValue(t *testing.T) { - requirePass(t) - - store, err := NewPassStore(testLogger()) - require.NoError(t, err) - - containerName := fmt.Sprintf("attachment-overwrite-%d", time.Now().UnixNano()) - keys := []string{"REDIS_PASSWORD"} - defer CleanupPassAttachment(t, containerName, keys) - - err = store.SetAttachment(containerName, map[string]string{"REDIS_PASSWORD": "first"}) - require.NoError(t, err) - - // Re-setting same key with a different value should deterministically overwrite. - err = store.SetAttachment(containerName, map[string]string{"REDIS_PASSWORD": "second"}) - require.NoError(t, err) - - values, err := store.GetAllAttachment(containerName) - require.NoError(t, err) - assert.Equal(t, "second", values["REDIS_PASSWORD"]) -} - -func TestPassStore_ListKeys_RecoversOrphanedEntries(t *testing.T) { - requirePass(t) - - store, err := NewPassStore(testLogger()) - require.NoError(t, err) - - domainName := fmt.Sprintf("orphan.%d.example.com", time.Now().UnixNano()) - keys := []string{"EXISTING", "ORIGIN"} - defer cleanupPassDomain(t, domainName, keys) - - err = store.Set(domainName, map[string]string{"EXISTING": "present"}) - require.NoError(t, err) - - safeDomain, err := domain.SanitizeDomainForEnvFile(domainName) - require.NoError(t, err) - - orphanPath := fmt.Sprintf("%s/%s/ORIGIN", PassDomainSecretsPath, safeDomain) - err = passInsertValue(orphanPath, "https://example.com") - require.NoError(t, err) - - manifestPath := fmt.Sprintf("%s/%s/.keys", PassDomainSecretsPath, safeDomain) - err = passInsertValue(manifestPath, "EXISTING\n") - require.NoError(t, err) - - listed, err := store.ListKeys(domainName) - require.NoError(t, err) - assert.ElementsMatch(t, []string{"EXISTING", "ORIGIN"}, listed) - - values, err := store.GetAll(domainName) - require.NoError(t, err) - assert.Equal(t, "present", values["EXISTING"]) - assert.Equal(t, "https://example.com", values["ORIGIN"]) - - manifest, err := passShow(manifestPath) - require.NoError(t, err) - assert.Contains(t, manifest, "EXISTING") - assert.Contains(t, manifest, "ORIGIN") -} - -func TestParsePassListOutput(t *testing.T) { - tests := []struct { - name string - basePath string - output string - want []passListEntry - }{ - { - name: "simple tree", - basePath: "domain", - output: "domain\n├── key1\n└── key2", - want: []passListEntry{{name: "key1", depth: 1}, {name: "key2", depth: 1}}, - }, - { - name: "nested tree", - basePath: "domain", - output: "domain\n│ ├── subkey1\n│ └── subkey2\n└── key1", - want: []passListEntry{{name: "subkey1", depth: 2}, {name: "subkey2", depth: 2}, {name: "key1", depth: 1}}, - }, - { - name: "empty output", - basePath: "domain", - output: "", - want: []passListEntry{}, - }, - { - name: "ASCII fallback chars", - basePath: "domain", - output: "domain\n| |-- subkey1\n`-- key1", - want: []passListEntry{{name: "subkey1", depth: 2}, {name: "key1", depth: 1}}, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - got := parsePassListOutput(tt.basePath, tt.output) - assert.Equal(t, tt.want, got) - }) - } -} diff --git a/internal/adapters/out/domainsecrets/store.go b/internal/adapters/out/domainsecrets/store.go deleted file mode 100644 index 0eeaa8169..000000000 --- a/internal/adapters/out/domainsecrets/store.go +++ /dev/null @@ -1,554 +0,0 @@ -// Package domainsecrets implements the DomainSecretStore adapter using filesystem-based env files. -package domainsecrets - -import ( - "bufio" - "fmt" - "os" - "path/filepath" - "sort" - "strings" - - "github.com/bnema/zerowrap" - - "github.com/bnema/gordon/internal/boundaries/out" - "github.com/bnema/gordon/internal/domain" -) - -// FileStore implements the DomainSecretStore interface using filesystem-based env files. -// This adapter is responsible only for file I/O operations; domain validation -// should be performed by the use case layer before calling these methods. -type FileStore struct { - envDir string - log zerowrap.Logger -} - -// NewFileStore creates a new file-based domain secret store. -func NewFileStore(envDir string, log zerowrap.Logger) (*FileStore, error) { - // Ensure env directory exists - if err := os.MkdirAll(envDir, 0700); err != nil { - return nil, fmt.Errorf("failed to create env directory %s: %w", envDir, err) - } - - log.Debug(). - Str(zerowrap.FieldLayer, "adapter"). - Str(zerowrap.FieldAdapter, "domainsecrets"). - Str("env_dir", envDir). - Msg("domain secret store initialized") - - return &FileStore{ - envDir: envDir, - log: log, - }, nil -} - -// ListKeys returns the list of secret keys for a domain (not values). -func (s *FileStore) ListKeys(domainName string) ([]string, error) { - envFile, err := s.validateEnvFilePath(domainName) - if err != nil { - return nil, err - } - - file, err := os.Open(envFile) - if err != nil { - if os.IsNotExist(err) { - return []string{}, nil - } - return nil, fmt.Errorf("failed to open env file: %w", err) - } - defer file.Close() - - var keys []string - scanner := bufio.NewScanner(file) - for scanner.Scan() { - line := strings.TrimSpace(scanner.Text()) - // Skip empty lines and comments - if line == "" || strings.HasPrefix(line, "#") { - continue - } - // Extract key from KEY=value - if idx := strings.Index(line, "="); idx > 0 { - keys = append(keys, line[:idx]) - } - } - - if err := scanner.Err(); err != nil { - return nil, fmt.Errorf("failed to read env file: %w", err) - } - - return keys, nil -} - -// GetAll returns all secrets for a domain as a key-value map. -func (s *FileStore) GetAll(domainName string) (map[string]string, error) { - envFile, err := s.validateEnvFilePath(domainName) - if err != nil { - return nil, err - } - - data, err := os.ReadFile(envFile) - if err != nil { - if os.IsNotExist(err) { - return map[string]string{}, nil - } - return nil, fmt.Errorf("failed to read env file: %w", err) - } - - secrets, err := domain.ParseEnvData(data) - if err != nil { - return nil, fmt.Errorf("failed to parse env file: %w", err) - } - - return secrets, nil -} - -// Set sets or updates multiple secrets for a domain, merging with existing. -func (s *FileStore) Set(domainName string, secrets map[string]string) error { - // Validate domain first - if _, err := s.validateEnvFilePath(domainName); err != nil { - return err - } - - // Ensure env directory exists - if err := os.MkdirAll(s.envDir, 0700); err != nil { - return fmt.Errorf("failed to create env directory: %w", err) - } - - // Read existing secrets - existing, err := s.GetAll(domainName) - if err != nil { - return err - } - - // Merge new secrets with existing - for key, value := range secrets { - existing[key] = value - } - - // Write back atomically - return s.writeSecretsAtomic(domainName, existing) -} - -// Delete removes a specific secret key from a domain. -func (s *FileStore) Delete(domainName, key string) error { - // Validate domain first - if _, err := s.validateEnvFilePath(domainName); err != nil { - return err - } - - // Read existing secrets - existing, err := s.GetAll(domainName) - if err != nil { - return err - } - - // Remove the key - delete(existing, key) - - // Write back atomically - return s.writeSecretsAtomic(domainName, existing) -} - -// SetAttachment sets or updates multiple secrets for an attachment container. -func (s *FileStore) SetAttachment(containerName string, secrets map[string]string) error { - // Validate container name first - if _, err := s.validateAttachmentEnvFilePath(containerName); err != nil { - return err - } - - // Ensure env directory exists - if err := os.MkdirAll(s.envDir, 0700); err != nil { - return fmt.Errorf("failed to create env directory: %w", err) - } - - // Read existing secrets - existing, err := s.GetAllAttachment(containerName) - if err != nil { - return err - } - - // Merge new secrets with existing - for key, value := range secrets { - existing[key] = value - } - - // Write back atomically - return s.writeAttachmentSecretsAtomic(containerName, existing) -} - -// DeleteAttachment removes a specific secret key from an attachment container. -func (s *FileStore) DeleteAttachment(containerName, key string) error { - // Validate container name first - if _, err := s.validateAttachmentEnvFilePath(containerName); err != nil { - return err - } - - // Read existing secrets - existing, err := s.GetAllAttachment(containerName) - if err != nil { - return err - } - - // Remove the key (no-op if not present) - delete(existing, key) - - // Write back atomically - return s.writeAttachmentSecretsAtomic(containerName, existing) -} - -// GetAllAttachment returns all secrets for an attachment container as a key-value map. -func (s *FileStore) GetAllAttachment(containerName string) (map[string]string, error) { - envFile, err := s.validateAttachmentEnvFilePath(containerName) - if err != nil { - return nil, err - } - - data, err := os.ReadFile(envFile) - if err != nil { - if os.IsNotExist(err) { - return map[string]string{}, nil - } - return nil, fmt.Errorf("failed to read attachment env file: %w", err) - } - - secrets, err := domain.ParseEnvData(data) - if err != nil { - return nil, fmt.Errorf("failed to parse attachment env file: %w", err) - } - - return secrets, nil -} - -// validateSecretValues checks that no secret value contains newline characters. -func validateSecretValues(secrets map[string]string) error { - for key, value := range secrets { - if strings.ContainsAny(value, "\n\r") { - return fmt.Errorf("secret value for %q contains newline characters", key) - } - } - return nil -} - -// writeSecretsAtomic writes all secrets to the domain's env file atomically. -// It writes to a temporary file first, syncs it, then renames to the final path. -func (s *FileStore) writeSecretsAtomic(domainName string, secrets map[string]string) error { - if err := validateSecretValues(secrets); err != nil { - return fmt.Errorf("invalid domain secret values for %q: %w", domainName, err) - } - - envFile, err := s.validateEnvFilePath(domainName) - if err != nil { - return err - } - tmpFile := envFile + ".tmp" - - file, err := os.OpenFile(tmpFile, os.O_WRONLY|os.O_CREATE|os.O_TRUNC, 0600) - if err != nil { - return fmt.Errorf("failed to create temp env file: %w", err) - } - - // Write header comment - if _, err := fmt.Fprintf(file, "# Environment variables for %s\n", domainName); err != nil { - file.Close() - os.Remove(tmpFile) - return fmt.Errorf("failed to write header: %w", err) - } - if _, err := fmt.Fprintf(file, "# Managed by Gordon admin API\n\n"); err != nil { - file.Close() - os.Remove(tmpFile) - return fmt.Errorf("failed to write header: %w", err) - } - - // Write each secret in sorted order for deterministic output - keys := make([]string, 0, len(secrets)) - for key := range secrets { - keys = append(keys, key) - } - sort.Strings(keys) - - for _, key := range keys { - if _, err := fmt.Fprintf(file, "%s=%s\n", key, secrets[key]); err != nil { - file.Close() - os.Remove(tmpFile) - return fmt.Errorf("failed to write secret %s: %w", key, err) - } - } - - // Sync to ensure data is on disk before rename - if err := file.Sync(); err != nil { - file.Close() - os.Remove(tmpFile) - return fmt.Errorf("failed to sync env file: %w", err) - } - - if err := file.Close(); err != nil { - os.Remove(tmpFile) - return fmt.Errorf("failed to close env file: %w", err) - } - - // Atomic rename - if err := os.Rename(tmpFile, envFile); err != nil { - os.Remove(tmpFile) - return fmt.Errorf("failed to rename env file: %w", err) - } - - return nil -} - -// getEnvFilePath converts a domain to its env file path. -// This must match the naming convention in envloader.FileLoader.getEnvFilePath. -// -// SECURITY: Validates domain and ensures the resulting path stays within envDir. -func (s *FileStore) getEnvFilePath(domainName string) (string, error) { - storageKey, err := domain.NewEnvStorageKey(domainName) - if err != nil { - s.log.Warn(). - Str(zerowrap.FieldLayer, "adapter"). - Str(zerowrap.FieldAdapter, "domainsecrets"). - Str("domain", domainName). - Err(err). - Msg("rejected invalid domain") - return "", err - } - - fullPath := filepath.Join(s.envDir, storageKey.FileName()) - - // SECURITY: Final validation - ensure path stays within envDir - cleanPath := filepath.Clean(fullPath) - cleanEnvDir := filepath.Clean(s.envDir) - if !strings.HasPrefix(cleanPath, cleanEnvDir+string(filepath.Separator)) { - s.log.Error(). - Str(zerowrap.FieldLayer, "adapter"). - Str(zerowrap.FieldAdapter, "domainsecrets"). - Str("domain", domainName). - Str("attempted_path", fullPath). - Msg("path traversal attempt blocked - path escapes env directory") - return "", domain.ErrPathTraversal - } - - return fullPath, nil -} - -// validateEnvFilePath validates that a domain produces a valid env file path. -// Returns an error if the domain is invalid or would result in path traversal. -func (s *FileStore) validateEnvFilePath(domainName string) (string, error) { - return s.getEnvFilePath(domainName) -} - -// getAttachmentEnvFilePath converts a container name to its attachment env file path. -// Attachment env files follow the pattern: gordon-.env -// -// SECURITY: Validates container name and ensures the resulting path stays within envDir. -func (s *FileStore) getAttachmentEnvFilePath(containerName string) (string, error) { - if err := domain.ValidateContainerName(containerName); err != nil { - s.log.Warn(). - Str(zerowrap.FieldLayer, "adapter"). - Str(zerowrap.FieldAdapter, "domainsecrets"). - Str("container", containerName). - Err(err). - Msg("rejected invalid container name") - return "", err - } - - fullPath := filepath.Join(s.envDir, "gordon-"+containerName+".env") - - // SECURITY: Final validation - ensure path stays within envDir - cleanPath := filepath.Clean(fullPath) - cleanEnvDir := filepath.Clean(s.envDir) - if !strings.HasPrefix(cleanPath, cleanEnvDir+string(filepath.Separator)) { - s.log.Error(). - Str(zerowrap.FieldLayer, "adapter"). - Str(zerowrap.FieldAdapter, "domainsecrets"). - Str("container", containerName). - Str("attempted_path", fullPath). - Msg("path traversal attempt blocked - path escapes env directory") - return "", domain.ErrPathTraversal - } - - return fullPath, nil -} - -// validateAttachmentEnvFilePath validates that a container name produces a valid attachment env file path. -// Returns an error if the container name is invalid or would result in path traversal. -func (s *FileStore) validateAttachmentEnvFilePath(containerName string) (string, error) { - return s.getAttachmentEnvFilePath(containerName) -} - -// writeAttachmentSecretsAtomic writes all secrets to the attachment's env file atomically. -// It writes to a temporary file first, syncs it, then renames to the final path. -func (s *FileStore) writeAttachmentSecretsAtomic(containerName string, secrets map[string]string) error { - if err := validateSecretValues(secrets); err != nil { - return fmt.Errorf("invalid attachment secret values for %q: %w", containerName, err) - } - - envFile, err := s.validateAttachmentEnvFilePath(containerName) - if err != nil { - return err - } - tmpFile := envFile + ".tmp" - - file, err := os.OpenFile(tmpFile, os.O_WRONLY|os.O_CREATE|os.O_TRUNC, 0600) - if err != nil { - return fmt.Errorf("failed to create temp attachment env file: %w", err) - } - - // Write header comment - if _, err := fmt.Fprintf(file, "# Environment variables for attachment %s\n", containerName); err != nil { - file.Close() - os.Remove(tmpFile) - return fmt.Errorf("failed to write header: %w", err) - } - if _, err := fmt.Fprintf(file, "# Managed by Gordon admin API\n\n"); err != nil { - file.Close() - os.Remove(tmpFile) - return fmt.Errorf("failed to write header: %w", err) - } - - // Write each secret in sorted order for deterministic output - keys := make([]string, 0, len(secrets)) - for key := range secrets { - keys = append(keys, key) - } - sort.Strings(keys) - - for _, key := range keys { - if _, err := fmt.Fprintf(file, "%s=%s\n", key, secrets[key]); err != nil { - file.Close() - os.Remove(tmpFile) - return fmt.Errorf("failed to write secret %s: %w", key, err) - } - } - - // Sync to ensure data is on disk before rename - if err := file.Sync(); err != nil { - file.Close() - os.Remove(tmpFile) - return fmt.Errorf("failed to sync attachment env file: %w", err) - } - - if err := file.Close(); err != nil { - os.Remove(tmpFile) - return fmt.Errorf("failed to close attachment env file: %w", err) - } - - // Atomic rename - if err := os.Rename(tmpFile, envFile); err != nil { - os.Remove(tmpFile) - return fmt.Errorf("failed to rename attachment env file: %w", err) - } - - return nil -} - -// ListAttachmentKeys finds attachment env files for a domain and returns their keys. -// Supports both new (collision-resistant) and legacy container naming for backwards compatibility. -// Attachment env files follow the naming pattern: gordon-{sanitized-domain}-{service}.env -func (s *FileStore) ListAttachmentKeys(domainName string) ([]out.AttachmentSecrets, error) { - // Try both new and legacy sanitization for backwards compatibility - sanitized := domain.SanitizeDomainForContainer(domainName) - sanitizedLegacy := domain.SanitizeDomainForContainerLegacy(domainName) - prefixes := []string{ - "gordon-" + sanitized + "-", // New format (collision-resistant) - "gordon-" + sanitizedLegacy + "-", // Old format (buggy but backwards compatible) - } - - // List all files in env directory - entries, err := os.ReadDir(s.envDir) - if err != nil { - if os.IsNotExist(err) { - return nil, nil - } - return nil, fmt.Errorf("failed to read env directory: %w", err) - } - - seen := make(map[string]bool) - var results []out.AttachmentSecrets - for _, entry := range entries { - if entry.IsDir() { - continue - } - name := entry.Name() - - // Check if this matches any of our prefixes - var matchedPrefix string - for _, prefix := range prefixes { - if strings.HasPrefix(name, prefix) && strings.HasSuffix(name, ".env") { - matchedPrefix = prefix - break - } - } - if matchedPrefix == "" { - continue - } - - // Extract container name from filename - containerName := strings.TrimSuffix(name, ".env") - - // Deduplicate results - if seen[containerName] { - continue - } - seen[containerName] = true - - // Extract service name from filename - // e.g., "gordon-git-bnema-dev-gitea-postgres.env" → "gitea-postgres" - serviceName := strings.TrimPrefix(name, matchedPrefix) - serviceName = strings.TrimSuffix(serviceName, ".env") - if serviceName == "" { - continue - } - - // Read keys from this file using existing method - // Note: We use the container name directly since it matches the env file naming - keys, err := s.listKeysFromFile(filepath.Join(s.envDir, name)) - if err != nil { - s.log.Warn(). - Err(err). - Str("file", name). - Str("domain", domainName). - Msg("failed to read attachment secrets file") - continue - } - - if len(keys) > 0 { - results = append(results, out.AttachmentSecrets{ - Service: containerName, - Keys: keys, - }) - } - } - - return results, nil -} - -// listKeysFromFile reads secret keys from a specific env file path. -func (s *FileStore) listKeysFromFile(filePath string) ([]string, error) { - file, err := os.Open(filePath) - if err != nil { - if os.IsNotExist(err) { - return []string{}, nil - } - return nil, fmt.Errorf("failed to open env file: %w", err) - } - defer file.Close() - - var keys []string - scanner := bufio.NewScanner(file) - for scanner.Scan() { - line := strings.TrimSpace(scanner.Text()) - // Skip empty lines and comments - if line == "" || strings.HasPrefix(line, "#") { - continue - } - // Extract key from KEY=value - if idx := strings.Index(line, "="); idx > 0 { - keys = append(keys, line[:idx]) - } - } - - if err := scanner.Err(); err != nil { - return nil, fmt.Errorf("failed to read env file: %w", err) - } - - return keys, nil -} diff --git a/internal/adapters/out/domainsecrets/store_test.go b/internal/adapters/out/domainsecrets/store_test.go deleted file mode 100644 index 8f67f26cf..000000000 --- a/internal/adapters/out/domainsecrets/store_test.go +++ /dev/null @@ -1,432 +0,0 @@ -package domainsecrets - -import ( - "errors" - "os" - "path/filepath" - "testing" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" - - "github.com/bnema/gordon/internal/domain" -) - -func TestFileStore_PathTraversal(t *testing.T) { - tmpDir, err := os.MkdirTemp("", "domainsecrets-test-*") - require.NoError(t, err) - defer os.RemoveAll(tmpDir) - - store, err := NewFileStore(tmpDir, testLogger()) - require.NoError(t, err) - - tests := []struct { - name string - domain string - wantErr bool - }{ - // Valid domains - {"simple domain", "example.com", false}, - {"subdomain", "sub.example.com", false}, - {"domain with port", "example.com:8080", false}, - {"domain with path", "example.com/path", false}, - {"hyphenated domain", "my-app.example.com", false}, - {"single char", "a", false}, - - // Path traversal attempts - should fail - {"path traversal with dots", "../../../etc/passwd", true}, - {"path traversal in middle", "example.com/../../../etc/passwd", true}, - {"double dots only", "..", true}, - {"path traversal at end", "example.com/..", true}, - - // Invalid format - should fail - {"empty domain", "", true}, - {"starts with dot", ".example.com", true}, - {"ends with dot", "example.com.", true}, - {"special chars semicolon", "example;rm -rf /", true}, - {"null byte", "example\x00.com", true}, - {"newline", "example\n.com", true}, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - // Test ListKeys - _, err := store.ListKeys(tt.domain) - if tt.wantErr { - require.Error(t, err) - assert.True(t, errors.Is(err, domain.ErrPathTraversal), "expected ErrPathTraversal, got: %v", err) - } else { - // No error, or file not found (which is fine) - if err != nil { - assert.False(t, errors.Is(err, domain.ErrPathTraversal), "unexpected ErrPathTraversal") - } - } - - // Test Set - err = store.Set(tt.domain, map[string]string{"KEY": "value"}) - if tt.wantErr { - require.Error(t, err) - assert.True(t, errors.Is(err, domain.ErrPathTraversal), "expected ErrPathTraversal, got: %v", err) - } else { - assert.NoError(t, err) - } - - // Test GetAll - _, err = store.GetAll(tt.domain) - if tt.wantErr { - require.Error(t, err) - assert.True(t, errors.Is(err, domain.ErrPathTraversal), "expected ErrPathTraversal, got: %v", err) - } - - // Test Delete - err = store.Delete(tt.domain, "KEY") - if tt.wantErr { - require.Error(t, err) - assert.True(t, errors.Is(err, domain.ErrPathTraversal), "expected ErrPathTraversal, got: %v", err) - } - }) - } -} - -func TestFileStore_PathContainment(t *testing.T) { - tmpDir, err := os.MkdirTemp("", "domainsecrets-test-*") - require.NoError(t, err) - defer os.RemoveAll(tmpDir) - - store, err := NewFileStore(tmpDir, testLogger()) - require.NoError(t, err) - - // Valid domain - file should be created inside tmpDir - err = store.Set("example.com", map[string]string{"KEY": "value"}) - require.NoError(t, err) - - // Verify file exists inside tmpDir - expectedFile := filepath.Join(tmpDir, "ZXhhbXBsZS5jb20.env") - _, err = os.Stat(expectedFile) - assert.NoError(t, err, "expected env file to exist at %s", expectedFile) - - // Verify no file was created outside tmpDir - entries, err := filepath.Glob(tmpDir + "/../*") - require.NoError(t, err) - for _, entry := range entries { - if filepath.Base(entry) != filepath.Base(tmpDir) { - // Check it's not our test file escaped - assert.NotContains(t, entry, ".env", "unexpected .env file outside tmpDir: %s", entry) - } - } -} - -func TestFileStore_DistinctDomainsUseDistinctFiles(t *testing.T) { - store, err := NewFileStore(t.TempDir(), testLogger()) - require.NoError(t, err) - - dotPath, err := store.validateEnvFilePath("app.example.com") - require.NoError(t, err) - - portPath, err := store.validateEnvFilePath("app:example:com") - require.NoError(t, err) - - pathPath, err := store.validateEnvFilePath("app/example/com") - require.NoError(t, err) - - assert.NotEqual(t, dotPath, portPath) - assert.NotEqual(t, dotPath, pathPath) - assert.NotEqual(t, portPath, pathPath) -} - -func TestSanitizeDomainForContainer(t *testing.T) { - tests := []struct { - domain string - expected string - description string - }{ - { - domain: "git.example.com", - expected: "git__example__com", - description: "Dots become double underscores", - }, - { - domain: "git-example.com", - expected: "git-example__com", - description: "Hyphens preserved, dots become underscores", - }, - { - domain: "app:8080.example.com", - expected: "app-_8080__example__com", - description: "Colons become hyphen-underscore", - }, - { - domain: "git.example.com:3000", - expected: "git__example__com-_3000", - description: "Multiple separators handled distinctly", - }, - { - domain: "simple.com", - expected: "simple__com", - description: "Simple domain", - }, - } - - for _, tt := range tests { - t.Run(tt.domain, func(t *testing.T) { - result := domain.SanitizeDomainForContainer(tt.domain) - assert.Equal(t, tt.expected, result, "sanitization should match expected") - }) - } - - // Verify no collisions between potentially conflicting domains - t.Run("NoCollisions", func(t *testing.T) { - domains := []string{ - "git.example.com", - "git-example.com", - "app:8080.example.com", - "app-8080-example.com", - } - - results := make(map[string]string) - for _, d := range domains { - result := domain.SanitizeDomainForContainer(d) - if original, exists := results[result]; exists { - t.Errorf("COLLISION: %q and %q both sanitize to %q", original, d, result) - } - results[result] = d - } - }) -} - -func TestFileStore_AttachmentOperations(t *testing.T) { - tmpDir, err := os.MkdirTemp("", "domainsecrets-test-*") - require.NoError(t, err) - defer os.RemoveAll(tmpDir) - - store, err := NewFileStore(tmpDir, testLogger()) - require.NoError(t, err) - - containerName := "gitea-postgres" - - t.Run("SetAttachment creates new file", func(t *testing.T) { - err := store.SetAttachment(containerName, map[string]string{ - "POSTGRES_DB": "gitea", - "POSTGRES_PASSWORD": "secret123", - }) - require.NoError(t, err) - - // Verify file exists - expectedFile := filepath.Join(tmpDir, "gordon-"+containerName+".env") - _, err = os.Stat(expectedFile) - assert.NoError(t, err, "expected attachment env file to exist at %s", expectedFile) - }) - - t.Run("GetAllAttachment retrieves secrets", func(t *testing.T) { - secrets, err := store.GetAllAttachment(containerName) - require.NoError(t, err) - assert.Equal(t, "gitea", secrets["POSTGRES_DB"]) - assert.Equal(t, "secret123", secrets["POSTGRES_PASSWORD"]) - }) - - t.Run("SetAttachment merges with existing", func(t *testing.T) { - err := store.SetAttachment(containerName, map[string]string{ - "NEW_VAR": "new_value", - }) - require.NoError(t, err) - - // Should have both old and new secrets - secrets, err := store.GetAllAttachment(containerName) - require.NoError(t, err) - assert.Equal(t, "gitea", secrets["POSTGRES_DB"]) - assert.Equal(t, "secret123", secrets["POSTGRES_PASSWORD"]) - assert.Equal(t, "new_value", secrets["NEW_VAR"]) - }) - - t.Run("GetAllAttachment returns empty map for non-existent container", func(t *testing.T) { - secrets, err := store.GetAllAttachment("non-existent") - require.NoError(t, err) - assert.Empty(t, secrets) - }) - - t.Run("SetAttachment with hyphenated container name", func(t *testing.T) { - hyphenatedContainer := "gordon-git-example-com-gitea-postgres" - err := store.SetAttachment(hyphenatedContainer, map[string]string{ - "KEY": "value", - }) - require.NoError(t, err) - - secrets, err := store.GetAllAttachment(hyphenatedContainer) - require.NoError(t, err) - assert.Equal(t, "value", secrets["KEY"]) - }) -} - -func TestFileStore_AttachmentPathTraversal(t *testing.T) { - tmpDir, err := os.MkdirTemp("", "domainsecrets-test-*") - require.NoError(t, err) - defer os.RemoveAll(tmpDir) - - store, err := NewFileStore(tmpDir, testLogger()) - require.NoError(t, err) - - tests := []struct { - name string - container string - wantSetErr bool - wantGetErr bool - }{ - // Valid container names - {"simple", "postgres", false, false}, - {"with hyphen", "gitea-postgres", false, false}, - {"with underscore", "gitea_postgres", false, false}, - {"complex real container", "gordon-git-example-com-gitea-postgres", false, false}, - - // Invalid container names - should fail - {"empty", "", true, true}, - {"starts with number", "1container", true, true}, - {"starts with hyphen", "-container", true, true}, - {"starts with underscore", "_container", true, true}, - {"path traversal", "../etc", true, true}, - {"contains slash", "container/name", true, true}, - {"contains backslash", "container\\name", true, true}, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - // Test SetAttachment - err := store.SetAttachment(tt.container, map[string]string{"KEY": "value"}) - if tt.wantSetErr { - require.Error(t, err) - } else { - assert.NoError(t, err) - } - - // Test GetAllAttachment - _, err = store.GetAllAttachment(tt.container) - if tt.wantGetErr { - require.Error(t, err) - } else { - if !tt.wantSetErr { - // Only check no error if we successfully set it - assert.NoError(t, err) - } - } - }) - } -} - -func TestFileStore_ValidDomainOperations(t *testing.T) { - tmpDir, err := os.MkdirTemp("", "domainsecrets-test-*") - require.NoError(t, err) - defer os.RemoveAll(tmpDir) - - store, err := NewFileStore(tmpDir, testLogger()) - require.NoError(t, err) - - domain := "test.example.com" - - // Set secrets - err = store.Set(domain, map[string]string{ - "DB_HOST": "localhost", - "DB_PASSWORD": "secret123", - }) - require.NoError(t, err) - - // Get all secrets - secrets, err := store.GetAll(domain) - require.NoError(t, err) - assert.Equal(t, "localhost", secrets["DB_HOST"]) - assert.Equal(t, "secret123", secrets["DB_PASSWORD"]) - - // List keys - keys, err := store.ListKeys(domain) - require.NoError(t, err) - assert.Len(t, keys, 2) - assert.Contains(t, keys, "DB_HOST") - assert.Contains(t, keys, "DB_PASSWORD") - - // Delete a key - err = store.Delete(domain, "DB_PASSWORD") - require.NoError(t, err) - - // Verify deletion - secrets, err = store.GetAll(domain) - require.NoError(t, err) - assert.Len(t, secrets, 1) - assert.Equal(t, "localhost", secrets["DB_HOST"]) - _, exists := secrets["DB_PASSWORD"] - assert.False(t, exists) -} - -func TestFileStore_DeleteAttachment(t *testing.T) { - tmpDir, err := os.MkdirTemp("", "domainsecrets-test-*") - require.NoError(t, err) - defer os.RemoveAll(tmpDir) - - store, err := NewFileStore(tmpDir, testLogger()) - require.NoError(t, err) - - containerName := "gitea-postgres" - - // Seed with two secrets - err = store.SetAttachment(containerName, map[string]string{ - "POSTGRES_USER": "gitea", - "POSTGRES_PASSWORD": "secret123", - }) - require.NoError(t, err) - - // Delete one key - err = store.DeleteAttachment(containerName, "POSTGRES_PASSWORD") - require.NoError(t, err) - - // Verify remaining - secrets, err := store.GetAllAttachment(containerName) - require.NoError(t, err) - assert.Len(t, secrets, 1) - assert.Equal(t, "gitea", secrets["POSTGRES_USER"]) - _, exists := secrets["POSTGRES_PASSWORD"] - assert.False(t, exists) - - // Delete nonexistent key - should be no-op - err = store.DeleteAttachment(containerName, "NONEXISTENT_KEY") - assert.NoError(t, err) - - // Verify secrets unchanged after no-op delete - secrets, err = store.GetAllAttachment(containerName) - require.NoError(t, err) - assert.Len(t, secrets, 1) - assert.Equal(t, "gitea", secrets["POSTGRES_USER"]) -} - -func TestGetEnvFilePath(t *testing.T) { - tmpDir, err := os.MkdirTemp("", "domainsecrets-test-*") - require.NoError(t, err) - defer os.RemoveAll(tmpDir) - - store, err := NewFileStore(tmpDir, testLogger()) - require.NoError(t, err) - - tests := []struct { - domain string - wantEmpty bool - wantFilename string - }{ - {"example.com", false, "ZXhhbXBsZS5jb20.env"}, - {"sub.example.com", false, "c3ViLmV4YW1wbGUuY29t.env"}, - {"example.com:8080", false, "ZXhhbXBsZS5jb206ODA4MA.env"}, - {"example.com/path", false, "ZXhhbXBsZS5jb20vcGF0aA.env"}, - {"../evil", true, ""}, - {"", true, ""}, - } - - for _, tt := range tests { - t.Run(tt.domain, func(t *testing.T) { - path, err := store.getEnvFilePath(tt.domain) - - if tt.wantEmpty { - assert.Error(t, err) - assert.Empty(t, path) - } else { - require.NoError(t, err) - assert.NotEmpty(t, path) - assert.Equal(t, filepath.Join(tmpDir, tt.wantFilename), path) - } - }) - } -} diff --git a/internal/adapters/out/envloader/file.go b/internal/adapters/out/envloader/file.go deleted file mode 100644 index b9f0903aa..000000000 --- a/internal/adapters/out/envloader/file.go +++ /dev/null @@ -1,253 +0,0 @@ -// Package envloader implements the environment variable loader adapter. -package envloader - -import ( - "bufio" - "context" - "fmt" - "os" - "path/filepath" - "strings" - - "github.com/bnema/zerowrap" - - "github.com/bnema/gordon/internal/boundaries/out" - "github.com/bnema/gordon/internal/domain" -) - -// FileLoader implements the EnvLoader interface using filesystem-based env files. -type FileLoader struct { - envDir string - secretProviders map[string]out.SecretProvider - log zerowrap.Logger -} - -// NewFileLoader creates a new file-based environment loader. -func NewFileLoader(envDir string, log zerowrap.Logger) (*FileLoader, error) { - // Ensure env directory exists - if err := os.MkdirAll(envDir, 0700); err != nil { - return nil, fmt.Errorf("failed to create env directory %s: %w", envDir, err) - } - - log.Debug(). - Str(zerowrap.FieldLayer, "adapter"). - Str(zerowrap.FieldAdapter, "envloader"). - Str("env_dir", envDir). - Msg("env loader initialized") - - return &FileLoader{ - envDir: envDir, - secretProviders: make(map[string]out.SecretProvider), - log: log, - }, nil -} - -// RegisterSecretProvider registers a secret provider for resolving secrets. -func (l *FileLoader) RegisterSecretProvider(provider out.SecretProvider) { - l.secretProviders[provider.Name()] = provider - l.log.Debug(). - Str(zerowrap.FieldLayer, "adapter"). - Str(zerowrap.FieldAdapter, "envloader"). - Str("provider", provider.Name()). - Msg("secret provider registered") -} - -// LoadEnv loads environment variables for a given domain. -func (l *FileLoader) LoadEnv(ctx context.Context, domain string) ([]string, error) { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "adapter", - zerowrap.FieldAdapter: "envloader", - zerowrap.FieldAction: "LoadEnv", - "domain": domain, - }) - log := zerowrap.FromCtx(ctx) - - envVars := []string{} - envFile, err := l.getEnvFilePath(domain) - if err != nil { - return nil, log.WrapErr(err, "invalid env file domain") - } - - // Check if env file exists - if _, err := os.Stat(envFile); os.IsNotExist(err) { - log.Debug().Str("env_file", envFile).Msg("no env file found for route") - return envVars, nil - } - - log.Debug().Str("env_file", envFile).Msg("loading env file for route") - - file, err := os.Open(envFile) - if err != nil { - return nil, log.WrapErr(err, "failed to open env file") - } - defer file.Close() - - scanner := bufio.NewScanner(file) - lineNum := 0 - - for scanner.Scan() { - lineNum++ - line := strings.TrimSpace(scanner.Text()) - - // Skip empty lines and comments - if line == "" || strings.HasPrefix(line, "#") { - continue - } - - // Parse KEY=VALUE format - parts := strings.SplitN(line, "=", 2) - if len(parts) != 2 { - log.Warn(). - Int("line", lineNum). - Msg("invalid env file line format, skipping") - continue - } - - key := strings.TrimSpace(parts[0]) - value := strings.TrimSpace(parts[1]) - - // Remove quotes if present - if len(value) >= 2 { - if (strings.HasPrefix(value, "\"") && strings.HasSuffix(value, "\"")) || - (strings.HasPrefix(value, "'") && strings.HasSuffix(value, "'")) { - value = value[1 : len(value)-1] - } - } - - // Resolve secrets if value contains secret syntax - resolvedValue, err := l.resolveSecrets(ctx, value) - if err != nil { - return nil, log.WrapErrWithFields( - fmt.Errorf("key %s on line %d: %w", key, lineNum, err), - "failed to resolve secret", - map[string]any{ - "key": key, - "line": lineNum, - }, - ) - } - - envVars = append(envVars, fmt.Sprintf("%s=%s", key, resolvedValue)) - } - - if err := scanner.Err(); err != nil { - return nil, log.WrapErr(err, "error reading env file") - } - - log.Info().Int(zerowrap.FieldCount, len(envVars)).Msg("loaded environment variables for route") - - return envVars, nil -} - -// CreateEnvFile creates an empty environment file for a new domain. -func (l *FileLoader) CreateEnvFile(ctx context.Context, domain string) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "adapter", - zerowrap.FieldAdapter: "envloader", - zerowrap.FieldAction: "CreateEnvFile", - "domain": domain, - }) - log := zerowrap.FromCtx(ctx) - - envFile, err := l.getEnvFilePath(domain) - if err != nil { - return log.WrapErr(err, "invalid env file domain") - } - - // Check if file already exists - if _, err := os.Stat(envFile); err == nil { - log.Debug().Str("env_file", envFile).Msg("env file already exists, skipping creation") - return nil - } else if !os.IsNotExist(err) { - return log.WrapErr(err, "error checking env file") - } - - // Create empty env file with helpful comments - content := fmt.Sprintf(`# Environment variables for route: %s -# Add KEY=VALUE pairs, one per line -# Secrets can be referenced with ${provider:path} syntax - -`, domain) - - if err := os.WriteFile(envFile, []byte(content), 0600); err != nil { - return log.WrapErr(err, "failed to create env file") - } - - log.Info().Str("env_file", envFile).Msg("created empty env file for route") - - return nil -} - -// EnvFileExists checks if an environment file exists for a domain. -func (l *FileLoader) EnvFileExists(domain string) (bool, error) { - envFile, err := l.getEnvFilePath(domain) - if err != nil { - return false, err - } - - _, err = os.Stat(envFile) - if os.IsNotExist(err) { - return false, nil - } - if err != nil { - return false, err - } - - return true, nil -} - -// resolveSecrets resolves any secret references in the value. -// Secret syntax: ${provider:path} -func (l *FileLoader) resolveSecrets(ctx context.Context, value string) (string, error) { - // Check if value contains secret syntax: ${provider:path} - if !strings.Contains(value, "${") { - return value, nil - } - - result := value - for { - start := strings.Index(result, "${") - if start == -1 { - break - } - - end := strings.Index(result[start:], "}") - if end == -1 { - return "", fmt.Errorf("unclosed secret syntax") - } - end += start - - secretRef := result[start+2 : end] - parts := strings.SplitN(secretRef, ":", 2) - if len(parts) != 2 { - return "", fmt.Errorf("invalid secret syntax: expected 'provider:path'") - } - - providerName := parts[0] - secretPath := parts[1] - - provider, exists := l.secretProviders[providerName] - if !exists { - return "", fmt.Errorf("unknown secret provider: %s", providerName) - } - - secretValue, err := provider.GetSecret(ctx, secretPath) - if err != nil { - return "", fmt.Errorf("failed to get secret from provider %s", providerName) - } - - // Replace the secret reference with the actual value - result = result[:start] + secretValue + result[end+1:] - } - - return result, nil -} - -func (l *FileLoader) getEnvFilePath(domainName string) (string, error) { - storageKey, err := domain.NewEnvStorageKey(domainName) - if err != nil { - return "", err - } - - return filepath.Join(l.envDir, storageKey.FileName()), nil -} diff --git a/internal/adapters/out/envloader/file_test.go b/internal/adapters/out/envloader/file_test.go deleted file mode 100644 index 91d137791..000000000 --- a/internal/adapters/out/envloader/file_test.go +++ /dev/null @@ -1,368 +0,0 @@ -package envloader - -import ( - "bytes" - "context" - "errors" - "os" - "path/filepath" - "testing" - - "github.com/bnema/zerowrap" - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" - - "github.com/bnema/gordon/internal/boundaries/out/mocks" - "github.com/bnema/gordon/internal/domain" -) - -func TestFileLoader_ResolveSecrets(t *testing.T) { - // Create a temp directory for tests - tmpDir := t.TempDir() - log := zerowrap.New(zerowrap.Config{Level: "fatal"}) - - loader, err := NewFileLoader(tmpDir, log) - require.NoError(t, err) - - ctx := context.Background() - - t.Run("no secret syntax returns value unchanged", func(t *testing.T) { - result, err := loader.resolveSecrets(ctx, "plain-value") - assert.NoError(t, err) - assert.Equal(t, "plain-value", result) - }) - - t.Run("unclosed secret syntax returns error", func(t *testing.T) { - _, err := loader.resolveSecrets(ctx, "${pass:secret") - assert.Error(t, err) - assert.Contains(t, err.Error(), "unclosed secret syntax") - }) - - t.Run("invalid secret syntax without colon returns error", func(t *testing.T) { - _, err := loader.resolveSecrets(ctx, "${invalid}") - assert.Error(t, err) - assert.Contains(t, err.Error(), "invalid secret syntax") - }) - - t.Run("unknown provider returns error", func(t *testing.T) { - _, err := loader.resolveSecrets(ctx, "${unknown:path}") - assert.Error(t, err) - assert.Contains(t, err.Error(), "unknown secret provider") - }) - - t.Run("pass provider resolves secret", func(t *testing.T) { - mockPass := mocks.NewMockSecretProvider(t) - mockPass.EXPECT().Name().Return("pass") - mockPass.EXPECT().GetSecret(ctx, "company/api-key").Return("secret-value-123", nil) - - loader.RegisterSecretProvider(mockPass) - - result, err := loader.resolveSecrets(ctx, "${pass:company/api-key}") - assert.NoError(t, err) - assert.Equal(t, "secret-value-123", result) - }) - - t.Run("sops provider resolves secret", func(t *testing.T) { - mockSops := mocks.NewMockSecretProvider(t) - mockSops.EXPECT().Name().Return("sops") - mockSops.EXPECT().GetSecret(ctx, "secrets.yaml:app.password").Return("sops-secret", nil) - - loader.RegisterSecretProvider(mockSops) - - result, err := loader.resolveSecrets(ctx, "${sops:secrets.yaml:app.password}") - assert.NoError(t, err) - assert.Equal(t, "sops-secret", result) - }) - - t.Run("multiple secrets in one value", func(t *testing.T) { - mockPass := mocks.NewMockSecretProvider(t) - mockPass.EXPECT().Name().Return("pass") - mockPass.EXPECT().GetSecret(ctx, "db/user").Return("admin", nil) - mockPass.EXPECT().GetSecret(ctx, "db/pass").Return("secret123", nil) - - loader.RegisterSecretProvider(mockPass) - - result, err := loader.resolveSecrets(ctx, "postgresql://${pass:db/user}:${pass:db/pass}@localhost:5432/db") - assert.NoError(t, err) - assert.Equal(t, "postgresql://admin:secret123@localhost:5432/db", result) - }) - - t.Run("provider error is propagated", func(t *testing.T) { - mockPass := mocks.NewMockSecretProvider(t) - mockPass.EXPECT().Name().Return("pass") - mockPass.EXPECT().GetSecret(ctx, "missing/secret").Return("", assert.AnError) - - loader.RegisterSecretProvider(mockPass) - - _, err := loader.resolveSecrets(ctx, "${pass:missing/secret}") - assert.Error(t, err) - assert.Contains(t, err.Error(), "failed to get secret from provider") - }) -} - -func TestFileLoader_LoadEnv(t *testing.T) { - log := zerowrap.New(zerowrap.Config{Level: "fatal"}) - ctx := context.Background() - - t.Run("missing env file returns empty slice", func(t *testing.T) { - tmpDir := t.TempDir() - loader, err := NewFileLoader(tmpDir, log) - require.NoError(t, err) - - envVars, err := loader.LoadEnv(ctx, "nonexistent.example.com") - assert.NoError(t, err) - assert.Empty(t, envVars) - }) - - t.Run("loads simple env file", func(t *testing.T) { - tmpDir := t.TempDir() - loader, err := NewFileLoader(tmpDir, log) - require.NoError(t, err) - - // Create env file - envContent := `# Comment line -NODE_ENV=production -API_URL=https://api.example.com -` - storageKey, err := domain.NewEnvStorageKey("app.example.com") - require.NoError(t, err) - envFile := filepath.Join(tmpDir, storageKey.FileName()) - require.NoError(t, os.WriteFile(envFile, []byte(envContent), 0600)) - - envVars, err := loader.LoadEnv(ctx, "app.example.com") - assert.NoError(t, err) - assert.Len(t, envVars, 2) - assert.Contains(t, envVars, "NODE_ENV=production") - assert.Contains(t, envVars, "API_URL=https://api.example.com") - }) - - t.Run("loads env file with secrets", func(t *testing.T) { - tmpDir := t.TempDir() - loader, err := NewFileLoader(tmpDir, log) - require.NoError(t, err) - - // Register mock provider - mockPass := mocks.NewMockSecretProvider(t) - mockPass.EXPECT().Name().Return("pass") - mockPass.EXPECT().GetSecret(ctx, "app/api-key").Return("super-secret-key", nil) - loader.RegisterSecretProvider(mockPass) - - // Create env file with secret reference - envContent := `NODE_ENV=production -API_KEY=${pass:app/api-key} -` - storageKey, err := domain.NewEnvStorageKey("app.example.com") - require.NoError(t, err) - envFile := filepath.Join(tmpDir, storageKey.FileName()) - require.NoError(t, os.WriteFile(envFile, []byte(envContent), 0600)) - - envVars, err := loader.LoadEnv(ctx, "app.example.com") - assert.NoError(t, err) - assert.Len(t, envVars, 2) - assert.Contains(t, envVars, "NODE_ENV=production") - assert.Contains(t, envVars, "API_KEY=super-secret-key") - }) - - t.Run("handles quoted values", func(t *testing.T) { - tmpDir := t.TempDir() - loader, err := NewFileLoader(tmpDir, log) - require.NoError(t, err) - - envContent := `DOUBLE_QUOTED="hello world" -SINGLE_QUOTED='hello world' -` - storageKey, err := domain.NewEnvStorageKey("app.example.com") - require.NoError(t, err) - envFile := filepath.Join(tmpDir, storageKey.FileName()) - require.NoError(t, os.WriteFile(envFile, []byte(envContent), 0600)) - - envVars, err := loader.LoadEnv(ctx, "app.example.com") - assert.NoError(t, err) - assert.Len(t, envVars, 2) - assert.Contains(t, envVars, "DOUBLE_QUOTED=hello world") - assert.Contains(t, envVars, "SINGLE_QUOTED=hello world") - }) - - t.Run("skips invalid lines", func(t *testing.T) { - tmpDir := t.TempDir() - loader, err := NewFileLoader(tmpDir, log) - require.NoError(t, err) - - envContent := `VALID=value -invalid line without equals -ALSO_VALID=another -` - storageKey, err := domain.NewEnvStorageKey("app.example.com") - require.NoError(t, err) - envFile := filepath.Join(tmpDir, storageKey.FileName()) - require.NoError(t, os.WriteFile(envFile, []byte(envContent), 0600)) - - envVars, err := loader.LoadEnv(ctx, "app.example.com") - assert.NoError(t, err) - assert.Len(t, envVars, 2) - assert.Contains(t, envVars, "VALID=value") - assert.Contains(t, envVars, "ALSO_VALID=another") - }) -} - -func TestFileLoader_LoadEnv_RedactsSecretBearingDiagnostics(t *testing.T) { - ctx := context.Background() - - t.Run("invalid line warning omits raw line content", func(t *testing.T) { - var logs bytes.Buffer - log := zerowrap.New(zerowrap.Config{Level: "warn", Format: "json", Output: &logs}) - loader, err := NewFileLoader(t.TempDir(), log) - require.NoError(t, err) - ctx := zerowrap.WithCtx(ctx, log) - - envFile, err := loader.getEnvFilePath("app.example.com") - require.NoError(t, err) - require.NoError(t, os.WriteFile(envFile, []byte("valid line leaks super-secret-token\nVALID=value\n"), 0600)) - - envVars, err := loader.LoadEnv(ctx, "app.example.com") - require.NoError(t, err) - assert.Equal(t, []string{"VALID=value"}, envVars) - assert.NotContains(t, logs.String(), "super-secret-token") - assert.NotContains(t, logs.String(), "valid line leaks") - assert.Contains(t, logs.String(), "line") - }) - - t.Run("secret resolution error omits unclosed raw value", func(t *testing.T) { - log := zerowrap.New(zerowrap.Config{Level: "fatal"}) - loader, err := NewFileLoader(t.TempDir(), log) - require.NoError(t, err) - - envFile, err := loader.getEnvFilePath("app.example.com") - require.NoError(t, err) - require.NoError(t, os.WriteFile(envFile, []byte("DATABASE_URL=postgres://user:${pass:db-password-super-secret\n"), 0600)) - - _, err = loader.LoadEnv(ctx, "app.example.com") - require.Error(t, err) - assert.Contains(t, err.Error(), "failed to resolve secret") - assert.Contains(t, err.Error(), "DATABASE_URL") - assert.NotContains(t, err.Error(), "postgres://user") - assert.NotContains(t, err.Error(), "db-password-super-secret") - }) - - t.Run("secret resolution error omits malformed closed secret ref", func(t *testing.T) { - log := zerowrap.New(zerowrap.Config{Level: "fatal"}) - loader, err := NewFileLoader(t.TempDir(), log) - require.NoError(t, err) - - envFile, err := loader.getEnvFilePath("app.example.com") - require.NoError(t, err) - require.NoError(t, os.WriteFile(envFile, []byte("TOKEN=${db-password-super-secret}\n"), 0600)) - - _, err = loader.LoadEnv(ctx, "app.example.com") - require.Error(t, err) - assert.Contains(t, err.Error(), "failed to resolve secret") - assert.Contains(t, err.Error(), "TOKEN") - assert.NotContains(t, err.Error(), "db-password-super-secret") - }) - - t.Run("secret resolution error omits provider error details", func(t *testing.T) { - log := zerowrap.New(zerowrap.Config{Level: "fatal"}) - loader, err := NewFileLoader(t.TempDir(), log) - require.NoError(t, err) - - mockPass := mocks.NewMockSecretProvider(t) - mockPass.EXPECT().Name().Return("pass") - mockPass.EXPECT().GetSecret(ctx, "db-password-super-secret").Return("", errors.New("backend leaked db-password-super-secret")) - loader.RegisterSecretProvider(mockPass) - - envFile, err := loader.getEnvFilePath("app.example.com") - require.NoError(t, err) - require.NoError(t, os.WriteFile(envFile, []byte("TOKEN=${pass:db-password-super-secret}\n"), 0600)) - - _, err = loader.LoadEnv(ctx, "app.example.com") - require.Error(t, err) - assert.Contains(t, err.Error(), "failed to resolve secret") - assert.Contains(t, err.Error(), "TOKEN") - assert.Contains(t, err.Error(), "provider pass") - assert.NotContains(t, err.Error(), "db-password-super-secret") - }) -} - -func TestFileLoader_GetEnvFilePath(t *testing.T) { - tmpDir := t.TempDir() - log := zerowrap.New(zerowrap.Config{Level: "fatal"}) - - loader, err := NewFileLoader(tmpDir, log) - require.NoError(t, err) - - tests := []struct { - domain string - expected string - }{ - {"app.example.com", "YXBwLmV4YW1wbGUuY29t.env"}, - {"api.sub.example.com", "YXBpLnN1Yi5leGFtcGxlLmNvbQ.env"}, - {"localhost:8080", "bG9jYWxob3N0OjgwODA.env"}, - } - - for _, tt := range tests { - t.Run(tt.domain, func(t *testing.T) { - result, err := loader.getEnvFilePath(tt.domain) - require.NoError(t, err) - assert.Equal(t, filepath.Join(tmpDir, tt.expected), result) - }) - } -} - -func TestFileLoader_GetEnvFilePath_DistinguishesSeparators(t *testing.T) { - tmpDir := t.TempDir() - log := zerowrap.New(zerowrap.Config{Level: "fatal"}) - - loader, err := NewFileLoader(tmpDir, log) - require.NoError(t, err) - - dotPath, err := loader.getEnvFilePath("app.example.com") - require.NoError(t, err) - portPath, err := loader.getEnvFilePath("app:example:com") - require.NoError(t, err) - slashPath, err := loader.getEnvFilePath("app/example/com") - require.NoError(t, err) - - assert.NotEqual(t, dotPath, portPath) - assert.NotEqual(t, dotPath, slashPath) - assert.NotEqual(t, portPath, slashPath) -} - -func TestFileLoader_GetEnvFilePath_RejectsInvalidDomain(t *testing.T) { - tmpDir := t.TempDir() - log := zerowrap.New(zerowrap.Config{Level: "fatal"}) - - loader, err := NewFileLoader(tmpDir, log) - require.NoError(t, err) - - _, err = loader.getEnvFilePath("../etc/passwd") - require.ErrorIs(t, err, domain.ErrPathTraversal) -} - -func TestFileLoader_EnvFileExists(t *testing.T) { - log := zerowrap.New(zerowrap.Config{Level: "fatal"}) - - t.Run("returns false nil when file is missing", func(t *testing.T) { - tmpDir := t.TempDir() - loader, err := NewFileLoader(tmpDir, log) - require.NoError(t, err) - - exists, err := loader.EnvFileExists("app.example.com") - require.NoError(t, err) - assert.False(t, exists) - }) - - t.Run("returns true nil when file exists", func(t *testing.T) { - tmpDir := t.TempDir() - loader, err := NewFileLoader(tmpDir, log) - require.NoError(t, err) - - storageKey, err := domain.NewEnvStorageKey("app.example.com") - require.NoError(t, err) - envFile := filepath.Join(tmpDir, storageKey.FileName()) - require.NoError(t, os.WriteFile(envFile, []byte("KEY=value\n"), 0600)) - - exists, err := loader.EnvFileExists("app.example.com") - require.NoError(t, err) - assert.True(t, exists) - }) -} diff --git a/internal/adapters/out/envloader/pass_loader.go b/internal/adapters/out/envloader/pass_loader.go deleted file mode 100644 index 9b4880d6b..000000000 --- a/internal/adapters/out/envloader/pass_loader.go +++ /dev/null @@ -1,117 +0,0 @@ -// Package envloader implements the environment variable loader adapter. -package envloader - -import ( - "context" - "fmt" - "sort" - "strings" - - "github.com/bnema/zerowrap" - - "github.com/bnema/gordon/internal/adapters/out/domainsecrets" -) - -// PassLoader implements the EnvLoader interface using pass-backed secrets. -type PassLoader struct { - store *domainsecrets.PassStore - log zerowrap.Logger -} - -// NewPassLoader creates a new pass-based environment loader. -func NewPassLoader(store *domainsecrets.PassStore, log zerowrap.Logger) (*PassLoader, error) { - if store == nil { - return nil, fmt.Errorf("pass store is required") - } - - log.Debug(). - Str(zerowrap.FieldLayer, "adapter"). - Str(zerowrap.FieldAdapter, "envloader"). - Str("provider", "pass"). - Msg("env loader initialized") - - return &PassLoader{ - store: store, - log: log, - }, nil -} - -// LoadEnv loads environment variables for a given domain. -func (l *PassLoader) LoadEnv(ctx context.Context, domain string) ([]string, error) { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "adapter", - zerowrap.FieldAdapter: "envloader", - zerowrap.FieldAction: "LoadEnv", - "domain": domain, - }) - log := zerowrap.FromCtx(ctx) - - var secretsMap map[string]string - var err error - - if isAttachmentContainer(domain) { - secretsMap, err = l.store.GetAllAttachment(domain) - if err != nil { - return nil, log.WrapErr(err, "failed to load attachment env from pass") - } - } else { - secretsMap, err = l.store.GetAll(domain) - if err != nil { - return nil, log.WrapErr(err, "failed to load env from pass") - } - } - - keys := make([]string, 0, len(secretsMap)) - for key := range secretsMap { - keys = append(keys, key) - } - sort.Strings(keys) - - envVars := make([]string, 0, len(keys)) - for _, key := range keys { - envVars = append(envVars, fmt.Sprintf("%s=%s", key, secretsMap[key])) - } - - log.Info().Int(zerowrap.FieldCount, len(envVars)).Msg("loaded environment variables for route") - - return envVars, nil -} - -// isAttachmentContainer determines whether a given name refers to an attachment container -// or a domain. It uses a simple heuristic: attachment container names always start with -// the "gordon-" prefix (e.g., "gordon-git-example-com-gitea-postgres"). -// -// LIMITATION: This heuristic assumes domains never start with "gordon-". A domain like -// "gordon-app.example.com" would be incorrectly classified as an attachment container. -// This is an acceptable trade-off for simplicity, but if such domains become common, -// this should be replaced with a more robust detection mechanism (e.g., checking both -// GetAll() and GetAllAttachment() and seeing which returns results). -func isAttachmentContainer(name string) bool { - // Attachment container names start with "gordon-" and never contain dots. - // Domain names (including preview domains like "gordon--test.bnema.dev") - // always contain dots, so checking for dots distinguishes them. - return strings.HasPrefix(name, "gordon-") && !strings.Contains(name, ".") -} - -// CreateEnvFile is a no-op for pass-backed loader. -func (l *PassLoader) CreateEnvFile(ctx context.Context, domain string) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "adapter", - zerowrap.FieldAdapter: "envloader", - zerowrap.FieldAction: "CreateEnvFile", - "domain": domain, - }) - log := zerowrap.FromCtx(ctx) - log.Debug().Msg("pass loader does not create env files") - return nil -} - -// EnvFileExists checks if a manifest exists for a domain in pass. -func (l *PassLoader) EnvFileExists(domain string) (bool, error) { - exists, err := l.store.ManifestExists(domain) - if err != nil { - l.log.Warn().Err(err).Str("domain", domain).Msg("failed to check pass manifest") - return false, err - } - return exists, nil -} diff --git a/internal/adapters/out/envloader/pass_loader_test.go b/internal/adapters/out/envloader/pass_loader_test.go deleted file mode 100644 index 059a84687..000000000 --- a/internal/adapters/out/envloader/pass_loader_test.go +++ /dev/null @@ -1,108 +0,0 @@ -package envloader - -import ( - "context" - "fmt" - "os/exec" - "testing" - "time" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" - - "github.com/bnema/gordon/internal/adapters/out/domainsecrets" - "github.com/bnema/zerowrap" -) - -func passCmd(args ...string) error { - ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second) - defer cancel() - return exec.CommandContext(ctx, "pass", args...).Run() -} - -func requirePass(t *testing.T) { - if err := passCmd("version"); err != nil { - t.Skip("pass not available") - } - if err := passCmd("ls"); err != nil { - t.Skip("pass store not initialized") - } -} - -func cleanupPassEnv(t *testing.T, path string) { - _ = passCmd("rm", "-rf", path) -} - -func TestPassLoader_LoadEnv_Domain(t *testing.T) { - requirePass(t) - - log := zerowrap.New(zerowrap.Config{Level: "fatal"}) - store, err := domainsecrets.NewPassStore(log) - require.NoError(t, err) - - loader, err := NewPassLoader(store, log) - require.NoError(t, err) - - domainName := fmt.Sprintf("test.%d.example.com", time.Now().UnixNano()) - defer cleanupPassEnv(t, fmt.Sprintf("%s/%s", domainsecrets.PassDomainSecretsPath, domainName)) - - secretsMap := map[string]string{ - "API_KEY": "test123", - "DB_HOST": "localhost", - } - err = store.Set(domainName, secretsMap) - require.NoError(t, err) - - ctx := context.Background() - envVars, err := loader.LoadEnv(ctx, domainName) - require.NoError(t, err) - assert.Len(t, envVars, 2) - assert.Contains(t, envVars, "API_KEY=test123") - assert.Contains(t, envVars, "DB_HOST=localhost") -} - -func TestPassLoader_LoadEnv_Attachment(t *testing.T) { - requirePass(t) - - log := zerowrap.New(zerowrap.Config{Level: "fatal"}) - store, err := domainsecrets.NewPassStore(log) - require.NoError(t, err) - - loader, err := NewPassLoader(store, log) - require.NoError(t, err) - - containerName := fmt.Sprintf("gordon-gitea-postgres-%d", time.Now().UnixNano()) - defer cleanupPassEnv(t, fmt.Sprintf("%s/%s", domainsecrets.PassAttachmentPath, containerName)) - - secretsMap := map[string]string{ - "POSTGRES_USER": "gitea", - "POSTGRES_PASSWORD": "secret123", - } - err = store.SetAttachment(containerName, secretsMap) - require.NoError(t, err) - - ctx := context.Background() - envVars, err := loader.LoadEnv(ctx, containerName) - require.NoError(t, err) - assert.Len(t, envVars, 2) - assert.Contains(t, envVars, "POSTGRES_USER=gitea") - assert.Contains(t, envVars, "POSTGRES_PASSWORD=secret123") -} - -func TestPassLoader_LoadEnv_EmptyDomain(t *testing.T) { - requirePass(t) - - log := zerowrap.New(zerowrap.Config{Level: "fatal"}) - store, err := domainsecrets.NewPassStore(log) - require.NoError(t, err) - - loader, err := NewPassLoader(store, log) - require.NoError(t, err) - - domainName := fmt.Sprintf("empty.%d.example.com", time.Now().UnixNano()) - - ctx := context.Background() - envVars, err := loader.LoadEnv(ctx, domainName) - require.NoError(t, err) - assert.Len(t, envVars, 0) -} diff --git a/internal/adapters/out/eventbus/inmemory.go b/internal/adapters/out/eventbus/inmemory.go index 1f12aaef4..1482ccd80 100644 --- a/internal/adapters/out/eventbus/inmemory.go +++ b/internal/adapters/out/eventbus/inmemory.go @@ -9,10 +9,7 @@ import ( "github.com/bnema/zerowrap" "github.com/google/uuid" - "go.opentelemetry.io/otel/attribute" - "go.opentelemetry.io/otel/metric" - "github.com/bnema/gordon/internal/adapters/out/telemetry" "github.com/bnema/gordon/internal/boundaries/out" "github.com/bnema/gordon/internal/domain" ) @@ -27,12 +24,12 @@ type InMemory struct { cancel context.CancelFunc bufferSize int log zerowrap.Logger - metrics *telemetry.Metrics + metrics out.Metrics } // SetMetrics sets the telemetry metrics for the event bus. // Must be called before Start() to avoid data races on bus.metrics reads. -func (bus *InMemory) SetMetrics(m *telemetry.Metrics) { +func (bus *InMemory) SetMetrics(m out.Metrics) { bus.mu.Lock() bus.metrics = m bus.mu.Unlock() @@ -98,9 +95,7 @@ func (bus *InMemory) Publish(eventType domain.EventType, payload any) error { // Record dropped event metric if bus.metrics != nil { - bus.metrics.EventsDropped.Add(context.Background(), 1, metric.WithAttributes( - attribute.String("event_type", string(event.Type)), - )) + bus.metrics.RecordEventDropped(context.Background(), event.Type) } return fmt.Errorf("event channel is full, dropping event %s", event.ID) } @@ -239,9 +234,7 @@ func (bus *InMemory) handleEvent(event domain.Event) { // Record processed event metric if bus.metrics != nil { - bus.metrics.EventsProcessed.Add(context.Background(), 1, metric.WithAttributes( - attribute.String("event_type", string(event.Type)), - )) + bus.metrics.RecordEventProcessed(context.Background(), event.Type) } } case <-ctx.Done(): diff --git a/internal/adapters/out/filesystem/backup_storage.go b/internal/adapters/out/filesystem/backup_storage.go index 0ae5dd7ed..36d36569e 100644 --- a/internal/adapters/out/filesystem/backup_storage.go +++ b/internal/adapters/out/filesystem/backup_storage.go @@ -50,16 +50,17 @@ func expandTilde(path string) string { } // Store saves backup data and returns the absolute storage path. -func (s *BackupStorage) Store(_ context.Context, domainName, dbName string, schedule domain.BackupSchedule, timestamp time.Time, data io.Reader) (string, error) { - domainPart := sanitizeBackupPathComponent(domainName) - dbPart := sanitizeBackupPathComponent(dbName) +func (s *BackupStorage) Store(_ context.Context, app, service, database string, schedule domain.BackupSchedule, timestamp time.Time, data io.Reader) (string, error) { + appPart := sanitizeBackupPathComponent(app) + servicePart := sanitizeBackupPathComponent(service) + dbPart := sanitizeBackupPathComponent(database) schedulePart := string(schedule) if schedulePart == "" { schedulePart = "manual" } schedulePart = sanitizeBackupPathComponent(schedulePart) - backupDir := filepath.Join(s.rootDir, domainPart, dbPart, schedulePart) + backupDir := filepath.Join(s.rootDir, appPart, servicePart, dbPart, schedulePart) if err := os.MkdirAll(backupDir, 0750); err != nil { return "", fmt.Errorf("failed to create backup path: %w", err) } @@ -107,9 +108,9 @@ func (s *BackupStorage) Get(_ context.Context, path string) (io.ReadCloser, erro } // List returns backups for a domain, optionally filtered by schedule. -func (s *BackupStorage) List(_ context.Context, domainName string, schedule *domain.BackupSchedule) ([]domain.BackupJob, error) { - domainRoot := filepath.Join(s.rootDir, sanitizeBackupPathComponent(domainName)) - if _, err := os.Stat(domainRoot); err != nil { +func (s *BackupStorage) List(_ context.Context, app string, schedule *domain.BackupSchedule) ([]domain.BackupJob, error) { + appRoot := filepath.Join(s.rootDir, sanitizeBackupPathComponent(app)) + if _, err := os.Stat(appRoot); err != nil { if os.IsNotExist(err) { return []domain.BackupJob{}, nil } @@ -117,7 +118,7 @@ func (s *BackupStorage) List(_ context.Context, domainName string, schedule *dom } jobs := make([]domain.BackupJob, 0) - err := filepath.WalkDir(domainRoot, func(path string, d os.DirEntry, walkErr error) error { + err := filepath.WalkDir(appRoot, func(path string, d os.DirEntry, walkErr error) error { if walkErr != nil { return walkErr } @@ -125,17 +126,21 @@ func (s *BackupStorage) List(_ context.Context, domainName string, schedule *dom return nil } - rel, err := filepath.Rel(domainRoot, path) + rel, err := filepath.Rel(appRoot, path) if err != nil { return nil } parts := strings.Split(rel, string(filepath.Separator)) - if len(parts) != 3 { + if len(parts) != 4 { + // Pre-cutover app/database/schedule records do not contain a + // canonical service identity, so they cannot safely join a + // current app/service/database history. return nil } - dbName := parts[0] - sched := domain.BackupSchedule(parts[1]) + service := parts[0] + dbName := parts[1] + sched := domain.BackupSchedule(parts[2]) if schedule != nil && *schedule != sched { return nil } @@ -153,7 +158,8 @@ func (s *BackupStorage) List(_ context.Context, domainName string, schedule *dom jobs = append(jobs, domain.BackupJob{ ID: base, - Domain: domainName, + App: app, + Service: service, DBName: dbName, Schedule: sched, Type: domain.BackupTypeLogical, @@ -185,15 +191,15 @@ func (s *BackupStorage) Delete(_ context.Context, path string) error { } // ApplyRetention removes old backups according to the schedule policy. -func (s *BackupStorage) ApplyRetention(ctx context.Context, domainName string, policy domain.RetentionPolicy) (int, error) { - jobs, err := s.List(ctx, domainName, nil) +func (s *BackupStorage) ApplyRetention(ctx context.Context, app string, policy domain.RetentionPolicy) (int, error) { + jobs, err := s.List(ctx, app, nil) if err != nil { return 0, err } groups := make(map[string][]domain.BackupJob) for _, job := range jobs { - key := fmt.Sprintf("%s|%s", job.DBName, job.Schedule) + key := fmt.Sprintf("%s|%s|%s", job.Service, job.DBName, job.Schedule) groups[key] = append(groups[key], job) } @@ -215,7 +221,7 @@ func (s *BackupStorage) ApplyRetention(ctx context.Context, domainName string, p } return deleted, err } - if err := s.removeEmptyBackupDirs(group[idx].FilePath, domainName); err != nil { + if err := s.removeEmptyBackupDirs(group[idx].FilePath, app); err != nil { return deleted, err } deleted++ diff --git a/internal/adapters/out/filesystem/backup_storage_test.go b/internal/adapters/out/filesystem/backup_storage_test.go index e91d70531..4f1116d50 100644 --- a/internal/adapters/out/filesystem/backup_storage_test.go +++ b/internal/adapters/out/filesystem/backup_storage_test.go @@ -4,6 +4,7 @@ import ( "bytes" "context" "io" + "os" "path/filepath" "testing" "time" @@ -19,7 +20,7 @@ func TestBackupStorage_StoreAndGet(t *testing.T) { now := time.Date(2026, 2, 7, 11, 0, 0, 0, time.UTC) payload := []byte("backup-content") - path, err := storage.Store(context.Background(), "app.example.com", "postgres", domain.ScheduleDaily, now, bytes.NewReader(payload)) + path, err := storage.Store(context.Background(), "app.example.com", "postgres", "postgres", domain.ScheduleDaily, now, bytes.NewReader(payload)) require.NoError(t, err) rc, err := storage.Get(context.Background(), path) @@ -40,9 +41,9 @@ func TestBackupStorage_StoreSameSecondBackupsUseDistinctFinalAndTempPaths(t *tes firstPayload := []byte("first-backup") secondPayload := []byte("second-backup") - firstPath, err := storage.Store(context.Background(), "app.example.com", "postgres", domain.ScheduleDaily, now, bytes.NewReader(firstPayload)) + firstPath, err := storage.Store(context.Background(), "app.example.com", "postgres", "postgres", domain.ScheduleDaily, now, bytes.NewReader(firstPayload)) require.NoError(t, err) - secondPath, err := storage.Store(context.Background(), "app.example.com", "postgres", domain.ScheduleDaily, now, bytes.NewReader(secondPayload)) + secondPath, err := storage.Store(context.Background(), "app.example.com", "postgres", "postgres", domain.ScheduleDaily, now, bytes.NewReader(secondPayload)) require.NoError(t, err) assert.NotEqual(t, firstPath, secondPath) @@ -64,16 +65,54 @@ func TestBackupStorage_StoreSameSecondBackupsUseDistinctFinalAndTempPaths(t *tes assert.Equal(t, secondPayload, secondData) } +func TestBackupStorage_PreservesCanonicalIdentityAndSeparatesSameNamedDatabases(t *testing.T) { + storage, err := NewBackupStorage(t.TempDir(), testLogger()) + require.NoError(t, err) + + started := time.Date(2026, 9, 12, 12, 0, 0, 0, time.UTC) + first, err := storage.Store(context.Background(), "shop", "api", "main", domain.ScheduleDaily, started, bytes.NewReader([]byte("api"))) + require.NoError(t, err) + second, err := storage.Store(context.Background(), "shop", "worker", "main", domain.ScheduleDaily, started, bytes.NewReader([]byte("worker"))) + require.NoError(t, err) + assert.NotEqual(t, first, second) + + jobs, err := storage.List(context.Background(), "shop", nil) + require.NoError(t, err) + require.Len(t, jobs, 2) + assert.ElementsMatch(t, []string{"api", "worker"}, []string{jobs[0].Service, jobs[1].Service}) + for _, job := range jobs { + assert.Equal(t, "shop", job.App) + assert.Equal(t, "main", job.DBName) + } +} + +func TestBackupStorage_DoesNotMergePreCutoverRecords(t *testing.T) { + root := t.TempDir() + storage, err := NewBackupStorage(root, testLogger()) + require.NoError(t, err) + + legacyDir := filepath.Join(root, "shop", "main", "daily") + require.NoError(t, os.MkdirAll(legacyDir, 0750)) + require.NoError(t, os.WriteFile(filepath.Join(legacyDir, "20260912T120000Z.bak"), []byte("legacy"), 0600)) + _, err = storage.Store(context.Background(), "shop", "api", "main", domain.ScheduleDaily, time.Now(), bytes.NewReader([]byte("current"))) + require.NoError(t, err) + + jobs, err := storage.List(context.Background(), "shop", nil) + require.NoError(t, err) + require.Len(t, jobs, 1) + assert.Equal(t, "api", jobs[0].Service) +} + func TestBackupStorage_ListWithScheduleFilter(t *testing.T) { storage, err := NewBackupStorage(t.TempDir(), testLogger()) require.NoError(t, err) base := time.Date(2026, 2, 7, 10, 0, 0, 0, time.UTC) - _, err = storage.Store(context.Background(), "app.example.com", "postgres", domain.ScheduleDaily, base, bytes.NewReader([]byte("d1"))) + _, err = storage.Store(context.Background(), "app.example.com", "postgres", "postgres", domain.ScheduleDaily, base, bytes.NewReader([]byte("d1"))) require.NoError(t, err) - _, err = storage.Store(context.Background(), "app.example.com", "postgres", domain.ScheduleDaily, base.Add(time.Hour), bytes.NewReader([]byte("d2"))) + _, err = storage.Store(context.Background(), "app.example.com", "postgres", "postgres", domain.ScheduleDaily, base.Add(time.Hour), bytes.NewReader([]byte("d2"))) require.NoError(t, err) - _, err = storage.Store(context.Background(), "app.example.com", "postgres", domain.ScheduleWeekly, base.Add(2*time.Hour), bytes.NewReader([]byte("w1"))) + _, err = storage.Store(context.Background(), "app.example.com", "postgres", "postgres", domain.ScheduleWeekly, base.Add(2*time.Hour), bytes.NewReader([]byte("w1"))) require.NoError(t, err) schedule := domain.ScheduleDaily @@ -89,7 +128,7 @@ func TestBackupStorage_Delete(t *testing.T) { storage, err := NewBackupStorage(t.TempDir(), testLogger()) require.NoError(t, err) - path, err := storage.Store(context.Background(), "app.example.com", "postgres", domain.ScheduleDaily, time.Now().UTC(), bytes.NewReader([]byte("data"))) + path, err := storage.Store(context.Background(), "app.example.com", "postgres", "postgres", domain.ScheduleDaily, time.Now().UTC(), bytes.NewReader([]byte("data"))) require.NoError(t, err) err = storage.Delete(context.Background(), path) @@ -105,7 +144,7 @@ func TestBackupStorage_ApplyRetention(t *testing.T) { base := time.Date(2026, 2, 7, 6, 0, 0, 0, time.UTC) for i := range 4 { - _, err := storage.Store(context.Background(), "app.example.com", "postgres", domain.ScheduleDaily, base.Add(time.Duration(i)*time.Hour), bytes.NewReader([]byte("data"))) + _, err := storage.Store(context.Background(), "app.example.com", "postgres", "postgres", domain.ScheduleDaily, base.Add(time.Duration(i)*time.Hour), bytes.NewReader([]byte("data"))) require.NoError(t, err) } @@ -123,7 +162,7 @@ func TestBackupStorage_StoreSanitizesDotOnlyPathComponents(t *testing.T) { storage, err := NewBackupStorage(t.TempDir(), testLogger()) require.NoError(t, err) - path, err := storage.Store(context.Background(), "..", "...", domain.ScheduleDaily, time.Now().UTC(), bytes.NewReader([]byte("data"))) + path, err := storage.Store(context.Background(), "..", "service", "...", domain.ScheduleDaily, time.Now().UTC(), bytes.NewReader([]byte("data"))) require.NoError(t, err) assert.NotContains(t, path, "..") } @@ -133,7 +172,7 @@ func TestBackupStorage_StoreSanitizesSchedulePathComponent(t *testing.T) { storage, err := NewBackupStorage(rootDir, testLogger()) require.NoError(t, err) - path, err := storage.Store(context.Background(), "app.example.com", "postgres", domain.BackupSchedule("../../escape"), time.Now().UTC(), bytes.NewReader([]byte("data"))) + path, err := storage.Store(context.Background(), "app.example.com", "postgres", "postgres", domain.BackupSchedule("../../escape"), time.Now().UTC(), bytes.NewReader([]byte("data"))) require.NoError(t, err) rel, err := filepath.Rel(rootDir, path) diff --git a/internal/adapters/out/filesystem/blob.go b/internal/adapters/out/filesystem/blob.go index 0a0a811de..77c2e3a2a 100644 --- a/internal/adapters/out/filesystem/blob.go +++ b/internal/adapters/out/filesystem/blob.go @@ -34,6 +34,7 @@ func NewBlobStorage(rootDir string, log zerowrap.Logger) (*BlobStorage, error) { dirs := []string{ filepath.Join(rootDir, "blobs"), filepath.Join(rootDir, "uploads"), + filepath.Join(rootDir, "ownership"), } for _, dir := range dirs { @@ -261,6 +262,14 @@ func (s *BlobStorage) StartBlobUpload(name string) (string, error) { } file.Close() + // Bind the upload to the repository that started it. Every later + // operation on this UUID must present the same repository, so a known + // upload UUID is not a bearer token for another repository. + if err := os.WriteFile(uploadRepoPath(uploadPath), []byte(name), 0600); err != nil { + _ = os.Remove(uploadPath) + return "", fmt.Errorf("failed to record upload repository: %w", err) + } + s.log.Info(). Str(zerowrap.FieldLayer, "adapter"). Str(zerowrap.FieldAdapter, "filesystem"). @@ -271,12 +280,36 @@ func (s *BlobStorage) StartBlobUpload(name string) (string, error) { return uploadID, nil } +// verifyUploadRepository reports the repository that started an upload and +// fails closed when it does not match the requesting repository. +func (s *BlobStorage) verifyUploadRepository(uploadPath, name string) error { + raw, err := os.ReadFile(uploadRepoPath(uploadPath)) + if err != nil { + if os.IsNotExist(err) { + return fmt.Errorf("%w: %s", domain.ErrUploadNotFound, filepath.Base(uploadPath)) + } + return fmt.Errorf("read upload repository: %w", err) + } + if string(raw) != name { + return fmt.Errorf("%w: %s", domain.ErrUploadNotFound, filepath.Base(uploadPath)) + } + return nil +} + +// uploadRepoPath returns the sidecar recording an upload's repository. +func uploadRepoPath(uploadPath string) string { + return uploadPath + ".repo" +} + // AppendBlobChunk appends data to an in-progress upload. func (s *BlobStorage) AppendBlobChunk(name, uuid string, data io.Reader, contentLength, maxBlobSize int64) (int64, error) { uploadPath, err := s.getUploadPath(uuid) if err != nil { return 0, fmt.Errorf("invalid upload path: %w", err) } + if err := s.verifyUploadRepository(uploadPath, name); err != nil { + return 0, err + } mu := s.getUploadLock(uuid) mu.Lock() @@ -379,12 +412,18 @@ func (lw *lockedWriteCloser) Close() error { return lw.closeErr } -// FinishBlobUpload completes an upload and moves it to blob storage. -func (s *BlobStorage) FinishBlobUpload(uuid, digest string) error { +// FinishBlobUpload completes an upload for the named repository, moves it +// to blob storage, and records the repository/blob association. The +// association is written only after the content-addressed blob exists and +// only for the repository that started the upload. +func (s *BlobStorage) FinishBlobUpload(name, uuid, digest string) error { uploadPath, err := s.getUploadPath(uuid) if err != nil { return fmt.Errorf("invalid upload path: %w", err) } + if err := s.verifyUploadRepository(uploadPath, name); err != nil { + return err + } mu := s.getUploadLock(uuid) mu.Lock() @@ -423,23 +462,33 @@ func (s *BlobStorage) FinishBlobUpload(uuid, digest string) error { return fmt.Errorf("failed to move upload to blob location: %w", err) } finalized = true + _ = os.Remove(uploadRepoPath(uploadPath)) + + if err := s.recordBlobOwnership(name, digest); err != nil { + return err + } s.log.Info(). Str(zerowrap.FieldLayer, "adapter"). Str(zerowrap.FieldAdapter, "filesystem"). Str("uuid", uuid). + Str("name", name). Str("digest", digest). Msg("blob upload finished") return nil } -// CancelBlobUpload cancels an in-progress upload. -func (s *BlobStorage) CancelBlobUpload(uuid string) error { +// CancelBlobUpload cancels an in-progress upload owned by the named +// repository. +func (s *BlobStorage) CancelBlobUpload(name, uuid string) error { uploadPath, err := s.getUploadPath(uuid) if err != nil { return fmt.Errorf("invalid upload path: %w", err) } + if err := s.verifyUploadRepository(uploadPath, name); err != nil { + return err + } mu := s.getUploadLock(uuid) mu.Lock() @@ -454,21 +503,59 @@ func (s *BlobStorage) CancelBlobUpload(uuid string) error { if err := os.Remove(uploadPath); err != nil { if os.IsNotExist(err) { s.cleanupUploadLock(uuid) - return fmt.Errorf("upload not found: %s", uuid) + return fmt.Errorf("%w: %s", domain.ErrUploadNotFound, uuid) } return fmt.Errorf("failed to cancel upload: %w", err) } + _ = os.Remove(uploadRepoPath(uploadPath)) finalized = true s.log.Info(). Str(zerowrap.FieldLayer, "adapter"). Str(zerowrap.FieldAdapter, "filesystem"). Str("uuid", uuid). + Str("name", name). Msg("blob upload cancelled") return nil } +// BlobOwnedByRepository reports whether the repository completed an upload +// of the digest. Ownership comes only from that authenticated completion; +// a manifest that merely names a digest is not evidence of ownership. +func (s *BlobStorage) BlobOwnedByRepository(name, digest string) (bool, error) { + if name == "" || digest == "" { + return false, nil + } + path, err := s.getOwnershipPath(name, digest) + if err != nil { + return false, fmt.Errorf("invalid ownership path: %w", err) + } + if _, err := os.Stat(path); err != nil { + if os.IsNotExist(err) { + return false, nil + } + return false, fmt.Errorf("stat blob ownership: %w", err) + } + return true, nil +} + +// recordBlobOwnership persists the repository/blob association as a durable +// marker file. +func (s *BlobStorage) recordBlobOwnership(name, digest string) error { + path, err := s.getOwnershipPath(name, digest) + if err != nil { + return fmt.Errorf("invalid ownership path: %w", err) + } + if err := os.MkdirAll(filepath.Dir(path), 0750); err != nil { + return fmt.Errorf("failed to create ownership directory: %w", err) + } + if err := os.WriteFile(path, []byte(name), 0600); err != nil { + return fmt.Errorf("failed to record blob ownership: %w", err) + } + return nil +} + // CleanupStaleUploads removes upload files older than maxAge. func (s *BlobStorage) CleanupStaleUploads(maxAge time.Duration) (int, int64, error) { uploadsDir := filepath.Join(s.rootDir, "uploads") @@ -489,6 +576,15 @@ func (s *BlobStorage) CleanupStaleUploads(maxAge time.Duration) (int, int64, err if entry.IsDir() { continue } + if strings.HasSuffix(entry.Name(), ".repo") { + uploadPath := filepath.Join(uploadsDir, strings.TrimSuffix(entry.Name(), ".repo")) + if _, err := os.Stat(uploadPath); os.IsNotExist(err) { + if err := os.Remove(filepath.Join(uploadsDir, entry.Name())); err != nil && !os.IsNotExist(err) { + s.log.Warn().Err(err).Str("file", entry.Name()).Msg("failed to remove orphan upload sidecar") + } + } + continue + } info, err := entry.Info() if err != nil { @@ -519,6 +615,7 @@ func (s *BlobStorage) CleanupStaleUploads(maxAge time.Duration) (int, int64, err s.log.Warn().Err(err).Str("file", entry.Name()).Msg("failed to remove stale upload") continue } + _ = os.Remove(uploadRepoPath(path)) mu.Unlock() s.cleanupUploadLock(entry.Name()) @@ -586,6 +683,25 @@ func (s *BlobStorage) getUploadPath(uuid string) (string, error) { return path, nil } +// getOwnershipPath returns the durable marker path for a repository/blob +// association. The repository is hashed so its name never becomes a path +// component. +func (s *BlobStorage) getOwnershipPath(name, digest string) (string, error) { + if err := validation.ValidateRepositoryName(name); err != nil { + return "", fmt.Errorf("invalid repository name: %w", err) + } + if err := validation.ValidateDigest(digest); err != nil { + return "", fmt.Errorf("invalid digest: %w", err) + } + sum := sha256.Sum256([]byte(name)) + parts := strings.SplitN(digest, ":", 2) + path := filepath.Join(s.rootDir, "ownership", hex.EncodeToString(sum[:]), parts[0], parts[1]) + if err := validation.ValidatePathWithinRoot(s.rootDir, path); err != nil { + return "", fmt.Errorf("path validation failed: %w", err) + } + return path, nil +} + func (s *BlobStorage) getUploadLock(uuid string) *sync.Mutex { v, _ := s.uploadLocks.LoadOrStore(uuid, &sync.Mutex{}) return v.(*sync.Mutex) diff --git a/internal/adapters/out/filesystem/blob_test.go b/internal/adapters/out/filesystem/blob_test.go index 641ccb50f..383f4d7d6 100644 --- a/internal/adapters/out/filesystem/blob_test.go +++ b/internal/adapters/out/filesystem/blob_test.go @@ -439,7 +439,7 @@ func TestBlobStorage_FinishBlobUpload(t *testing.T) { // Finish upload - use testDigest1 (validation will pass, but hash won't match) // This tests the flow, actual digest verification happens in verifyUploadDigest - err = storage.FinishBlobUpload(uuid, testDigest1) + err = storage.FinishBlobUpload("myapp", uuid, testDigest1) // This will fail because content doesn't match digest assert.Error(t, err) assert.Contains(t, err.Error(), "digest") @@ -462,7 +462,7 @@ func TestBlobStorage_CancelBlobUpload(t *testing.T) { require.NoError(t, err) // Cancel upload - err = storage.CancelBlobUpload(uuid) + err = storage.CancelBlobUpload("myapp", uuid) require.NoError(t, err) // Verify upload file is gone @@ -477,7 +477,7 @@ func TestBlobStorage_CancelBlobUpload_NotFound(t *testing.T) { storage, err := NewBlobStorage(tmpDir, log) require.NoError(t, err) - err = storage.CancelBlobUpload("00000000-0000-4000-a000-000000000000") + err = storage.CancelBlobUpload("myapp", "00000000-0000-4000-a000-000000000000") assert.Error(t, err) assert.Contains(t, err.Error(), "upload not found") @@ -570,7 +570,7 @@ func TestBlobStorage_CompleteUploadFlow(t *testing.T) { assert.Equal(t, expectedTotal, totalSize) // 3. Cancel upload (since we can't easily compute the real digest) - err = storage.CancelBlobUpload(uuid) + err = storage.CancelBlobUpload(repoName, uuid) require.NoError(t, err) } @@ -611,7 +611,7 @@ func TestBlobStorage_ConcurrentAppendAndFinalize(t *testing.T) { wg.Add(1) go func() { defer wg.Done() - finalizeErrs <- storage.FinishBlobUpload(uuid, digestForData(fullyAppended)) + finalizeErrs <- storage.FinishBlobUpload("myapp", uuid, digestForData(fullyAppended)) }() wg.Wait() @@ -642,7 +642,7 @@ func TestBlobStorage_ConcurrentAppendAndFinalize(t *testing.T) { require.Equal(t, appenders, appendCount) retryDigest := digestForData(fullyAppended) - err = storage.FinishBlobUpload(uuid, retryDigest) + err = storage.FinishBlobUpload("myapp", uuid, retryDigest) require.NoError(t, err) reader, err := storage.GetBlob(retryDigest) @@ -673,7 +673,7 @@ func TestBlobStorage_AppendAfterFinalize(t *testing.T) { _, err = storage.AppendBlobChunk("myapp", uuid, bytes.NewReader(blobData), -1, 0) require.NoError(t, err) - err = storage.FinishBlobUpload(uuid, digestForData(blobData)) + err = storage.FinishBlobUpload("myapp", uuid, digestForData(blobData)) require.NoError(t, err) _, err = storage.AppendBlobChunk("myapp", uuid, bytes.NewReader([]byte("more")), -1, 0) @@ -696,7 +696,7 @@ func TestBlobStorage_FinishBlobUploadMarksBlobAsRecentlyFinalized(t *testing.T) old := time.Now().Add(-2 * 24 * time.Hour) require.NoError(t, os.Chtimes(uploadPath, old, old)) - require.NoError(t, storage.FinishBlobUpload(uuid, digestForData(blobData))) + require.NoError(t, storage.FinishBlobUpload("myapp", uuid, digestForData(blobData))) modTime, err := storage.GetBlobModTime(digestForData(blobData)) require.NoError(t, err) assert.WithinDuration(t, time.Now(), modTime, time.Minute) @@ -716,11 +716,11 @@ func TestBlobStorage_FailedFinalizeRetryable(t *testing.T) { _, err = storage.AppendBlobChunk("myapp", uuid, bytes.NewReader(blobData), -1, 0) require.NoError(t, err) - err = storage.FinishBlobUpload(uuid, testDigest1) + err = storage.FinishBlobUpload("myapp", uuid, testDigest1) assert.Error(t, err) assert.Contains(t, err.Error(), "digest verification failed") - err = storage.FinishBlobUpload(uuid, digestForData(blobData)) + err = storage.FinishBlobUpload("myapp", uuid, digestForData(blobData)) require.NoError(t, err) } @@ -748,7 +748,7 @@ func TestBlobStorage_SequentialFlow(t *testing.T) { } digest := digestForData(blobData) - err = storage.FinishBlobUpload(uuid, digest) + err = storage.FinishBlobUpload("myapp", uuid, digest) require.NoError(t, err) reader, err := storage.GetBlob(digest) @@ -760,6 +760,75 @@ func TestBlobStorage_SequentialFlow(t *testing.T) { assert.Equal(t, blobData, storedData) } +func TestBlobStorage_UploadBoundToRepository(t *testing.T) { + storage, err := NewBlobStorage(t.TempDir(), testLogger()) + require.NoError(t, err) + + uuid, err := storage.StartBlobUpload("owner") + require.NoError(t, err) + + // A known upload UUID is not a bearer token for another repository. + data := []byte("foreign attempt") + _, err = storage.AppendBlobChunk("intruder", uuid, bytes.NewReader(data), -1, 0) + assert.ErrorIs(t, err, domain.ErrUploadNotFound) + + err = storage.FinishBlobUpload("intruder", uuid, digestForData(data)) + assert.ErrorIs(t, err, domain.ErrUploadNotFound) + + err = storage.CancelBlobUpload("intruder", uuid) + assert.ErrorIs(t, err, domain.ErrUploadNotFound) + + // The owner can still complete the upload. + _, err = storage.AppendBlobChunk("owner", uuid, bytes.NewReader(data), -1, 0) + require.NoError(t, err) + require.NoError(t, storage.FinishBlobUpload("owner", uuid, digestForData(data))) +} + +func TestBlobStorage_BlobOwnershipPersistsAcrossRestart(t *testing.T) { + dir := t.TempDir() + storage, err := NewBlobStorage(dir, testLogger()) + require.NoError(t, err) + + uuid, err := storage.StartBlobUpload("myapp") + require.NoError(t, err) + data := []byte("owned content") + _, err = storage.AppendBlobChunk("myapp", uuid, bytes.NewReader(data), -1, 0) + require.NoError(t, err) + digest := digestForData(data) + require.NoError(t, storage.FinishBlobUpload("myapp", uuid, digest)) + + owned, err := storage.BlobOwnedByRepository("myapp", digest) + require.NoError(t, err) + assert.True(t, owned) + + foreign, err := storage.BlobOwnedByRepository("other", digest) + require.NoError(t, err) + assert.False(t, foreign) + + // Ownership survives a restart of the storage adapter. + reopened, err := NewBlobStorage(dir, testLogger()) + require.NoError(t, err) + owned, err = reopened.BlobOwnedByRepository("myapp", digest) + require.NoError(t, err) + assert.True(t, owned) +} + +func TestBlobStorage_FailedFinishDoesNotRecordOwnership(t *testing.T) { + storage, err := NewBlobStorage(t.TempDir(), testLogger()) + require.NoError(t, err) + + uuid, err := storage.StartBlobUpload("myapp") + require.NoError(t, err) + data := []byte("mismatched content") + _, err = storage.AppendBlobChunk("myapp", uuid, bytes.NewReader(data), -1, 0) + require.NoError(t, err) + + require.Error(t, storage.FinishBlobUpload("myapp", uuid, testDigest1)) + owned, err := storage.BlobOwnedByRepository("myapp", testDigest1) + require.NoError(t, err) + assert.False(t, owned) +} + func digestForData(data []byte) string { sum := sha256.Sum256(data) return "sha256:" + hex.EncodeToString(sum[:]) diff --git a/internal/adapters/out/filesystem/manifest.go b/internal/adapters/out/filesystem/manifest.go index 338e45ab5..46eac0517 100644 --- a/internal/adapters/out/filesystem/manifest.go +++ b/internal/adapters/out/filesystem/manifest.go @@ -1,10 +1,12 @@ package filesystem import ( + "crypto/sha256" "encoding/json" "fmt" "os" "path/filepath" + "strings" "time" "github.com/bnema/zerowrap" @@ -110,6 +112,23 @@ func (s *ManifestStorage) PutManifest(name, reference, contentType string, data Msg("failed to update tags list") } + // Mirror tag-addressed manifests under their content digest so + // digest-addressed pulls (the pinned deploy path) resolve. + if !validation.IsDigest(reference) { + digest := fmt.Sprintf("sha256:%x", sha256.Sum256(data)) + if digest != reference { + if err := s.putManifestByDigest(name, digest, contentType, data); err != nil { + s.log.Warn(). + Str(zerowrap.FieldLayer, "adapter"). + Str(zerowrap.FieldAdapter, "filesystem"). + Err(err). + Str("name", name). + Str("digest", digest). + Msg("failed to mirror manifest by digest") + } + } + } + s.log.Info(). Str(zerowrap.FieldLayer, "adapter"). Str(zerowrap.FieldAdapter, "filesystem"). @@ -245,6 +264,9 @@ func (s *ManifestStorage) GetManifestModTime(name, reference string) (time.Time, func (s *ManifestStorage) getManifestPath(name, reference string) (string, error) { // Validate name to prevent path traversal (defense in depth) + if strings.Contains(name, "..") || strings.Contains(reference, "..") { + return "", fmt.Errorf("path traversal not allowed") + } if _, err := validation.ValidatePath(name); err != nil { return "", fmt.Errorf("invalid repository name: %w", err) } @@ -308,6 +330,22 @@ func (s *ManifestStorage) putManifestContentType(name, reference, contentType st return os.WriteFile(contentTypePath, []byte(contentType), 0600) } +// putManifestByDigest mirrors one manifest under its content digest so +// digest-addressed pulls resolve. +func (s *ManifestStorage) putManifestByDigest(name, digest, contentType string, data []byte) error { + manifestPath, err := s.getManifestPath(name, digest) + if err != nil { + return err + } + if err := os.MkdirAll(filepath.Dir(manifestPath), 0750); err != nil { + return fmt.Errorf("failed to create manifest directory: %w", err) + } + if err := os.WriteFile(manifestPath, data, 0600); err != nil { + return fmt.Errorf("failed to write digest manifest: %w", err) + } + return s.putManifestContentType(name, digest, contentType) +} + func (s *ManifestStorage) deleteManifestContentType(name, reference string) error { contentTypePath, err := s.getManifestContentTypePath(name, reference) if err != nil { diff --git a/internal/adapters/out/filesystem/manifest_test.go b/internal/adapters/out/filesystem/manifest_test.go index 697347fcf..50338d0df 100644 --- a/internal/adapters/out/filesystem/manifest_test.go +++ b/internal/adapters/out/filesystem/manifest_test.go @@ -1,6 +1,8 @@ package filesystem import ( + "crypto/sha256" + "fmt" "os" "path/filepath" "testing" @@ -53,6 +55,24 @@ func TestManifestStorage_PutAndGetManifest(t *testing.T) { assert.Equal(t, contentType, ct) } +func TestManifestStorage_PutMirrorsDigest(t *testing.T) { + tmpDir := t.TempDir() + log := testLogger() + + storage, err := NewManifestStorage(tmpDir, log) + require.NoError(t, err) + + manifestData := []byte(`{"schemaVersion": 2}`) + contentType := "application/vnd.docker.distribution.manifest.v2+json" + require.NoError(t, storage.PutManifest("myapp", "v1", contentType, manifestData)) + + digest := fmt.Sprintf("sha256:%x", sha256.Sum256(manifestData)) + data, ct, err := storage.GetManifest("myapp", digest) + require.NoError(t, err) + assert.Equal(t, manifestData, data) + assert.Equal(t, contentType, ct) +} + func TestManifestStorage_GetManifest_NotFound(t *testing.T) { tmpDir := t.TempDir() log := testLogger() diff --git a/internal/adapters/out/filesystem/previewstore.go b/internal/adapters/out/filesystem/previewstore.go deleted file mode 100644 index 99879a38d..000000000 --- a/internal/adapters/out/filesystem/previewstore.go +++ /dev/null @@ -1,92 +0,0 @@ -package filesystem - -import ( - "context" - "encoding/json" - "errors" - "io/fs" - "os" - "path/filepath" - "sync" - - "github.com/bnema/gordon/internal/domain" -) - -type previewStoreData struct { - Previews []domain.PreviewRoute `json:"previews"` -} - -// PreviewStore persists preview routes to a JSON file. -type PreviewStore struct { - path string - mu sync.Mutex -} - -// NewPreviewStore creates a new filesystem-backed preview store. -func NewPreviewStore(path string) *PreviewStore { - return &PreviewStore{path: path} -} - -func (s *PreviewStore) Load(_ context.Context) ([]domain.PreviewRoute, error) { - s.mu.Lock() - defer s.mu.Unlock() - - data, err := os.ReadFile(s.path) - if err != nil { - if errors.Is(err, fs.ErrNotExist) { - return nil, nil - } - return nil, err - } - - var store previewStoreData - if err := json.Unmarshal(data, &store); err != nil { - return nil, err - } - return store.Previews, nil -} - -func (s *PreviewStore) Save(_ context.Context, previews []domain.PreviewRoute) error { - s.mu.Lock() - defer s.mu.Unlock() - - if len(previews) == 0 { - err := os.Remove(s.path) - if errors.Is(err, fs.ErrNotExist) { - return nil - } - return err - } - - data, err := json.MarshalIndent(previewStoreData{Previews: previews}, "", " ") - if err != nil { - return err - } - - // Atomic write: temp file → rename - dir := filepath.Dir(s.path) - if err := os.MkdirAll(dir, 0o750); err != nil { - return err - } - tmp, err := os.CreateTemp(dir, ".previews-*.json.tmp") - if err != nil { - return err - } - tmpName := tmp.Name() - - if _, err := tmp.Write(data); err != nil { - tmp.Close() - os.Remove(tmpName) - return err - } - if err := tmp.Sync(); err != nil { - tmp.Close() - os.Remove(tmpName) - return err - } - if err := tmp.Close(); err != nil { - os.Remove(tmpName) - return err - } - return os.Rename(tmpName, s.path) -} diff --git a/internal/adapters/out/filesystem/previewstore_test.go b/internal/adapters/out/filesystem/previewstore_test.go deleted file mode 100644 index f9513035f..000000000 --- a/internal/adapters/out/filesystem/previewstore_test.go +++ /dev/null @@ -1,61 +0,0 @@ -package filesystem - -import ( - "context" - "os" - "path/filepath" - "testing" - "time" - - "github.com/bnema/gordon/internal/domain" - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" -) - -func TestPreviewStore_SaveAndLoad(t *testing.T) { - dir := t.TempDir() - store := NewPreviewStore(filepath.Join(dir, "previews.json")) - - previews := []domain.PreviewRoute{ - { - Domain: "myapp--feat.example.com", - Image: "myapp:preview-feat", - BaseRoute: "myapp.example.com", - Name: "feat", - CreatedAt: time.Now().UTC().Truncate(time.Second), - ExpiresAt: time.Now().UTC().Add(48 * time.Hour).Truncate(time.Second), - HTTPS: true, - Volumes: []string{"preview-feat-pgdata"}, - Containers: []string{"preview-feat-postgres"}, - }, - } - - ctx := context.Background() - err := store.Save(ctx, previews) - require.NoError(t, err) - - loaded, err := store.Load(ctx) - require.NoError(t, err) - assert.Equal(t, previews, loaded) -} - -func TestPreviewStore_LoadEmpty(t *testing.T) { - dir := t.TempDir() - store := NewPreviewStore(filepath.Join(dir, "previews.json")) - - loaded, err := store.Load(context.Background()) - require.NoError(t, err) - assert.Empty(t, loaded) -} - -func TestPreviewStore_SaveEmpty(t *testing.T) { - dir := t.TempDir() - path := filepath.Join(dir, "previews.json") - store := NewPreviewStore(path) - - err := store.Save(context.Background(), []domain.PreviewRoute{}) - require.NoError(t, err) - - _, err = os.Stat(path) - assert.True(t, os.IsNotExist(err)) -} diff --git a/internal/adapters/out/imageref/resolver.go b/internal/adapters/out/imageref/resolver.go new file mode 100644 index 000000000..c9654f1c8 --- /dev/null +++ b/internal/adapters/out/imageref/resolver.go @@ -0,0 +1,114 @@ +// Package imageref resolves image references to pinned digests without +// pulling or running anything. Installation-registry refs resolve locally +// through manifest storage; external refs resolve through the registry +// transfer path (allowlist + require-digest policy enforced by callers). +package imageref + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "fmt" + "strings" + + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/pkg/validation" +) + +// ManifestReader is the manifest-storage subset the resolver needs. +type ManifestReader interface { + GetManifest(name, reference string) ([]byte, string, error) +} + +// RemoteResolver resolves external refs (allowlisted registries). +type RemoteResolver interface { + ResolveDigest(ctx context.Context, ref string) (string, error) +} + +// Resolver implements out.ImageResolver. +type Resolver struct { + registryDomain string + manifests ManifestReader + remote RemoteResolver + policy domain.ImageSourcePolicy +} + +// NewResolver wires local manifest resolution with an optional remote. +func NewResolver(registryDomain string, manifests ManifestReader, remote RemoteResolver) *Resolver { + return &Resolver{ + registryDomain: registryDomain, + manifests: manifests, + remote: remote, + policy: domain.ImageSourcePolicy{InstallationRegistry: registryDomain}, + } +} + +// WithPolicy installs the installation image-source policy. Every +// reference, including an already-pinned digest, is validated against it +// before resolution. +func (r *Resolver) WithPolicy(policy domain.ImageSourcePolicy) *Resolver { + r.policy = policy + return r +} + +// ResolveDigest implements out.ImageResolver. +func (r *Resolver) ResolveDigest(ctx context.Context, ref string) (string, error) { + if err := ctx.Err(); err != nil { + return "", err + } + trimmed := strings.TrimSpace(ref) + if trimmed == "" { + return "", fmt.Errorf("imageref: empty reference: %w", domain.ErrAppImageUnresolvable) + } + // Already pinned: digest refs pass through after validation. + if _, digest, ok := strings.Cut(trimmed, "@"); ok { + if err := validation.ValidateImageDigest(digest); err != nil { + return "", fmt.Errorf("imageref: reference %q has invalid digest: %w", ref, domain.ErrAppImageUnresolvable) + } + } + // Every reference, pinned or not, is validated against installation + // policy before any resolution or pull. + if err := r.policy.ValidateImageSource(trimmed); err != nil { + return "", fmt.Errorf("imageref: %w", err) + } + if _, digest, ok := strings.Cut(trimmed, "@"); ok { + return digest, nil + } + // Installation-registry refs resolve locally. + if name, tag, ok := splitLocalRef(r.registryDomain, trimmed); ok { + raw, _, err := r.manifests.GetManifest(name, tag) + if err != nil { + return "", fmt.Errorf("imageref: local manifest %s:%s: %w", name, tag, domain.ErrAppImageUnresolvable) + } + sum := sha256.Sum256(raw) + return "sha256:" + hex.EncodeToString(sum[:]), nil + } + // External refs need the remote resolver. + if r.remote == nil { + return "", fmt.Errorf("imageref: external reference %q needs remote resolution: %w", ref, domain.ErrAppImageUnresolvable) + } + digest, err := r.remote.ResolveDigest(ctx, trimmed) + if err != nil { + return "", fmt.Errorf("imageref: remote %q: %w", ref, domain.ErrAppImageUnresolvable) + } + if err := validation.ValidateImageDigest(digest); err != nil { + return "", fmt.Errorf("imageref: remote %q returned invalid digest: %w", ref, domain.ErrAppImageUnresolvable) + } + return digest, nil +} + +// splitLocalRef splits registry-domain refs into name + tag. +func splitLocalRef(registryDomain, ref string) (string, string, bool) { + host, rest, ok := strings.Cut(ref, "/") + if !ok || host != registryDomain { + return "", "", false + } + name, tag, _ := strings.Cut(rest, ":") + if name == "" { + return "", "", false + } + if tag == "" { + tag = "latest" + } + return name, tag, true +} diff --git a/internal/adapters/out/imageref/resolver_test.go b/internal/adapters/out/imageref/resolver_test.go new file mode 100644 index 000000000..78c16b1ea --- /dev/null +++ b/internal/adapters/out/imageref/resolver_test.go @@ -0,0 +1,135 @@ +package imageref + +import ( + "context" + "errors" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +type stubManifests struct { + raw map[string][]byte + err error +} + +func (f *stubManifests) GetManifest(name, reference string) ([]byte, string, error) { + if f.err != nil { + return nil, "", f.err + } + raw, ok := f.raw[name+":"+reference] + if !ok { + return nil, "", errors.New("missing") + } + return raw, "application/vnd.oci.image.manifest.v1+json", nil +} + +func TestResolver_PinnedPassesThrough(t *testing.T) { + r := NewResolver("reg.example.com", &stubManifests{}, nil) + valid := "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef" + got, err := r.ResolveDigest(context.Background(), "img@"+valid) + require.NoError(t, err) + assert.Equal(t, valid, got) + + _, err = r.ResolveDigest(context.Background(), "img@sha256:short") + require.ErrorIs(t, err, domain.ErrAppImageUnresolvable) +} + +func TestResolver_LocalManifestHashes(t *testing.T) { + r := NewResolver("reg.example.com", &stubManifests{raw: map[string][]byte{"blog/web:1.0": []byte("{}")}}, nil) + got, err := r.ResolveDigest(context.Background(), "reg.example.com/blog/web:1.0") + require.NoError(t, err) + assert.True(t, len(got) == 7+64) + + _, err = r.ResolveDigest(context.Background(), "reg.example.com/blog/missing:1.0") + require.ErrorIs(t, err, domain.ErrAppImageUnresolvable) +} + +func TestResolver_ExternalNeedsRemote(t *testing.T) { + r := NewResolver("reg.example.com", &stubManifests{}, nil) + _, err := r.ResolveDigest(context.Background(), "docker.io/library/nginx:1.0") + require.ErrorIs(t, err, domain.ErrAppImageUnresolvable) +} + +func TestResolver_RejectsMalformedRemoteDigest(t *testing.T) { + r := NewResolver("reg.example.com", &stubManifests{}, &recordingRemote{digest: "sha256:not-a-digest"}) + _, err := r.ResolveDigest(context.Background(), "docker.io/library/nginx:1.0") + require.ErrorIs(t, err, domain.ErrAppImageUnresolvable) +} + +// recordingRemote records whether the connector was consulted. +type recordingRemote struct { + called bool + digest string +} + +func (r *recordingRemote) ResolveDigest(context.Context, string) (string, error) { + r.called = true + if r.digest != "" { + return r.digest, nil + } + return "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", nil +} + +// TestResolver_RejectsDisallowedRefsWithoutConnecting proves policy is +// enforced before the remote connector is consulted, including for +// already-pinned digest references. +func TestResolver_RejectsDisallowedRefsWithoutConnecting(t *testing.T) { + const digest = "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + remote := &recordingRemote{} + r := NewResolver("reg.example.com", &stubManifests{}, remote).WithPolicy(domain.ImageSourcePolicy{ + InstallationRegistry: "reg.example.com", + AllowedRegistries: []string{"registry.example.com"}, + }) + + _, err := r.ResolveDigest(context.Background(), "127.0.0.1:12345/private@"+digest) + require.ErrorIs(t, err, domain.ErrAppImageNotAllowed) + + _, err = r.ResolveDigest(context.Background(), "other.example.com/team/app@"+digest) + require.ErrorIs(t, err, domain.ErrAppImageNotAllowed) + + _, err = r.ResolveDigest(context.Background(), "other.example.com/team/app:1.0") + require.ErrorIs(t, err, domain.ErrAppImageNotAllowed) + + assert.False(t, remote.called, "policy rejection must happen before connecting") +} + +func TestResolver_AllowsInstallationAndAllowlistedRefs(t *testing.T) { + const digest = "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + manifests := &stubManifests{raw: map[string][]byte{"blog/web:1.0": []byte("{}")}} + remote := &recordingRemote{} + r := NewResolver("reg.example.com", manifests, remote).WithPolicy(domain.ImageSourcePolicy{ + InstallationRegistry: "reg.example.com", + AllowedRegistries: []string{"registry.example.com"}, + }) + + got, err := r.ResolveDigest(context.Background(), "reg.example.com/blog/web:1.0") + require.NoError(t, err) + assert.True(t, len(got) == 7+64) + + got, err = r.ResolveDigest(context.Background(), "registry.example.com/team/app@"+digest) + require.NoError(t, err) + assert.Equal(t, digest, got) + assert.False(t, remote.called, "digest refs resolve without contacting the connector") +} + +// TestResolver_RequireDigestRejectsTags proves the require-digest policy is +// enforced on the manifest reference before it is resolved. +func TestResolver_RequireDigestRejectsTags(t *testing.T) { + const digest = "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + r := NewResolver("reg.example.com", &stubManifests{}, &recordingRemote{}).WithPolicy(domain.ImageSourcePolicy{ + AllowedRegistries: []string{"registry.example.com"}, + RequireDigest: true, + InstallationRegistry: "reg.example.com", + }) + + _, err := r.ResolveDigest(context.Background(), "registry.example.com/team/app:1.0") + require.ErrorIs(t, err, domain.ErrAppImageNotAllowed) + + got, err := r.ResolveDigest(context.Background(), "registry.example.com/team/app@"+digest) + require.NoError(t, err) + assert.Equal(t, digest, got) +} diff --git a/internal/adapters/out/s3/volume_backup_storage.go b/internal/adapters/out/s3/volume_backup_storage.go index cac144ef6..4adac6683 100644 --- a/internal/adapters/out/s3/volume_backup_storage.go +++ b/internal/adapters/out/s3/volume_backup_storage.go @@ -84,8 +84,8 @@ func (s *VolumeBackupStorage) StoreVolumeArchive(ctx context.Context, job domain if s.bucket == "" { return "", fmt.Errorf("s3 bucket is required") } - if job.Domain == "" { - return "", fmt.Errorf("backup domain is required") + if job.App == "" { + return "", fmt.Errorf("backup app is required") } if job.VolumeName == "" { return "", fmt.Errorf("backup volume name is required") @@ -105,7 +105,8 @@ func (s *VolumeBackupStorage) StoreVolumeArchive(ctx context.Context, job domain Body: data, ContentType: aws.String(contentTypeForCompression(compression)), Metadata: map[string]string{ - "gordon-domain": job.Domain, + "gordon-app": job.App, + "gordon-service": job.Service, "gordon-volume": job.VolumeName, "gordon-container": job.ContainerName, "gordon-mount-path": job.MountPath, @@ -143,8 +144,8 @@ func (s *VolumeBackupStorage) GetVolumeArchive(ctx context.Context, artifactRef } // ListVolumeArchives lists completed volume archive backups for a domain. -func (s *VolumeBackupStorage) ListVolumeArchives(ctx context.Context, domainName string) ([]domain.VolumeBackupJob, error) { - listPrefix := s.domainPrefix(domainName) +func (s *VolumeBackupStorage) ListVolumeArchives(ctx context.Context, app string) ([]domain.VolumeBackupJob, error) { + listPrefix := s.appPrefix(app) jobs := make([]domain.VolumeBackupJob, 0) var token *string for { @@ -160,7 +161,7 @@ func (s *VolumeBackupStorage) ListVolumeArchives(ctx context.Context, domainName if obj.Key == nil { continue } - job, ok := s.jobFromKey(domainName, *obj.Key) + job, ok := s.jobFromKey(app, *obj.Key) if !ok { continue } @@ -186,6 +187,7 @@ func (s *VolumeBackupStorage) hydrateVolumeBackupJobMetadata(ctx context.Context if err != nil || out == nil || out.Metadata == nil { return } + job.Service = out.Metadata["gordon-service"] job.ContainerName = out.Metadata["gordon-container"] job.MountPath = out.Metadata["gordon-mount-path"] if compression := out.Metadata["gordon-compression"]; compression != "" { @@ -212,7 +214,7 @@ func (s *VolumeBackupStorage) DeleteVolumeArchive(ctx context.Context, artifactR } // ApplyVolumeRetention deletes old completed volume archives according to the keep count. -func (s *VolumeBackupStorage) ApplyVolumeRetention(ctx context.Context, domainName string, policy domain.VolumeBackupRetentionPolicy) (int, error) { +func (s *VolumeBackupStorage) ApplyVolumeRetention(ctx context.Context, app string, policy domain.VolumeBackupRetentionPolicy) (int, error) { if policy.Keep < 0 { return 0, fmt.Errorf("volume backup retention keep cannot be negative") } @@ -220,7 +222,7 @@ func (s *VolumeBackupStorage) ApplyVolumeRetention(ctx context.Context, domainNa return 0, nil } - jobs, err := s.ListVolumeArchives(ctx, domainName) + jobs, err := s.ListVolumeArchives(ctx, app) if err != nil { return 0, err } @@ -258,15 +260,18 @@ func (s *VolumeBackupStorage) objectKey(job domain.VolumeBackupJob) string { ext = "tar.gz" } fileName := fmt.Sprintf("%s-%s.%s", started.Format(volumeBackupTimestampLayout), sanitizeS3KeyComponent(id), ext) - parts := []string{s.prefix, "domains", sanitizeS3KeyComponent(job.Domain), "volumes", sanitizeS3KeyComponent(job.VolumeName), fileName} + parts := []string{s.prefix, "apps", sanitizeS3KeyComponent(job.App), "volumes", sanitizeS3KeyComponent(job.VolumeName), fileName} return joinS3Key(parts...) } -func (s *VolumeBackupStorage) domainPrefix(domainName string) string { - if strings.TrimSpace(domainName) == "" { - return joinS3Key(s.prefix, "domains") + "/" +// appPrefix is the object prefix of one app's volume archives, or of every +// app when app is empty. Object keys are app-based; artifacts written by +// older versions under a domain prefix are never read or migrated here. +func (s *VolumeBackupStorage) appPrefix(app string) string { + if strings.TrimSpace(app) == "" { + return joinS3Key(s.prefix, "apps") + "/" } - return joinS3Key(s.prefix, "domains", sanitizeS3KeyComponent(domainName), "volumes") + "/" + return joinS3Key(s.prefix, "apps", sanitizeS3KeyComponent(app), "volumes") + "/" } func (s *VolumeBackupStorage) artifactRef(key string) string { @@ -290,18 +295,18 @@ func (s *VolumeBackupStorage) keyFromArtifactRef(artifactRef string) (string, er return strings.TrimPrefix(artifactRef, "/"), nil } -func (s *VolumeBackupStorage) jobFromKey(domainName, key string) (domain.VolumeBackupJob, bool) { +func (s *VolumeBackupStorage) jobFromKey(app, key string) (domain.VolumeBackupJob, bool) { var rel string - if strings.TrimSpace(domainName) == "" { - rel = strings.TrimPrefix(key, joinS3Key(s.prefix, "domains")+"/") + if strings.TrimSpace(app) == "" { + rel = strings.TrimPrefix(key, joinS3Key(s.prefix, "apps")+"/") parts := strings.Split(rel, "/") if len(parts) != 4 || parts[1] != "volumes" { return domain.VolumeBackupJob{}, false } - domainName = parts[0] + app = parts[0] rel = strings.Join(parts[2:], "/") } else { - rel = strings.TrimPrefix(key, joinS3Key(s.prefix, "domains", sanitizeS3KeyComponent(domainName), "volumes")+"/") + rel = strings.TrimPrefix(key, joinS3Key(s.prefix, "apps", sanitizeS3KeyComponent(app), "volumes")+"/") } parts := strings.Split(rel, "/") if len(parts) != 2 { @@ -318,7 +323,7 @@ func (s *VolumeBackupStorage) jobFromKey(domainName, key string) (domain.VolumeB } return domain.VolumeBackupJob{ ID: id, - Domain: domainName, + App: app, VolumeName: volumeName, Type: domain.BackupTypeVolumeArchive, Status: domain.BackupStatusCompleted, diff --git a/internal/adapters/out/s3/volume_backup_storage_test.go b/internal/adapters/out/s3/volume_backup_storage_test.go index 53802dbae..ae001dd99 100644 --- a/internal/adapters/out/s3/volume_backup_storage_test.go +++ b/internal/adapters/out/s3/volume_backup_storage_test.go @@ -89,7 +89,8 @@ func TestVolumeBackupStorageStoreListGet(t *testing.T) { artifact, err := storage.StoreVolumeArchive(context.Background(), domain.VolumeBackupJob{ ID: "job/1", - Domain: "app.example.com", + App: "app.example.com", + Service: "api", ContainerName: "app", VolumeName: "gordon-app-data", MountPath: "/data", @@ -98,13 +99,14 @@ func TestVolumeBackupStorageStoreListGet(t *testing.T) { }, bytes.NewReader([]byte("archive"))) require.NoError(t, err) - assert.Equal(t, "s3://gordon-backups/prod/gordon/domains/app.example.com/volumes/gordon-app-data/20260619T020000Z-job_1.tar.zst", artifact) + assert.Equal(t, "s3://gordon-backups/prod/gordon/apps/app.example.com/volumes/gordon-app-data/20260619T020000Z-job_1.tar.zst", artifact) assert.Equal(t, "application/zstd", aws.ToString(uploader.last.ContentType)) jobs, err := storage.ListVolumeArchives(context.Background(), "app.example.com") require.NoError(t, err) require.Len(t, jobs, 1) assert.Equal(t, domain.BackupStatusCompleted, jobs[0].Status) + assert.Equal(t, "api", jobs[0].Service) assert.Equal(t, "app", jobs[0].ContainerName) assert.Equal(t, "/data", jobs[0].MountPath) assert.Equal(t, string(domain.VolumeBackupCompressionZstd), jobs[0].Metadata["compression"]) @@ -123,10 +125,10 @@ func TestVolumeBackupStorageRetentionUsesParsedKeyTimestamp(t *testing.T) { client := newFakeVolumeS3Client() storage := NewVolumeBackupStorageWithClients(domain.VolumeBackupConfig{S3Bucket: "bucket"}, client, &fakeVolumeUploader{client: client}) keys := []string{ - "domains/app.example.com/volumes/gordon-data/20260619T010000Z-a.tar.zst", - "domains/app.example.com/volumes/gordon-data/20260619T020000Z-b.tar.zst", - "domains/app.example.com/volumes/gordon-data/20260619T030000Z-c.tar.zst", - "domains/app.example.com/volumes/gordon-data/not-a-backup.tar.zst", + "apps/app.example.com/volumes/gordon-data/20260619T010000Z-a.tar.zst", + "apps/app.example.com/volumes/gordon-data/20260619T020000Z-b.tar.zst", + "apps/app.example.com/volumes/gordon-data/20260619T030000Z-c.tar.zst", + "apps/app.example.com/volumes/gordon-data/not-a-backup.tar.zst", } for _, key := range keys { client.objects[key] = []byte("x") @@ -136,8 +138,8 @@ func TestVolumeBackupStorageRetentionUsesParsedKeyTimestamp(t *testing.T) { require.NoError(t, err) assert.Equal(t, 1, deleted) - assert.Equal(t, []string{"domains/app.example.com/volumes/gordon-data/20260619T010000Z-a.tar.zst"}, client.deleted) - assert.Contains(t, client.objects, "domains/app.example.com/volumes/gordon-data/not-a-backup.tar.zst") + assert.Equal(t, []string{"apps/app.example.com/volumes/gordon-data/20260619T010000Z-a.tar.zst"}, client.deleted) + assert.Contains(t, client.objects, "apps/app.example.com/volumes/gordon-data/not-a-backup.tar.zst") } func TestVolumeBackupStorageRejectsWrongBucketArtifact(t *testing.T) { diff --git a/internal/adapters/out/telemetry/logs.go b/internal/adapters/out/telemetry/logs.go new file mode 100644 index 000000000..be59731d6 --- /dev/null +++ b/internal/adapters/out/telemetry/logs.go @@ -0,0 +1,169 @@ +package telemetry + +import ( + "context" + "sync" + + "go.opentelemetry.io/otel/attribute" + otellog "go.opentelemetry.io/otel/log" + sdklog "go.opentelemetry.io/otel/sdk/log" + "go.opentelemetry.io/otel/sdk/resource" + semconv "go.opentelemetry.io/otel/semconv/v1.26.0" + + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/domain" +) + +// Attribute keys shared by every exported record. +const ( + attrGordonApp = "gordon.app" + attrGordonService = "gordon.service" + attrLogType = "log.type" + attrLogStream = "log.iostream" +) + +var _ out.LogExporter = (*LogExporter)(nil) + +// LogExporter implements out.LogExporter over OTLP. +// +// OTLP carries service identity on the resource, not on the record, so +// each source (Gordon or one app service) gets its own lazily created +// LoggerProvider and resource. All providers share one exporter; the +// batch processors of every provider serialize through it. +type LogExporter struct { + exporter sdklog.Exporter + shared *serialExporter + base *resource.Resource + + mu sync.Mutex + providers map[domain.LogSource]*sdklog.LoggerProvider + closed bool +} + +// NewLogExporter wraps exporter. base carries host/OS/version attributes +// merged into every source resource; it may be nil. +func NewLogExporter(exporter sdklog.Exporter, base *resource.Resource) *LogExporter { + shared := &serialExporter{Exporter: exporter} + return &LogExporter{ + exporter: exporter, + shared: shared, + base: base, + providers: map[domain.LogSource]*sdklog.LoggerProvider{}, + } +} + +// Export implements out.LogExporter. It never blocks on the network: +// the batch processor queues records and drops the oldest on overload. +func (e *LogExporter) Export(ctx context.Context, record domain.LogRecord) { + provider := e.provider(record.Source) + if provider == nil { + return + } + var r otellog.Record + r.SetTimestamp(record.Time) + r.SetObservedTimestamp(record.Time) + r.SetBody(attribute.StringValue(record.Body)) + if sev, text := severity(record.Severity); sev != otellog.SeverityUndefined { + r.SetSeverity(sev) + r.SetSeverityText(text) + } + attrs := make([]attribute.KeyValue, 0, len(record.Attributes)+2) + if record.Type != "" { + attrs = append(attrs, attribute.String(attrLogType, record.Type)) + } + if record.Stream != "" { + attrs = append(attrs, attribute.String(attrLogStream, record.Stream)) + } + for key, value := range record.Attributes { + attrs = append(attrs, attribute.String(key, value)) + } + r.AddAttributes(attrs...) + provider.Logger("gordon").Emit(ctx, r) +} + +// Shutdown flushes every source provider, then closes the exporter. +func (e *LogExporter) Shutdown(ctx context.Context) error { + e.mu.Lock() + e.closed = true + providers := e.providers + e.providers = map[domain.LogSource]*sdklog.LoggerProvider{} + e.mu.Unlock() + + var firstErr error + for _, provider := range providers { + if err := provider.Shutdown(ctx); err != nil && firstErr == nil { + firstErr = err + } + } + if err := e.exporter.Shutdown(ctx); err != nil && firstErr == nil { + firstErr = err + } + return firstErr +} + +func (e *LogExporter) provider(source domain.LogSource) *sdklog.LoggerProvider { + e.mu.Lock() + defer e.mu.Unlock() + if e.closed { + return nil + } + if provider, ok := e.providers[source]; ok { + return provider + } + // sourceResource is schemaless, so Merge cannot fail on schema + // conflicts; its attributes win over base (service.name). + res, err := resource.Merge(e.base, sourceResource(source)) + if err != nil { + res = sourceResource(source) + } + provider := sdklog.NewLoggerProvider( + sdklog.WithProcessor(sdklog.NewBatchProcessor(e.shared)), + sdklog.WithResource(res), + ) + e.providers[source] = provider + return provider +} + +// sourceResource builds the resource identity of one source: +// service.name "." plus namespace and Gordon attributes. +func sourceResource(source domain.LogSource) *resource.Resource { + attrs := []attribute.KeyValue{semconv.ServiceName(source.ServiceName())} + if !source.IsGordon() { + attrs = append(attrs, + semconv.ServiceNamespace(source.App), + attribute.String(attrGordonApp, source.App), + attribute.String(attrGordonService, source.Service), + ) + } + return resource.NewSchemaless(attrs...) +} + +func severity(level domain.LogSeverity) (otellog.Severity, string) { + switch level { + case domain.LogSeverityInfo: + return otellog.SeverityInfo, "INFO" + case domain.LogSeverityWarn: + return otellog.SeverityWarn, "WARN" + case domain.LogSeverityError: + return otellog.SeverityError, "ERROR" + default: + return otellog.SeverityUndefined, "" + } +} + +// serialExporter serializes Export calls: the SDK forbids concurrent +// Export on one exporter, and each source provider runs its own batcher. +// Shutdown is ignored so one source provider shutting down cannot close +// the exporter shared by the others; the owner shuts it down once. +type serialExporter struct { + sdklog.Exporter + mu sync.Mutex +} + +func (s *serialExporter) Export(ctx context.Context, records []sdklog.Record) error { + s.mu.Lock() + defer s.mu.Unlock() + return s.Exporter.Export(ctx, records) +} + +func (s *serialExporter) Shutdown(context.Context) error { return nil } diff --git a/internal/adapters/out/telemetry/logs_test.go b/internal/adapters/out/telemetry/logs_test.go new file mode 100644 index 000000000..9372f48ba --- /dev/null +++ b/internal/adapters/out/telemetry/logs_test.go @@ -0,0 +1,112 @@ +package telemetry + +import ( + "context" + "sync" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "go.opentelemetry.io/otel/attribute" + otellog "go.opentelemetry.io/otel/log" + sdklog "go.opentelemetry.io/otel/sdk/log" + "go.opentelemetry.io/otel/sdk/resource" + + "github.com/bnema/gordon/internal/domain" +) + +type recordingExporter struct { + mu sync.Mutex + records []sdklog.Record + shutdown int +} + +func (e *recordingExporter) Export(_ context.Context, records []sdklog.Record) error { + e.mu.Lock() + defer e.mu.Unlock() + for _, r := range records { + e.records = append(e.records, r.Clone()) + } + return nil +} + +func (e *recordingExporter) Shutdown(context.Context) error { + e.mu.Lock() + defer e.mu.Unlock() + e.shutdown++ + return nil +} + +func (e *recordingExporter) ForceFlush(context.Context) error { return nil } + +func resourceAttr(r sdklog.Record, key string) string { + res := r.Resource() + value, _ := res.Set().Value(attribute.Key(key)) + return value.AsString() +} + +func recordAttr(r sdklog.Record, key string) string { + var found string + r.WalkAttributes(func(kv attribute.KeyValue) bool { + if string(kv.Key) == key { + found = kv.Value.AsString() + return false + } + return true + }) + return found +} + +func TestLogExporter_ExportsPerSourceIdentity(t *testing.T) { + rec := &recordingExporter{} + base := resource.NewSchemaless(attribute.String("service.name", "gordon"), attribute.String("host.name", "node-1")) + exp := NewLogExporter(rec, base) + ts := time.Date(2026, 1, 2, 3, 4, 5, 0, time.UTC) + + exp.Export(context.Background(), domain.LogRecord{ + Time: ts, + Source: domain.LogSource{App: "blog", Service: "web"}, + Type: domain.LogTypeContainer, + Stream: domain.LogStreamStderr, + Severity: domain.LogSeverityError, + Body: "boom", + }) + exp.Export(context.Background(), domain.LogRecord{ + Time: ts, + Source: domain.LogSource{App: "shop", Service: "web"}, + Body: "hello", + }) + require.NoError(t, exp.Shutdown(context.Background())) + + require.Len(t, rec.records, 2) + byService := map[string]sdklog.Record{} + for _, r := range rec.records { + byService[resourceAttr(r, "service.name")] = r + } + blog, ok := byService["blog.web"] + require.True(t, ok, "services with the same name stay distinct per app") + _, ok = byService["shop.web"] + require.True(t, ok) + + assert.Equal(t, "blog", resourceAttr(blog, "service.namespace")) + assert.Equal(t, "blog", resourceAttr(blog, attrGordonApp)) + assert.Equal(t, "web", resourceAttr(blog, attrGordonService)) + assert.Equal(t, "node-1", resourceAttr(blog, "host.name"), "base attributes are kept") + assert.Equal(t, "boom", blog.Body().AsString()) + assert.Equal(t, ts, blog.Timestamp()) + assert.Equal(t, otellog.SeverityError, blog.Severity()) + assert.Equal(t, domain.LogTypeContainer, recordAttr(blog, attrLogType)) + assert.Equal(t, domain.LogStreamStderr, recordAttr(blog, attrLogStream)) + assert.Equal(t, 1, rec.shutdown, "shared exporter is shut down exactly once") +} + +func TestLogExporter_DropsAfterShutdown(t *testing.T) { + rec := &recordingExporter{} + exp := NewLogExporter(rec, nil) + require.NoError(t, exp.Shutdown(context.Background())) + + exp.Export(context.Background(), domain.LogRecord{Source: domain.LogSource{App: "blog", Service: "web"}, Body: "late"}) + + assert.Empty(t, rec.records) +} diff --git a/internal/adapters/out/telemetry/metrics.go b/internal/adapters/out/telemetry/metrics.go index b450d974e..1b66e7440 100644 --- a/internal/adapters/out/telemetry/metrics.go +++ b/internal/adapters/out/telemetry/metrics.go @@ -1,12 +1,22 @@ package telemetry import ( + "context" + "go.opentelemetry.io/otel" + "go.opentelemetry.io/otel/attribute" "go.opentelemetry.io/otel/metric" + + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/domain" ) +var _ out.Metrics = (*Metrics)(nil) + // Metrics holds Gordon-specific OTel metrics instruments. type Metrics struct { + meter metric.Meter + // Deployments DeployTotal metric.Int64Counter DeployDuration metric.Float64Histogram @@ -15,7 +25,6 @@ type Metrics struct { // Container lifecycle ContainerRestarts metric.Int64Counter ContainerCrashLoops metric.Int64Counter - ManagedContainers metric.Int64UpDownCounter // Registry ImagePushTotal metric.Int64Counter @@ -31,7 +40,7 @@ type Metrics struct { // (OTel returns noop instruments when no MeterProvider is set). func NewMetrics() (*Metrics, error) { meter := otel.Meter("gordon") - m := &Metrics{} + m := &Metrics{meter: meter} var err error if m.DeployTotal, err = meter.Int64Counter("gordon.deploy.total", @@ -56,10 +65,6 @@ func NewMetrics() (*Metrics, error) { metric.WithDescription("Total crash loop detections")); err != nil { return nil, err } - if m.ManagedContainers, err = meter.Int64UpDownCounter("gordon.container.managed", - metric.WithDescription("Currently managed containers")); err != nil { - return nil, err - } if m.ImagePushTotal, err = meter.Int64Counter("gordon.registry.push.total", metric.WithDescription("Total image pushes")); err != nil { return nil, err @@ -80,3 +85,41 @@ func NewMetrics() (*Metrics, error) { return m, nil } + +// ObserveManagedContainers registers the gordon.container.managed gauge. +// count is called at each collection, so the exported value always reflects +// current state instead of drifting like an incremental counter. A count +// error skips that observation rather than exporting a wrong value. +func (m *Metrics) ObserveManagedContainers(count func(context.Context) (int64, error)) error { + _, err := m.meter.Int64ObservableGauge("gordon.container.managed", + metric.WithDescription("Currently managed containers"), + metric.WithInt64Callback(func(ctx context.Context, o metric.Int64Observer) error { + n, err := count(ctx) + if err != nil { + return err + } + o.Observe(n) + return nil + })) + return err +} + +// RecordImagePush implements out.Metrics. +func (m *Metrics) RecordImagePush(ctx context.Context, name, reference string, sizeBytes int64) { + attrs := metric.WithAttributes( + attribute.String("name", name), + attribute.String("reference", reference), + ) + m.ImagePushTotal.Add(ctx, 1, attrs) + m.ImagePushSize.Add(ctx, sizeBytes, attrs) +} + +// RecordEventProcessed implements out.Metrics. +func (m *Metrics) RecordEventProcessed(ctx context.Context, eventType domain.EventType) { + m.EventsProcessed.Add(ctx, 1, metric.WithAttributes(attribute.String("event_type", string(eventType)))) +} + +// RecordEventDropped implements out.Metrics. +func (m *Metrics) RecordEventDropped(ctx context.Context, eventType domain.EventType) { + m.EventsDropped.Add(ctx, 1, metric.WithAttributes(attribute.String("event_type", string(eventType)))) +} diff --git a/internal/adapters/out/telemetry/metrics_test.go b/internal/adapters/out/telemetry/metrics_test.go new file mode 100644 index 000000000..5aab2072c --- /dev/null +++ b/internal/adapters/out/telemetry/metrics_test.go @@ -0,0 +1,67 @@ +package telemetry + +import ( + "context" + "errors" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "go.opentelemetry.io/otel" + sdkmetric "go.opentelemetry.io/otel/sdk/metric" + "go.opentelemetry.io/otel/sdk/metric/metricdata" +) + +func collectManaged(t *testing.T, reader *sdkmetric.ManualReader) (int64, bool) { + t.Helper() + var rm metricdata.ResourceMetrics + require.NoError(t, reader.Collect(context.Background(), &rm)) + for _, sm := range rm.ScopeMetrics { + for _, m := range sm.Metrics { + if m.Name != "gordon.container.managed" { + continue + } + gauge, ok := m.Data.(metricdata.Gauge[int64]) + require.True(t, ok) + require.Len(t, gauge.DataPoints, 1) + assert.Equal(t, 0, gauge.DataPoints[0].Attributes.Len(), "gauge must stay label-free") + return gauge.DataPoints[0].Value, true + } + } + return 0, false +} + +func TestObserveManagedContainers(t *testing.T) { + reader := sdkmetric.NewManualReader() + prev := otel.GetMeterProvider() + otel.SetMeterProvider(sdkmetric.NewMeterProvider(sdkmetric.WithReader(reader))) + t.Cleanup(func() { otel.SetMeterProvider(prev) }) + + m, err := NewMetrics() + require.NoError(t, err) + + current := int64(3) + var countErr error + require.NoError(t, m.ObserveManagedContainers(func(context.Context) (int64, error) { + return current, countErr + })) + + v, ok := collectManaged(t, reader) + require.True(t, ok) + assert.Equal(t, int64(3), v) + + current = 1 + v, ok = collectManaged(t, reader) + require.True(t, ok) + assert.Equal(t, int64(1), v) + + // A failed count skips the observation instead of exporting a wrong value. + countErr = errors.New("state unavailable") + var rm metricdata.ResourceMetrics + _ = reader.Collect(context.Background(), &rm) + for _, sm := range rm.ScopeMetrics { + for _, metric := range sm.Metrics { + assert.NotEqual(t, "gordon.container.managed", metric.Name) + } + } +} diff --git a/internal/adapters/out/telemetry/provider.go b/internal/adapters/out/telemetry/provider.go index 2f72d6d37..edb3f7ad5 100644 --- a/internal/adapters/out/telemetry/provider.go +++ b/internal/adapters/out/telemetry/provider.go @@ -26,7 +26,7 @@ type Config struct { AuthToken string `mapstructure:"auth_token"` // Basic auth token (base64 encoded user:pass) Traces bool `mapstructure:"traces"` // Enable trace export Metrics bool `mapstructure:"metrics"` // Enable metric export - Logs bool `mapstructure:"logs"` // Bridge zerowrap logs to OTLP + Logs bool `mapstructure:"logs"` // Export Gordon, proxy access, and app container logs to OTLP TraceSampleRate float64 `mapstructure:"trace_sample_rate"` // 0=disabled, (0,1)=ratio-based, >=1=always sample } @@ -35,6 +35,9 @@ type Provider struct { TracerProvider *trace.TracerProvider MeterProvider *metric.MeterProvider LogProvider *otellog.LoggerProvider + // WorkloadLogs exports proxy access and app container logs, each + // under its own service identity. Nil when log export is disabled. + WorkloadLogs *LogExporter } // endpointConfig holds parsed endpoint details shared across exporter setup. @@ -218,5 +221,20 @@ func (p *Provider) setupLogging(ctx context.Context, ep *endpointConfig, res *re ) p.LogProvider = lp *shutdowns = append(*shutdowns, lp.Shutdown) + + // Workload logs use a dedicated exporter: the SDK forbids concurrent + // Export calls, and Gordon's own provider already owns logExp. + workloadExp, err := otlploghttp.New(ctx, logOpts...) + if err != nil { + return fmt.Errorf("create workload log exporter: %w", err) + } + // Host and OS only: app records must not inherit Gordon's + // service.name or service.version. + base, err := resource.New(ctx, resource.WithOS(), resource.WithHost()) + if err != nil { + return fmt.Errorf("create workload log resource: %w", err) + } + p.WorkloadLogs = NewLogExporter(workloadExp, base) + *shutdowns = append(*shutdowns, p.WorkloadLogs.Shutdown) return nil } diff --git a/internal/app/app_deploy_handler_lifecycle_test.go b/internal/app/app_deploy_handler_lifecycle_test.go new file mode 100644 index 000000000..9cf646948 --- /dev/null +++ b/internal/app/app_deploy_handler_lifecycle_test.go @@ -0,0 +1,138 @@ +package app + +import ( + "context" + "encoding/json" + "net/http" + "net/http/httptest" + "sync" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/dto" + "github.com/bnema/gordon/internal/adapters/in/http/admin" + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/apps" + "github.com/bnema/gordon/internal/usecase/deployment" +) + +// blockingHandlerDeployEngine implements the deployment-engine port the +// daemon-owned app service executes through. ExecuteDeploy reports the +// context it received and blocks until released, so request-cancellation +// independence is observable at the HTTP handler boundary. +type blockingHandlerDeployEngine struct { + started chan context.Context + release chan struct{} +} + +func (e *blockingHandlerDeployEngine) StartDeploy(context.Context, deployment.DeployInput) (*deployment.StartDeployResult, error) { + op := domain.AppOperation{Op: "op-1", Kind: "deploy", App: "blog", InputRevision: "rev-1"} + return &deployment.StartDeployResult{ + Claim: deployment.DeployClaim{App: "blog", Op: "op-1", Service: "web", Revision: "rev-1", Journal: op}, + Owned: true, + }, nil +} + +func (e *blockingHandlerDeployEngine) ExecuteDeploy(ctx context.Context, _ deployment.DeployClaim) (*deployment.DeployResult, error) { + e.started <- ctx + <-e.release + return &deployment.DeployResult{Op: "op-1", App: "blog"}, nil +} + +func (e *blockingHandlerDeployEngine) AbandonDeploy(context.Context, deployment.DeployClaim) error { + return nil +} + +func (e *blockingHandlerDeployEngine) Stop(context.Context, string, string) (*deployment.LifecycleResult, error) { + return nil, nil +} + +func (e *blockingHandlerDeployEngine) Start(context.Context, string, string) (*deployment.LifecycleResult, error) { + return nil, nil +} + +func (e *blockingHandlerDeployEngine) Restart(context.Context, string, string, string) (*deployment.LifecycleResult, error) { + return nil, nil +} + +func (e *blockingHandlerDeployEngine) Remove(context.Context, string, string) (*deployment.LifecycleResult, error) { + return nil, nil +} + +// TestAppDeployHandler_RequestCancellationDoesNotAbortExecution drives the +// admin deploy handler over the real daemon-owned AppServiceImpl and proves +// the request boundary is inert: the handler answers 202 with the persisted +// running journal promptly, and cancelling the request context afterwards +// never reaches the in-flight replacement. The handler owns no goroutine; the +// execution derives from the daemon context. +func TestAppDeployHandler_RequestCancellationDoesNotAbortExecution(t *testing.T) { + store := outmocks.NewMockAppState(t) + running := domain.AppOperation{Op: "op-1", Kind: "deploy", App: "blog", InputRevision: "rev-1"} + store.EXPECT().LoadOperation(mock.Anything, "blog", "op-1").Return(running, nil).Once() + // Show resolves no app: the response carries the journal alone. + store.EXPECT().AppExists(mock.Anything, "blog").Return(false, nil).Once() + // The app-wide service scope check sees no services and lets it through. + store.EXPECT().LoadDesired(mock.Anything, "blog").Return(domain.AppDesiredRevision{}, false, nil).Once() + store.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{}, false, nil).Once() + + release := make(chan struct{}) + var releaseOnce sync.Once + unblock := func() { releaseOnce.Do(func() { close(release) }) } + + engine := &blockingHandlerDeployEngine{started: make(chan context.Context, 1), release: release} + svc := apps.NewAppServiceImpl(store, engine, nil, zerowrap.Default()).WithDaemonContext(context.Background()) + t.Cleanup(func() { + unblock() + _ = svc.Shutdown(context.Background()) + }) + + handler := admin.NewHandler(admin.HandlerDeps{Log: zerowrap.Default(), AppSvc: svc}) + + reqCtx, cancelRequest := context.WithCancel( + context.WithValue(context.Background(), domain.ContextKeyScopes, []string{"admin:apps:write"})) + defer cancelRequest() + req := httptest.NewRequest(http.MethodPost, "/admin/apps/blog/deploy", nil) + req.Header.Set("Idempotency-Key", "key-1") + req = req.WithContext(reqCtx) + rec := httptest.NewRecorder() + + serveDone := make(chan struct{}) + go func() { + defer close(serveDone) + handler.ServeHTTP(rec, req) + }() + + var execCtx context.Context + select { + case execCtx = <-engine.started: + case <-time.After(3 * time.Second): + t.Fatal("background deploy execution never started") + } + select { + case <-serveDone: + case <-time.After(3 * time.Second): + t.Fatal("handler did not answer promptly") + } + require.Equal(t, http.StatusAccepted, rec.Code) + // The persisted journal is what the handler answers with: the client can + // poll the same op through operations/by-key. + var resp dto.AppDeployResponse + require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &resp)) + assert.Equal(t, "op-1", resp.Op) + assert.Equal(t, dto.AppStatusRunning, resp.Status) + + // The request is over and its context is cancelled: the daemon-owned + // execution must be unaffected. + cancelRequest() + select { + case <-execCtx.Done(): + t.Fatal("request cancellation aborted the daemon-owned execution") + case <-time.After(50 * time.Millisecond): + } +} diff --git a/internal/app/app_devices.go b/internal/app/app_devices.go new file mode 100644 index 000000000..da71079ed --- /dev/null +++ b/internal/app/app_devices.go @@ -0,0 +1,107 @@ +package app + +import ( + "fmt" + "strings" + + "github.com/spf13/viper" + + "github.com/bnema/gordon/internal/domain" +) + +// AppDevicePolicy is one administrative device grant declared under +// [app_devices.]. The map key is the stable logical device name an +// app manifest device references. CDI holds the explicit CDI device IDs +// granted (aggregate "=all" selectors are rejected). AllowedApps and +// AllowedServices are exact, non-empty allowlists. +type AppDevicePolicy struct { + CDI []string `mapstructure:"cdi"` + AllowedApps []string `mapstructure:"allowed_apps"` + AllowedServices []string `mapstructure:"allowed_services"` +} + +// knownAppDeviceKeys is the allowlist of [app_devices.] sub-keys. +// Unknown keys (for example a typo like "cdis") fail closed instead of +// silently dropping a grant. +var knownAppDeviceKeys = map[string]struct{}{ + "cdi": {}, + "allowed_apps": {}, + "allowed_services": {}, +} + +// validateAppDeviceKeys rejects unknown sub-keys inside [app_devices.*]. +// Viper unmarshals leniently, so a typo would otherwise drop the grant +// silently. settings is the decoded app_devices subtree (viper lowercases +// all keys) used only for key inspection: values still come from Config. +func validateAppDeviceKeys(devices map[string]any) error { + for name, raw := range devices { + entry, ok := raw.(map[string]any) + if !ok { + return fmt.Errorf("invalid app_devices.%s: entry must be a table", name) + } + for key := range entry { + if _, ok := knownAppDeviceKeys[key]; !ok { + return fmt.Errorf("invalid app_devices.%s: unknown key %q (allowed: cdi, allowed_apps, allowed_services)", name, key) + } + } + } + return nil +} + +// buildAppDevicePolicies converts configured app devices into validated +// domain policies. Failures name only the device and the field-level +// reason: host inventory never appears in config errors. +func buildAppDevicePolicies(cfg Config) (map[string]domain.AppDevicePolicy, error) { + policies := make(map[string]domain.AppDevicePolicy, len(cfg.AppDevices)) + for name, device := range cfg.AppDevices { + policy := domain.AppDevicePolicy{ + Name: name, + CDI: append([]string(nil), device.CDI...), + AllowedApps: append([]string(nil), device.AllowedApps...), + AllowedServices: append([]string(nil), device.AllowedServices...), + } + if err := policy.Validate(); err != nil { + return nil, fmt.Errorf("invalid app_devices.%s: %w: %s", name, domain.ErrDevicePolicy, redactedDevicePolicyReason(err)) + } + policies[name] = policy + } + return policies, nil +} + +// redactedDevicePolicyReason keeps the field-level reason from a domain +// device policy validation error while stripping every quoted value so an +// invalid config never echoes host inventory (CDI IDs, app names). +func redactedDevicePolicyReason(err error) string { + reason := policyQuotedValue.ReplaceAllString(err.Error(), "") + reason = strings.Join(strings.Fields(reason), " ") + // Keep only the innermost detail: the outer wrap repeats the generic + // violation and the policy name, which is already part of the device key. + if idx := strings.LastIndex(reason, ": "); idx >= 0 { + reason = reason[idx+2:] + } + return reason +} + +// validateDeviceConfig validates the [app_devices] subtree: unknown keys +// fail closed, then the decoded policies validate. Split from +// applyLoadedConfig to keep its complexity within budget. +func validateDeviceConfig(v *viper.Viper, cfg Config) error { + if err := validateAppDeviceKeys(v.GetStringMap("app_devices")); err != nil { + return err + } + if _, err := buildAppDevicePolicies(cfg); err != nil { + return err + } + return nil +} + +// validateAppPolicies validates both administrative app policy subtrees +// (bind mounts and device grants) before publication. Failed validation +// keeps the previous policies live. Split from applyLoadedConfig to keep +// its complexity within budget. +func validateAppPolicies(v *viper.Viper, cfg Config) error { + if _, err := buildAppMountPolicies(cfg); err != nil { + return err + } + return validateDeviceConfig(v, cfg) +} diff --git a/internal/app/app_devices_test.go b/internal/app/app_devices_test.go new file mode 100644 index 000000000..ffb88b74d --- /dev/null +++ b/internal/app/app_devices_test.go @@ -0,0 +1,118 @@ +package app + +import ( + "strings" + "testing" + + "github.com/spf13/viper" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +func TestBuildAppDevicePolicies_ValidConfigUnmarshal(t *testing.T) { + v := viper.New() + v.SetConfigType("toml") + require.NoError(t, v.ReadConfig(strings.NewReader(` +[app_devices.test_gpu] +cdi = ["example.com/gpu=GPU-test-uuid"] +allowed_apps = ["demo"] +allowed_services = ["worker"] +`))) + + var cfg Config + require.NoError(t, v.Unmarshal(&cfg)) + require.NoError(t, validateAppDeviceKeys(v.GetStringMap("app_devices"))) + + policies, err := buildAppDevicePolicies(cfg) + require.NoError(t, err) + require.Contains(t, policies, "test_gpu") + policy := policies["test_gpu"] + assert.Equal(t, "test_gpu", policy.Name) + assert.Equal(t, []string{"example.com/gpu=GPU-test-uuid"}, policy.CDI) + assert.Equal(t, []string{"demo"}, policy.AllowedApps) + assert.Equal(t, []string{"worker"}, policy.AllowedServices) +} + +func TestBuildAppDevicePolicies_Invalid(t *testing.T) { + cases := []struct { + name string + device AppDevicePolicy + reason string + }{ + {"empty cdi", AppDevicePolicy{AllowedApps: []string{"demo"}, AllowedServices: []string{"worker"}}, "cdi must not be empty"}, + {"raw path", AppDevicePolicy{CDI: []string{"/dev/card0"}, AllowedApps: []string{"demo"}, AllowedServices: []string{"worker"}}, "not a host path"}, + {"aggregate", AppDevicePolicy{CDI: []string{"example.com/gpu=all"}, AllowedApps: []string{"demo"}, AllowedServices: []string{"worker"}}, "not allowed"}, + {"empty allowed apps", AppDevicePolicy{CDI: []string{"example.com/gpu=0"}, AllowedServices: []string{"worker"}}, "allowed apps must not be empty"}, + {"empty allowed services", AppDevicePolicy{CDI: []string{"example.com/gpu=0"}, AllowedApps: []string{"demo"}}, "allowed services must not be empty"}, + {"duplicate cdi", AppDevicePolicy{CDI: []string{"example.com/gpu=0", "example.com/gpu=0"}, AllowedApps: []string{"demo"}, AllowedServices: []string{"worker"}}, "duplicate cdi device"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + cfg := Config{AppDevices: map[string]AppDevicePolicy{"test_gpu": tc.device}} + _, err := buildAppDevicePolicies(cfg) + require.Error(t, err) + assert.Contains(t, err.Error(), "app_devices.test_gpu") + assert.Contains(t, err.Error(), tc.reason, "the field-level reason must survive redaction") + assert.ErrorIs(t, err, domain.ErrDevicePolicy) + for _, id := range tc.device.CDI { + assert.NotContains(t, err.Error(), id, "config errors must not echo CDI IDs") + } + }) + } +} + +func TestValidateAppDeviceKeys_UnknownKey(t *testing.T) { + v := viper.New() + v.SetConfigType("toml") + require.NoError(t, v.ReadConfig(strings.NewReader(` +[app_devices.test_gpu] +cdis = ["example.com/gpu=0"] +allowed_apps = ["demo"] +allowed_services = ["worker"] +`))) + + err := validateAppDeviceKeys(v.GetStringMap("app_devices")) + require.Error(t, err) + assert.Contains(t, err.Error(), "unknown key") + assert.Contains(t, err.Error(), "app_devices.test_gpu") +} + +func TestValidateAppDeviceKeys_Valid(t *testing.T) { + v := viper.New() + v.SetConfigType("toml") + require.NoError(t, v.ReadConfig(strings.NewReader(` +[app_devices.test_gpu] +cdi = ["example.com/gpu=0"] +allowed_apps = ["demo"] +allowed_services = ["worker"] +`))) + + require.NoError(t, validateAppDeviceKeys(v.GetStringMap("app_devices"))) + require.NoError(t, validateAppDeviceKeys(nil)) +} + +func TestBuildAppDevicePolicies_CopiesSlices(t *testing.T) { + cfg := Config{AppDevices: map[string]AppDevicePolicy{ + "test_gpu": { + CDI: []string{"example.com/gpu=0"}, + AllowedApps: []string{"demo"}, + AllowedServices: []string{"worker"}, + }, + }} + policies, err := buildAppDevicePolicies(cfg) + require.NoError(t, err) + // Mutating the published policy must not affect later builds from the + // same config: reloads always rebuild from Config. + policies["test_gpu"].CDI[0] = "mutated" + again, err := buildAppDevicePolicies(cfg) + require.NoError(t, err) + assert.Equal(t, []string{"example.com/gpu=0"}, again["test_gpu"].CDI) +} + +func TestConfig_AppDevicesDefaultsEmpty(t *testing.T) { + policies, err := buildAppDevicePolicies(Config{}) + require.NoError(t, err) + assert.Empty(t, policies) +} diff --git a/internal/app/app_monitor.go b/internal/app/app_monitor.go new file mode 100644 index 000000000..65fc3b9bb --- /dev/null +++ b/internal/app/app_monitor.go @@ -0,0 +1,92 @@ +package app + +import ( + "context" + "sync" + "time" + + "github.com/bnema/zerowrap" + + "github.com/bnema/gordon/internal/boundaries/in" +) + +// appMonitorInterval is the fixed periodic recovery cadence. +const appMonitorInterval = 15 * time.Second + +// appMonitor is the daemon-owned driver of periodic app recovery. One +// instance lives for the whole process: the loop is created once after +// boot reconciliation and reload never starts another. Passes never +// overlap because a single goroutine runs them. +type appMonitor struct { + reconciler in.AppReconciler + interval time.Duration + // ticks lets tests drive the cadence deterministically. Nil uses a + // real ticker. + ticks func() <-chan time.Time + log zerowrap.Logger + + mu sync.Mutex + cancel context.CancelFunc + done chan struct{} +} + +func newAppMonitor(reconciler in.AppReconciler, log zerowrap.Logger) *appMonitor { + return &appMonitor{reconciler: reconciler, interval: appMonitorInterval, log: log} +} + +// Start launches the recovery loop under the daemon context. It is +// idempotent: a running monitor is left untouched. +func (m *appMonitor) Start(ctx context.Context) { + if m == nil || m.reconciler == nil { + return + } + m.mu.Lock() + defer m.mu.Unlock() + if m.cancel != nil { + return + } + runCtx, cancel := context.WithCancel(ctx) + m.cancel = cancel + done := make(chan struct{}) + m.done = done + go m.loop(runCtx, done) +} + +// Stop cancels the loop and joins it, so shutdown never races runtime, +// traffic, or state teardown. Safe to call more than once. +func (m *appMonitor) Stop() { + if m == nil { + return + } + m.mu.Lock() + cancel, done := m.cancel, m.done + m.cancel, m.done = nil, nil + m.mu.Unlock() + if cancel == nil { + return + } + cancel() + <-done +} + +func (m *appMonitor) loop(ctx context.Context, done chan struct{}) { + defer close(done) + var ticks <-chan time.Time + if m.ticks != nil { + ticks = m.ticks() + } else { + ticker := time.NewTicker(m.interval) + defer ticker.Stop() + ticks = ticker.C + } + for { + select { + case <-ctx.Done(): + return + case <-ticks: + if err := m.reconciler.ReconcileRunning(ctx); err != nil { + m.log.Warn().Err(err).Msg("app reconciliation pass failed") + } + } + } +} diff --git a/internal/app/app_monitor_test.go b/internal/app/app_monitor_test.go new file mode 100644 index 000000000..2293780fb --- /dev/null +++ b/internal/app/app_monitor_test.go @@ -0,0 +1,403 @@ +package app + +import ( + "context" + "sync" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/apptraffic" +) + +func httpAppSpec() domain.AppService { + return domain.AppService{ + Name: "web", + HTTP: []domain.AppHTTPInterface{{Host: "blog.example.com", Port: 8080, TLS: "auto"}}, + } +} + +// newPublisherServices wires the minimum app wiring the publisher needs: +// a state source, the ACTIVE-derived host index, and the activator. +func newPublisherServices(state *outmocks.MockAppState) (*services, *apptraffic.HostIndex) { + index := apptraffic.NewHostIndex() + return &services{ + appState: state, + appActivator: apptraffic.NewActivator(zerowrap.Default()), + appHostIndex: index, + }, index +} + +func TestAppTrafficPublisher_WithdrawServiceIsFailClosedInStateAndIndex(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + svc, index := newPublisherServices(state) + + served := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", Spec: httpAppSpec(), BackendBinds: map[int]int{8080: 32770}}, + }} + withdrawn := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", Spec: httpAppSpec()}, + }} + state.EXPECT().LoadActive(mock.Anything, "blog").Return(served, true, nil).Once() + state.EXPECT().SaveActive(mock.Anything, mock.MatchedBy(func(active domain.AppActive) bool { + service := active.Services["web"] + return service.BackendBinds == nil && service.UDPBackendBinds == nil + })).Return(nil).Once() + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(withdrawn, true, nil).Once() + + // Serve the host first so the withdrawal has something to remove. + require.NoError(t, svc.appActivator.RebuildHostIndex(ctx, index, stubAppState{active: served}, nil)) + backend, ok := index.LookupHost("blog.example.com") + require.True(t, ok) + require.True(t, backend.Resolved(), "precondition: the host resolves to a loopback backend") + + publisher := newAppTrafficPublisher(svc, Config{}) + require.NoError(t, publisher.WithdrawService(ctx, "blog", "web")) + + // The proxy fails closed on an unresolved backend (proxy/service.go + // resolveAppTarget), so an unresolved projection is a withdrawn one. + backend, ok = index.LookupHost("blog.example.com") + if ok { + assert.False(t, backend.Resolved(), "a withdrawn service must not resolve to a backend") + } +} + +func TestAppTrafficPublisher_WithdrawServiceSurfacesStateFailure(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + svc, index := newPublisherServices(state) + + served := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", Spec: httpAppSpec(), BackendBinds: map[int]int{8080: 32770}}, + }} + state.EXPECT().LoadActive(mock.Anything, "blog").Return(served, true, nil).Once() + state.EXPECT().SaveActive(mock.Anything, mock.Anything).Return(assert.AnError).Once() + + require.NoError(t, svc.appActivator.RebuildHostIndex(ctx, index, stubAppState{active: served}, nil)) + + publisher := newAppTrafficPublisher(svc, Config{}) + err := publisher.WithdrawService(ctx, "blog", "web") + require.Error(t, err) + assert.Contains(t, err.Error(), "persist fail-closed state") + // The projection could not be rebuilt, so stale forwarding may remain: + // the caller keeps its publication inhibition and retries. + _, ok := index.LookupHost("blog.example.com") + assert.True(t, ok) +} + +func TestAppTrafficPublisher_WithdrawUnknownServiceIsANoop(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + svc, _ := newPublisherServices(state) + + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{App: "blog"}, true, nil).Once() + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{App: "blog"}, true, nil).Once() + + publisher := newAppTrafficPublisher(svc, Config{}) + require.NoError(t, publisher.WithdrawService(ctx, "blog", "web")) +} + +func TestAppTrafficPublisher_RebuildTrafficReprojectsVerifiedState(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + svc, index := newPublisherServices(state) + + served := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", Spec: httpAppSpec(), BackendBinds: map[int]int{8080: 32770}}, + }} + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(served, true, nil).Once() + + publisher := newAppTrafficPublisher(svc, Config{}) + require.NoError(t, publisher.RebuildTraffic(ctx)) + + backend, ok := index.LookupHost("blog.example.com") + require.True(t, ok) + assert.Equal(t, 32770, backend.Port) +} + +func TestAppTrafficPublisher_SerializesConcurrentPublication(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + svc, index := newPublisherServices(state) + + served := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", Spec: httpAppSpec(), BackendBinds: map[int]int{8080: 32770}}, + }} + state.EXPECT().LoadActive(mock.Anything, "blog").Return(served, true, nil) + state.EXPECT().SaveActive(mock.Anything, mock.Anything).Return(nil) + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil) + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil) + + publisher := newAppTrafficPublisher(svc, Config{}) + done := make(chan struct{}) + for range 4 { + go func() { + defer func() { done <- struct{}{} }() + _ = publisher.RebuildTraffic(ctx) + _ = publisher.WithdrawService(ctx, "blog", "web") + }() + } + for range 4 { + <-done + } + assert.NotNil(t, index) +} + +// stubAppState is a read-only app state double for index projections. +// The zero value reports running intent. +type stubAppState struct { + active domain.AppActive + stopped bool +} + +func (s stubAppState) ListApps(context.Context) ([]string, error) { + return []string{s.active.App}, nil +} + +func (s stubAppState) LoadActive(context.Context, string) (domain.AppActive, bool, error) { + return s.active, true, nil +} + +func (s stubAppState) LoadIntent(context.Context, string) (domain.AppStopIntent, error) { + return domain.AppStopIntent{App: s.active.App, Stopped: s.stopped}, nil +} + +func TestAppTrafficPublisher_WithdrawalIsIdempotent(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + svc, _ := newPublisherServices(state) + + alreadyWithdrawn := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", Spec: httpAppSpec()}, + }} + state.EXPECT().LoadActive(mock.Anything, "blog").Return(alreadyWithdrawn, true, nil).Once() + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(alreadyWithdrawn, true, nil).Once() + + publisher := newAppTrafficPublisher(svc, Config{}) + require.NoError(t, publisher.WithdrawService(ctx, "blog", "web")) +} + +func TestAppMonitor_RunsOnePassPerTick(t *testing.T) { + reconciler := &fakeReconciler{} + monitor := newAppMonitor(reconciler, zerowrap.Default()) + ticks := make(chan time.Time, 8) + monitor.ticks = func() <-chan time.Time { return ticks } + monitor.Start(context.Background()) + + ticks <- time.Now() + require.Eventually(t, func() bool { return reconciler.callCount() == 1 }, time.Second, time.Millisecond) + ticks <- time.Now() + require.Eventually(t, func() bool { return reconciler.callCount() == 2 }, time.Second, time.Millisecond) + + monitor.Stop() + assert.Equal(t, 2, reconciler.callCount()) +} + +func TestAppMonitor_DoesNotOverlapPasses(t *testing.T) { + reconciler := &fakeReconciler{enter: make(chan struct{}), release: make(chan struct{})} + monitor := newAppMonitor(reconciler, zerowrap.Default()) + ticks := make(chan time.Time, 8) + monitor.ticks = func() <-chan time.Time { return ticks } + monitor.Start(context.Background()) + + ticks <- time.Now() + <-reconciler.entered() + // Additional ticks while a pass is in flight queue at most one + // further pass, never a concurrent one. + for range 3 { + ticks <- time.Now() + } + assert.Equal(t, 1, reconciler.callCount()) + assert.Equal(t, 1, reconciler.maxConcurrent()) + + close(reconciler.release) + monitor.Stop() + assert.GreaterOrEqual(t, reconciler.callCount(), 1) +} + +func TestAppMonitor_ReconciliationErrorDoesNotStopTheLoop(t *testing.T) { + reconciler := &fakeReconciler{err: assert.AnError} + monitor := newAppMonitor(reconciler, zerowrap.Default()) + ticks := make(chan time.Time, 8) + monitor.ticks = func() <-chan time.Time { return ticks } + monitor.Start(context.Background()) + + ticks <- time.Now() + ticks <- time.Now() + require.Eventually(t, func() bool { return reconciler.callCount() == 2 }, time.Second, time.Millisecond) + monitor.Stop() +} + +func TestAppMonitor_StartIsIdempotent(t *testing.T) { + reconciler := &fakeReconciler{} + monitor := newAppMonitor(reconciler, zerowrap.Default()) + ticks := make(chan time.Time, 8) + monitor.ticks = func() <-chan time.Time { return ticks } + + monitor.Start(context.Background()) + monitor.Start(context.Background()) + monitor.Start(context.Background()) + + ticks <- time.Now() + require.Eventually(t, func() bool { return reconciler.callCount() == 1 }, time.Second, time.Millisecond) + monitor.Stop() + assert.Equal(t, 1, reconciler.callCount(), "reload must never create a second loop") +} + +func TestAppMonitor_StopCancelsAndJoins(t *testing.T) { + blocking := &fakeReconciler{enter: make(chan struct{}), release: make(chan struct{})} + monitor := newAppMonitor(blocking, zerowrap.Default()) + ticks := make(chan time.Time, 8) + monitor.ticks = func() <-chan time.Time { return ticks } + monitor.Start(context.Background()) + + ticks <- time.Now() + <-blocking.entered() + + stopped := make(chan struct{}) + go func() { + monitor.Stop() + close(stopped) + }() + select { + case <-stopped: + t.Fatal("Stop returned while a bounded pass was still running") + case <-time.After(30 * time.Millisecond): + } + close(blocking.release) + select { + case <-stopped: + case <-time.After(time.Second): + t.Fatal("Stop never joined the loop") + } + monitor.Stop() // idempotent +} + +func TestAppMonitor_NilReconcilerNeverStarts(t *testing.T) { + monitor := newAppMonitor(nil, zerowrap.Default()) + monitor.Start(context.Background()) + monitor.Stop() +} + +// fakeReconciler records passes and can hold one open. +type fakeReconciler struct { + mu sync.Mutex + calls int + inflight int + maxInflight int + enter chan struct{} + release chan struct{} + enteredOnce sync.Once + err error +} + +func (f *fakeReconciler) ReconcileRunning(context.Context) error { + f.mu.Lock() + f.calls++ + f.inflight++ + if f.inflight > f.maxInflight { + f.maxInflight = f.inflight + } + f.mu.Unlock() + if f.enter != nil { + f.enteredOnce.Do(func() { close(f.enter) }) + <-f.release + } + f.mu.Lock() + f.inflight-- + f.mu.Unlock() + return f.err +} + +func (f *fakeReconciler) callCount() int { + f.mu.Lock() + defer f.mu.Unlock() + return f.calls +} + +func (f *fakeReconciler) maxConcurrent() int { + f.mu.Lock() + defer f.mu.Unlock() + return f.maxInflight +} + +func (f *fakeReconciler) entered() <-chan struct{} { return f.enter } + +// TestAppTrafficPublisher_RebuildSkipsStoppedApps proves durable stopped +// intent is honored by every publication, including the plain rebuild a +// stopped-app convergence pass performs: a stopped service is never +// projected back into the proxy index. +func TestAppTrafficPublisher_RebuildSkipsStoppedApps(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + svc, index := newPublisherServices(state) + + served := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", Spec: httpAppSpec(), BackendBinds: map[int]int{8080: 32770}}, + }} + // Precondition: the host is routable while the app runs. + require.NoError(t, svc.appActivator.RebuildHostIndex(ctx, index, stubAppState{active: served}, nil)) + backend, ok := index.LookupHost("blog.example.com") + require.True(t, ok) + require.True(t, backend.Resolved()) + + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog"). + Return(domain.AppStopIntent{App: "blog", Stopped: true}, nil).Once() + + publisher := newAppTrafficPublisher(svc, Config{}) + require.NoError(t, publisher.RebuildTraffic(ctx)) + + backend, ok = index.LookupHost("blog.example.com") + if ok { + assert.False(t, backend.Resolved(), "a stopped app must not stay routable") + } +} + +// TestAppTrafficPublisher_AdmissionHonorsContextCancellation proves a +// caller whose deadline expired never waits behind an in-flight +// publication: a monitor pass must not hold its app lock for a graph +// application someone else started, and shutdown must not block on it. +func TestAppTrafficPublisher_AdmissionHonorsContextCancellation(t *testing.T) { + state := outmocks.NewMockAppState(t) + svc, _ := newPublisherServices(state) + + release := make(chan struct{}) + entered := make(chan struct{}) + state.EXPECT().ListApps(mock.Anything).RunAndReturn(func(context.Context) ([]string, error) { + close(entered) + <-release + return nil, nil + }).Once() + + publisher := newAppTrafficPublisher(svc, Config{}) + holding := make(chan struct{}) + go func() { + defer close(holding) + _ = publisher.RebuildTraffic(context.Background()) + }() + <-entered + + expired, cancel := context.WithCancel(context.Background()) + cancel() + err := publisher.RebuildTraffic(expired) + require.ErrorIs(t, err, context.Canceled) + + close(release) + <-holding +} diff --git a/internal/app/app_mounts.go b/internal/app/app_mounts.go new file mode 100644 index 000000000..1ffeb3c2a --- /dev/null +++ b/internal/app/app_mounts.go @@ -0,0 +1,70 @@ +package app + +import ( + "fmt" + "path/filepath" + "regexp" + "strings" + + "github.com/bnema/gordon/internal/domain" +) + +// AppMountPolicy is one administrative bind policy declared under +// [[app_mounts.]]. The map key is the stable mount name an app manifest +// bind references. Source is the only host path a bind may come from. Root, +// when set, is the administrative boundary the resolved source must stay +// under. AllowedApps and AllowedServices are exact, non-empty allowlists. +type AppMountPolicy struct { + Source string `mapstructure:"source"` + ReadOnly bool `mapstructure:"read_only"` + AllowedApps []string `mapstructure:"allowed_apps"` + AllowedServices []string `mapstructure:"allowed_services"` + Root string `mapstructure:"root"` +} + +// policyQuotedValue matches the quoted values domain validation embeds in +// its messages, such as configured source paths or roots. +var policyQuotedValue = regexp.MustCompile(`"[^"]*"`) + +// redactedBindPolicyReason keeps the field-level reason from a domain policy +// validation error (for example "source must be absolute") while stripping +// every quoted value so an invalid config never echoes host paths. +func redactedBindPolicyReason(err error) string { + reason := policyQuotedValue.ReplaceAllString(err.Error(), "") + reason = strings.Join(strings.Fields(reason), " ") + // Keep only the innermost detail: the outer wrap repeats the generic + // violation and the policy name, which is already part of the mount key. + if idx := strings.LastIndex(reason, ": "); idx >= 0 { + reason = reason[idx+2:] + } + return reason +} + +// buildAppMountPolicies converts configured app mounts into validated domain +// policies. Failures name only the mount and the field-level reason: source +// paths are never echoed. +func buildAppMountPolicies(cfg Config) (map[string]domain.AppBindPolicy, error) { + policies := make(map[string]domain.AppBindPolicy, len(cfg.AppMounts)) + for name, mount := range cfg.AppMounts { + root := mount.Root + if root == "" && filepath.IsAbs(mount.Source) { + // The source's parent is the least surprising implicit + // administrative boundary for the documented minimal form. + // A symlink source may resolve within it, never escape it. + root = filepath.Dir(mount.Source) + } + policy := domain.AppBindPolicy{ + Name: name, + Source: mount.Source, + ReadOnly: mount.ReadOnly, + AllowedApps: mount.AllowedApps, + AllowedServices: mount.AllowedServices, + Root: root, + } + if err := policy.Validate(); err != nil { + return nil, fmt.Errorf("invalid app_mounts.%s: %w: %s", name, domain.ErrBindPolicy, redactedBindPolicyReason(err)) + } + policies[name] = policy + } + return policies, nil +} diff --git a/internal/app/app_mounts_test.go b/internal/app/app_mounts_test.go new file mode 100644 index 000000000..040b14e7a --- /dev/null +++ b/internal/app/app_mounts_test.go @@ -0,0 +1,103 @@ +package app + +import ( + "strings" + "testing" + + "github.com/spf13/viper" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +func TestBuildAppMountPolicies_ValidConfigUnmarshal(t *testing.T) { + v := viper.New() + v.SetConfigType("toml") + require.NoError(t, v.ReadConfig(strings.NewReader(` +[app_mounts.config] +source = "/srv/gordon/binds/blog/config" +read_only = true +allowed_apps = ["blog", "docs"] +allowed_services = ["web"] +root = "/srv/gordon" +`))) + + var cfg Config + require.NoError(t, v.Unmarshal(&cfg)) + + policies, err := buildAppMountPolicies(cfg) + require.NoError(t, err) + require.Contains(t, policies, "config") + policy := policies["config"] + assert.Equal(t, "config", policy.Name) + assert.Equal(t, "/srv/gordon/binds/blog/config", policy.Source) + assert.True(t, policy.ReadOnly) + assert.Equal(t, []string{"blog", "docs"}, policy.AllowedApps) + assert.Equal(t, []string{"web"}, policy.AllowedServices) + assert.Equal(t, "/srv/gordon", policy.Root) +} + +func TestBuildAppMountPolicies_Invalid(t *testing.T) { + cases := []struct { + name string + mount AppMountPolicy + reason string + }{ + {"empty source", AppMountPolicy{AllowedApps: []string{"blog"}, AllowedServices: []string{"web"}}, "source must not be empty"}, + {"relative source", AppMountPolicy{Source: "binds", AllowedApps: []string{"blog"}, AllowedServices: []string{"web"}}, "source must be absolute"}, + {"empty allowed apps", AppMountPolicy{Source: "/srv/binds", AllowedServices: []string{"web"}}, "allowed apps must not be empty"}, + {"empty allowed services", AppMountPolicy{Source: "/srv/binds", AllowedApps: []string{"blog"}}, "allowed services must not be empty"}, + {"unclean root", AppMountPolicy{Source: "/srv/binds", AllowedApps: []string{"blog"}, AllowedServices: []string{"web"}, Root: "/srv/../etc"}, "root must be a normalized clean path"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + cfg := Config{AppMounts: map[string]AppMountPolicy{"config": tc.mount}} + _, err := buildAppMountPolicies(cfg) + require.Error(t, err) + assert.Contains(t, err.Error(), "app_mounts.config") + assert.Contains(t, err.Error(), tc.reason, "the field-level reason must survive redaction") + assert.ErrorIs(t, err, domain.ErrBindPolicy) + if tc.mount.Source != "" { + assert.NotContains(t, err.Error(), tc.mount.Source, "config errors must not echo the source path") + } + if tc.mount.Root != "" { + assert.NotContains(t, err.Error(), tc.mount.Root, "config errors must not echo the root path") + } + }) + } +} + +func TestRedactedBindPolicyReason_RedactsConfiguredValues(t *testing.T) { + _, err := buildAppMountPolicies(Config{AppMounts: map[string]AppMountPolicy{ + "config": { + Source: "/srv/binds/blog/config", + AllowedApps: []string{"Blog"}, + AllowedServices: []string{"web"}, + }, + }}) + + require.Error(t, err) + assert.Contains(t, err.Error(), "app name must be a DNS label") + assert.NotContains(t, err.Error(), "/srv/binds/blog/config", "configured paths must never be echoed") + assert.NotContains(t, err.Error(), "Blog", "configured names must never be echoed") +} + +func TestBuildAppMountPolicies_DefaultRootIsSourceParent(t *testing.T) { + cfg := Config{AppMounts: map[string]AppMountPolicy{ + "config": { + Source: "/srv/binds/config", + AllowedApps: []string{"blog"}, + AllowedServices: []string{"web"}, + }, + }} + policies, err := buildAppMountPolicies(cfg) + require.NoError(t, err) + assert.Equal(t, "/srv/binds", policies["config"].Root) +} + +func TestConfig_AppMountsDefaultsEmpty(t *testing.T) { + policies, err := buildAppMountPolicies(Config{}) + require.NoError(t, err) + assert.Empty(t, policies) +} diff --git a/internal/app/app_traffic.go b/internal/app/app_traffic.go new file mode 100644 index 000000000..3e3cf7e6f --- /dev/null +++ b/internal/app/app_traffic.go @@ -0,0 +1,189 @@ +package app + +import ( + "context" + "fmt" + + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/usecase/apptraffic" +) + +// appTrafficPublisher is the single serialized HTTP/L4 publication +// boundary for apps. Activation rebuilds, fail-closed recovery +// withdrawal, and config reload all take the same mutex: the host index +// replacement and the runtime graph application always run as one +// operation, so no caller can observe an index that disagrees with the +// applied graph, and a stale config can never win a race. +type appTrafficPublisher struct { + // gate is a one-slot semaphore instead of a mutex so a waiting + // caller can abandon the wait when its context ends: a monitor pass + // must never hold an app lock for a full graph application that a + // sibling publication started, and shutdown must not block behind + // it. + gate chan struct{} + svc *services + cfg Config + // applyGraph applies one prepared configuration. The publication + // projects and validates a candidate against this call and swaps the + // live host index only after it returns nil, so a rejected graph can + // never leave the index ahead of the dataplane. It is a field so + // publication ordering is testable without a live traffic manager. + applyGraph func(ctx context.Context, cfg Config, hosts appHostRoutes) error +} + +var _ out.AppTrafficRefresher = (*appTrafficPublisher)(nil) + +// newAppTrafficPublisher wires the boundary. The installation config is +// stored so activation rebuilds use the same validated snapshot that +// reload applied. +func newAppTrafficPublisher(svc *services, cfg Config) *appTrafficPublisher { + publisher := &appTrafficPublisher{gate: make(chan struct{}, 1), svc: svc, cfg: cfg} + publisher.applyGraph = func(ctx context.Context, cfg Config, hosts appHostRoutes) error { + return applyTrafficRuntimeConfig(ctx, svc.trafficManager, cfg, svc.configSvc, hosts) + } + publisher.gate <- struct{}{} + return publisher +} + +// acquire waits for the publication gate or the caller's deadline. +func (p *appTrafficPublisher) acquire(ctx context.Context) error { + select { + case <-ctx.Done(): + return ctx.Err() + case <-p.gate: + return nil + } +} + +// release returns the publication gate to the next caller. +func (p *appTrafficPublisher) release() { + p.gate <- struct{}{} +} + +// RebuildTraffic re-projects verified ACTIVE state and applies the full +// HTTP/L4 graph. +func (p *appTrafficPublisher) RebuildTraffic(ctx context.Context) error { + if err := p.acquire(ctx); err != nil { + return fmt.Errorf("traffic publisher: %w", err) + } + defer p.release() + return p.rebuild(ctx, p.cfg) +} + +// RebuildWithConfig applies the current validated config through the +// same serialization point. Startup and reload call it so they cannot +// interleave with a recovery withdrawal. +func (p *appTrafficPublisher) RebuildWithConfig(ctx context.Context, cfg Config) error { + if err := p.acquire(ctx); err != nil { + return fmt.Errorf("traffic publisher: %w", err) + } + defer p.release() + p.cfg = cfg + return p.rebuild(ctx, cfg) +} + +// WithdrawService makes one service non-forwardable and republishes. +// The durable ACTIVE record loses the service's binds first: every +// later projection is then fail-closed even if the graph application +// itself fails. A non-nil result means stale forwarding may remain and +// the caller must retain its publication inhibition. +// WithdrawServiceState is the canonical state-only ACTIVE bind +// withdrawal: it clears the recorded binds through the same +// serialization point as WithdrawService but applies no graph. +func (p *appTrafficPublisher) WithdrawServiceState(ctx context.Context, app, service string) error { + if err := p.acquire(ctx); err != nil { + return fmt.Errorf("traffic publisher: %w", err) + } + defer p.release() + return p.withdrawFromState(ctx, app, service) +} + +func (p *appTrafficPublisher) WithdrawService(ctx context.Context, app, service string) error { + if err := p.acquire(ctx); err != nil { + return fmt.Errorf("traffic publisher: %w", err) + } + defer p.release() + if err := p.withdrawFromState(ctx, app, service); err != nil { + return err + } + err := p.rebuild(ctx, p.cfg) + if err == nil { + return nil + } + // The graph apply failed while the durable ACTIVE record already + // dropped the service's binds. The read model still has to follow: a + // stale index entry could dial a loopback port that has been + // recycled by another workload, so the proxy must fail closed. The + // error is returned unchanged so the caller keeps its publication + // inhibition and retries the graph. + if candidate, projErr := p.projectCandidate(ctx, p.cfg); projErr == nil && candidate != nil { + p.svc.appHostIndex.Replace(candidate.Entries()) + } + return err +} + +// withdrawFromState drops the recorded binds of one service in ACTIVE. +// The container itself is untouched: withdrawal is a traffic decision, +// never a workload mutation. +func (p *appTrafficPublisher) withdrawFromState(ctx context.Context, app, service string) error { + if p.svc.appState == nil { + return nil + } + active, ok, err := p.svc.appState.LoadActive(ctx, app) + if err != nil { + return fmt.Errorf("withdraw %s/%s: load active: %w", app, service, err) + } + if !ok { + return nil + } + current, ok := active.Services[service] + if !ok || (len(current.BackendBinds) == 0 && len(current.UDPBackendBinds) == 0) { + return nil + } + current.BackendBinds = nil + current.UDPBackendBinds = nil + active.Services[service] = current + if err := p.svc.appState.SaveActive(ctx, active); err != nil { + return fmt.Errorf("withdraw %s/%s: persist fail-closed state: %w", app, service, err) + } + return nil +} + +// rebuild projects ACTIVE into a candidate host view, applies the full +// graph against that candidate, and publishes the candidate only after +// the dataplane accepted it. The caller holds the publication gate. On +// any failure the live index and the applied graph both stay at their +// previous generation: the proxy never serves routing the dataplane +// rejected, and no retirement ever follows an unpublished switch. +func (p *appTrafficPublisher) rebuild(ctx context.Context, cfg Config) error { + candidate, err := p.projectCandidate(ctx, cfg) + if err != nil { + return err + } + var hosts appHostRoutes + if candidate != nil { + hosts = candidate + } + if err := p.applyGraph(ctx, cfg, hosts); err != nil { + return fmt.Errorf("apply traffic graph: %w", err) + } + if candidate != nil { + p.svc.appHostIndex.Replace(candidate.Entries()) + } + return nil +} + +// projectCandidate builds the detached host view for one publication. +// The candidate never touches the live index: a failing projection leaves +// what the proxy serves untouched. It returns nil when app state is not +// wired, in which case the graph carries configuration routes only. +func (p *appTrafficPublisher) projectCandidate(ctx context.Context, cfg Config) (*apptraffic.HostIndex, error) { + if p.svc.appActivator == nil || p.svc.appState == nil || p.svc.appHostIndex == nil { + return nil, nil + } + candidate := apptraffic.NewHostIndex() + if err := p.svc.appActivator.RebuildHostIndex(ctx, candidate, p.svc.appState, appEntrypointPolicies(cfg)); err != nil { + return nil, fmt.Errorf("project app host index: %w", err) + } + return candidate, nil +} diff --git a/internal/app/app_traffic_publication_test.go b/internal/app/app_traffic_publication_test.go new file mode 100644 index 000000000..8da7a6c52 --- /dev/null +++ b/internal/app/app_traffic_publication_test.go @@ -0,0 +1,180 @@ +package app + +import ( + "context" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/apptraffic" +) + +// servedActive builds an ACTIVE record for one served host with a +// resolved loopback backend. +func servedActive(container string, port int) domain.AppActive { + return domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: container, Spec: httpAppSpec(), BackendBinds: map[int]int{8080: port}}, + }} +} + +// TestAppTrafficPublisher_PublishesIndexOnlyAfterGraphApply proves the +// prepare/commit contract: the graph is built and applied against a +// detached candidate while the live index still serves the previous +// generation, and only a successful apply publishes the candidate. +func TestAppTrafficPublisher_PublishesIndexOnlyAfterGraphApply(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + svc, index := newPublisherServices(state) + + first := servedActive("c-1", 32770) + second := servedActive("c-2", 32771) + + // First generation: one successful publication. + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(first, true, nil).Once() + + publisher := newAppTrafficPublisher(svc, Config{}) + require.NoError(t, publisher.RebuildTraffic(ctx)) + backend, ok := index.LookupHost("blog.example.com") + require.True(t, ok) + assert.Equal(t, 32770, backend.Port) + + // Second generation: the applier observes the candidate and the live + // index side by side. + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(second, true, nil).Once() + + applies := 0 + publisher.applyGraph = func(_ context.Context, _ Config, hosts appHostRoutes) error { + applies++ + require.NotNil(t, hosts, "the graph is built against the candidate view") + require.Len(t, hosts.AppHosts(), 1) + // The candidate carries the new backend while the live index + // still carries the old one: nothing is published yet. + live, ok := index.LookupHost("blog.example.com") + require.True(t, ok) + assert.Equal(t, 32770, live.Port, "the live index must not change before the graph apply succeeds") + candidateHosts := hosts.(*apptraffic.HostIndex) + candidateBackend, ok := candidateHosts.LookupHost("blog.example.com") + require.True(t, ok) + assert.Equal(t, 32771, candidateBackend.Port) + return nil + } + + require.NoError(t, publisher.RebuildTraffic(ctx)) + assert.Equal(t, 1, applies) + backend, ok = index.LookupHost("blog.example.com") + require.True(t, ok) + assert.Equal(t, 32771, backend.Port, "the accepted candidate is published exactly once") +} + +// TestAppTrafficPublisher_GraphFailureKeepsPreviousGeneration proves a +// rejected graph leaves the live index exactly where it was. +func TestAppTrafficPublisher_GraphFailureKeepsPreviousGeneration(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + svc, index := newPublisherServices(state) + + first := servedActive("c-1", 32770) + second := servedActive("c-2", 32771) + + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(first, true, nil).Once() + + publisher := newAppTrafficPublisher(svc, Config{}) + require.NoError(t, publisher.RebuildTraffic(ctx)) + + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(second, true, nil).Once() + + publisher.applyGraph = func(context.Context, Config, appHostRoutes) error { + return assert.AnError + } + + require.ErrorIs(t, publisher.RebuildTraffic(ctx), assert.AnError) + backend, ok := index.LookupHost("blog.example.com") + require.True(t, ok) + assert.Equal(t, 32770, backend.Port, "a rejected graph leaves the previous generation published") +} + +// TestAppTrafficPublisher_WithdrawalStaysFailClosedWhenGraphApplyFails +// proves a rejected graph cannot leave a withdrawn backend published: the +// durable binds are already gone, so the read model must drop them even +// though the error is reported for a retry. +func TestAppTrafficPublisher_WithdrawalStaysFailClosedWhenGraphApplyFails(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + svc, index := newPublisherServices(state) + + served := servedActive("c-1", 32770) + withdrawn := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", Spec: httpAppSpec()}, + }} + + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(served, true, nil).Once() + + publisher := newAppTrafficPublisher(svc, Config{}) + require.NoError(t, publisher.RebuildTraffic(ctx)) + _, ok := index.LookupHost("blog.example.com") + require.True(t, ok) + + // The withdrawal clears the binds, then the graph apply is rejected. + state.EXPECT().LoadActive(mock.Anything, "blog").Return(served, true, nil).Once() + state.EXPECT().SaveActive(mock.Anything, mock.Anything).Return(nil).Once() + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(withdrawn, true, nil).Once() + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(withdrawn, true, nil).Once() + + publisher.applyGraph = func(context.Context, Config, appHostRoutes) error { + return assert.AnError + } + + require.Error(t, publisher.WithdrawService(ctx, "blog", "web")) + backend, ok := index.LookupHost("blog.example.com") + if ok { + assert.False(t, backend.Resolved(), "a withdrawn service must not keep a resolved backend") + } +} + +// TestAppTrafficPublisher_ProjectionFailureAppliesNoGraph proves a +// failing projection never reaches the dataplane: the graph apply is not +// called at all and the live index is untouched. +func TestAppTrafficPublisher_ProjectionFailureAppliesNoGraph(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + svc, index := newPublisherServices(state) + + first := servedActive("c-1", 32770) + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(first, true, nil).Once() + + publisher := newAppTrafficPublisher(svc, Config{}) + require.NoError(t, publisher.RebuildTraffic(ctx)) + + state.EXPECT().ListApps(mock.Anything).Return(nil, assert.AnError).Once() + applied := false + publisher.applyGraph = func(context.Context, Config, appHostRoutes) error { + applied = true + return nil + } + + require.Error(t, publisher.RebuildTraffic(ctx)) + assert.False(t, applied, "a failed projection must not reach the dataplane") + backend, ok := index.LookupHost("blog.example.com") + require.True(t, ok) + assert.Equal(t, 32770, backend.Port) +} diff --git a/internal/app/app_wiring.go b/internal/app/app_wiring.go new file mode 100644 index 000000000..614423adc --- /dev/null +++ b/internal/app/app_wiring.go @@ -0,0 +1,252 @@ +package app + +import ( + "context" + "fmt" + "net" + "strconv" + + "github.com/bnema/zerowrap" + + "github.com/bnema/gordon/internal/adapters/out/appsecrets" + "github.com/bnema/gordon/internal/adapters/out/appstate" + "github.com/bnema/gordon/internal/adapters/out/httpprober" + "github.com/bnema/gordon/internal/adapters/out/imageref" + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/apps" + "github.com/bnema/gordon/internal/usecase/apptraffic" + "github.com/bnema/gordon/internal/usecase/deployment" + "github.com/bnema/gordon/internal/usecase/health" +) + +// appDeployEngine is the deployment-engine subset the daemon-owned app +// service executes through. It mirrors the usecase's own port so the +// composition root stays decoupled from engine internals. +type appDeployEngine interface { + StartDeploy(ctx context.Context, input deployment.DeployInput) (*deployment.StartDeployResult, error) + AbandonDeploy(ctx context.Context, claim deployment.DeployClaim) error + ExecuteDeploy(ctx context.Context, claim deployment.DeployClaim) (*deployment.DeployResult, error) + Stop(ctx context.Context, app, opID string) (*deployment.LifecycleResult, error) + Start(ctx context.Context, app, opID string) (*deployment.LifecycleResult, error) + Restart(ctx context.Context, app, service, opID string) (*deployment.LifecycleResult, error) + Remove(ctx context.Context, app, opID string) (*deployment.LifecycleResult, error) +} + +// newAppDaemonService builds the daemon-owned app administration service. +// ctx is the daemon/supervisor lifecycle context: background deploy +// executions derive from it and are cancelled by Shutdown during graceful +// teardown. It is deliberately never an HTTP request context, so request +// cancellation cannot abort an in-flight replacement. +func newAppDaemonService(ctx context.Context, store out.AppState, deploy appDeployEngine, secrets out.SecretWriter, log zerowrap.Logger) *apps.AppServiceImpl { + return apps.NewAppServiceImpl(store, deploy, secrets, log).WithDaemonContext(ctx) +} + +// initApps wires the single v3 app engine: bbolt state, image digests, +// pass-backed secrets, deployment/lifecycle execution, and traffic +// activation. The daemon is the sole app-state writer; CLI reaches the +// engine only through the admin /apps surface. +func (si *serviceInit) initApps() error { + dataDir := resolveDataDir(si.cfg.Server.DataDir) + store, err := appstate.NewStore(dataDir, si.log) + if err != nil { + return si.log.WrapErr(err, "failed to open app state") + } + store.WithRevisionRetention(si.cfg.Apps.RevisionRetention) + si.svc.appState = store + observeManagedContainers(si.svc.metrics, store, si.log) + // One process-wide GC barrier owns the ordering between resource + // acquisition/publication and destructive prune. + si.svc.gcBarrier = newGCBarrier() + if si.svc.backupSvc != nil { + si.svc.backupSvc.WithAppState(store) + } + if si.svc.volumeBackupSvc != nil { + si.svc.volumeBackupSvc.WithAppState(store) + } + if si.svc.imageSvc != nil { + si.svc.imageSvc.WithPrunePorts(store, si.svc.runtime, si.svc.gcBarrier) + } + if si.svc.volumeSvc != nil { + si.svc.volumeSvc.WithPrunePorts(store, si.svc.runtime, si.svc.gcBarrier) + } + registryDomain := si.svc.configSvc.GetRegistryDomain() + imagePolicy := domain.ImageSourcePolicy{ + AllowedRegistries: si.cfg.Images.AllowedRegistries, + RequireDigest: si.cfg.Images.RequireDigest, + InstallationRegistry: registryDomain, + } + if err := imagePolicy.Validate(); err != nil { + return fmt.Errorf("invalid image registry policy: %w", err) + } + resolver := imageref.NewResolver(registryDomain, si.svc.manifestStorage, nil).WithPolicy(imagePolicy) + // The publisher is the single serialized HTTP/L4 boundary: raw + // activations, fail-closed recovery withdrawal, and reload all go + // through it. It resolves writers/permission paths late-bound. + si.svc.appTrafficPublisher = newAppTrafficPublisher(si.svc, si.cfg) + limits, err := containerResourceLimits(si.cfg) + if err != nil { + return err + } + si.svc.appDeploySvc = deployment.NewService(deployment.Deps{ + State: store, + Runtime: si.svc.runtime, + Images: resolver, + Secrets: si.svc.serviceSecretProvider, + Registry: deployment.RegistryConfig{ + Domain: registryDomain, + PullAddress: net.JoinHostPort("127.0.0.1", strconv.Itoa(si.svc.configSvc.GetRegistryPort())), + Username: si.svc.internalRegUser, + Password: si.svc.internalRegPass, + }, + Networks: deployment.NetworkConfig{ + Prefix: si.cfg.NetworkIsolation.Prefix, + Internal: si.cfg.NetworkIsolation.Internal, + }, + Limits: limits, + ImagePolicy: imagePolicy, + Traffic: si.svc.appTrafficPublisher, + }, si.log).WithGCBarrier(si.svc.gcBarrier) + si.svc.appActivator = apptraffic.NewActivator(si.log) + mountPolicies, err := buildAppMountPolicies(si.cfg) + if err != nil { + return err + } + devicePolicies, err := buildAppDevicePolicies(si.cfg) + if err != nil { + return err + } + si.svc.appDeploySvc.SetBindPolicies(mountPolicies) + si.svc.appDeploySvc.SetDevicePolicies(devicePolicies) + appSvcImpl := newAppDaemonService(si.ctx, store, si.svc.appDeploySvc, appsecrets.NewStore(si.log), si.log). + WithEntrypoints(appEntrypointListeners(si.cfg)). + WithGCBarrier(si.svc.gcBarrier). + WithImagePolicy(imagePolicy). + WithBindPolicies(mountPolicies). + WithDevicePolicies(devicePolicies) + si.svc.appSvcImpl = appSvcImpl + si.svc.appSvc = appSvcImpl + // Health checks resolve from ACTIVE state (loopback backends). + si.svc.healthSvc = health.NewService(store, si.svc.runtime, httpprober.New(), si.log) + // ACTIVE-derived host index for the proxy: rebuilt after every + // activation and at boot. Wired here (store is open); the proxy + // itself was built earlier in initRuntimeAndProxy. + si.svc.appHostIndex = apptraffic.NewHostIndex() + if si.svc.proxySvc != nil { + si.svc.proxySvc.WithAppTargets(si.svc.appHostIndex) + } + return nil +} + +// appRouteSource adapts the ACTIVE-derived host index plus installation +// external routes to the route-source interfaces (PKI AppRoutes, +// public-TLS RouteSource). bbolt stays authoritative; this adapter only +// projects. Fail-closed: unwired index yields zero app hosts. The index +// is late-bound (built after PKI/public-TLS), so it resolves per call. +type appRouteSource struct { + index func() *apptraffic.HostIndex + external func() map[string]string +} + +// AppHosts implements out.AppHostSource. +func (s *appRouteSource) AppHosts() []out.AppHost { + if s == nil || s.index == nil { + return nil + } + index := s.index() + if index == nil { + return nil + } + return index.AppHosts() +} + +// GetExternalRoutes implements the installation half of out.AppRoutes +// and the public-TLS RouteSource. +func (s *appRouteSource) GetExternalRoutes() map[string]string { + if s == nil || s.external == nil { + return nil + } + return s.external() +} + +// GetRoutes implements the public-TLS RouteSource as a consumer-local +// adapter: host-only entries, Image explicitly unused (no app meaning). +func (s *appRouteSource) GetRoutes(context.Context) []domain.Route { + hosts := s.AppHosts() + routes := make([]domain.Route, 0, len(hosts)) + for _, h := range hosts { + routes = append(routes, domain.Route{Domain: h.Host}) + } + return routes +} + +// appRoutes returns the shared ACTIVE-derived route source, late-bound +// to the host index built in initApps. +func (si *serviceInit) appRoutes() *appRouteSource { + return &appRouteSource{ + index: func() *apptraffic.HostIndex { return si.svc.appHostIndex }, + external: func() map[string]string { return si.svc.configSvc.GetExternalRoutes() }, + } +} + +// appEntrypointPolicies maps installation entrypoints to the projection +// policy (trusted CIDRs, raw-fallback behavior, transport limits). +func appEntrypointPolicies(cfg Config) map[string]apptraffic.EntrypointPolicy { + policies := make(map[string]apptraffic.EntrypointPolicy, len(cfg.EntryPoints)) + for name, entry := range cfg.EntryPoints { + policies[name] = apptraffic.EntrypointPolicy{ + Name: name, + Address: entry.Address, + Protocol: entry.Protocol, + TrustedCIDRs: entry.TrustedCIDRs, + RawFallback: entry.RawFallback, + RawFallbackTrustedCIDRs: entry.RawFallbackTrustedCIDRs, + AllowPublicRawFallback: entry.AllowPublicRawFallback, + } + } + return policies +} + +// appEntrypointListeners maps installation entrypoints to the canonical +// listener an app L4 publish declaration must match exactly. +func appEntrypointListeners(cfg Config) map[string]domain.EntryPointListener { + listeners := make(map[string]domain.EntryPointListener, len(cfg.EntryPoints)) + for name, entry := range cfg.EntryPoints { + listeners[name] = domain.EntryPointListener{Address: entry.Address, Protocol: entry.Protocol} + } + return listeners +} + +// rebuildAppHostIndex re-projects ACTIVE state into the proxy host index +// and applies the full HTTP/L4 traffic graph through the serialized +// publisher. Call after every activation (deploy/start/restart/stop/ +// remove) and at boot; callers decide whether a failure is fatal to the +// mutation. +func rebuildAppHostIndex(ctx context.Context, svc *services) error { + if svc.appTrafficPublisher == nil { + return nil + } + return svc.appTrafficPublisher.RebuildTraffic(ctx) +} + +// reconcileAppsAtBoot reconciles declarative apps intended to run, +// rebuilds the proxy host index from reconciled ACTIVE state, then +// starts the single periodic recovery monitor. Failures only warn: the +// monitor still starts so a partially failed boot self-heals. +func reconcileAppsAtBoot(ctx context.Context, svc *services, log zerowrap.Logger) { + if svc.appDeploySvc == nil { + return + } + if err := svc.appDeploySvc.ReconcileBoot(ctx); err != nil { + log.Warn().Err(err).Msg("failed to reconcile apps at boot") + } + if err := rebuildAppHostIndex(ctx, svc); err != nil { + log.Warn().Err(err).Msg("failed to rebuild app host index at boot") + } + // Exactly one monitor per process: a second boot reconciliation must + // not orphan a running loop in the daemon context. + if svc.appMonitor == nil { + svc.appMonitor = newAppMonitor(svc.appDeploySvc, log) + } + svc.appMonitor.Start(ctx) +} diff --git a/internal/app/auth_secrets.go b/internal/app/auth_secrets.go new file mode 100644 index 000000000..683088a22 --- /dev/null +++ b/internal/app/auth_secrets.go @@ -0,0 +1,300 @@ +package app + +import ( + "context" + "fmt" + "os" + "path/filepath" + "strings" + "time" + + "github.com/bnema/zerowrap" + "golang.org/x/sys/unix" + + "github.com/bnema/gordon/internal/adapters/out/secrets" + "github.com/bnema/gordon/internal/adapters/out/tokenstore" + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/auth" + "github.com/bnema/gordon/pkg/duration" +) + +// createAuthService creates the authentication service and token store. +func createAuthService(ctx context.Context, cfg Config, log zerowrap.Logger) (out.TokenStore, *auth.Service, error) { + if !cfg.Auth.Enabled { + log.Warn().Msg("auth.enabled=false detected: running in local-only mode (registry loopback-only, admin API disabled)") + return nil, nil, nil + } + + authType, err := resolveAuthType(cfg) + if err != nil { + return nil, nil, fmt.Errorf("resolve auth type: %w", err) + } + backend, err := resolveSecretsBackend(cfg.Auth.SecretsBackend) + if err != nil { + return nil, nil, log.WrapErr(err, "failed to resolve secrets backend") + } + dataDir := resolveDataDir(cfg.Server.DataDir) + + store, err := createTokenStore(backend, dataDir, log) + if err != nil { + return nil, nil, err + } + + authConfig, err := buildAuthConfig(ctx, cfg, authType, backend, dataDir, log) + if err != nil { + return nil, nil, err + } + + authSvc := auth.NewService(authConfig, store, log) + + log.Info(). + Str("type", string(authType)). + Str("backend", string(backend)). + Msg("registry authentication enabled") + + return store, authSvc, nil +} + +// resolveAuthType determines the auth type from config. +// Token-only authentication is the only supported mode. +func resolveAuthType(cfg Config) (domain.AuthType, error) { + if cfg.Auth.Type != "" && cfg.Auth.Type != "token" { + return "", fmt.Errorf("unsupported auth.type %q; only \"token\" is supported", cfg.Auth.Type) + } + return domain.AuthTypeToken, nil +} + +func resolveSecretsBackend(backend string) (domain.SecretsBackend, error) { + switch backend { + case "pass": + return domain.SecretsBackendPass, nil + case "sops": + return domain.SecretsBackendSops, nil + case "unsafe": + return domain.SecretsBackendUnsafe, nil + case "": + return "", fmt.Errorf("auth.secrets_backend is required") + default: + return "", fmt.Errorf("unsupported auth.secrets_backend %q", backend) + } +} + +func createTokenStore(backend domain.SecretsBackend, dataDir string, log zerowrap.Logger) (out.TokenStore, error) { + // Token store is always created since tokens work in both auth modes + store, err := tokenstore.NewStore(backend, dataDir, log) + if err != nil { + return nil, log.WrapErr(err, "failed to create token store") + } + return store, nil +} + +func buildAuthConfig(ctx context.Context, cfg Config, authType domain.AuthType, backend domain.SecretsBackend, dataDir string, log zerowrap.Logger) (auth.Config, error) { + authConfig := auth.Config{ + Enabled: cfg.Auth.Enabled, + AuthType: authType, + Username: cfg.Auth.Username, + } + + // Token config is always required (tokens work in all auth modes) + secret, expiry, err := loadTokenConfig(ctx, cfg, backend, dataDir, log) + if err != nil { + return auth.Config{}, err + } + authConfig.TokenSecret = secret + authConfig.TokenExpiry = expiry + + accessTokenTTL := 15 * time.Minute // default + if cfg.Auth.AccessTokenTTL != "" { + parsed, err := time.ParseDuration(cfg.Auth.AccessTokenTTL) + if err != nil { + return auth.Config{}, fmt.Errorf("invalid auth.access_token_ttl %q: %w", cfg.Auth.AccessTokenTTL, err) + } + if parsed <= 0 { + return auth.Config{}, fmt.Errorf("auth.access_token_ttl must be positive") + } + if parsed > auth.MaxAccessTokenLifetime { + return auth.Config{}, fmt.Errorf("auth.access_token_ttl must not exceed %v", auth.MaxAccessTokenLifetime) + } + accessTokenTTL = parsed + } + authConfig.AccessTokenTTL = accessTokenTTL + + return authConfig, nil +} + +func loadTokenConfig(ctx context.Context, cfg Config, backend domain.SecretsBackend, dataDir string, log zerowrap.Logger) ([]byte, time.Duration, error) { + secret, err := loadTokenSecret(ctx, cfg, backend, dataDir, log) + if err != nil { + return nil, 0, err + } + + expiry, err := parseTokenExpiry(cfg.Auth.TokenExpiry) + if err != nil { + return nil, 0, err + } + + return secret, expiry, nil +} + +// TokenSecretEnvVar is the environment variable for the JWT signing secret. +// SECURITY: This takes priority over config file to allow secure secret injection. +const TokenSecretEnvVar = "GORDON_AUTH_TOKEN_SECRET" //nolint:gosec // This is an env var name, not a credential + +func loadTokenSecret(ctx context.Context, cfg Config, backend domain.SecretsBackend, dataDir string, log zerowrap.Logger) ([]byte, error) { + // SECURITY: Priority order for token secret: + // 1. Environment variable (most secure - no disk exposure) + // 2. Secrets backend (pass/sops - encrypted) + // 3. Config file path (least preferred) + + const minTokenSecretLength = 32 + + // Check environment variable first + if envSecret := os.Getenv(TokenSecretEnvVar); envSecret != "" { + if len(envSecret) < minTokenSecretLength { + return nil, fmt.Errorf("token secret from %s must be at least %d bytes (got %d)", TokenSecretEnvVar, minTokenSecretLength, len(envSecret)) + } + log.Debug().Msg("using token secret from environment variable") + return []byte(envSecret), nil + } + + // Fall back to config-specified path via secrets backend + if cfg.Auth.TokenSecret == "" { + return nil, fmt.Errorf("token_secret is required for JWT token generation; set %s environment variable or configure auth.token_secret", TokenSecretEnvVar) + } + + secret, err := loadSecret(ctx, backend, cfg.Auth.TokenSecret, dataDir, log) + if err != nil { + return nil, log.WrapErr(err, "failed to load token secret") + } + + if len(secret) < minTokenSecretLength { + return nil, fmt.Errorf("token_secret must be at least %d bytes (got %d); use a strong random secret", minTokenSecretLength, len(secret)) + } + + return []byte(secret), nil +} + +func parseTokenExpiry(expiry string) (time.Duration, error) { + if expiry == "" { + return 0, nil + } + + parsed, err := duration.Parse(expiry) + if err != nil { + return 0, fmt.Errorf("invalid token_expiry: %w", err) + } + + return parsed, nil +} + +// loadSecret loads a secret from the configured backend. +func loadSecret(ctx context.Context, backend domain.SecretsBackend, path, dataDir string, log zerowrap.Logger) (string, error) { + switch backend { + case domain.SecretsBackendPass: + provider := secrets.NewPassProvider(log) + return provider.GetSecret(ctx, path) + case domain.SecretsBackendSops: + provider := secrets.NewSopsProvider(log) + return provider.GetSecret(ctx, path) + case domain.SecretsBackendUnsafe: + // For unsafe backend, path is relative to dataDir/secrets/. + return readUnsafeSecret(dataDir, path) + default: + return "", fmt.Errorf("unknown secrets backend: %s", backend) + } +} + +func readFileBeneath(root, cleanedRelPath string) ([]byte, error) { + rootFD, err := unix.Open(root, unix.O_RDONLY|unix.O_DIRECTORY|unix.O_CLOEXEC|unix.O_NOFOLLOW, 0) + if err != nil { + return nil, fmt.Errorf("failed to open secrets root: %w", err) + } + defer unix.Close(rootFD) + + parts := strings.Split(filepath.ToSlash(cleanedRelPath), "/") + dirFD := rootFD + var closeDirFDs []int + defer func() { + for i := len(closeDirFDs) - 1; i >= 0; i-- { + _ = unix.Close(closeDirFDs[i]) + } + }() + + for i, part := range parts { + if part == "" || part == "." || part == ".." { + return nil, fmt.Errorf("invalid secret path: path must stay under dataDir/secrets") + } + last := i == len(parts)-1 + if last { + fd, err := unix.Openat(dirFD, part, unix.O_RDONLY|unix.O_NONBLOCK|unix.O_CLOEXEC|unix.O_NOFOLLOW, 0) + if err != nil { + return nil, fmt.Errorf("failed to read secret file: %w", err) + } + defer unix.Close(fd) + var st unix.Stat_t + if err := unix.Fstat(fd, &st); err != nil { + return nil, fmt.Errorf("failed to stat secret file: %w", err) + } + if st.Mode&unix.S_IFMT != unix.S_IFREG { + return nil, fmt.Errorf("invalid secret path: secret must be a regular file") + } + data, err := os.ReadFile(fmt.Sprintf("/proc/self/fd/%d", fd)) + if err != nil { + return nil, fmt.Errorf("failed to read secret file: %w", err) + } + return data, nil + } + + nextFD, err := unix.Openat(dirFD, part, unix.O_RDONLY|unix.O_DIRECTORY|unix.O_CLOEXEC|unix.O_NOFOLLOW, 0) + if err != nil { + return nil, fmt.Errorf("failed to open secret path component: %w", err) + } + closeDirFDs = append(closeDirFDs, nextFD) + dirFD = nextFD + } + + return nil, fmt.Errorf("invalid secret path: empty path") +} + +type unsafeSecretProvider struct { + dataDir string +} + +func (p unsafeSecretProvider) Name() string { return string(domain.SecretsBackendUnsafe) } + +func (p unsafeSecretProvider) IsAvailable() bool { return true } + +func (p unsafeSecretProvider) GetSecret(_ context.Context, path string) (string, error) { + return readUnsafeSecret(p.dataDir, path) +} + +func createStandaloneServiceSecretProvider(backend domain.SecretsBackend, dataDir string, log zerowrap.Logger) out.SecretProvider { + switch backend { + case domain.SecretsBackendPass: + return secrets.NewPassProvider(log) + case domain.SecretsBackendSops: + return secrets.NewSopsProvider(log) + case domain.SecretsBackendUnsafe: + return unsafeSecretProvider{dataDir: dataDir} + default: + return nil + } +} + +func readUnsafeSecret(dataDir, secretPath string) (string, error) { + if filepath.IsAbs(secretPath) { + return "", fmt.Errorf("invalid secret path: absolute paths are not allowed") + } + cleaned := filepath.Clean(secretPath) + if cleaned == "." || cleaned == ".." || strings.HasPrefix(cleaned, ".."+string(filepath.Separator)) { + return "", fmt.Errorf("invalid secret path: path must stay under dataDir/secrets") + } + + root := filepath.Clean(filepath.Join(dataDir, "secrets")) + data, err := readFileBeneath(root, cleaned) + if err != nil { + return "", err + } + return string(data), nil +} diff --git a/internal/app/backup_wiring.go b/internal/app/backup_wiring.go new file mode 100644 index 000000000..cf297585f --- /dev/null +++ b/internal/app/backup_wiring.go @@ -0,0 +1,250 @@ +package app + +import ( + "context" + "fmt" + "path/filepath" + "strings" + "time" + + "github.com/bnema/zerowrap" + + "github.com/bnema/gordon/internal/adapters/out/filesystem" + s3storage "github.com/bnema/gordon/internal/adapters/out/s3" + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/backup" +) + +type databaseBackupSettingsConfig struct { + Enabled bool + Schedule string + StorageDir string + Retention struct { + Hourly int + Daily int + Weekly int + Monthly int + } +} + +func databaseBackupSettings(cfg Config) databaseBackupSettingsConfig { + out := databaseBackupSettingsConfig{ + Enabled: cfg.Backups.Databases.Enabled, + Schedule: cfg.Backups.Databases.Schedule, + StorageDir: cfg.Backups.Databases.StorageDir, + } + out.Retention.Hourly = cfg.Backups.Databases.Retention.Hourly + out.Retention.Daily = cfg.Backups.Databases.Retention.Daily + out.Retention.Weekly = cfg.Backups.Databases.Retention.Weekly + out.Retention.Monthly = cfg.Backups.Databases.Retention.Monthly + // Legacy [backups] keys intentionally override new database defaults when + // backups.enabled is true, preserving existing working pg_dump schedules. + // Otherwise, prefer the already-populated backups.databases.* values. + if cfg.Backups.Enabled { + out.Enabled = true + if cfg.Backups.Schedule != "" { + out.Schedule = cfg.Backups.Schedule + } + if cfg.Backups.StorageDir != "" { + out.StorageDir = cfg.Backups.StorageDir + } + } else { + if out.Schedule == "" { + out.Schedule = cfg.Backups.Schedule + } + if out.StorageDir == "" { + out.StorageDir = cfg.Backups.StorageDir + } + } + if out.Retention.Hourly == 0 { + out.Retention.Hourly = cfg.Backups.Retention.Hourly + } + if out.Retention.Daily == 0 { + out.Retention.Daily = cfg.Backups.Retention.Daily + } + if out.Retention.Weekly == 0 { + out.Retention.Weekly = cfg.Backups.Retention.Weekly + } + if out.Retention.Monthly == 0 { + out.Retention.Monthly = cfg.Backups.Retention.Monthly + } + return out +} + +func createBackupService(cfg Config, svc *services, log zerowrap.Logger) (*filesystem.BackupStorage, *backup.Service, error) { + dbCfg := databaseBackupSettings(cfg) + if !dbCfg.Enabled { + return nil, nil, nil + } + + storageDir := dbCfg.StorageDir + if storageDir == "" { + dataDir := resolveDataDir(cfg.Server.DataDir) + storageDir = filepath.Join(dataDir, "backups") + } + + backupStorage, err := filesystem.NewBackupStorage(storageDir, log) + if err != nil { + return nil, nil, log.WrapErr(err, "failed to create backup storage") + } + + retention, err := validateBackupRetention(cfg) + if err != nil { + return nil, nil, log.WrapErr(err, "invalid backup retention policy") + } + + backupCfg := domain.BackupConfig{ + Enabled: dbCfg.Enabled, + StorageDir: storageDir, + Retention: retention, + } + + backupSvc := backup.NewService(svc.runtime, backupStorage, backupCfg, log) + + log.Info(). + Str("storage_dir", storageDir). + Msg("backup service initialized") + + return backupStorage, backupSvc, nil +} + +func createVolumeBackupService(ctx context.Context, cfg Config, svc *services, log zerowrap.Logger) (out.VolumeBackupStorage, *backup.VolumeService, domain.VolumeBackupConfig, error) { + if !cfg.Backups.Volumes.Enabled { + return nil, nil, domain.VolumeBackupConfig{}, nil + } + volumeCfg, err := validateVolumeBackupConfig(cfg) + if err != nil { + return nil, nil, domain.VolumeBackupConfig{}, log.WrapErr(err, "invalid volume backup configuration") + } + + storage, err := s3storage.NewVolumeBackupStorage(ctx, volumeCfg) + if err != nil { + return nil, nil, domain.VolumeBackupConfig{}, log.WrapErr(err, "failed to create volume backup storage") + } + + volumeSvc := backup.NewVolumeService(svc.runtime, storage, volumeCfg, log) + log.Info(). + Str("bucket", volumeCfg.S3Bucket). + Str("prefix", volumeCfg.S3Prefix). + Msg("volume backup service initialized") + + return storage, volumeSvc, volumeCfg, nil +} + +func validateBackupRetention(cfg Config) (domain.RetentionPolicy, error) { + dbCfg := databaseBackupSettings(cfg) + if dbCfg.Retention.Hourly < 0 { + return domain.RetentionPolicy{}, fmt.Errorf("backups.databases.retention.hourly cannot be negative") + } + if dbCfg.Retention.Daily < 0 { + return domain.RetentionPolicy{}, fmt.Errorf("backups.databases.retention.daily cannot be negative") + } + if dbCfg.Retention.Weekly < 0 { + return domain.RetentionPolicy{}, fmt.Errorf("backups.databases.retention.weekly cannot be negative") + } + if dbCfg.Retention.Monthly < 0 { + return domain.RetentionPolicy{}, fmt.Errorf("backups.databases.retention.monthly cannot be negative") + } + + return domain.RetentionPolicy{ + Hourly: dbCfg.Retention.Hourly, + Daily: dbCfg.Retention.Daily, + Weekly: dbCfg.Retention.Weekly, + Monthly: dbCfg.Retention.Monthly, + }, nil +} + +func validateVolumeBackupConfig(cfg Config) (domain.VolumeBackupConfig, error) { + volumeCfg := cfg.Backups.Volumes + interval, err := parsePositiveDurationDefault(volumeCfg.Interval, "24h", "backups.volumes.interval") + if err != nil { + return domain.VolumeBackupConfig{}, err + } + timeout, err := parsePositiveDurationDefault(volumeCfg.Timeout, "2h", "backups.volumes.timeout") + if err != nil { + return domain.VolumeBackupConfig{}, err + } + compression, err := parseVolumeBackupCompression(volumeCfg.Compression) + if err != nil { + return domain.VolumeBackupConfig{}, err + } + maxConcurrency := volumeCfg.MaxConcurrency + if maxConcurrency == 0 { + maxConcurrency = 2 + } + helperImage := strings.TrimSpace(volumeCfg.HelperImage) + if helperImage == "" { + helperImage = "alpine:3.20" + } + volumePrefix := strings.TrimSpace(cfg.Volumes.Prefix) + if volumePrefix == "" { + volumePrefix = "gordon" + } + if compression == domain.VolumeBackupCompressionZstd && helperImage == "alpine:3.20" { + return domain.VolumeBackupConfig{}, fmt.Errorf("backups.volumes.compression zstd requires a helper_image that provides zstd") + } + if err := validateVolumeBackupS3Settings(volumeCfg.Enabled, volumeCfg.Retention.Keep, maxConcurrency, volumeCfg.S3.Bucket, volumeCfg.S3.Region); err != nil { + return domain.VolumeBackupConfig{}, err + } + return domain.VolumeBackupConfig{ + Enabled: volumeCfg.Enabled, + Interval: interval, + Compression: compression, + Retention: domain.VolumeBackupRetentionPolicy{Keep: volumeCfg.Retention.Keep}, + Timeout: timeout, + MaxConcurrency: maxConcurrency, + HelperImage: helperImage, + VolumePrefix: volumePrefix, + S3Bucket: strings.TrimSpace(volumeCfg.S3.Bucket), + S3Region: strings.TrimSpace(volumeCfg.S3.Region), + S3Prefix: strings.TrimSpace(volumeCfg.S3.Prefix), + S3Endpoint: strings.TrimSpace(volumeCfg.S3.Endpoint), + S3PathStyle: volumeCfg.S3.PathStyle, + S3SSEAlgorithm: strings.TrimSpace(volumeCfg.S3.SSEAlgorithm), + S3SSEKMSKeyID: strings.TrimSpace(volumeCfg.S3.SSEKMSKeyID), + }, nil +} + +func parsePositiveDurationDefault(raw, defaultValue, field string) (time.Duration, error) { + if strings.TrimSpace(raw) == "" { + raw = defaultValue + } + d, err := time.ParseDuration(strings.TrimSpace(raw)) + if err != nil || d <= 0 { + return 0, fmt.Errorf("%s must be a positive duration", field) + } + return d, nil +} + +func parseVolumeBackupCompression(raw string) (domain.VolumeBackupCompression, error) { + if strings.TrimSpace(raw) == "" { + raw = string(domain.VolumeBackupCompressionGzip) + } + compression := domain.VolumeBackupCompression(strings.ToLower(strings.TrimSpace(raw))) + switch compression { + case domain.VolumeBackupCompressionGzip, domain.VolumeBackupCompressionZstd: + return compression, nil + default: + return "", fmt.Errorf("backups.volumes.compression must be one of: gzip, zstd") + } +} + +func validateVolumeBackupS3Settings(enabled bool, keep, maxConcurrency int, bucket, region string) error { + if keep < 0 { + return fmt.Errorf("backups.volumes.retention.keep cannot be negative") + } + if enabled && keep == 0 { + return fmt.Errorf("backups.volumes.retention.keep must be positive when volume backups are enabled") + } + if maxConcurrency < 1 { + return fmt.Errorf("backups.volumes.max_concurrency must be at least 1") + } + if enabled && strings.TrimSpace(bucket) == "" { + return fmt.Errorf("backups.volumes.s3.bucket is required when volume backups are enabled") + } + if enabled && strings.TrimSpace(region) == "" { + return fmt.Errorf("backups.volumes.s3.region is required when volume backups are enabled") + } + return nil +} diff --git a/internal/app/config.go b/internal/app/config.go new file mode 100644 index 000000000..a4970042b --- /dev/null +++ b/internal/app/config.go @@ -0,0 +1,455 @@ +package app + +import ( + "fmt" + "os" + "path/filepath" + "strings" + + "github.com/bnema/zerowrap" + "github.com/spf13/viper" + + "github.com/bnema/gordon/internal/adapters/out/telemetry" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/publictls" + servicecfg "github.com/bnema/gordon/internal/usecase/services" + "github.com/bnema/gordon/internal/usecase/traffic" +) + +// Config holds the application configuration. +type Config struct { + Server struct { + Port int `mapstructure:"port"` + RegistryPort int `mapstructure:"registry_port"` + GordonDomain string `mapstructure:"gordon_domain"` + RegistryDomain string `mapstructure:"registry_domain"` + LegacyRegistryDomains []string `mapstructure:"legacy_registry_domains"` + TLSPort int `mapstructure:"tls_port"` + TLSCertFile string `mapstructure:"tls_cert_file"` + TLSKeyFile string `mapstructure:"tls_key_file"` + ForceHTTPSRedirect bool `mapstructure:"force_https_redirect"` + DataDir string `mapstructure:"data_dir"` + MaxProxyBodySize string `mapstructure:"max_proxy_body_size"` // e.g., "512MB", "1GB" + MaxBlobChunkSize string `mapstructure:"max_blob_chunk_size"` // e.g., "512MB", "1GB" + MaxBlobSize string `mapstructure:"max_blob_size"` // e.g., "1GB", "2GB" + MaxProxyResponseSize string `mapstructure:"max_proxy_response_size"` // e.g., "1GB", "0" for no limit + MaxConcurrentConns int `mapstructure:"max_concurrent_connections"` + RegistryAllowedIPs []string `mapstructure:"registry_allowed_ips"` + ProxyAllowedIPs []string `mapstructure:"proxy_allowed_ips"` + RegistryListenAddr string `mapstructure:"registry_listen_address"` + } `mapstructure:"server"` + + Apps struct { + // RevisionRetention bounds unreferenced-revision GC per app. + // Zero or negative restores the compiled default (8). + RevisionRetention int `mapstructure:"revision_retention"` + } `mapstructure:"apps"` + + // AppMounts declares administrative bind policies keyed by stable mount + // name, referenced by name from an app manifest's [[services..bind]]. + AppMounts map[string]AppMountPolicy `mapstructure:"app_mounts"` + + // AppDevices declares administrative device grants keyed by stable + // logical device name, referenced by name from an app manifest's + // `devices` list. + AppDevices map[string]AppDevicePolicy `mapstructure:"app_devices"` + + Logging struct { + Level string `mapstructure:"level"` + Format string `mapstructure:"format"` + File struct { + Enabled bool `mapstructure:"enabled"` + Path string `mapstructure:"path"` + MaxSize int `mapstructure:"max_size"` + MaxBackups int `mapstructure:"max_backups"` + MaxAge int `mapstructure:"max_age"` + } `mapstructure:"file"` + ContainerLogs struct { + Enabled bool `mapstructure:"enabled"` + Dir string `mapstructure:"dir"` + MaxSize int `mapstructure:"max_size"` + MaxBackups int `mapstructure:"max_backups"` + MaxAge int `mapstructure:"max_age"` + } `mapstructure:"container_logs"` + AccessLog struct { + Enabled bool `mapstructure:"enabled"` + Format string `mapstructure:"format"` + Output string `mapstructure:"output"` + FilePath string `mapstructure:"file_path"` + MaxSize int `mapstructure:"max_size"` + MaxBackups int `mapstructure:"max_backups"` + MaxAge int `mapstructure:"max_age"` + ExcludeHealthChecks bool `mapstructure:"exclude_health_checks"` + SyslogIdentifier string `mapstructure:"syslog_identifier"` + } `mapstructure:"access_log"` + } `mapstructure:"logging"` + + Volumes struct { + AutoCreate bool `mapstructure:"auto_create"` + Prefix string `mapstructure:"prefix"` + Preserve bool `mapstructure:"preserve"` + } `mapstructure:"volumes"` + + Auth struct { + Enabled bool `mapstructure:"enabled"` + Type string `mapstructure:"type"` // only "token" is supported + SecretsBackend string `mapstructure:"secrets_backend"` // "pass", "sops", or "unsafe" + Username string `mapstructure:"username"` + TokenSecret string `mapstructure:"token_secret"` // path in secrets backend + TokenExpiry string `mapstructure:"token_expiry"` // e.g., "720h", "30d" + AccessTokenTTL string `mapstructure:"access_token_ttl"` // e.g., "15m", "30m" (default: 15m) + } `mapstructure:"auth"` + + API struct { + RateLimit struct { + Enabled bool `mapstructure:"enabled"` + GlobalRPS float64 `mapstructure:"global_rps"` + PerIPRPS float64 `mapstructure:"per_ip_rps"` + Burst int `mapstructure:"burst"` + TrustedProxies []string `mapstructure:"trusted_proxies"` + } `mapstructure:"rate_limit"` + } `mapstructure:"api"` + + EntryPoints map[string]traffic.EntryPointConfig `mapstructure:"entrypoints"` + Traffic traffic.Config `mapstructure:"traffic"` + NetworkServices []traffic.NetworkServiceConfig `mapstructure:"network_services"` + Services []servicecfg.Config `mapstructure:"services"` + + Backups struct { + // Legacy database backup keys. Prefer backups.databases.* for new configs. + Enabled bool `mapstructure:"enabled"` + Schedule string `mapstructure:"schedule"` + StorageDir string `mapstructure:"storage_dir"` + Retention struct { + Hourly int `mapstructure:"hourly"` + Daily int `mapstructure:"daily"` + Weekly int `mapstructure:"weekly"` + Monthly int `mapstructure:"monthly"` + } `mapstructure:"retention"` + Databases struct { + Enabled bool `mapstructure:"enabled"` + Schedule string `mapstructure:"schedule"` + StorageDir string `mapstructure:"storage_dir"` + Retention struct { + Hourly int `mapstructure:"hourly"` + Daily int `mapstructure:"daily"` + Weekly int `mapstructure:"weekly"` + Monthly int `mapstructure:"monthly"` + } `mapstructure:"retention"` + } `mapstructure:"databases"` + Volumes struct { + Enabled bool `mapstructure:"enabled"` + Interval string `mapstructure:"interval"` + Compression string `mapstructure:"compression"` + Timeout string `mapstructure:"timeout"` + MaxConcurrency int `mapstructure:"max_concurrency"` + HelperImage string `mapstructure:"helper_image"` + S3 struct { + Bucket string `mapstructure:"bucket"` + Region string `mapstructure:"region"` + Prefix string `mapstructure:"prefix"` + Endpoint string `mapstructure:"endpoint"` + PathStyle bool `mapstructure:"path_style"` + SSEAlgorithm string `mapstructure:"sse_algorithm"` + SSEKMSKeyID string `mapstructure:"sse_kms_key_id"` + } `mapstructure:"s3"` + Retention struct { + Keep int `mapstructure:"keep"` + } `mapstructure:"retention"` + } `mapstructure:"volumes"` + } `mapstructure:"backups"` + + Images struct { + AllowedRegistries []string `mapstructure:"allowed_registries"` + RequireDigest bool `mapstructure:"require_digest"` + Prune struct { + Enabled bool `mapstructure:"enabled"` + Schedule string `mapstructure:"schedule"` + KeepLast int `mapstructure:"keep_last"` + } `mapstructure:"prune"` + } `mapstructure:"images"` + + Containers struct { + MemoryLimit string `mapstructure:"memory_limit"` // e.g., "512MB", "1GB" + CPULimit float64 `mapstructure:"cpu_limit"` // CPU cores, e.g., 1.0 = 1 core + PidsLimit int64 `mapstructure:"pids_limit"` // e.g., 512 + SecurityProfile string `mapstructure:"security_profile"` // compat or strict + } `mapstructure:"containers"` + + NetworkIsolation struct { + Enabled bool `mapstructure:"enabled"` + Prefix string `mapstructure:"network_prefix"` + Internal bool `mapstructure:"internal"` + } `mapstructure:"network_isolation"` + + Telemetry telemetry.Config `mapstructure:"telemetry"` + + TLS struct { + ACME struct { + Enabled bool `mapstructure:"enabled"` + Email string `mapstructure:"email"` + Challenge string `mapstructure:"challenge"` + ObtainBatchSize int `mapstructure:"obtain_batch_size"` + } `mapstructure:"acme"` + } `mapstructure:"tls"` + + DNS struct { + Resolvers []string `mapstructure:"resolvers"` + PropagationTimeout string `mapstructure:"propagation_timeout"` + PollingInterval string `mapstructure:"polling_interval"` + } `mapstructure:"dns"` +} + +func warnDeprecatedConfigKeys(v *viper.Viper, log zerowrap.Logger) { + for _, key := range []string{"server.tls_enabled", "server.force_hsts"} { + if v.IsSet(key) { + log.Warn().Str("key", key).Msg("deprecated config key — Gordon now uses an internal CA with automatic TLS; remove this from your config") + } + } +} + +// initConfig loads configuration from file. +func initConfig(configPath string) (*viper.Viper, Config, error) { + v := viper.New() + if err := loadConfig(v, configPath); err != nil { + return nil, Config{}, fmt.Errorf("failed to load config: %w", err) + } + + var cfg Config + if err := v.Unmarshal(&cfg); err != nil { + return nil, Config{}, fmt.Errorf("failed to unmarshal config: %w", err) + } + if err := validateEntrypointMigration(v, cfg); err != nil { + return nil, Config{}, err + } + if err := validateRetiredAppConfig(v); err != nil { + return nil, Config{}, err + } + if _, err := buildAppMountPolicies(cfg); err != nil { + return nil, Config{}, err + } + if err := validateDeviceConfig(v, cfg); err != nil { + return nil, Config{}, err + } + + return v, cfg, nil +} + +// retiredAppConfigKeys are pre-v3 application keys removed with the +// declarative-apps cutover. Presence fails boot/reload closed with a +// config-retired diagnostic naming the fix — never a silent migration. +// Keys that remain live installation-level settings are deliberately +// absent: top-level "services" still declares standalone L4 workloads +// (see config.services), so it must never be rejected here. +var retiredAppConfigKeys = []struct { + key string + hint string +}{ + {"routes", "declare [[services..http]] in an app file, then 'gordon apps apply'"}, + {"attachments", "declare [services.] + volumes; attachments are removed"}, + {"network_groups", "declare [[network.shared]] with ownership verification"}, + {"service_routes", "declare [[services..http]] instead"}, + {"auto", "feature removed; declare explicit interfaces"}, + {"auto_route", "feature removed; declare explicit interfaces"}, + {"auto_route_allowed_domains", "feature removed; declare explicit interfaces"}, + {"previews", "staging is an ordinary app file"}, + {"env", "remove the [env] section; declare [env] and [services..secrets] in app files"}, +} + +// validateRetiredAppConfig rejects obsolete application configuration +// BEFORE any app/runtime mutation, at startup and reload. InConfig +// (not IsSet) targets explicit file keys only — installation defaults +// registered via SetDefault must not trip the rejection. +func validateRetiredAppConfig(v *viper.Viper) error { + diagnostics := make([]string, 0, len(retiredAppConfigKeys)) + for _, retired := range retiredAppConfigKeys { + if v.InConfig(retired.key) { + diagnostics = append(diagnostics, fmt.Sprintf("key %q was removed in v3; %s", retired.key, retired.hint)) + } + } + if len(diagnostics) > 0 { + return fmt.Errorf("config-retired: %s", strings.Join(diagnostics, "; ")) + } + return nil +} + +func validateEntrypointMigration(v *viper.Viper, cfg Config) error { + if len(cfg.EntryPoints) > 0 { + return nil + } + + legacyKeys := make([]string, 0, 2) + for _, key := range []string{"server.port", "server.tls_port"} { + if v.IsSet(key) { + legacyKeys = append(legacyKeys, key) + } + } + if len(legacyKeys) == 0 { + return nil + } + + return fmt.Errorf("legacy %s configuration requires at least one [entrypoints] entry; see docs/upgrading.md", strings.Join(legacyKeys, " or ")) +} + +// resolveLogFilePath returns the configured log file path or a default. +func resolveLogFilePath(cfg Config) string { + if cfg.Logging.File.Path != "" { + return cfg.Logging.File.Path + } + dataDir := cfg.Server.DataDir + if dataDir == "" { + dataDir = DefaultDataDir() + } + return filepath.Join(dataDir, "logs", "gordon.log") +} + +// resolveRuntimeConfig converts a server.runtime config value to a socket path. +// "auto" or "" means auto-detect. +// Named runtimes ("podman", "docker") are resolved to well-known socket paths. +// URI schemes (unix://) are stripped so callers receive a bare path. +func resolveRuntimeConfig(value string) string { + if value == "" || value == "auto" { + return "" + } + // Named runtimes: resolve to well-known socket paths. + switch value { + case "podman": + // Check XDG_RUNTIME_DIR first (rootless Podman). + if xdg := os.Getenv("XDG_RUNTIME_DIR"); xdg != "" { + candidate := filepath.Join(xdg, "podman", "podman.sock") + if _, err := os.Stat(candidate); err == nil { + return candidate + } + } + // Fallback to system-wide Podman socket. + return "/run/podman/podman.sock" + case "docker": + return "/var/run/docker.sock" + } + // Explicit socket path — strip URI scheme if present. + if socketPath, ok := strings.CutPrefix(value, "unix://"); ok { + return socketPath + } + return value +} + +func resolveDataDir(dataDir string) string { + if dataDir == "" { + return DefaultDataDir() + } + return dataDir +} + +func resolveRegistryDomains(cfg Config) (string, []string) { + registryDomain := cfg.Server.RegistryDomain + if registryDomain == "" { + registryDomain = cfg.Server.GordonDomain + } + return registryDomain, append([]string{}, cfg.Server.LegacyRegistryDomains...) +} + +// loadConfig loads configuration from file and sets defaults. +func loadConfig(v *viper.Viper, configPath string) error { + v.SetDefault("server.registry_port", 5000) + v.SetDefault("server.legacy_registry_domains", []string{}) + v.SetDefault("server.tls_cert_file", "") + v.SetDefault("server.tls_key_file", "") + v.SetDefault("tls.acme.enabled", false) + v.SetDefault("tls.acme.email", "") + v.SetDefault("tls.acme.challenge", "auto") + v.SetDefault("tls.acme.obtain_batch_size", 1) + v.SetDefault("dns.resolvers", publictls.DefaultDNSResolvers) + v.SetDefault("dns.propagation_timeout", "5m") + v.SetDefault("dns.polling_interval", "5s") + v.SetDefault("server.force_https_redirect", false) + v.SetDefault("server.data_dir", DefaultDataDir()) + v.SetDefault("server.runtime", "auto") + v.SetDefault("logging.level", "info") + v.SetDefault("logging.format", "console") + v.SetDefault("logging.file.enabled", false) + v.SetDefault("logging.file.max_size", 100) + v.SetDefault("logging.file.max_backups", 3) + v.SetDefault("logging.file.max_age", 28) + v.SetDefault("logging.container_logs.enabled", true) + v.SetDefault("logging.container_logs.dir", "") + v.SetDefault("logging.container_logs.max_size", 100) + v.SetDefault("logging.container_logs.max_backups", 3) + v.SetDefault("logging.container_logs.max_age", 28) + v.SetDefault("logging.access_log.enabled", false) + v.SetDefault("logging.access_log.format", "json") + v.SetDefault("logging.access_log.output", "stdout") + v.SetDefault("logging.access_log.file_path", "") + v.SetDefault("logging.access_log.max_size", 100) + v.SetDefault("logging.access_log.max_backups", 3) + v.SetDefault("logging.access_log.max_age", 28) + v.SetDefault("logging.access_log.exclude_health_checks", true) + v.SetDefault("logging.access_log.syslog_identifier", "gordon-access") + v.SetDefault("auth.enabled", true) + // Note: auth.type defaults to "token" (the only supported mode) + v.SetDefault("auth.secrets_backend", "") + v.SetDefault("auth.token_expiry", "720h") + v.SetDefault("api.rate_limit.enabled", true) + v.SetDefault("api.rate_limit.global_rps", 500) + v.SetDefault("api.rate_limit.per_ip_rps", 50) + v.SetDefault("api.rate_limit.burst", 100) + v.SetDefault("auto_route.enabled", false) + v.SetDefault("network_isolation.enabled", true) + v.SetDefault("network_isolation.network_prefix", "gordon") + v.SetDefault("network_isolation.internal", false) + v.SetDefault("volumes.auto_create", true) + v.SetDefault("volumes.prefix", "gordon") + v.SetDefault("volumes.preserve", true) + v.SetDefault("backups.databases.enabled", false) + v.SetDefault("backups.databases.schedule", string(domain.ScheduleDaily)) + v.SetDefault("backups.databases.storage_dir", "") + v.SetDefault("backups.databases.retention.hourly", 0) + v.SetDefault("backups.databases.retention.daily", 0) + v.SetDefault("backups.databases.retention.weekly", 0) + v.SetDefault("backups.databases.retention.monthly", 0) + v.SetDefault("backups.volumes.enabled", false) + v.SetDefault("backups.volumes.interval", "24h") + v.SetDefault("backups.volumes.compression", string(domain.VolumeBackupCompressionGzip)) + v.SetDefault("backups.volumes.timeout", "2h") + v.SetDefault("backups.volumes.max_concurrency", 2) + v.SetDefault("backups.volumes.helper_image", "alpine:3.20") + v.SetDefault("backups.volumes.s3.bucket", "") + v.SetDefault("backups.volumes.s3.region", "") + v.SetDefault("backups.volumes.s3.prefix", "") + v.SetDefault("backups.volumes.s3.endpoint", "") + v.SetDefault("backups.volumes.s3.path_style", false) + v.SetDefault("backups.volumes.s3.sse_algorithm", "") + v.SetDefault("backups.volumes.s3.sse_kms_key_id", "") + v.SetDefault("backups.volumes.retention.keep", 14) + v.SetDefault("images.allowed_registries", []string{}) + v.SetDefault("images.require_digest", false) + v.SetDefault("images.prune.enabled", false) + v.SetDefault("images.prune.schedule", string(domain.ScheduleDaily)) + v.SetDefault("images.prune.keep_last", domain.DefaultImagePruneKeepLast) + v.SetDefault("containers.security_profile", "compat") + v.SetDefault("telemetry.enabled", false) + v.SetDefault("telemetry.endpoint", "") + v.SetDefault("telemetry.auth_token", "") + v.SetDefault("telemetry.traces", true) + v.SetDefault("telemetry.metrics", true) + v.SetDefault("telemetry.logs", true) + v.SetDefault("telemetry.trace_sample_rate", 1.0) + + v.SetDefault("server.max_concurrent_connections", -1) // -1 = use default (10000), 0 = no limit + v.SetDefault("server.registry_allowed_ips", []string{}) + v.SetDefault("server.proxy_allowed_ips", []string{}) + v.SetDefault("server.registry_listen_address", "") + + ConfigureViper(v, configPath) + + if err := v.ReadInConfig(); err != nil { + if _, ok := err.(viper.ConfigFileNotFoundError); !ok { + return fmt.Errorf("failed to read config file: %w", err) + } + } + + v.SetEnvPrefix("GORDON") + v.SetEnvKeyReplacer(strings.NewReplacer(".", "_")) + v.AutomaticEnv() + + return nil +} diff --git a/internal/app/config_validation.go b/internal/app/config_validation.go new file mode 100644 index 000000000..3dd3d76b2 --- /dev/null +++ b/internal/app/config_validation.go @@ -0,0 +1,13 @@ +package app + +import "fmt" + +// ValidateConfigFile loads and statically validates an explicit configuration +// file without initializing runtime services or binding listeners. +func ValidateConfigFile(path string) error { + if path == "" { + return fmt.Errorf("configuration file path is required") + } + _, _, err := initConfig(path) + return err +} diff --git a/internal/app/credentials.go b/internal/app/credentials.go new file mode 100644 index 000000000..3b4d67e8c --- /dev/null +++ b/internal/app/credentials.go @@ -0,0 +1,195 @@ +package app + +import ( + "crypto/rand" + "encoding/hex" + "encoding/json" + "fmt" + "os" + "path/filepath" + "time" + + "github.com/bnema/zerowrap" +) + +const ( + internalRegistryUsername = "gordon-internal" + serviceTokenSubject = "gordon-service" + serviceTokenDefaultTTL = 30 * 24 * time.Hour +) + +func generateInternalRegistryAuth() (string, string, error) { + password, err := randomTokenHex(32) + if err != nil { + return "", "", err + } + return internalRegistryUsername, password, nil +} + +func randomTokenHex(size int) (string, error) { + buf := make([]byte, size) + if _, err := rand.Read(buf); err != nil { + return "", err + } + return hex.EncodeToString(buf), nil +} + +// InternalCredentials holds the internal registry credentials for CLI access. +type InternalCredentials struct { + Username string `json:"username"` + Password string `json:"password"` +} + +// getSecureRuntimeDir returns a secure directory for runtime files. +// Priority: XDG_RUNTIME_DIR > ~/.gordon/run +func getSecureRuntimeDir() (string, error) { + // Try XDG_RUNTIME_DIR first (typically /run/user/ on Linux) + if runtimeDir := os.Getenv("XDG_RUNTIME_DIR"); runtimeDir != "" { + gordonDir := filepath.Join(runtimeDir, "gordon") + if err := os.MkdirAll(gordonDir, 0700); err == nil { + return gordonDir, nil + } + } + + // Fall back to ~/.gordon/run + homeDir, err := os.UserHomeDir() + if err != nil { + return "", fmt.Errorf("failed to get home directory: %w", err) + } + + gordonDir := filepath.Join(homeDir, ".gordon", "run") + if err := os.MkdirAll(gordonDir, 0700); err != nil { + return "", fmt.Errorf("failed to create runtime directory: %w", err) + } + + return gordonDir, nil +} + +// getInternalCredentialsFile returns the path to the internal credentials file. +// SECURITY: Credentials are stored in a secure location with restricted permissions. +func getInternalCredentialsFile() string { + runtimeDir, err := getSecureRuntimeDir() + if err != nil { + // Fall back to temp dir if we can't get secure dir (shouldn't happen) + return filepath.Join(os.TempDir(), "gordon-internal-creds.json") + } + return filepath.Join(runtimeDir, "internal-creds.json") +} + +// persistInternalCredentials saves the internal registry credentials to a secure file. +// SECURITY: Credentials are stored in XDG_RUNTIME_DIR or ~/.gordon/run with 0600 permissions. +// The file is cleaned up on graceful shutdown but may persist if Gordon crashes. +// These credentials are for internal loopback communication only and are regenerated on each start. +func persistInternalCredentials(username, password string) error { + creds := InternalCredentials{ + Username: username, + Password: password, + } + data, err := json.Marshal(creds) + if err != nil { + return fmt.Errorf("failed to marshal credentials: %w", err) + } + + credFile := getInternalCredentialsFile() + + // Ensure parent directory exists with secure permissions + if err := os.MkdirAll(filepath.Dir(credFile), 0700); err != nil { + return fmt.Errorf("failed to create credentials directory: %w", err) + } + + // Write file with restrictive permissions (owner read/write only) + if err := os.WriteFile(credFile, data, 0600); err != nil { + return fmt.Errorf("failed to write credentials file: %w", err) + } + return nil +} + +// cleanupInternalCredentials removes the internal credentials file. +func cleanupInternalCredentials() { + _ = os.Remove(getInternalCredentialsFile()) +} + +// getInternalCredentialsCandidates returns candidate file paths in priority order: +// 1. XDG_RUNTIME_DIR/gordon/ (set by systemd for the daemon) +// 2. /run/user//gordon/ (well-known systemd default, for CLI in shells without XDG_RUNTIME_DIR) +// 3. ~/.gordon/run/ (fallback for non-systemd environments) +// 4. os.TempDir() (last resort, matches getInternalCredentialsFile fallback path) +func getInternalCredentialsCandidates() []string { + var candidates []string + + // 1. XDG_RUNTIME_DIR (set in daemon's environment) + if runtimeDir := os.Getenv("XDG_RUNTIME_DIR"); runtimeDir != "" { + candidates = append(candidates, filepath.Join(runtimeDir, "gordon", "internal-creds.json")) + } + + // 2. /run/user//gordon/ (systemd default, may not be in CLI's env) + uid := os.Getuid() + sysRuntime := filepath.Join("/run/user", fmt.Sprintf("%d", uid), "gordon", "internal-creds.json") + // Avoid duplicate if XDG_RUNTIME_DIR already points here + if len(candidates) == 0 || candidates[0] != sysRuntime { + candidates = append(candidates, sysRuntime) + } + + // 3. ~/.gordon/run/ fallback + if homeDir, err := os.UserHomeDir(); err == nil { + candidates = append(candidates, filepath.Join(homeDir, ".gordon", "run", "internal-creds.json")) + } + + // 4. os.TempDir() last resort — matches the fallback path in getInternalCredentialsFile, + // ensuring GetInternalCredentials can find credentials even when getSecureRuntimeDir fails. + candidates = append(candidates, filepath.Join(os.TempDir(), "gordon-internal-creds.json")) + + return candidates +} + +// GetInternalCredentialsFromCandidates reads credentials from the first candidate file that exists. +// Exported for testing. +func GetInternalCredentialsFromCandidates(candidates []string) (*InternalCredentials, error) { + var lastErr error + for _, path := range candidates { + data, err := os.ReadFile(path) + if os.IsNotExist(err) { + continue + } + if err != nil { + // Non-permission errors (e.g. EACCES) may be transient or path-specific; + // record and try the next candidate rather than failing immediately. + lastErr = fmt.Errorf("failed to read credentials file %s: %w", path, err) + continue + } + var creds InternalCredentials + if err := json.Unmarshal(data, &creds); err != nil { + // Corrupt file — record and fall through to lower-priority candidates. + lastErr = fmt.Errorf("failed to parse credentials at %s: %w", path, err) + continue + } + return &creds, nil + } + if lastErr != nil { + return nil, lastErr + } + return nil, fmt.Errorf("no credentials file found (is Gordon running?): checked %v", candidates) +} + +// GetInternalCredentials reads the internal registry credentials from file. +// Probes all candidate runtime directories so CLI works regardless of whether +// XDG_RUNTIME_DIR is set in the current shell environment. +func GetInternalCredentials() (*InternalCredentials, error) { + return GetInternalCredentialsFromCandidates(getInternalCredentialsCandidates()) +} + +func setupInternalRegistryAuth(svc *services, log zerowrap.Logger) error { + var err error + svc.internalRegUser, svc.internalRegPass, err = generateInternalRegistryAuth() + if err != nil { + return log.WrapErr(err, "failed to generate internal registry credentials") + } + + // Persist credentials to file for CLI access (gordon auth internal) + if err := persistInternalCredentials(svc.internalRegUser, svc.internalRegPass); err != nil { + log.Warn().Err(err).Msg("failed to persist internal credentials for CLI access") + } + + log.Debug().Msg("internal registry auth generated for loopback pulls") + return nil +} diff --git a/internal/app/gcbarrier.go b/internal/app/gcbarrier.go new file mode 100644 index 000000000..b303db01b --- /dev/null +++ b/internal/app/gcbarrier.go @@ -0,0 +1,109 @@ +package app + +import ( + "context" + "sync" + + "github.com/bnema/gordon/internal/boundaries/out" +) + +// gcBarrier is the process-wide GC barrier. +// +// Shared leases cover the window in which a deploy, start, restart, +// remove, recovery, or restore path selects, acquires, and durably +// publishes the protection of a runtime resource. One exclusive lease +// covers prune's protection snapshot, planning, and exact deletion. +// +// Lock order is fixed: GC barrier → per-app coordinator → registry +// mutation lock → short bbolt transaction. Callers therefore take a +// shared lease BEFORE the per-app coordinator, never the other way +// round. +type gcBarrier struct { + mu sync.Mutex + shared int + exclusive bool + // changed is closed and replaced whenever a lease is released, so + // waiters re-check availability. + changed chan struct{} +} + +// newGCBarrier creates an idle process-wide barrier. +func newGCBarrier() *gcBarrier { + return &gcBarrier{changed: make(chan struct{})} +} + +// AcquireShared implements out.GCBarrier. +func (b *gcBarrier) AcquireShared(ctx context.Context) (out.GCLease, error) { + return b.acquire(ctx, false) +} + +// AcquireExclusive implements out.GCBarrier. +func (b *gcBarrier) AcquireExclusive(ctx context.Context) (out.GCLease, error) { + return b.acquire(ctx, true) +} + +func (b *gcBarrier) acquire(ctx context.Context, exclusive bool) (out.GCLease, error) { + for { + if err := ctx.Err(); err != nil { + return nil, err + } + b.mu.Lock() + if b.available(exclusive) { + if exclusive { + b.exclusive = true + } else { + b.shared++ + } + b.mu.Unlock() + return &gcLease{barrier: b, exclusive: exclusive}, nil + } + wait := b.changed + b.mu.Unlock() + + select { + case <-ctx.Done(): + return nil, ctx.Err() + case <-wait: + } + } +} + +// available reports whether a lease of the requested kind can be taken +// without violating the exclusion rules. The caller holds mu. +func (b *gcBarrier) available(exclusive bool) bool { + if exclusive { + return !b.exclusive && b.shared == 0 + } + return !b.exclusive +} + +// release drops one held lease and wakes waiters. The caller must hold +// no lock. +func (b *gcBarrier) release(exclusive bool) { + b.mu.Lock() + defer b.mu.Unlock() + if exclusive { + b.exclusive = false + } else if b.shared > 0 { + b.shared-- + } + close(b.changed) + b.changed = make(chan struct{}) +} + +// gcLease is one held barrier lease. Release is idempotent. +type gcLease struct { + barrier *gcBarrier + exclusive bool + once sync.Once +} + +// Release implements out.GCLease. +func (l *gcLease) Release() { + if l == nil || l.barrier == nil { + return + } + l.once.Do(func() { l.barrier.release(l.exclusive) }) +} + +var _ out.GCBarrier = (*gcBarrier)(nil) diff --git a/internal/app/gcbarrier_test.go b/internal/app/gcbarrier_test.go new file mode 100644 index 000000000..14d04d339 --- /dev/null +++ b/internal/app/gcbarrier_test.go @@ -0,0 +1,238 @@ +package app + +import ( + "context" + "errors" + "sync" + "testing" + "time" + + "github.com/bnema/gordon/internal/boundaries/out" +) + +func TestGCBarrierExcludesExclusiveDuringShared(t *testing.T) { + barrier := newGCBarrier() + ctx := context.Background() + + shared, err := barrier.AcquireShared(ctx) + if err != nil { + t.Fatalf("AcquireShared: %v", err) + } + + // An exclusive lease must wait until the last shared holder is + // done, so a prune cannot plan against a half-acquired resource. + waitCtx, cancel := context.WithTimeout(ctx, 50*time.Millisecond) + defer cancel() + if _, err := barrier.AcquireExclusive(waitCtx); !errors.Is(err, context.DeadlineExceeded) { + t.Fatalf("exclusive lease during a shared lease returned %v, want deadline exceeded", err) + } + + exclusive, err := acquireExclusiveAfter(barrier, shared) + if err != nil { + t.Fatalf("exclusive lease after release: %v", err) + } + exclusive.Release() +} + +// acquireExclusiveAfter releases shared, then takes the exclusive lease +// that must now be grantable. +func acquireExclusiveAfter(barrier *gcBarrier, shared out.GCLease) (out.GCLease, error) { + shared.Release() + return barrier.AcquireExclusive(context.Background()) +} + +func TestGCBarrierExcludesSharedDuringExclusive(t *testing.T) { + barrier := newGCBarrier() + ctx := context.Background() + + exclusive, err := barrier.AcquireExclusive(ctx) + if err != nil { + t.Fatalf("AcquireExclusive: %v", err) + } + defer exclusive.Release() + + waitCtx, cancel := context.WithTimeout(ctx, 50*time.Millisecond) + defer cancel() + if _, err := barrier.AcquireShared(waitCtx); !errors.Is(err, context.DeadlineExceeded) { + t.Fatalf("shared lease during an exclusive lease returned %v, want deadline exceeded", err) + } + + // A second exclusive lease also waits. + if _, err := barrier.AcquireExclusive(waitCtx); !errors.Is(err, context.DeadlineExceeded) { + t.Fatalf("second exclusive lease returned %v, want deadline exceeded", err) + } + + exclusive.Release() + shared, err := barrier.AcquireShared(ctx) + if err != nil { + t.Fatalf("shared lease after release: %v", err) + } + shared.Release() +} + +func TestGCBarrierCancellation(t *testing.T) { + barrier := newGCBarrier() + held, err := barrier.AcquireExclusive(context.Background()) + if err != nil { + t.Fatalf("AcquireExclusive: %v", err) + } + defer held.Release() + + ctx, cancel := context.WithCancel(context.Background()) + cancel() + if _, err := barrier.AcquireShared(ctx); !errors.Is(err, context.Canceled) { + t.Fatalf("canceled acquisition returned %v, want context.Canceled", err) + } + + // A waiter canceled while blocked returns instead of wedging + // shutdown. + waitCtx, cancelWait := context.WithCancel(context.Background()) + done := make(chan error, 1) + go func() { + _, err := barrier.AcquireShared(waitCtx) + done <- err + }() + cancelWait() + select { + case err := <-done: + if !errors.Is(err, context.Canceled) { + t.Fatalf("blocked waiter returned %v, want context.Canceled", err) + } + case <-time.After(2 * time.Second): + t.Fatal("canceled waiter did not return") + } +} + +func TestGCBarrierDeterministicHandoff(t *testing.T) { + barrier := newGCBarrier() + ctx := context.Background() + + // Simulate a workload mutation holding the lease across resource + // acquisition and durable publication. + shared, err := barrier.AcquireShared(ctx) + if err != nil { + t.Fatalf("AcquireShared: %v", err) + } + + acquired := make(chan struct{}) + releaseExclusive := make(chan struct{}) + go func() { + exclusive, err := barrier.AcquireExclusive(ctx) + if err != nil { + close(acquired) + return + } + close(acquired) + <-releaseExclusive + exclusive.Release() + }() + + // The exclusive lease must not be granted while the shared lease is + // held, no matter how the goroutine is scheduled. + select { + case <-acquired: + t.Fatal("exclusive lease granted while a shared lease was held") + case <-time.After(50 * time.Millisecond): + } + + shared.Release() + select { + case <-acquired: + case <-time.After(2 * time.Second): + t.Fatal("exclusive lease was never granted after the shared lease was released") + } + + // Once the exclusive lease is held, a new shared lease waits. + waitCtx, cancel := context.WithTimeout(ctx, 50*time.Millisecond) + defer cancel() + if _, err := barrier.AcquireShared(waitCtx); !errors.Is(err, context.DeadlineExceeded) { + t.Fatalf("shared lease during handoff returned %v, want deadline exceeded", err) + } + + close(releaseExclusive) + // The barrier must drain back to idle. + deadline := time.Now().Add(2 * time.Second) + for { + lease, err := barrier.AcquireExclusive(context.Background()) + if err == nil { + lease.Release() + break + } + if time.Now().After(deadline) { + t.Fatal("barrier never returned to idle") + } + time.Sleep(time.Millisecond) + } +} + +func TestGCBarrierConcurrentSharedHolders(t *testing.T) { + barrier := newGCBarrier() + ctx := context.Background() + + const holders = 16 + var wg sync.WaitGroup + start := make(chan struct{}) + for range holders { + wg.Add(1) + go func() { + defer wg.Done() + <-start + lease, err := barrier.AcquireShared(ctx) + if err != nil { + t.Errorf("AcquireShared: %v", err) + return + } + time.Sleep(time.Millisecond) + lease.Release() + }() + } + close(start) + + // While any shared holder may be active, an exclusive lease must + // never be granted concurrently with one. + exclusiveDone := make(chan struct{}) + go func() { + defer close(exclusiveDone) + for range 200 { + lease, err := barrier.AcquireExclusive(ctx) + if err != nil { + t.Errorf("AcquireExclusive: %v", err) + return + } + barrier.mu.Lock() + shared := barrier.shared + holding := barrier.exclusive + barrier.mu.Unlock() + if !holding || shared != 0 { + t.Errorf("exclusive lease held with %d shared holders (exclusive=%v)", shared, holding) + } + lease.Release() + } + }() + + wg.Wait() + <-exclusiveDone +} + +func TestGCBarrierLeaseReleaseIsIdempotent(t *testing.T) { + barrier := newGCBarrier() + lease, err := barrier.AcquireShared(context.Background()) + if err != nil { + t.Fatalf("AcquireShared: %v", err) + } + lease.Release() + lease.Release() + + barrier.mu.Lock() + shared := barrier.shared + barrier.mu.Unlock() + if shared != 0 { + t.Fatalf("shared count after double release = %d, want 0", shared) + } + + exclusive, err := barrier.AcquireExclusive(context.Background()) + if err != nil { + t.Fatalf("AcquireExclusive after double release: %v", err) + } + exclusive.Release() +} diff --git a/internal/app/http_edge.go b/internal/app/http_edge.go new file mode 100644 index 000000000..03ea2fffa --- /dev/null +++ b/internal/app/http_edge.go @@ -0,0 +1,509 @@ +package app + +import ( + "context" + "encoding/json" + "net" + "net/http" + "strings" + + "github.com/bnema/zerowrap" + "go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp" + + "github.com/bnema/gordon/internal/adapters/dto" + acmehttp "github.com/bnema/gordon/internal/adapters/in/http/acme" + "github.com/bnema/gordon/internal/adapters/in/http/admin" + "github.com/bnema/gordon/internal/adapters/in/http/httphelper" + "github.com/bnema/gordon/internal/adapters/in/http/middleware" + "github.com/bnema/gordon/internal/adapters/in/http/onboarding" + proxyadapter "github.com/bnema/gordon/internal/adapters/in/http/proxy" + "github.com/bnema/gordon/internal/adapters/in/http/registry" + pkiadapter "github.com/bnema/gordon/internal/adapters/out/pki" + "github.com/bnema/gordon/internal/adapters/out/ratelimit" + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/domain" +) + +func loopbackOnly(next http.Handler, log zerowrap.Logger) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + host, _, err := net.SplitHostPort(r.RemoteAddr) + if err != nil { + host = r.RemoteAddr + } + + ip := net.ParseIP(host) + if ip == nil || !ip.IsLoopback() { + log.Warn(). + Str("path", r.URL.Path). + Str("remote_addr", r.RemoteAddr). + Msg("blocked non-loopback access on internal admin route") + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(http.StatusForbidden) + _ = json.NewEncoder(w).Encode(dto.ErrorResponse{Error: "Forbidden"}) + return + } + + next.ServeHTTP(w, r) + }) +} + +// hostRedirectEligible reports whether a known host should be redirected +// from plaintext to HTTPS: app hosts declared tls=never stay plain HTTP. +func hostRedirectEligible(svc *services, host string) bool { + if svc.appHostIndex == nil { + return true + } + entry, ok := svc.appHostIndex.Lookup(host) + return !ok || entry.TLSMode != domain.AppTLSNever +} + +// hostRequiresTLS reports whether an app host declared tls=always, which +// must never be served over plaintext. An unwired index has no always hosts. +func hostRequiresTLS(svc *services, host string) bool { + if svc.appHostIndex == nil { + return false + } + entry, ok := svc.appHostIndex.Lookup(host) + return ok && entry.TLSMode == domain.AppTLSAlways +} + +// createHTTPHandlers creates HTTP handlers with middleware. +// Returns three handlers: registry, HTTP proxy (with CIDR + onboarding), and HTTPS proxy. +func createHTTPHandlers(svc *services, cfg Config, log zerowrap.Logger, accessWriter out.AccessLogWriter) (http.Handler, http.Handler, http.Handler) { + // Parse trusted proxies once for all middleware chains. + // This ensures consistent IP extraction across logging, rate limiting, and auth. + trustedNets := httphelper.ParseTrustedProxies(cfg.API.RateLimit.TrustedProxies) + + // Registry handler + registryHandler := registry.NewHandler(svc.registrySvc, log, svc.maxBlobChunkSize, svc.maxBlobSize) + svc.registryHandler = registryHandler + if svc.reloadCoordinator != nil { + svc.reloadCoordinator.SetRegistryLimits(registryHandler) + } + registryWithMiddleware, cidrAllowlistMiddleware, rateLimitMiddleware := buildRegistryHandlerWithMiddleware( + svc, + cfg, + trustedNets, + registryHandler, + log, + ) + + registryMux := http.NewServeMux() + registerAuthRoutes(registryMux, svc, trustedNets, cidrAllowlistMiddleware, rateLimitMiddleware, cfg, log) + registryMux.Handle("/v2/", wrapRegistryForLocalMode(registryWithMiddleware, cfg, log)) + registerAdminRoutes(registryMux, svc, cfg, trustedNets, log) + + // Proxy handler. Registry-domain forwarding is gated on auth: with + // auth disabled the registry stays local-only and public ingress can + // never reach it through this proxy. + proxyHandler := proxyadapter.NewHandler(svc.proxySvc, trustedNets, log). + WithRegistryForwarding(cfg.Auth.Enabled) + + // HTTP proxy handler chain: HTTPS redirect for non-proxy clients, then CIDR allowlist + proxyAllowedNets, proxyCIDRMiddleware := buildProxyCIDRAllowlistMiddleware(cfg, trustedNets, log) + + httpProxyMiddlewares := []func(http.Handler) http.Handler{ + middleware.PanicRecovery(log), + middleware.RequestLogger(log, trustedNets), + middleware.SecurityHeaders, + middleware.HTTPSRedirectWithEligibility(proxyAllowedNets, effectiveProxyHTTPPort(cfg), effectiveProxyTLSPort(cfg), cfg.Server.ForceHTTPSRedirect, log, func(host string) bool { + return svc.proxySvc.IsKnownHost(context.Background(), host) + }, func(host string) bool { + return hostRedirectEligible(svc, host) + }, func(host string) bool { + return hostRequiresTLS(svc, host) + }), + } + if proxyCIDRMiddleware != nil { + httpProxyMiddlewares = append(httpProxyMiddlewares, proxyCIDRMiddleware) + } + + httpProxyWithMiddleware := otelhttp.NewHandler( + middleware.Chain(httpProxyMiddlewares...)(proxyHandler), + "gordon.proxy", + ) + + // Build the onboarding handler once if internal CA is available and TLS is enabled. + var obHandler *onboarding.Handler + if svc.caAdapter != nil && effectiveProxyTLSPort(cfg) != 0 { + mobileconfigBytes := pkiadapter.GenerateMobileconfig( + svc.caAdapter.RootCertificateDER(), + svc.caAdapter.RootCommonName(), + ) + obHandler = onboarding.NewHandler( + svc.caAdapter.RootCertificate(), + mobileconfigBytes, + svc.caAdapter.RootFingerprint(), + effectiveProxyHTTPPort(cfg), + effectiveProxyTLSPort(cfg), + ) + } + + // HTTP proxyMux: trusted proxy traffic flows through the normal proxy chain. + // Direct clients get an onboarding gate (when CA is available) placed BEFORE + // HTTPSRedirect so force_https_redirect cannot bypass onboarding. + // ACME HTTP-01 challenge handler is registered before the catch-all "/" so + // it gets first chance regardless of source IP. + proxyMux := http.NewServeMux() + + // Register ACME HTTP-01 challenge handler before all other routes so + // Let's Encrypt validation always succeeds, even for onboarding clients. + if svc.publicTLSSvc != nil { + proxyMux.Handle(acmehttp.Prefix, acmehttp.NewHandler(svc.publicTLSSvc)) + } + + if proxyCIDRMiddleware != nil && proxyAllowedNets == nil { + // Invalid proxy_allowed_ips: deny all traffic (fail-closed). + proxyMux.Handle("/", proxyCIDRMiddleware(httpProxyWithMiddleware)) + } else if obHandler != nil { + proxyMux.Handle("/", directHTTPOnboardingGate(obHandler, proxyAllowedNets, httpProxyWithMiddleware, log)) + } else { + proxyMux.Handle("/", httpProxyWithMiddleware) + } + + // HTTPS proxy handler chain: security headers + proxy + CA onboarding + // Onboarding routes live on the TLS port so Tailnet / direct clients + // can click through the initial cert warning, install the CA, and + // then trust all subsequent connections. + // The middleware chain wraps the entire mux so onboarding routes also + // get PanicRecovery, RequestLogger, and SecurityHeaders. + httpsProxyMiddlewares := []func(http.Handler) http.Handler{ + middleware.PanicRecovery(log), + middleware.RequestLogger(log, trustedNets), + middleware.SecurityHeaders, + } + + httpsMux := http.NewServeMux() + if obHandler != nil && cfg.Server.GordonDomain != "" { + gordonDomain := strings.TrimSuffix(strings.ToLower(strings.TrimSpace(cfg.Server.GordonDomain)), ".") + onboardingMux := http.NewServeMux() + registerOnboardingRoutes(onboardingMux, obHandler) + // Register onboarding paths host-gated so normal traffic hits + // proxyHandler directly through the catch-all / pattern. + httpsMux.Handle("GET /.well-known/gordon/", gordonDomainOnboardingGate(gordonDomain, onboardingMux, proxyHandler)) + httpsMux.Handle("GET /.well-known/gordon/ca", gordonDomainOnboardingGate(gordonDomain, onboardingMux, proxyHandler)) + httpsMux.Handle("GET /.well-known/gordon/ca.crt", gordonDomainOnboardingGate(gordonDomain, onboardingMux, proxyHandler)) + httpsMux.Handle("GET /.well-known/gordon/ca.mobileconfig", gordonDomainOnboardingGate(gordonDomain, onboardingMux, proxyHandler)) + httpsMux.Handle("/", proxyHandler) + } else { + httpsMux.Handle("/", proxyHandler) + } + + httpsHandler := otelhttp.NewHandler(middleware.Chain(httpsProxyMiddlewares...)(httpsMux), "gordon.proxy.tls") + + // Wrap top-level handlers with access logging outside all gates + // (loopbackOnly, denyAllHandler, CIDR allowlist) so every request — + // including rejected probes — produces exactly one access-log line. + var registryOut, proxyOut, httpsOut http.Handler = registryMux, proxyMux, httpsHandler + if accessWriter != nil { + excludeHC := cfg.Logging.AccessLog.ExcludeHealthChecks + registryOut = middleware.AccessLogger(accessWriter, excludeHC, log, trustedNets)(registryOut) + proxyOut = middleware.AccessLogger(accessWriter, excludeHC, log, trustedNets)(proxyOut) + httpsOut = middleware.AccessLogger(accessWriter, excludeHC, log, trustedNets)(httpsOut) + } + svc.httpsProxyHandler = httpsOut + + return registryOut, proxyOut, httpsOut +} + +// registerOnboardingRoutes registers CA onboarding well-known HTTP routes on +// the given mux. Both direct-HTTP and Gordon-domain HTTPS onboarding use this. +func registerOnboardingRoutes(mux *http.ServeMux, ob *onboarding.Handler) { + mux.HandleFunc("GET /.well-known/gordon/", ob.ServeOnboardingPage) + mux.HandleFunc("GET /.well-known/gordon/ca", ob.ServeOnboardingPage) + mux.HandleFunc("GET /.well-known/gordon/ca.crt", ob.ServeCACert) + mux.HandleFunc("GET /.well-known/gordon/ca.mobileconfig", ob.ServeMobileconfig) +} + +// directHTTPOnboardingGate returns an http.Handler that splits HTTP traffic +// by source IP. Trusted proxy IPs flow through to the normal proxy chain. +// Direct clients are served the CA onboarding flow on allowed paths and +// receive 403 on everything else. This gate runs BEFORE HTTPSRedirect so +// force_https_redirect cannot bypass onboarding for direct clients. +func directHTTPOnboardingGate(ob *onboarding.Handler, proxyNets []*net.IPNet, proxyChain http.Handler, log zerowrap.Logger) http.Handler { + // Build a small mux for direct-client onboarding paths. + onboardingMux := http.NewServeMux() + registerOnboardingRoutes(onboardingMux, ob) + + // Reserve ACME challenge path for future use. + onboardingMux.HandleFunc("/.well-known/acme-challenge/", func(w http.ResponseWriter, _ *http.Request) { + http.Error(w, "not found", http.StatusNotFound) + }) + + // Catch-all: reject any other direct HTTP request. + // Uses a method-aware split: GET writes a body, HEAD gets an empty 403. + onboardingMux.HandleFunc("/", directHTTPForbidden) + + onboardingWithMiddleware := middleware.Chain( + middleware.PanicRecovery(log), + middleware.RequestLogger(log), // Intentionally omit trusted proxy nets so direct onboarding logs use RemoteAddr only. + middleware.SecurityHeaders, + )(onboardingMux) + + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + remoteIP := httphelper.ExtractRemoteIP(r.RemoteAddr) + if httphelper.IsTrustedOrLocal(remoteIP, proxyNets) { + proxyChain.ServeHTTP(w, r) + return + } + onboardingWithMiddleware.ServeHTTP(w, r) + }) +} + +// canonicalHostsEqual compares two hosts after normalising both: stripping +// port, trimming spaces, lowercasing, and removing trailing dot. +func canonicalHostsEqual(host, expected string) bool { + host = strings.TrimSpace(host) + if h, _, err := net.SplitHostPort(host); err == nil { + host = h + } + host = strings.TrimSuffix(strings.ToLower(host), ".") + expected = strings.TrimSuffix(strings.ToLower(strings.TrimSpace(expected)), ".") + return host == expected +} + +// gordonDomainOnboardingGate returns a handler that serves onboarding routes +// only when the request host matches gordonDomain. For mismatched hosts it +// delegates to proxyHandler. gordonDomain must already be canonicalised +// (trimmed, lowered, trailing dot removed). +func gordonDomainOnboardingGate(gordonDomain string, onboardingMux, proxyHandler http.Handler) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if gordonDomain != "" && canonicalHostsEqual(r.Host, gordonDomain) { + onboardingMux.ServeHTTP(w, r) + return + } + proxyHandler.ServeHTTP(w, r) + }) +} + +// directHTTPForbidden responds with 403 for non-onboarding HTTP paths. +// HEAD requests get an empty body per HTTP semantics. +func directHTTPForbidden(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "text/plain; charset=utf-8") + w.WriteHeader(http.StatusForbidden) + if r.Method != http.MethodHead { + _, _ = w.Write([]byte("Only certificate onboarding is available over HTTP.\n")) + } +} + +func buildRegistryHandlerWithMiddleware( + svc *services, + cfg Config, + trustedNets []*net.IPNet, + registryHandler http.Handler, + log zerowrap.Logger, +) (http.Handler, func(http.Handler) http.Handler, func(http.Handler) http.Handler) { + registryMiddlewares := []func(http.Handler) http.Handler{ + middleware.PanicRecovery(log), + middleware.RequestLogger(log, trustedNets), + middleware.SecurityHeaders, + } + + cidrAllowlistMiddleware := buildRegistryCIDRAllowlistMiddleware(cfg, trustedNets, log) + if cidrAllowlistMiddleware != nil { + registryMiddlewares = append(registryMiddlewares, cidrAllowlistMiddleware) + } + + rateLimitMiddleware := buildRegistryRateLimitMiddleware(cfg, log) + registryMiddlewares = append(registryMiddlewares, rateLimitMiddleware) + + appendRegistryAuthMiddleware(®istryMiddlewares, svc, cfg, trustedNets, log) + + registryWithOtel := otelhttp.NewHandler( + middleware.Chain(registryMiddlewares...)(registryHandler), + "gordon.registry", + ) + return registryWithOtel, cidrAllowlistMiddleware, rateLimitMiddleware +} + +// parseCIDRAllowlist parses a list of IPs/CIDRs, logs warnings for invalid entries, +// and returns the parsed nets. label is used in log messages (e.g. "registry_allowed_ips"). +func parseCIDRAllowlist(ips []string, label string, log zerowrap.Logger) ([]*net.IPNet, bool) { + if len(ips) == 0 { + return nil, false + } + + allowedNets := httphelper.ParseTrustedProxies(ips) + if len(allowedNets) != len(ips) { + for _, entry := range ips { + if nets := httphelper.ParseTrustedProxies([]string{entry}); len(nets) == 0 { + log.Warn().Str("entry", entry).Msgf("ignoring invalid %s entry", label) + } + } + } + + if len(allowedNets) == 0 { + log.Error(). + Strs(label, ips). + Msgf("%s is set but no valid entries were parsed; will deny all traffic (fail-closed)", label) + return nil, true // allInvalid + } + + return allowedNets, false +} + +// denyAllHandler returns a middleware that rejects every request with 403 Forbidden. +func denyAllHandler(label string, trustedNets []*net.IPNet, log zerowrap.Logger) func(http.Handler) http.Handler { + return func(next http.Handler) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + log.Warn(). + Str(zerowrap.FieldClientIP, middleware.GetClientIP(r, trustedNets)). + Msgf("access denied due to invalid %s configuration", label) + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(http.StatusForbidden) + _ = json.NewEncoder(w).Encode(dto.ErrorResponse{Error: "Forbidden"}) + }) + } +} + +func buildRegistryCIDRAllowlistMiddleware(cfg Config, trustedNets []*net.IPNet, log zerowrap.Logger) func(http.Handler) http.Handler { + allowedNets, allInvalid := parseCIDRAllowlist(cfg.Server.RegistryAllowedIPs, "registry_allowed_ips", log) + if allInvalid { + return denyAllHandler("registry_allowed_ips", trustedNets, log) + } + if allowedNets == nil { + return nil + } + return middleware.RegistryCIDRAllowlist(allowedNets, trustedNets, log) +} + +func buildProxyCIDRAllowlistMiddleware(cfg Config, trustedNets []*net.IPNet, log zerowrap.Logger) ([]*net.IPNet, func(http.Handler) http.Handler) { + allowedNets, allInvalid := parseCIDRAllowlist(cfg.Server.ProxyAllowedIPs, "proxy_allowed_ips", log) + if allInvalid { + return nil, denyAllHandler("proxy_allowed_ips", trustedNets, log) + } + if allowedNets == nil { + return nil, nil + } + + log.Info(). + Strs("proxy_allowed_ips", cfg.Server.ProxyAllowedIPs). + Msg("proxy origin IP allowlist enabled") + + return allowedNets, middleware.ProxyCIDRAllowlist(allowedNets, log) +} + +func buildRegistryRateLimitMiddleware(cfg Config, log zerowrap.Logger) func(http.Handler) http.Handler { + if cfg.API.RateLimit.Enabled { + globalLimiter := ratelimit.NewMemoryStore(cfg.API.RateLimit.GlobalRPS, cfg.API.RateLimit.Burst, log) + ipLimiter := ratelimit.NewMemoryStore(cfg.API.RateLimit.PerIPRPS, cfg.API.RateLimit.Burst, log) + return registry.RateLimitMiddleware( + globalLimiter, + ipLimiter, + cfg.API.RateLimit.TrustedProxies, + log, + ) + } + + return registry.RateLimitMiddleware(nil, nil, nil, log) +} + +func appendRegistryAuthMiddleware(registryMiddlewares *[]func(http.Handler) http.Handler, svc *services, cfg Config, trustedNets []*net.IPNet, log zerowrap.Logger) { + if svc.authSvc != nil { + internalAuth := middleware.InternalRegistryAuth{ + Username: svc.internalRegUser, + Password: svc.internalRegPass, + } + *registryMiddlewares = append(*registryMiddlewares, middleware.RegistryAuthV2(svc.authSvc, internalAuth, trustedNets, log)) + return + } + + if cfg.Auth.Enabled { + log.Error().Msg("authentication service unavailable; registry requests will be denied") + *registryMiddlewares = append(*registryMiddlewares, func(next http.Handler) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(http.StatusServiceUnavailable) + _ = json.NewEncoder(w).Encode(dto.ErrorResponse{Error: "authentication service unavailable"}) + }) + }) + } +} + +func registerAuthRoutes( + registryMux *http.ServeMux, + svc *services, + trustedNets []*net.IPNet, + cidrAllowlistMiddleware func(http.Handler) http.Handler, + rateLimitMiddleware func(http.Handler) http.Handler, + cfg Config, + log zerowrap.Logger, +) { + if svc.authHandler == nil { + return + } + + // Auth endpoints always get rate limiting, even if global rate limiting is disabled. + // This prevents brute-force attacks against password/token endpoints. + authRateLimitMiddleware := rateLimitMiddleware + if !cfg.API.RateLimit.Enabled { + authGlobalLimiter := ratelimit.NewMemoryStore(50, 100, log) + authIPLimiter := ratelimit.NewMemoryStore(5, 10, log) + authRateLimitMiddleware = registry.RateLimitMiddleware(authGlobalLimiter, authIPLimiter, cfg.API.RateLimit.TrustedProxies, log) + } + + // Auth endpoints are NOT protected by auth - they're where clients authenticate + // but still need rate limiting to prevent brute force attacks. + authMiddlewares := []func(http.Handler) http.Handler{ + middleware.PanicRecovery(log), + middleware.RequestLogger(log, trustedNets), + middleware.SecurityHeaders, + } + if cidrAllowlistMiddleware != nil { + authMiddlewares = append(authMiddlewares, cidrAllowlistMiddleware) + } + authMiddlewares = append(authMiddlewares, authRateLimitMiddleware) + authWithMiddleware := otelhttp.NewHandler( + middleware.Chain(authMiddlewares...)(svc.authHandler), + "gordon.auth", + ) + registryMux.Handle("/auth/", authWithMiddleware) +} + +func wrapRegistryForLocalMode(registryWithMiddleware http.Handler, cfg Config, log zerowrap.Logger) http.Handler { + if !cfg.Auth.Enabled { + return loopbackOnly(registryWithMiddleware, log) + } + return registryWithMiddleware +} + +func registerAdminRoutes(registryMux *http.ServeMux, svc *services, cfg Config, trustedNets []*net.IPNet, log zerowrap.Logger) { + if svc.adminHandler == nil { + return + } + + if !cfg.Auth.Enabled { + log.Warn().Msg("auth disabled: admin API endpoints are not registered") + return + } + + adminMiddlewares := []func(http.Handler) http.Handler{ + middleware.PanicRecovery(log), + middleware.RequestLogger(log, trustedNets), + middleware.SecurityHeaders, + } + + if svc.authSvc != nil { + // Create rate limiters for admin API - uses same config as registry. + var globalLimiter, ipLimiter out.RateLimiter + if cfg.API.RateLimit.Enabled { + globalLimiter = ratelimit.NewMemoryStore(cfg.API.RateLimit.GlobalRPS, cfg.API.RateLimit.Burst, log) + ipLimiter = ratelimit.NewMemoryStore(cfg.API.RateLimit.PerIPRPS, cfg.API.RateLimit.Burst, log) + } + adminMiddlewares = append(adminMiddlewares, admin.AuthMiddleware(svc.authSvc, globalLimiter, ipLimiter, trustedNets, log)) + } else { + adminMiddlewares = append(adminMiddlewares, func(next http.Handler) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(http.StatusServiceUnavailable) + _ = json.NewEncoder(w).Encode(dto.ErrorResponse{Error: "authentication service unavailable"}) + }) + }) + } + + adminWithMiddleware := otelhttp.NewHandler( + middleware.Chain(adminMiddlewares...)(svc.adminHandler), + "gordon.admin", + ) + registryMux.Handle("/admin/", adminWithMiddleware) +} diff --git a/internal/app/kernel.go b/internal/app/kernel.go index 28a74736d..13f274f58 100644 --- a/internal/app/kernel.go +++ b/internal/app/kernel.go @@ -10,7 +10,6 @@ import ( "github.com/bnema/gordon/internal/boundaries/in" configusecase "github.com/bnema/gordon/internal/usecase/config" - secretsusecase "github.com/bnema/gordon/internal/usecase/secrets" ) // Kernel provides in-process service access for local CLI execution. @@ -19,7 +18,6 @@ import ( type Kernel struct { authEnabled bool configSvc in.ConfigService - secretSvc in.SecretService containerSvc in.ContainerService backupSvc in.BackupService volumeBackupSvc in.VolumeBackupService @@ -28,7 +26,13 @@ type Kernel struct { logSvc in.LogService volumeSvc in.VolumeService publicTLSSvc in.PublicTLSService - cleanup func() + appSvc in.AppService + // appAdmin is the daemon-owned app lifecycle (when the full wiring is + // available). Close cancels and joins its in-flight background deploy + // executions before any other kernel resource is torn down. + appAdmin appAdministration + log zerowrap.Logger + cleanup func() } // NewKernel initializes local services without starting server listeners. @@ -78,10 +82,9 @@ func newKernel(configPath string, initLog kernelLoggerInit) (*Kernel, error) { cleanup() } - return &Kernel{ + kernel := &Kernel{ authEnabled: cfg.Auth.Enabled, configSvc: svc.configSvc, - secretSvc: svc.secretSvc, containerSvc: svc.containerSvc, backupSvc: svc.backupSvc, volumeBackupSvc: svc.volumeBackupSvc, @@ -90,8 +93,16 @@ func newKernel(configPath string, initLog kernelLoggerInit) (*Kernel, error) { logSvc: svc.logSvc, volumeSvc: svc.volumeSvc, publicTLSSvc: svc.publicTLSSvc, + appSvc: svc.appSvc, + log: log, cleanup: wrappedCleanup, - }, nil + } + // A nil *AppServiceImpl must not be stored in the interface: the + // interface would be non-nil and Close would call it. + if svc.appSvcImpl != nil { + kernel.appAdmin = svc.appSvcImpl + } + return kernel, nil } else { log.Warn().Err(fullErr).Msg("local kernel running in minimal mode") } @@ -102,18 +113,10 @@ func newKernel(configPath string, initLog kernelLoggerInit) (*Kernel, error) { return nil, fmt.Errorf("failed to load configuration: %w", err) } - _, _, _, domainSecretStore, err := createDomainSecretStore(cfg, log) - if err != nil { - cleanup() - return nil, fmt.Errorf("failed to create local secret store: %w", err) - } - - secretSvc := secretsusecase.NewService(domainSecretStore, log, nil) - return &Kernel{ authEnabled: cfg.Auth.Enabled, configSvc: configSvc, - secretSvc: secretSvc, + log: log, cleanup: cleanup, }, nil } @@ -122,18 +125,31 @@ func quietInitLogger(Config) (zerowrap.Logger, func(), error) { return zerowrap.New(zerowrap.Config{Level: "disabled", Output: io.Discard}), func() {}, nil } +// Close tears the kernel down. It first cancels and joins daemon-owned app +// administration on a bounded context, so a background deploy execution is +// never torn down mid-flight alongside the state and runtime it uses. When +// that quiescence times out the remaining cleanup is skipped and the error is +// returned: an unfinished execution may still be writing to state. func (k *Kernel) Close() error { - if k == nil || k.cleanup == nil { + if k == nil { return nil } - k.cleanup() + if k.appAdmin != nil { + ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second) + err := quiesceAppAdministration(ctx, k.appAdmin, k.log) + cancel() + if err != nil { + return err + } + } + if k.cleanup != nil { + k.cleanup() + } return nil } func (k *Kernel) Config() in.ConfigService { return k.configSvc } -func (k *Kernel) Secrets() in.SecretService { return k.secretSvc } - func (k *Kernel) Container() in.ContainerService { return k.containerSvc } func (k *Kernel) Backup() in.BackupService { return k.backupSvc } @@ -150,4 +166,6 @@ func (k *Kernel) Volumes() in.VolumeService { return k.volumeSvc } func (k *Kernel) PublicTLS() in.PublicTLSService { return k.publicTLSSvc } +func (k *Kernel) Apps() in.AppService { return k.appSvc } + func (k *Kernel) AuthEnabled() bool { return k != nil && k.authEnabled } diff --git a/internal/app/kernel_test.go b/internal/app/kernel_test.go index 3ea388fcd..1709c0f10 100644 --- a/internal/app/kernel_test.go +++ b/internal/app/kernel_test.go @@ -1,11 +1,14 @@ package app import ( + "errors" "fmt" "os" "path/filepath" "testing" + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) @@ -34,5 +37,57 @@ secrets_backend = "unsafe" t.Cleanup(func() { require.NoError(t, kernel.Close()) }) require.NotNil(t, kernel.Config()) - require.NotNil(t, kernel.Secrets()) +} + +// TestKernelClose_QuiescenceTimeoutSkipsCleanup proves the fail-closed kernel +// Close ordering: daemon-owned app administration is cancelled and joined on +// a bounded context before any remaining kernel resource is torn down, and +// when that quiescence times out the dependency teardown is skipped and the +// error is returned, so an unfinished deploy is never dropped onto a closed +// state or runtime. +func TestKernelClose_QuiescenceTimeoutSkipsCleanup(t *testing.T) { + t.Parallel() + + order := []string{} + admin := &orderRecordingAdmin{order: &order, err: errors.New("in-flight deploy did not unwind")} + cleanedUp := false + kernel := &Kernel{ + appAdmin: admin, + log: zerowrap.Default(), + cleanup: func() { cleanedUp = true; order = append(order, "cleanup") }, + } + + require.Error(t, kernel.Close()) + require.Equal(t, []string{"app-shutdown"}, order, "teardown must stop when quiescence times out") + assert.False(t, cleanedUp) + assert.True(t, admin.deadline, "app administration shutdown must run on a bounded context") +} + +// TestKernelClose_CleansUpAfterQuiescence keeps the normal path: once app +// administration joined cleanly, the kernel still tears its resources down. +func TestKernelClose_CleansUpAfterQuiescence(t *testing.T) { + t.Parallel() + + order := []string{} + admin := &orderRecordingAdmin{order: &order} + kernel := &Kernel{ + appAdmin: admin, + log: zerowrap.Default(), + cleanup: func() { order = append(order, "cleanup") }, + } + + require.NoError(t, kernel.Close()) + require.Equal(t, []string{"app-shutdown", "cleanup"}, order) +} + +// TestKernelClose_WithoutAppAdministrationStillCleansUp keeps the minimal +// kernel path (no full service wiring) closing without a nil-interface trap. +func TestKernelClose_WithoutAppAdministrationStillCleansUp(t *testing.T) { + t.Parallel() + + cleaned := false + kernel := &Kernel{log: zerowrap.Default(), cleanup: func() { cleaned = true }} + + require.NoError(t, kernel.Close()) + assert.True(t, cleaned) } diff --git a/internal/app/listeners.go b/internal/app/listeners.go new file mode 100644 index 000000000..69c5c2809 --- /dev/null +++ b/internal/app/listeners.go @@ -0,0 +1,418 @@ +package app + +import ( + "crypto/tls" + "crypto/x509" + "fmt" + "net" + "net/http" + "strconv" + "strings" + "time" + + "github.com/bnema/zerowrap" + + trafficadapter "github.com/bnema/gordon/internal/adapters/in/traffic" + "github.com/bnema/gordon/internal/boundaries/in" + "github.com/bnema/gordon/internal/domain" + pkiusecase "github.com/bnema/gordon/internal/usecase/pki" + "github.com/bnema/gordon/internal/usecase/publictls" + "github.com/bnema/gordon/internal/usecase/traffic" +) + +// startProxyServers sets up the HTTP proxy server and, when tls_port != 0, +// an HTTPS proxy server with on-demand TLS certificates from the internal CA. +// certificateSelector implements a multi-source TLS certificate lookup. +// Priority: static certs → public ACME TLS → local PKI (internal CA). +type certificateSelector struct { + staticCerts []staticTLSCertificate + publicTLS in.PublicTLSService + localPKI *pkiusecase.Service +} + +type staticTLSCertificate struct { + cert tls.Certificate + leaf *x509.Certificate +} + +// GetCertificate selects a TLS certificate based on the ClientHello SNI. +// +// Priority: +// 1. Static certs — exact SNI match (leaf VerifyHostname) +// 2. Public ACME TLS — if the host requires ACME coverage +// 3. Local PKI (internal CA) — fallback for all other hosts +// 4. nil, nil — if no source can serve the host +func (s *certificateSelector) GetCertificate(hello *tls.ClientHelloInfo) (*tls.Certificate, error) { + // 1. Static certs — exact match via leaf VerifyHostname. + if cert := matchingPreparedStaticCert(s.staticCerts, hello.ServerName); cert != nil { + return cert, nil + } + + // 2. Public ACME TLS. + if s.publicTLS != nil { + cert, err := s.publicTLS.GetCertificateForHost(hello.ServerName) + if err == nil && cert != nil { + return cert, nil + } + // nil, nil means this host is not an ACME-required route. Errors mean + // public ACME cannot currently serve this host. In both cases, fall + // through to local PKI instead of aborting the TLS handshake. + } + + // 3. Local PKI (internal CA). + if s.localPKI != nil { + return s.localPKI.GetCertificate(hello) + } + + return nil, nil +} + +func prepareStaticTLSCertificates(certs []tls.Certificate) []staticTLSCertificate { + prepared := make([]staticTLSCertificate, 0, len(certs)) + for _, cert := range certs { + if cert.Leaf == nil && len(cert.Certificate) > 0 { + leaf, err := x509.ParseCertificate(cert.Certificate[0]) + if err == nil { + cert.Leaf = leaf + } + } + prepared = append(prepared, staticTLSCertificate{cert: cert, leaf: cert.Leaf}) + } + return prepared +} + +func matchingPreparedStaticCert(certs []staticTLSCertificate, serverName string) *tls.Certificate { + if serverName == "" { + if len(certs) == 0 { + return nil + } + return &certs[0].cert + } + for i := range certs { + if certs[i].leaf == nil { + continue + } + if err := certs[i].leaf.VerifyHostname(serverName); err == nil { + return &certs[i].cert + } + } + return nil +} + +// matchingStaticCert returns a pointer to the first static certificate whose +// leaf verifies the given serverName. Returns nil if no match is found. +func matchingStaticCert(certs []tls.Certificate, serverName string) *tls.Certificate { + return matchingPreparedStaticCert(prepareStaticTLSCertificates(certs), serverName) +} + +func startProxyServers(cfg Config, httpHandler, httpsHandler http.Handler, pkiSvc *pkiusecase.Service, publicTLS in.PublicTLSService, trafficManager *trafficadapter.Manager, log zerowrap.Logger) (*http.Server, <-chan struct{}, *http.Server, <-chan struct{}, error) { + var httpSrv *http.Server + var httpReady <-chan struct{} + + needsTLS := hasTLSCapableEntrypoint(cfg) + if !hasSmartTCPEntrypoint(cfg) && !needsTLS { + return httpSrv, httpReady, nil, nil, nil + } + cleanupHTTP := func(err error) (*http.Server, <-chan struct{}, *http.Server, <-chan struct{}, error) { + return httpSrv, httpReady, nil, nil, err + } + + var tlsConfig *tls.Config + if needsTLS { + var err error + tlsConfig, err = proxyTLSConfig(cfg, pkiSvc, publicTLS, log) + if err != nil { + return cleanupHTTP(err) + } + } + if trafficManager == nil { + return cleanupHTTP(fmt.Errorf("traffic manager is required when traffic entrypoints are enabled")) + } + registerTLSMuxHTTPServers(trafficManager, cfg, httpsHandler, tlsConfig, nil) + registerSmartTCPHTTPServers(trafficManager, cfg, httpHandler, httpsHandler, tlsConfig, nil) + + return httpSrv, httpReady, nil, nil, nil +} + +func hasSmartTCPEntrypoint(cfg Config) bool { + for _, entryPoint := range cfg.EntryPoints { + if entryPoint.Protocol == domain.EntryPointProtocolSmartTCP { + return true + } + } + return false +} + +func hasTLSCapableEntrypoint(cfg Config) bool { + for _, entryPoint := range cfg.EntryPoints { + switch entryPoint.Protocol { + case domain.EntryPointProtocolSmartTCP, domain.EntryPointProtocolTLSMux: + return true + } + } + return false +} + +func effectivePublicTLSPort(cfg Config) int { + return effectiveEntrypointPort(cfg, tlsCapableEntryPoint) +} + +func effectiveProxyTLSPort(cfg Config) int { + return effectivePublicTLSPort(cfg) +} + +func effectiveProxyHTTPPort(cfg Config) int { + return effectiveEntrypointPort(cfg, func(protocol domain.EntryPointProtocol) bool { + return protocol == domain.EntryPointProtocolSmartTCP + }) +} + +func effectiveEntrypointPort(cfg Config, match func(domain.EntryPointProtocol) bool) int { + if entryPoint, ok := cfg.EntryPoints[traffic.DefaultEdgeEntryPointName]; ok && match(entryPoint.Protocol) { + if port := portFromAddress(entryPoint.Address); port > 0 { + return port + } + } + + var candidatePort int + candidates := 0 + for name, entryPoint := range cfg.EntryPoints { + if name == traffic.DefaultEdgeEntryPointName || !match(entryPoint.Protocol) { + continue + } + port := portFromAddress(entryPoint.Address) + if port == 0 { + continue + } + candidatePort = port + candidates++ + } + if candidates == 1 { + return candidatePort + } + return 0 +} + +func tlsCapableEntryPoint(protocol domain.EntryPointProtocol) bool { + switch protocol { + case domain.EntryPointProtocolSmartTCP, domain.EntryPointProtocolTLSMux: + return true + default: + return false + } +} + +func effectiveHTTP01Port(cfg Config) int { + if hasSmartTCPHTTP01Entrypoint(cfg) { + return 80 + } + return 0 +} + +func validatePublicTLSReadiness(cfg Config) error { + mode, err := domain.ParseACMEChallengeMode(cfg.TLS.ACME.Challenge) + if err != nil { + return err + } + switch mode { + case domain.ACMEChallengeCloudflareDNS01, domain.ACMEChallengeAuto: + return nil + case domain.ACMEChallengeHTTP01: + return validateHTTP01ChallengeReadiness(cfg) + default: + return fmt.Errorf("%w: %q", domain.ErrACMEChallengeInvalid, mode) + } +} + +func validateEffectivePublicTLSReadiness(cfg Config, effective publictls.EffectiveChallenge) error { + if effective.Mode != domain.ACMEChallengeHTTP01 { + return nil + } + return validateHTTP01ChallengeReadiness(cfg) +} + +func validateHTTP01ChallengeReadiness(cfg Config) error { + if hasBoundHTTP01ChallengeListener(cfg) { + return nil + } + return fmt.Errorf("%w: http-01 requires an actually bound HTTP-01 challenge listener on external :80", domain.ErrACMEChallengeInvalid) +} + +func hasBoundHTTP01ChallengeListener(cfg Config) bool { + return hasSmartTCPHTTP01Entrypoint(cfg) +} + +func hasSmartTCPHTTP01Entrypoint(cfg Config) bool { + for _, entryPoint := range cfg.EntryPoints { + if entryPoint.Protocol == domain.EntryPointProtocolSmartTCP && portFromAddress(entryPoint.Address) == 80 { + return true + } + } + return false +} + +func portFromAddress(address string) int { + _, portText, err := net.SplitHostPort(address) + if err != nil { + return 0 + } + port, err := strconv.Atoi(portText) + if err != nil { + return 0 + } + return port +} + +func proxyTLSConfig(cfg Config, pkiSvc *pkiusecase.Service, publicTLS in.PublicTLSService, log zerowrap.Logger) (*tls.Config, error) { + var staticCerts []tls.Certificate + if cfg.Server.TLSCertFile != "" { + staticCert, err := tls.LoadX509KeyPair(cfg.Server.TLSCertFile, cfg.Server.TLSKeyFile) + if err != nil { + return nil, fmt.Errorf("load TLS keypair: %w", err) + } + staticCerts = []tls.Certificate{staticCert} + log.Info(). + Str("cert", cfg.Server.TLSCertFile). + Str("key", cfg.Server.TLSKeyFile). + Msg("loaded static TLS certificate (public ACME and internal CA handle remaining domains)") + } + selector := &certificateSelector{ + staticCerts: prepareStaticTLSCertificates(staticCerts), + publicTLS: publicTLS, + localPKI: pkiSvc, + } + return &tls.Config{ + MinVersion: tls.VersionTLS12, + GetCertificate: selector.GetCertificate, + NextProtos: []string{"h2", "http/1.1"}, + }, nil +} + +func registerSmartTCPHTTPServers(manager *trafficadapter.Manager, cfg Config, httpHandler, httpsHandler http.Handler, tlsConfig *tls.Config, previous map[string]struct{}) map[string]struct{} { + if manager == nil { + return previous + } + next := smartTCPHTTPServerNames(cfg) + for name := range previous { + if _, ok := next[name]; !ok { + manager.SetSmartTCPHTTPServer(name, nil, nil) + manager.SetSmartTCPTLSServer(name, nil, nil) + } + } + var httpProtos http.Protocols + httpProtos.SetHTTP1(true) + httpProtos.SetUnencryptedHTTP2(true) + for name := range next { + if httpHandler != nil { + manager.SetSmartTCPHTTPServer(name, httpHandler, &httpProtos) + } + if httpsHandler != nil && tlsConfig != nil { + manager.SetSmartTCPTLSServer(name, httpsHandler, tlsConfig) + } + } + return next +} + +func smartTCPHTTPServerNames(cfg Config) map[string]struct{} { + names := map[string]struct{}{} + for name, entryPoint := range cfg.EntryPoints { + if entryPoint.Protocol == domain.EntryPointProtocolSmartTCP { + names[name] = struct{}{} + } + } + return names +} + +func registerTLSMuxHTTPServers(manager *trafficadapter.Manager, cfg Config, httpsHandler http.Handler, tlsConfig *tls.Config, previous map[string]struct{}) map[string]struct{} { + if manager == nil { + return previous + } + next := tlsMuxHTTPServerNames(cfg) + for name := range previous { + if _, ok := next[name]; !ok { + manager.SetTLSHTTPServer(name, nil, nil) + } + } + if httpsHandler == nil || tlsConfig == nil { + return next + } + for name := range next { + manager.SetTLSHTTPServer(name, httpsHandler, tlsConfig) + } + return next +} + +func tlsMuxHTTPServerNames(cfg Config) map[string]struct{} { + names := map[string]struct{}{} + for name, entryPoint := range cfg.EntryPoints { + domainEntryPoint := domain.EntryPoint{Name: name, Protocol: entryPoint.Protocol} + if trafficManagerOwnsEntryPoint(domainEntryPoint) && domainEntryPoint.Protocol == domain.EntryPointProtocolTLSMux { + names[name] = struct{}{} + } + } + return names +} + +func startRegistryServers(cfg Config, handler http.Handler, errChan chan<- error, log zerowrap.Logger) (*http.Server, <-chan struct{}, *http.Server, <-chan struct{}) { + port := strconv.Itoa(cfg.Server.RegistryPort) + externalAddr := net.JoinHostPort(cfg.Server.RegistryListenAddr, port) + externalSrv, externalReady := startServer(externalAddr, handler, "registry", nil, errChan, log) + internalReady := closedReadyChannel() + if !needsInternalRegistryListener(cfg.Server.RegistryListenAddr) { + return externalSrv, externalReady, nil, internalReady + } + internalAddr := net.JoinHostPort("127.0.0.1", port) + internalSrv, internalReady := startServer(internalAddr, handler, "internal registry", nil, errChan, log) + return externalSrv, externalReady, internalSrv, internalReady +} + +func closedReadyChannel() <-chan struct{} { + ready := make(chan struct{}) + close(ready) + return ready +} + +func needsInternalRegistryListener(listenAddr string) bool { + addr := net.ParseIP(strings.TrimSpace(listenAddr)) + if addr == nil { + return listenAddr != "" + } + return !addr.IsLoopback() && !addr.IsUnspecified() +} + +// startServer starts an HTTP server, returning the server instance and a channel +// that closes once the listening socket is bound. This lets callers wait for the +// port to be ready before taking actions that depend on it (e.g. auto-start +// pulling from the local registry). The returned *http.Server can be used for +// graceful shutdown. +func startServer(addr string, handler http.Handler, name string, protocols *http.Protocols, errChan chan<- error, log zerowrap.Logger) (*http.Server, <-chan struct{}) { + ready := make(chan struct{}) + + server := &http.Server{ + Addr: addr, + Handler: handler, + Protocols: protocols, + ReadHeaderTimeout: 10 * time.Second, + ReadTimeout: 5 * time.Minute, + WriteTimeout: 5 * time.Minute, + IdleTimeout: 120 * time.Second, + MaxHeaderBytes: 1 << 20, + } + + go func() { + log.Info().Str("address", addr).Msgf("%s server starting", name) + + ln, err := net.Listen("tcp", addr) + if err != nil { + errChan <- fmt.Errorf("%s server error: %w", name, err) + return + } + close(ready) // signal: port is bound and accepting connections + + if err := server.Serve(ln); err != nil && err != http.ErrServerClosed { + errChan <- fmt.Errorf("%s server error: %w", name, err) + } + }() + + return server, ready +} diff --git a/internal/app/local_admin_socket.go b/internal/app/local_admin_socket.go new file mode 100644 index 000000000..2a394b458 --- /dev/null +++ b/internal/app/local_admin_socket.go @@ -0,0 +1,104 @@ +package app + +import ( + "context" + "errors" + "fmt" + "net/http" + "os" + "sync" + "time" + + "github.com/bnema/zerowrap" + + "github.com/bnema/gordon/internal/adapters/in/http/middleware" + "github.com/bnema/gordon/internal/adapters/localadmin" +) + +// localAdminServer owns the daemon's owner-only admin Unix socket: one HTTP +// listener, one local-only authority handler, and the socket file it created. +type localAdminServer struct { + srv *http.Server + path string + info os.FileInfo + log zerowrap.Logger + once sync.Once +} + +// startLocalAdminServer binds the owner-only admin socket, composes the +// local-only authority over the admin handler, and serves it in the +// background. Any startup error aborts daemon startup; a later serve error is +// reported on errChan so the daemon fails rather than silently losing local +// administration. +func startLocalAdminServer(svc *services, errChan chan<- error, log zerowrap.Logger) (*localAdminServer, error) { + if svc == nil || svc.adminHandler == nil { + return nil, nil + } + + dir, err := localadmin.EnsureRuntimeDir() + if err != nil { + return nil, fmt.Errorf("local admin socket directory: %w", err) + } + ln, info, err := localadmin.Listen(dir) + if err != nil { + return nil, fmt.Errorf("local admin socket: %w", err) + } + + handler := middleware.Chain( + middleware.PanicRecovery(log), + middleware.SecurityHeaders, + )(svc.adminHandler.LocalAuthority()) + + // No WriteTimeout: app log streaming over the socket stays open for as long + // as the client follows it. + srv := &http.Server{ + Handler: handler, + ReadHeaderTimeout: 10 * time.Second, + IdleTimeout: 120 * time.Second, + MaxHeaderBytes: 1 << 20, + } + + s := &localAdminServer{ + srv: srv, + path: localadmin.SocketPath(dir), + info: info, + log: log, + } + + go func() { + if serveErr := srv.Serve(ln); serveErr != nil && !errors.Is(serveErr, http.ErrServerClosed) { + wrapped := fmt.Errorf("local admin socket server: %w", serveErr) + select { + case errChan <- wrapped: + default: + log.Error().Err(serveErr).Msg("local admin socket server error") + } + } + }() + + log.Info().Str("path", s.path).Msg("local admin socket listening") + return s, nil +} + +// Close drains and stops the socket server, then removes the socket file only +// when it is still the file this listener bound. It is safe to call twice. +func (s *localAdminServer) Close() { + if s == nil { + return + } + s.once.Do(func() { + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + if err := s.srv.Shutdown(ctx); err != nil { + s.log.Warn().Err(err).Msg("local admin socket shutdown error") + if closeErr := s.srv.Close(); closeErr != nil { + s.log.Warn().Err(closeErr).Msg("failed to force-close local admin socket") + } + } + if err := localadmin.RemoveOwnedSocket(s.path, s.info); err != nil { + s.log.Warn().Err(err).Str("path", s.path).Msg("failed to remove local admin socket") + return + } + s.log.Info().Msg("local admin socket stopped") + }) +} diff --git a/internal/app/local_admin_socket_test.go b/internal/app/local_admin_socket_test.go new file mode 100644 index 000000000..8b6ef7d95 --- /dev/null +++ b/internal/app/local_admin_socket_test.go @@ -0,0 +1,142 @@ +package app + +import ( + "context" + "io" + "net" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + adminhttp "github.com/bnema/gordon/internal/adapters/in/http/admin" + "github.com/bnema/gordon/internal/adapters/localadmin" + in "github.com/bnema/gordon/internal/boundaries/in" + inmocks "github.com/bnema/gordon/internal/boundaries/in/mocks" +) + +// unixHTTPClient returns a client that dials the given Unix socket and never +// consults environment proxies. +func unixHTTPClient(socketPath string) *http.Client { + dialer := &net.Dialer{Timeout: 5 * time.Second} + return &http.Client{ + Timeout: 10 * time.Second, + Transport: &http.Transport{ + Proxy: nil, + DialContext: func(ctx context.Context, _, _ string) (net.Conn, error) { + return dialer.DialContext(ctx, "unix", socketPath) + }, + }, + CheckRedirect: func(*http.Request, []*http.Request) error { + return http.ErrUseLastResponse + }, + } +} + +func localAdminTestServices(t *testing.T, appSvc in.AppService) *services { + t.Helper() + return &services{adminHandler: adminhttp.NewHandler(adminhttp.HandlerDeps{ + AppSvc: appSvc, + Log: zerowrap.Default(), + })} +} + +func TestLocalAdminSocketServesAppSurfaceOverUnix(t *testing.T) { + xdg := t.TempDir() + t.Setenv("XDG_RUNTIME_DIR", xdg) + + appSvc := inmocks.NewMockAppService(t) + appSvc.EXPECT().List(mock.Anything).Return([]in.AppSummary(nil), nil).Once() + + server, err := startLocalAdminServer(localAdminTestServices(t, appSvc), make(chan error, 4), zerowrap.Default()) + require.NoError(t, err) + require.NotNil(t, server) + + path := localadmin.SocketPath(filepath.Join(xdg, "gordon")) + require.NoError(t, localadmin.ValidateSocket(path)) + + client := unixHTTPClient(path) + + resp, err := client.Get("http://gordon.local/admin/apps") + require.NoError(t, err) + body, err := io.ReadAll(resp.Body) + require.NoError(t, err) + resp.Body.Close() + require.Equal(t, http.StatusOK, resp.StatusCode, "body: %s", body) + + resp, err = client.Get("http://gordon.local/admin/auth/tokens") + require.NoError(t, err) + resp.Body.Close() + assert.Equal(t, http.StatusForbidden, resp.StatusCode) + + server.Close() + assert.NoFileExists(t, path) +} + +func TestLocalAdminSocketTCPAdminStaysDisabled(t *testing.T) { + xdg := t.TempDir() + t.Setenv("XDG_RUNTIME_DIR", xdg) + + cfg := Config{} + cfg.Auth.Enabled = false + + svc := localAdminTestServices(t, nil) + registryHandler, _, _ := createHTTPHandlers(svc, cfg, zerowrap.Default(), nil) + + // The TCP registry mux must not expose /admin/* in local-only mode. + for _, path := range []string{"/admin/status", "/admin/apps", "/admin/apps/apply"} { + req := httptest.NewRequest(http.MethodGet, path, nil) + req.RemoteAddr = "192.0.2.10:12345" + rec := httptest.NewRecorder() + registryHandler.ServeHTTP(rec, req) + require.Equal(t, http.StatusNotFound, rec.Code, "path %s", path) + } + + // The owner-only socket does serve the app surface. + server, err := startLocalAdminServer(svc, make(chan error, 4), zerowrap.Default()) + require.NoError(t, err) + t.Cleanup(server.Close) + + path := localadmin.SocketPath(filepath.Join(xdg, "gordon")) + client := unixHTTPClient(path) + resp, err := client.Get("http://gordon.local/admin/apps") + require.NoError(t, err) + resp.Body.Close() + assert.Equal(t, http.StatusServiceUnavailable, resp.StatusCode) +} + +func TestLocalAdminSocketCloseLeavesReplacementFile(t *testing.T) { + xdg := t.TempDir() + t.Setenv("XDG_RUNTIME_DIR", xdg) + + server, err := startLocalAdminServer(localAdminTestServices(t, inmocks.NewMockAppService(t)), make(chan error, 4), zerowrap.Default()) + require.NoError(t, err) + + path := localadmin.SocketPath(filepath.Join(xdg, "gordon")) + require.NoError(t, os.Remove(path)) + require.NoError(t, os.WriteFile(path, []byte("replacement"), 0o600)) + + server.Close() + + content, err := os.ReadFile(path) + require.NoError(t, err) + assert.Equal(t, "replacement", string(content)) + + // Close is idempotent and must not remove the replacement on a second call. + server.Close() + assert.FileExists(t, path) +} + +func TestStartLocalAdminServerSkipsWithoutAdminHandler(t *testing.T) { + server, err := startLocalAdminServer(&services{}, make(chan error, 4), zerowrap.Default()) + require.NoError(t, err) + assert.Nil(t, server) + server.Close() +} diff --git a/internal/app/migrate_env.go b/internal/app/migrate_env.go deleted file mode 100644 index dba3bc207..000000000 --- a/internal/app/migrate_env.go +++ /dev/null @@ -1,228 +0,0 @@ -package app - -import ( - "errors" - "fmt" - "os" - "path/filepath" - "strings" - - "github.com/bnema/zerowrap" - - "github.com/bnema/gordon/internal/adapters/out/domainsecrets" - "github.com/bnema/gordon/internal/domain" -) - -func migrateEnvFilesToPass(envDir string, passStore *domainsecrets.PassStore, log zerowrap.Logger) error { - entries, err := readEnvDir(envDir) - if err != nil { - return err - } - - for _, entry := range entries { - if entry.IsDir() { - continue - } - name := entry.Name() - if !isPlainEnvFile(name) { - continue - } - if err := migrateEnvFile(envDir, name, passStore, log); err != nil { - return err - } - } - - if err := migrateAttachmentEnvFilesToPass(envDir, passStore, log); err != nil { - return err - } - - return nil -} - -func readEnvDir(envDir string) ([]os.DirEntry, error) { - entries, err := os.ReadDir(envDir) - if err != nil { - if os.IsNotExist(err) { - return nil, nil - } - return nil, fmt.Errorf("failed to read env directory: %w", err) - } - return entries, nil -} - -func isPlainEnvFile(name string) bool { - if !strings.HasSuffix(name, ".env") || strings.HasSuffix(name, ".env.migrated") { - return false - } - // Skip attachment env files: gordon--.env - if strings.HasPrefix(name, "gordon-") { - return false - } - return true -} - -func migrateEnvFile(envDir, name string, passStore *domainsecrets.PassStore, log zerowrap.Logger) error { - filePath := filepath.Join(envDir, name) - data, err := os.ReadFile(filePath) - if err != nil { - return fmt.Errorf("failed to read env file %s: %w", filePath, err) - } - - domainName := strings.TrimSuffix(name, ".env") - secrets, err := domain.ParseEnvData(data) - if err != nil { - return fmt.Errorf("failed to parse env file %s: %w", filePath, err) - } - if len(secrets) == 0 { - log.Info(). - Str("file", filePath). - Msg("no secrets found in env file; skipping migration") - return nil - } - - writtenKeys, err := passStore.SetIfEmpty(domainName, secrets) - if err != nil { - if errors.Is(err, domain.ErrSecretsAlreadyExist) { - return fmt.Errorf("%w; refusing to delete plaintext env file automatically", err) - } - return fmt.Errorf("failed to migrate secrets for %s: %w", domainName, err) - } - - if err := os.Remove(filePath); err != nil { - // Check if the stored secrets match what we just wrote - storedSecrets, getErr := passStore.GetAll(domainName) - if getErr == nil && secretsMatch(storedSecrets, secrets) { - // Migration was successful, removal failed but secrets are stored - log.Warn(). - Str("file", filePath). - Str("domain", domainName). - Msg("secrets migrated successfully but failed to remove original file; treating as success") - return nil - } - // Attempt rollback: remove only the pass entries this migration wrote. - for _, key := range writtenKeys { - if delErr := passStore.Delete(domainName, key); delErr != nil { - log.Error(). - Err(delErr). - Str("domain", domainName). - Str("key", key). - Msg("failed to rollback pass entry after file removal failure") - } - } - return fmt.Errorf("failed to remove migrated env file %s: %w", filePath, err) - } - - log.Info(). - Int(zerowrap.FieldCount, len(secrets)). - Str("domain", domainName). - Msg("migrated secrets for domain from plain text to pass and removed original env file") - - return nil -} - -func migrateAttachmentEnvFilesToPass(envDir string, passStore *domainsecrets.PassStore, log zerowrap.Logger) error { - entries, err := readEnvDir(envDir) - if err != nil { - return err - } - - for _, entry := range entries { - if entry.IsDir() { - continue - } - name := entry.Name() - if !isAttachmentEnvFile(name) { - continue - } - if err := migrateAttachmentEnvFile(envDir, name, passStore, log); err != nil { - return err - } - } - - return nil -} - -func isAttachmentEnvFile(name string) bool { - if !strings.HasSuffix(name, ".env") || strings.HasSuffix(name, ".env.migrated") { - return false - } - return strings.HasPrefix(name, "gordon-") -} - -func migrateAttachmentEnvFile(envDir, name string, passStore *domainsecrets.PassStore, log zerowrap.Logger) error { - filePath := filepath.Join(envDir, name) - data, err := os.ReadFile(filePath) - if err != nil { - return fmt.Errorf("failed to read attachment env file %s: %w", filePath, err) - } - - containerName := extractContainerNameFromAttachmentFile(name) - secrets, err := domain.ParseEnvData(data) - if err != nil { - return fmt.Errorf("failed to parse attachment env file %s: %w", filePath, err) - } - if len(secrets) == 0 { - log.Info(). - Str("file", filePath). - Msg("no secrets found in attachment env file; skipping migration") - return nil - } - - writtenKeys, err := passStore.SetAttachmentIfEmpty(containerName, secrets) - if err != nil { - if errors.Is(err, domain.ErrSecretsAlreadyExist) { - return fmt.Errorf("%w; refusing to delete plaintext env file automatically", err) - } - return fmt.Errorf("failed to migrate attachment secrets for %s: %w", containerName, err) - } - - if err := os.Remove(filePath); err != nil { - // Check if the stored secrets match what we just wrote - storedSecrets, getErr := passStore.GetAllAttachment(containerName) - if getErr == nil && secretsMatch(storedSecrets, secrets) { - // Migration was successful, removal failed but secrets are stored - log.Warn(). - Str("file", filePath). - Str("container", containerName). - Msg("attachment secrets migrated successfully but failed to remove original file; treating as success") - return nil - } - // Attempt rollback: remove only the pass entries this migration wrote. - for _, key := range writtenKeys { - if delErr := passStore.DeleteAttachment(containerName, key); delErr != nil { - log.Error(). - Err(delErr). - Str("container", containerName). - Str("key", key). - Msg("failed to rollback pass entry after file removal failure") - } - } - return fmt.Errorf("failed to remove migrated attachment env file %s: %w", filePath, err) - } - - log.Info(). - Int(zerowrap.FieldCount, len(secrets)). - Str("container", containerName). - Str("file", filePath). - Msg("migrated attachment secrets to pass and removed original env file") - - return nil -} - -func extractContainerNameFromAttachmentFile(filename string) string { - name := strings.TrimSuffix(filename, ".env") - return name -} - -func secretsMatch(a, b map[string]string) bool { - if len(a) != len(b) { - return false - } - for k, v := range a { - bv, ok := b[k] - if !ok || bv != v { - return false - } - } - return true -} diff --git a/internal/app/migrate_env_test.go b/internal/app/migrate_env_test.go deleted file mode 100644 index a40745d1a..000000000 --- a/internal/app/migrate_env_test.go +++ /dev/null @@ -1,214 +0,0 @@ -package app - -import ( - "context" - "fmt" - "os" - "os/exec" - "path/filepath" - "testing" - "time" - - "github.com/bnema/zerowrap" - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" - - "github.com/bnema/gordon/internal/adapters/out/domainsecrets" - "github.com/bnema/gordon/internal/domain" -) - -func passCmd(args ...string) error { - ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second) - defer cancel() - return exec.CommandContext(ctx, "pass", args...).Run() -} - -func requirePass(t *testing.T) { - if err := passCmd("version"); err != nil { - t.Skip("pass not available") - } - if err := passCmd("ls"); err != nil { - t.Skip("pass store not initialized") - } -} - -func cleanupPassDomain(domainName string, keys []string) { - safeDomain, err := domain.SanitizeDomainForEnvFile(domainName) - if err != nil { - return - } - - for _, key := range keys { - path := fmt.Sprintf("%s/%s/%s", domainsecrets.PassDomainSecretsPath, safeDomain, key) - _ = passCmd("rm", "-f", path) - } - - manifestPath := fmt.Sprintf("%s/%s/.keys", domainsecrets.PassDomainSecretsPath, safeDomain) - _ = passCmd("rm", "-f", manifestPath) -} - -func TestMigrateEnvFile(t *testing.T) { - requirePass(t) - - tmpDir, err := os.MkdirTemp("", "migrate-env-test-*") - require.NoError(t, err) - defer os.RemoveAll(tmpDir) - - logger := zerowrap.Default() - store, err := domainsecrets.NewPassStore(logger) - require.NoError(t, err) - - domainName := fmt.Sprintf("test.%d.example.com", time.Now().UnixNano()) - keys := []string{"API_KEY", "DB_PASSWORD"} - defer cleanupPassDomain(domainName, keys) - - envContent := "API_KEY=secret123\nDB_PASSWORD=pass456\n" - envFileName := domainName + ".env" - envFilePath := filepath.Join(tmpDir, envFileName) - err = os.WriteFile(envFilePath, []byte(envContent), 0600) - require.NoError(t, err) - - err = migrateEnvFile(tmpDir, envFileName, store, logger) - require.NoError(t, err) - - _, err = os.Stat(envFilePath) - assert.True(t, os.IsNotExist(err), "original .env file should be removed") - - _, err = os.Stat(envFilePath + ".migrated") - assert.True(t, os.IsNotExist(err), ".env.migrated plaintext should not be preserved by default") - - passKeys, err := store.ListKeys(domainName) - require.NoError(t, err) - assert.ElementsMatch(t, keys, passKeys) - - values, err := store.GetAll(domainName) - require.NoError(t, err) - assert.Equal(t, "secret123", values["API_KEY"]) - assert.Equal(t, "pass456", values["DB_PASSWORD"]) -} - -func cleanupPassAttachment(_ *testing.T, containerName string, keys []string) { - for _, key := range keys { - path := fmt.Sprintf("%s/%s/%s", domainsecrets.PassAttachmentPath, containerName, key) - _ = passCmd("rm", "-f", path) - } - - manifestPath := fmt.Sprintf("%s/%s/.keys", domainsecrets.PassAttachmentPath, containerName) - _ = passCmd("rm", "-f", manifestPath) -} - -func TestMigrateAttachmentEnvFile(t *testing.T) { - requirePass(t) - - tmpDir, err := os.MkdirTemp("", "migrate-attachment-env-test-*") - require.NoError(t, err) - defer os.RemoveAll(tmpDir) - - logger := zerowrap.Default() - store, err := domainsecrets.NewPassStore(logger) - require.NoError(t, err) - - containerName := fmt.Sprintf("gitea-postgres-%d", time.Now().UnixNano()) - envFileName := "gordon-" + containerName + ".env" - // extractContainerNameFromAttachmentFile strips the .env suffix, so the - // stored name includes the "gordon-" prefix from the filename. - storedContainerName := extractContainerNameFromAttachmentFile(envFileName) - keys := []string{"POSTGRES_USER", "POSTGRES_PASSWORD"} - defer cleanupPassAttachment(t, storedContainerName, keys) - - envContent := "POSTGRES_USER=gitea\nPOSTGRES_PASSWORD=secret123\n" - envFilePath := filepath.Join(tmpDir, envFileName) - err = os.WriteFile(envFilePath, []byte(envContent), 0600) - require.NoError(t, err) - - err = migrateAttachmentEnvFile(tmpDir, envFileName, store, logger) - require.NoError(t, err) - - _, err = os.Stat(envFilePath) - assert.True(t, os.IsNotExist(err), "original .env file should be removed") - - _, err = os.Stat(envFilePath + ".migrated") - assert.True(t, os.IsNotExist(err), ".env.migrated plaintext should not be preserved by default") - - values, err := store.GetAllAttachment(storedContainerName) - require.NoError(t, err) - assert.Equal(t, "gitea", values["POSTGRES_USER"]) - assert.Equal(t, "secret123", values["POSTGRES_PASSWORD"]) -} - -func TestIsAttachmentEnvFile(t *testing.T) { - tests := []struct { - name string - filename string - want bool - }{ - { - name: "attachment env file", - filename: "gordon-gitea-postgres.env", - want: true, - }, - { - name: "attachment env file with complex name", - filename: "gordon-git-example-com-gitea-postgres.env", - want: true, - }, - { - name: "plain domain env file", - filename: "example.com.env", - want: false, - }, - { - name: "migrated attachment file", - filename: "gordon-gitea-postgres.env.migrated", - want: false, - }, - { - name: "non-env file", - filename: "config.yaml", - want: false, - }, - { - name: "gordon prefix but no env suffix", - filename: "gordon-something", - want: false, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - got := isAttachmentEnvFile(tt.filename) - assert.Equal(t, tt.want, got) - }) - } -} - -func TestExtractContainerNameFromAttachmentFile(t *testing.T) { - tests := []struct { - name string - filename string - want string - }{ - { - name: "simple attachment file", - filename: "gordon-gitea-postgres.env", - want: "gordon-gitea-postgres", - }, - { - name: "complex attachment file", - filename: "gordon-git-example-com-gitea-redis.env", - want: "gordon-git-example-com-gitea-redis", - }, - { - name: "single word container", - filename: "gordon-redis.env", - want: "gordon-redis", - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - got := extractContainerNameFromAttachmentFile(tt.filename) - assert.Equal(t, tt.want, got) - }) - } -} diff --git a/internal/app/paths.go b/internal/app/paths.go index 52fc646bd..971b1f096 100644 --- a/internal/app/paths.go +++ b/internal/app/paths.go @@ -4,6 +4,7 @@ package app import ( "os" "path/filepath" + "strings" "github.com/spf13/viper" ) @@ -23,6 +24,16 @@ func DefaultDataDir() string { func ConfigureViper(v *viper.Viper, configPath string) { if configPath != "" { v.SetConfigFile(configPath) + switch ext := strings.ToLower(filepath.Ext(configPath)); ext { + case ".json": + v.SetConfigType("json") + case ".yaml", ".yml": + v.SetConfigType("yaml") + case ".toml": + v.SetConfigType("toml") + default: + v.SetConfigType("toml") + } } else { v.SetConfigName("gordon") v.SetConfigType("toml") diff --git a/internal/app/paths_test.go b/internal/app/paths_test.go new file mode 100644 index 000000000..106cb5159 --- /dev/null +++ b/internal/app/paths_test.go @@ -0,0 +1,47 @@ +package app + +import ( + "os" + "path/filepath" + "testing" + + "github.com/spf13/viper" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func TestConfigureViper_ExplicitPathExtensionIsCaseInsensitive(t *testing.T) { + tests := []struct { + name string + ext string + content string + }{ + {name: "lowercase yaml", ext: ".yaml", content: "foo: bar\n"}, + {name: "uppercase yaml", ext: ".YAML", content: "foo: bar\n"}, + {name: "mixed case yml", ext: ".YmL", content: "foo: bar\n"}, + {name: "uppercase json", ext: ".JSON", content: "{\"foo\":\"bar\"}\n"}, + {name: "uppercase toml", ext: ".TOML", content: "foo = \"bar\"\n"}, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + path := filepath.Join(t.TempDir(), "gordon"+tt.ext) + require.NoError(t, os.WriteFile(path, []byte(tt.content), 0o600)) + + v := viper.New() + ConfigureViper(v, path) + require.NoError(t, v.ReadInConfig()) + assert.Equal(t, "bar", v.GetString("foo")) + }) + } +} + +func TestConfigureViper_UnknownExtensionFallsBackToTOML(t *testing.T) { + path := filepath.Join(t.TempDir(), "gordon.conf") + require.NoError(t, os.WriteFile(path, []byte("foo = \"bar\"\n"), 0o600)) + + v := viper.New() + ConfigureViper(v, path) + require.NoError(t, v.ReadInConfig()) + assert.Equal(t, "bar", v.GetString("foo")) +} diff --git a/internal/app/pidfile.go b/internal/app/pidfile.go new file mode 100644 index 000000000..ea797bc60 --- /dev/null +++ b/internal/app/pidfile.go @@ -0,0 +1,181 @@ +package app + +import ( + "errors" + "fmt" + "os" + "path/filepath" + "strconv" + "syscall" + + "github.com/bnema/zerowrap" +) + +// SendReloadSignal sends SIGUSR1 to the running Gordon process. +func SendReloadSignal() error { + process, _, err := findRunningProcess() + if err != nil { + return err + } + + if err := process.Signal(syscall.SIGUSR1); err != nil { + return fmt.Errorf("failed to send reload signal: %w", err) + } + + return nil +} + +// createPidFile creates a PID file for the Gordon process. +// SECURITY: Prefers secure locations (XDG_RUNTIME_DIR, ~/.gordon/run) over /tmp +// to prevent symlink attacks and unauthorized access. +func createPidFile(log zerowrap.Logger) string { + pid := os.Getpid() + + // SECURITY: Prioritize secure locations over /tmp + var locations []string + + // Try secure runtime directory first + if runtimeDir, err := getSecureRuntimeDir(); err == nil { + locations = append(locations, filepath.Join(runtimeDir, "gordon.pid")) + } + + // Fall back to home directory + if homeDir, err := os.UserHomeDir(); err == nil { + locations = append(locations, filepath.Join(homeDir, ".gordon", "gordon.pid")) + } + + // Last resort: /tmp (least secure due to world-writable) + locations = append(locations, filepath.Join(os.TempDir(), "gordon.pid")) + + for _, location := range locations { + // Ensure parent directory exists with secure permissions + if err := os.MkdirAll(filepath.Dir(location), 0700); err != nil { + continue + } + if err := os.WriteFile(location, fmt.Appendf(nil, "%d", pid), 0600); err == nil { + log.Debug().Str("pid_file", location).Int("pid", pid).Msg("created PID file") + return location + } + } + + log.Warn().Int("pid", pid).Msg("failed to create PID file in any location") + return "" +} + +// removePidFile removes the PID file. +func removePidFile(pidFile string, log zerowrap.Logger) { + if err := os.Remove(pidFile); err != nil { + log.Warn().Err(err).Str("pid_file", pidFile).Msg("failed to remove PID file") + } else { + log.Debug().Str("pid_file", pidFile).Msg("removed PID file") + } +} + +func pidFileLocations() []string { + var locations []string + seen := make(map[string]struct{}) + add := func(path string) { + if path == "" { + return + } + if _, ok := seen[path]; ok { + return + } + seen[path] = struct{}{} + locations = append(locations, path) + } + + // Check secure runtime directory first + if runtimeDir, err := getSecureRuntimeDir(); err == nil { + add(filepath.Join(runtimeDir, "gordon.pid")) + } + + // Also check canonical /run/user/ runtime path. This handles cases where + // Gordon started under systemd user services with runtime dir available, but + // CLI invocations (e.g. non-interactive SSH) don't have XDG_RUNTIME_DIR set. + runtimeByUID := filepath.Join("/run/user", strconv.Itoa(os.Getuid()), "gordon", "gordon.pid") + add(runtimeByUID) + + // Check explicit XDG_RUNTIME_DIR if present in this process env. + if runtimeDir := os.Getenv("XDG_RUNTIME_DIR"); runtimeDir != "" { + add(filepath.Join(runtimeDir, "gordon.pid")) + } + + // Check home directory + if homeDir, err := os.UserHomeDir(); err == nil { + add(filepath.Join(homeDir, ".gordon", "gordon.pid")) + // Legacy location for backward compatibility + add(filepath.Join(homeDir, ".gordon.pid")) + } + + // Legacy /tmp locations for backward compatibility + add(filepath.Join(os.TempDir(), "gordon.pid")) + add("/tmp/gordon.pid") + + return locations +} + +// findRunningPidFile returns the first PID file whose PID belongs to a live process. +// Stale/invalid PID files are ignored and removed when possible. +func findRunningPidFile() (string, int, error) { + return findRunningPidFileInLocations(pidFileLocations()) +} + +func findRunningPidFileInLocations(locations []string) (string, int, error) { + foundAny := false + + for _, location := range locations { + pidBytes, err := os.ReadFile(location) + if err != nil { + continue + } + + foundAny = true + + var pid int + if _, err := fmt.Sscanf(string(pidBytes), "%d", &pid); err != nil || pid <= 0 { + _ = os.Remove(location) + continue + } + + if isProcessAlive(pid) { + return location, pid, nil + } + + _ = os.Remove(location) + } + + if foundAny { + return "", 0, fmt.Errorf("found stale gordon PID file(s), is Gordon running?") + } + + return "", 0, fmt.Errorf("gordon PID file not found, is Gordon running?") +} + +func findRunningProcess() (*os.Process, int, error) { + _, pid, err := findRunningPidFile() + if err != nil { + return nil, 0, err + } + + process, err := os.FindProcess(pid) + if err != nil { + return nil, 0, fmt.Errorf("failed to find process: %w", err) + } + + return process, pid, nil +} + +func isProcessAlive(pid int) bool { + process, err := os.FindProcess(pid) + if err != nil { + return false + } + + err = process.Signal(syscall.Signal(0)) + if err == nil { + return true + } + + return errors.Is(err, syscall.EPERM) +} diff --git a/internal/app/pki_integration_test.go b/internal/app/pki_integration_test.go index 1fbaf83a9..06036fd92 100644 --- a/internal/app/pki_integration_test.go +++ b/internal/app/pki_integration_test.go @@ -9,24 +9,28 @@ import ( "time" pkiadapter "github.com/bnema/gordon/internal/adapters/out/pki" - "github.com/bnema/gordon/internal/boundaries/out/mocks" - "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/boundaries/out" pkiusecase "github.com/bnema/gordon/internal/usecase/pki" "github.com/bnema/zerowrap" "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/mock" "github.com/stretchr/testify/require" ) -func newRoutesMock(t *testing.T, domains ...string) *mocks.MockRouteChecker { - m := mocks.NewMockRouteChecker(t) - routes := make([]domain.Route, len(domains)) - for i, d := range domains { - routes[i] = domain.Route{Domain: d} +// stubAppRoutes is an ACTIVE-derived host source for PKI tests. +type stubAppRoutes struct { + hosts []out.AppHost +} + +func (s *stubAppRoutes) AppHosts() []out.AppHost { return s.hosts } + +func (s *stubAppRoutes) GetExternalRoutes() map[string]string { return nil } + +func newRoutesMock(_ *testing.T, domains ...string) *stubAppRoutes { + hosts := make([]out.AppHost, 0, len(domains)) + for _, d := range domains { + hosts = append(hosts, out.AppHost{Host: d}) } - m.EXPECT().GetRoutes(mock.Anything).Return(routes).Maybe() - m.EXPECT().GetExternalRoutes().Return(nil).Maybe() - return m + return &stubAppRoutes{hosts: hosts} } func TestTLSHandshake_OnDemandCert(t *testing.T) { diff --git a/internal/app/prune_scheduler_test.go b/internal/app/prune_scheduler_test.go new file mode 100644 index 000000000..d33e892a7 --- /dev/null +++ b/internal/app/prune_scheduler_test.go @@ -0,0 +1,50 @@ +package app + +import ( + "context" + "testing" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/usecase/images" +) + +// TestStartImagePruneSchedulerRegistersWhenAppsExist proves the +// scheduler is no longer disabled by app state: a configured schedule +// registers even though the server has apps, and the job runs the same +// use case as manual execution. +func TestStartImagePruneSchedulerRegistersWhenAppsExist(t *testing.T) { + cfg := Config{} + cfg.Images.Prune.Enabled = true + cfg.Images.Prune.Schedule = "daily" + cfg.Images.Prune.KeepLast = 3 + + svc := &services{imageSvc: images.NewService(nil, nil, nil, zerowrap.Default())} + + scheduler, err := startImagePruneScheduler(context.Background(), cfg, svc, zerowrap.Default(), nil) + require.NoError(t, err) + require.NotNil(t, scheduler, "a configured schedule must register regardless of app state") + assert.NotEmpty(t, scheduler.List()) +} + +func TestStartImagePruneSchedulerDisabledWhenNotConfigured(t *testing.T) { + svc := &services{imageSvc: images.NewService(nil, nil, nil, zerowrap.Default())} + + scheduler, err := startImagePruneScheduler(context.Background(), Config{}, svc, zerowrap.Default(), nil) + require.NoError(t, err) + assert.Nil(t, scheduler) +} + +func TestStartImagePruneSchedulerRejectsNegativeKeepLast(t *testing.T) { + cfg := Config{} + cfg.Images.Prune.Enabled = true + cfg.Images.Prune.Schedule = "daily" + cfg.Images.Prune.KeepLast = 3 + + svc := &services{imageSvc: images.NewService(nil, nil, nil, zerowrap.Default())} + scheduler, err := startImagePruneScheduler(context.Background(), cfg, svc, zerowrap.Default(), func() int { return -1 }) + require.Error(t, err) + assert.Nil(t, scheduler) +} diff --git a/internal/app/public_tls_wiring_test.go b/internal/app/public_tls_wiring_test.go index 238de7fc3..1bd5640c8 100644 --- a/internal/app/public_tls_wiring_test.go +++ b/internal/app/public_tls_wiring_test.go @@ -13,13 +13,13 @@ import ( "github.com/spf13/viper" "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/mock" "github.com/stretchr/testify/require" pkiadapter "github.com/bnema/gordon/internal/adapters/out/pki" inmocks "github.com/bnema/gordon/internal/boundaries/in/mocks" - outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/boundaries/out" "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/apptraffic" config "github.com/bnema/gordon/internal/usecase/config" pkiusecase "github.com/bnema/gordon/internal/usecase/pki" "github.com/bnema/gordon/internal/usecase/publictls" @@ -228,6 +228,25 @@ func TestCertificateSelector_StaticCertWins(t *testing.T) { "static cert should be returned, not public TLS") } +// stubAppRoutes is an ACTIVE-derived host source for TLS selector tests. +type stubAppRoutes struct { + hosts []out.AppHost +} + +func (s *stubAppRoutes) AppHosts() []out.AppHost { return s.hosts } + +func (s *stubAppRoutes) L4Entries() []apptraffic.RouteEntry { return nil } + +func (s *stubAppRoutes) GetExternalRoutes() map[string]string { return nil } + +func newRoutesMock(_ *testing.T, domains ...string) *stubAppRoutes { + hosts := make([]out.AppHost, 0, len(domains)) + for _, d := range domains { + hosts = append(hosts, out.AppHost{Host: d}) + } + return &stubAppRoutes{hosts: hosts} +} + func TestCertificateSelector_PublicTLSErrorFallsThroughWithoutLocalPKI(t *testing.T) { publicTLS := inmocks.NewMockPublicTLSService(t) publicTLS.EXPECT().GetCertificateForHost("acme-required.example.com").Return(nil, domain.ErrTLSRouteNotCovered) @@ -244,9 +263,7 @@ func TestCertificateSelector_PublicTLSErrorFallsThroughToLocalPKI(t *testing.T) publicTLS := inmocks.NewMockPublicTLSService(t) publicTLS.EXPECT().GetCertificateForHost("acme-required.example.com").Return(nil, domain.ErrTLSRouteNotCovered) - routes := outmocks.NewMockRouteChecker(t) - routes.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{{Domain: "acme-required.example.com"}}).Maybe() - routes.EXPECT().GetExternalRoutes().Return(map[string]string{}).Maybe() + routes := newRoutesMock(t, "acme-required.example.com") ca, err := pkiadapter.NewCA(t.TempDir(), zerowrap.Default()) require.NoError(t, err) pkiSvc := pkiusecase.NewService(context.Background(), ca, routes, nil, zerowrap.Default()) diff --git a/internal/app/reload.go b/internal/app/reload.go new file mode 100644 index 000000000..5ee0a62ed --- /dev/null +++ b/internal/app/reload.go @@ -0,0 +1,353 @@ +package app + +import ( + "context" + "crypto/tls" + "fmt" + "sync" + "time" + + "github.com/bnema/zerowrap" + "github.com/spf13/viper" + + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/container" + "github.com/bnema/gordon/internal/usecase/proxy" +) + +type configWatcher interface { + Watch(ctx context.Context, onChange func()) error +} + +// publicTLSReconciler is the interface for reconciling public TLS certificates. +type publicTLSReconciler interface { + Reconcile(context.Context) error +} + +type configReloader interface { + Reload(ctx context.Context) error +} + +type proxyConfigUpdater interface { + UpdateConfig(config proxy.Config) +} + +type reloadTrigger interface { + Trigger(ctx context.Context) error +} + +type loadedConfigApplier interface { + ApplyLoadedConfig(ctx context.Context) error +} + +type reloadCoordinator struct { + mu sync.Mutex + lastRun time.Time + debounce time.Duration + trailingTimer *time.Timer + trailingGeneration uint64 + stopped bool + + // lifecycleCtx owns the context used by debounced trailing reloads. Its + // base is detached from any caller's cancellation so a short-lived request + // context cannot abort a coalesced apply, while lifecycleCancel lets Stop + // tear the coordinator's own lifecycle down explicitly. + lifecycleCtx context.Context + lifecycleCancel context.CancelFunc + + configSvc configReloader + v *viper.Viper + proxySvc proxyConfigUpdater + // applyRuntime applies one validated config to the running daemon + // (policies, management hosts, traffic, entrypoints, container + // config). Nil when no runtime is wired (tests). + applyRuntime func(context.Context, Config) error + registryLimits interface { + UpdateBlobLimits(maxBlobChunkSize, maxBlobSize int64) + } + eventBus out.EventPublisher + publicTLS publicTLSReconciler + log zerowrap.Logger +} + +func newReloadCoordinator(v *viper.Viper, configSvc configReloader, proxySvc proxyConfigUpdater, registryLimits interface { + UpdateBlobLimits(maxBlobChunkSize, maxBlobSize int64) +}, eventBus out.EventPublisher, publicTLS publicTLSReconciler, applyRuntime func(context.Context, Config) error, log zerowrap.Logger) *reloadCoordinator { + return &reloadCoordinator{ + debounce: 500 * time.Millisecond, + configSvc: configSvc, + v: v, + proxySvc: proxySvc, + applyRuntime: applyRuntime, + registryLimits: registryLimits, + eventBus: eventBus, + publicTLS: publicTLS, + log: log, + } +} + +func (c *reloadCoordinator) SetRegistryLimits(limits interface { + UpdateBlobLimits(maxBlobChunkSize, maxBlobSize int64) +}) { + c.mu.Lock() + defer c.mu.Unlock() + c.registryLimits = limits +} + +// Trigger requests a config reload that first re-reads config from disk. +func (c *reloadCoordinator) Trigger(ctx context.Context) error { + c.mu.Lock() + defer c.mu.Unlock() + + return c.reloadDebouncedLocked(ctx, true) +} + +// ApplyLoadedConfig requests a reload of the config the watcher already +// loaded from disk. It shares the debounce/coalescing policy with Trigger so +// a burst of fsnotify callbacks applies the final state exactly once. +func (c *reloadCoordinator) ApplyLoadedConfig(ctx context.Context) error { + c.mu.Lock() + defer c.mu.Unlock() + + return c.reloadDebouncedLocked(ctx, false) +} + +// reloadDebouncedLocked is the single owner of the debounce/coalescing policy +// shared by every reload entrypoint. Requests inside the debounce window are +// merged into one trailing reload carrying the latest loadConfig intent. +func (c *reloadCoordinator) reloadDebouncedLocked(ctx context.Context, loadConfig bool) error { + now := time.Now() + if !c.lastRun.IsZero() && now.Sub(c.lastRun) < c.debounce { + c.scheduleTrailingReloadLocked(ctx, loadConfig) + return nil + } + + return c.reloadLocked(ctx, loadConfig) +} + +func (c *reloadCoordinator) scheduleTrailingReloadLocked(ctx context.Context, loadConfig bool) { + if c.stopped { + return + } + if c.trailingTimer != nil { + c.trailingTimer.Stop() + } + if c.lifecycleCancel == nil { + c.lifecycleCtx, c.lifecycleCancel = context.WithCancel(context.WithoutCancel(ctx)) + } + c.trailingGeneration++ + generation := c.trailingGeneration + trailingCtx := c.lifecycleCtx + c.trailingTimer = time.AfterFunc(c.debounce, func() { + c.runTrailingReload(trailingCtx, generation, loadConfig) + }) + c.log.Debug().Msg("coalescing config reload trigger") +} + +func (c *reloadCoordinator) runTrailingReload(ctx context.Context, generation uint64, loadConfig bool) { + c.mu.Lock() + defer c.mu.Unlock() + + if c.stopped || generation != c.trailingGeneration { + return + } + c.trailingTimer = nil + if err := c.reloadLocked(ctx, loadConfig); err != nil { + c.log.Error().Err(err).Msg("failed trailing config reload") + } +} + +func (c *reloadCoordinator) Stop() { + c.mu.Lock() + defer c.mu.Unlock() + c.stopped = true + c.trailingGeneration++ + if c.trailingTimer != nil { + c.trailingTimer.Stop() + c.trailingTimer = nil + } + if c.lifecycleCancel != nil { + c.lifecycleCancel() + } +} + +func (c *reloadCoordinator) reloadLocked(ctx context.Context, loadConfig bool) error { + now := time.Now() + if loadConfig { + if err := c.configSvc.Reload(ctx); err != nil { + c.log.Error().Err(err).Msg("failed to reload config") + return fmt.Errorf("failed to reload config: %w", err) + } + } + + if err := c.applyLoadedConfig(ctx, now); err != nil { + return err + } + + return nil +} + +func (c *reloadCoordinator) applyLoadedConfig(ctx context.Context, now time.Time) error { + var reloadCfg Config + if err := c.v.Unmarshal(&reloadCfg); err != nil { + c.log.Error().Err(err).Msg("failed to unmarshal config on reload") + return fmt.Errorf("failed to unmarshal config on reload: %w", err) + } + if err := validateEntrypointMigration(c.v, reloadCfg); err != nil { + return err + } + if err := validateRetiredAppConfig(c.v); err != nil { + return err + } + if err := validateAppPolicies(c.v, reloadCfg); err != nil { + return err + } + + reloadedProxy, err := buildProxyConfig(reloadCfg, c.log) + if err != nil { + c.log.Error().Err(err).Msg("failed to parse proxy config on reload") + return fmt.Errorf("failed to parse proxy config on reload: %w", err) + } + + if c.applyRuntime != nil { + if err := c.applyRuntime(ctx, reloadCfg); err != nil { + c.log.Error().Err(err).Msg("failed to apply runtime config on reload") + return fmt.Errorf("failed to apply runtime config on reload: %w", err) + } + } + c.proxySvc.UpdateConfig(reloadedProxy.proxyConfig) + if c.registryLimits != nil { + c.registryLimits.UpdateBlobLimits(reloadedProxy.maxBlobChunkSize, reloadedProxy.maxBlobSize) + } + + // Reconcile public TLS before publishing reload events so certificate + // authorization reflects the loaded config even if event delivery fails. + // A transient ACME issue must not abort the rest of the reload. + if c.publicTLS != nil { + if err := c.publicTLS.Reconcile(ctx); err != nil { + c.log.Warn().Err(err).Msg("failed to reconcile public TLS certificates after reload, continuing") + } + } + + if c.eventBus != nil { + if err := c.eventBus.Publish(domain.EventConfigReload, nil); err != nil { + c.log.Error().Err(err).Msg("failed to publish config reload event") + return fmt.Errorf("failed to publish config reload event: %w", err) + } + } + + c.lastRun = now + + c.log.Debug().Msg("config hot reload complete") + return nil +} + +// reloadRuntime applies a validated config to the running daemon. It +// prepares every derived value before the first side effect, so a config +// that cannot be converted leaves the whole runtime at the previous +// version. Side effects then run in one fixed order: traffic graph (the +// only fallible step), management hosts, HTTP entrypoints, container +// config, app policies. +type reloadRuntime struct { + v *viper.Viper + svc *services + log zerowrap.Logger +} + +// reloadPlan holds the values derived from one config before any apply. +type reloadPlan struct { + containerCfg container.Config + tlsConfig *tls.Config + bindPolicies map[string]domain.AppBindPolicy + devicePolicies map[string]domain.AppDevicePolicy +} + +// Apply is the reloadCoordinator's runtime step. +func (r *reloadRuntime) Apply(ctx context.Context, cfg Config) error { + plan, err := r.prepare(ctx, cfg) + if err != nil { + return fmt.Errorf("prepare reload: %w", err) + } + if r.svc.containerSvc != nil { + if err := r.applyServing(ctx, cfg, plan); err != nil { + return fmt.Errorf("apply serving config: %w", err) + } + } + r.applyAppPolicies(plan) + return nil +} + +func (r *reloadRuntime) prepare(ctx context.Context, cfg Config) (reloadPlan, error) { + var plan reloadPlan + var err error + if r.svc.containerSvc != nil { + if plan.containerCfg, err = buildContainerServiceConfig(ctx, r.v, cfg, r.svc, r.log); err != nil { + return plan, err + } + // The TLS config selects certificates at handshake time, so it is + // safe to build before management hosts change; building it here + // surfaces keypair load errors before any side effect. + if hasTLSCapableEntrypoint(cfg) && r.svc.httpsProxyHandler != nil { + if plan.tlsConfig, err = proxyTLSConfig(cfg, r.svc.pkiSvc, r.svc.publicTLSSvc, r.log); err != nil { + return plan, err + } + } + } + if r.svc.appSvcImpl != nil { + if plan.bindPolicies, err = buildAppMountPolicies(cfg); err != nil { + return plan, err + } + if plan.devicePolicies, err = buildAppDevicePolicies(cfg); err != nil { + return plan, err + } + } + return plan, nil +} + +// applyServing updates everything that serves traffic. The serialized +// traffic graph rebuild is the only fallible step and runs first, so a +// rejected graph leaves management hosts, entrypoints, and container +// config at the previous version. +func (r *reloadRuntime) applyServing(ctx context.Context, cfg Config, plan reloadPlan) error { + svc := r.svc + if err := svc.appTrafficPublisher.RebuildWithConfig(ctx, cfg); err != nil { + return err + } + managementHosts := []string{cfg.Server.GordonDomain} + if svc.pkiSvc != nil { + svc.pkiSvc.SetAdditionalDomains(managementHosts) + } + if svc.publicTLSSvc != nil { + svc.publicTLSSvc.SetAdditionalHosts(ctx, managementHosts) + } + svc.tlsHTTPEntryPoints = registerTLSMuxHTTPServers(svc.trafficManager, cfg, svc.httpsProxyHandler, plan.tlsConfig, svc.tlsHTTPEntryPoints) + svc.smartHTTPEntryPoints = registerSmartTCPHTTPServers(svc.trafficManager, cfg, svc.httpProxyHandler, svc.httpsProxyHandler, plan.tlsConfig, svc.smartHTTPEntryPoints) + svc.containerSvc.UpdateConfig(plan.containerCfg) + return nil +} + +// applyAppPolicies publishes the reloaded bind and device policies to both +// app services that enforce them. +func (r *reloadRuntime) applyAppPolicies(plan reloadPlan) { + if r.svc.appSvcImpl == nil { + return + } + r.svc.appSvcImpl.SetBindPolicies(plan.bindPolicies) + r.svc.appSvcImpl.SetDevicePolicies(plan.devicePolicies) + if r.svc.appDeploySvc != nil { + r.svc.appDeploySvc.SetBindPolicies(plan.bindPolicies) + r.svc.appDeploySvc.SetDevicePolicies(plan.devicePolicies) + } +} + +// setupConfigHotReload sets up config hot reload. +func setupConfigHotReload(ctx context.Context, configSvc configWatcher, coordinator loadedConfigApplier) error { + if err := configSvc.Watch(ctx, func() { + _ = coordinator.ApplyLoadedConfig(ctx) + }); err != nil { + return fmt.Errorf("failed to watch config: %w", err) + } + + return nil +} diff --git a/internal/app/run.go b/internal/app/run.go index cec1c2c35..387dcf527 100644 --- a/internal/app/run.go +++ b/internal/app/run.go @@ -1,317 +1,35 @@ // Package app provides the application initialization and wiring. +// +// run.go owns the daemon lifecycle (Run, server start, readiness). Each +// concern it orders lives in its own file: config.go (load/validate), +// services.go (service construction), app_wiring.go (apps engine), +// auth_secrets.go and credentials.go (auth and internal registry +// credentials), http_edge.go (handlers and middleware), listeners.go +// (proxy/TLS/registry listeners), reload.go (hot reload), schedulers.go, +// shutdown.go, and pidfile.go. package app import ( "context" - "crypto/rand" - "crypto/tls" - "crypto/x509" - "encoding/hex" - "encoding/json" - "errors" "fmt" - "net" "net/http" "os" "os/signal" "path/filepath" - "strconv" - "strings" - "sync" "syscall" "time" "github.com/bnema/zerowrap" zerowrapotel "github.com/bnema/zerowrap/otel" "github.com/spf13/viper" - "golang.org/x/sys/unix" - // Adapters - Output "github.com/bnema/gordon/internal/adapters/out/accesslog" - acmelego "github.com/bnema/gordon/internal/adapters/out/acmelego" - acmestore "github.com/bnema/gordon/internal/adapters/out/acmestore" - "github.com/bnema/gordon/internal/adapters/out/docker" - "github.com/bnema/gordon/internal/adapters/out/domainsecrets" - "github.com/bnema/gordon/internal/adapters/out/envloader" - "github.com/bnema/gordon/internal/adapters/out/eventbus" - "github.com/bnema/gordon/internal/adapters/out/filesystem" - "github.com/bnema/gordon/internal/adapters/out/httpprober" - "github.com/bnema/gordon/internal/adapters/out/logwriter" - pkiadapter "github.com/bnema/gordon/internal/adapters/out/pki" - "github.com/bnema/gordon/internal/adapters/out/ratelimit" - s3storage "github.com/bnema/gordon/internal/adapters/out/s3" - "github.com/bnema/gordon/internal/adapters/out/secrets" "github.com/bnema/gordon/internal/adapters/out/telemetry" - "github.com/bnema/gordon/internal/adapters/out/tokenstore" - - // OTel - "go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp" - - // Adapters - Input - "github.com/bnema/gordon/internal/adapters/dto" - acmehttp "github.com/bnema/gordon/internal/adapters/in/http/acme" - "github.com/bnema/gordon/internal/adapters/in/http/admin" - authhandler "github.com/bnema/gordon/internal/adapters/in/http/auth" - "github.com/bnema/gordon/internal/adapters/in/http/httphelper" - "github.com/bnema/gordon/internal/adapters/in/http/middleware" - "github.com/bnema/gordon/internal/adapters/in/http/onboarding" - proxyadapter "github.com/bnema/gordon/internal/adapters/in/http/proxy" - "github.com/bnema/gordon/internal/adapters/in/http/registry" - trafficadapter "github.com/bnema/gordon/internal/adapters/in/traffic" - - // Boundaries - "github.com/bnema/gordon/internal/boundaries/in" "github.com/bnema/gordon/internal/boundaries/out" - - // Domain - "github.com/bnema/gordon/internal/domain" - - // Packages + "github.com/bnema/gordon/internal/usecase/logexport" "github.com/bnema/gordon/pkg/version" - - // Use cases - "github.com/bnema/gordon/internal/usecase/auth" - "github.com/bnema/gordon/internal/usecase/auto" - "github.com/bnema/gordon/internal/usecase/auto/preview" - "github.com/bnema/gordon/internal/usecase/backup" - "github.com/bnema/gordon/internal/usecase/config" - "github.com/bnema/gordon/internal/usecase/container" - cronSvc "github.com/bnema/gordon/internal/usecase/cron" - "github.com/bnema/gordon/internal/usecase/health" - "github.com/bnema/gordon/internal/usecase/images" - "github.com/bnema/gordon/internal/usecase/logs" - pkiusecase "github.com/bnema/gordon/internal/usecase/pki" - "github.com/bnema/gordon/internal/usecase/proxy" - "github.com/bnema/gordon/internal/usecase/publictls" - registrySvc "github.com/bnema/gordon/internal/usecase/registry" - "github.com/bnema/gordon/internal/usecase/registrystate" - secretsSvc "github.com/bnema/gordon/internal/usecase/secrets" - servicecfg "github.com/bnema/gordon/internal/usecase/services" - "github.com/bnema/gordon/internal/usecase/traffic" - volumesSvc "github.com/bnema/gordon/internal/usecase/volumes" - - // Pkg - "github.com/bnema/gordon/pkg/bytesize" - "github.com/bnema/gordon/pkg/duration" ) -// Config holds the application configuration. -type Config struct { - Server struct { - Port int `mapstructure:"port"` - RegistryPort int `mapstructure:"registry_port"` - GordonDomain string `mapstructure:"gordon_domain"` - RegistryDomain string `mapstructure:"registry_domain"` - LegacyRegistryDomains []string `mapstructure:"legacy_registry_domains"` - TLSPort int `mapstructure:"tls_port"` - TLSCertFile string `mapstructure:"tls_cert_file"` - TLSKeyFile string `mapstructure:"tls_key_file"` - ForceHTTPSRedirect bool `mapstructure:"force_https_redirect"` - DataDir string `mapstructure:"data_dir"` - MaxProxyBodySize string `mapstructure:"max_proxy_body_size"` // e.g., "512MB", "1GB" - MaxBlobChunkSize string `mapstructure:"max_blob_chunk_size"` // e.g., "512MB", "1GB" - MaxBlobSize string `mapstructure:"max_blob_size"` // e.g., "1GB", "2GB" - MaxProxyResponseSize string `mapstructure:"max_proxy_response_size"` // e.g., "1GB", "0" for no limit - MaxConcurrentConns int `mapstructure:"max_concurrent_connections"` - RegistryAllowedIPs []string `mapstructure:"registry_allowed_ips"` - ProxyAllowedIPs []string `mapstructure:"proxy_allowed_ips"` - RegistryListenAddr string `mapstructure:"registry_listen_address"` - } `mapstructure:"server"` - - Logging struct { - Level string `mapstructure:"level"` - Format string `mapstructure:"format"` - File struct { - Enabled bool `mapstructure:"enabled"` - Path string `mapstructure:"path"` - MaxSize int `mapstructure:"max_size"` - MaxBackups int `mapstructure:"max_backups"` - MaxAge int `mapstructure:"max_age"` - } `mapstructure:"file"` - ContainerLogs struct { - Enabled bool `mapstructure:"enabled"` - Dir string `mapstructure:"dir"` - MaxSize int `mapstructure:"max_size"` - MaxBackups int `mapstructure:"max_backups"` - MaxAge int `mapstructure:"max_age"` - } `mapstructure:"container_logs"` - AccessLog struct { - Enabled bool `mapstructure:"enabled"` - Format string `mapstructure:"format"` - Output string `mapstructure:"output"` - FilePath string `mapstructure:"file_path"` - MaxSize int `mapstructure:"max_size"` - MaxBackups int `mapstructure:"max_backups"` - MaxAge int `mapstructure:"max_age"` - ExcludeHealthChecks bool `mapstructure:"exclude_health_checks"` - SyslogIdentifier string `mapstructure:"syslog_identifier"` - } `mapstructure:"access_log"` - } `mapstructure:"logging"` - - Env struct { - Dir string `mapstructure:"dir"` - } `mapstructure:"env"` - - Volumes struct { - AutoCreate bool `mapstructure:"auto_create"` - Prefix string `mapstructure:"prefix"` - Preserve bool `mapstructure:"preserve"` - } `mapstructure:"volumes"` - - Auth struct { - Enabled bool `mapstructure:"enabled"` - Type string `mapstructure:"type"` // only "token" is supported - SecretsBackend string `mapstructure:"secrets_backend"` // "pass", "sops", or "unsafe" - Username string `mapstructure:"username"` - TokenSecret string `mapstructure:"token_secret"` // path in secrets backend - TokenExpiry string `mapstructure:"token_expiry"` // e.g., "720h", "30d" - AccessTokenTTL string `mapstructure:"access_token_ttl"` // e.g., "15m", "30m" (default: 15m) - } `mapstructure:"auth"` - - API struct { - RateLimit struct { - Enabled bool `mapstructure:"enabled"` - GlobalRPS float64 `mapstructure:"global_rps"` - PerIPRPS float64 `mapstructure:"per_ip_rps"` - Burst int `mapstructure:"burst"` - TrustedProxies []string `mapstructure:"trusted_proxies"` - } `mapstructure:"rate_limit"` - } `mapstructure:"api"` - - EntryPoints map[string]traffic.EntryPointConfig `mapstructure:"entrypoints"` - Traffic traffic.Config `mapstructure:"traffic"` - NetworkServices []traffic.NetworkServiceConfig `mapstructure:"network_services"` - Services []servicecfg.Config `mapstructure:"services"` - - Backups struct { - // Legacy database backup keys. Prefer backups.databases.* for new configs. - Enabled bool `mapstructure:"enabled"` - Schedule string `mapstructure:"schedule"` - StorageDir string `mapstructure:"storage_dir"` - Retention struct { - Hourly int `mapstructure:"hourly"` - Daily int `mapstructure:"daily"` - Weekly int `mapstructure:"weekly"` - Monthly int `mapstructure:"monthly"` - } `mapstructure:"retention"` - Databases struct { - Enabled bool `mapstructure:"enabled"` - Schedule string `mapstructure:"schedule"` - StorageDir string `mapstructure:"storage_dir"` - Retention struct { - Hourly int `mapstructure:"hourly"` - Daily int `mapstructure:"daily"` - Weekly int `mapstructure:"weekly"` - Monthly int `mapstructure:"monthly"` - } `mapstructure:"retention"` - } `mapstructure:"databases"` - Volumes struct { - Enabled bool `mapstructure:"enabled"` - Interval string `mapstructure:"interval"` - Compression string `mapstructure:"compression"` - Timeout string `mapstructure:"timeout"` - MaxConcurrency int `mapstructure:"max_concurrency"` - HelperImage string `mapstructure:"helper_image"` - S3 struct { - Bucket string `mapstructure:"bucket"` - Region string `mapstructure:"region"` - Prefix string `mapstructure:"prefix"` - Endpoint string `mapstructure:"endpoint"` - PathStyle bool `mapstructure:"path_style"` - SSEAlgorithm string `mapstructure:"sse_algorithm"` - SSEKMSKeyID string `mapstructure:"sse_kms_key_id"` - } `mapstructure:"s3"` - Retention struct { - Keep int `mapstructure:"keep"` - } `mapstructure:"retention"` - } `mapstructure:"volumes"` - } `mapstructure:"backups"` - - Images struct { - AllowedRegistries []string `mapstructure:"allowed_registries"` - RequireDigest bool `mapstructure:"require_digest"` - Prune struct { - Enabled bool `mapstructure:"enabled"` - Schedule string `mapstructure:"schedule"` - KeepLast int `mapstructure:"keep_last"` - } `mapstructure:"prune"` - } `mapstructure:"images"` - - Containers struct { - MemoryLimit string `mapstructure:"memory_limit"` // e.g., "512MB", "1GB" - CPULimit float64 `mapstructure:"cpu_limit"` // CPU cores, e.g., 1.0 = 1 core - PidsLimit int64 `mapstructure:"pids_limit"` // e.g., 512 - SecurityProfile string `mapstructure:"security_profile"` // compat or strict - } `mapstructure:"containers"` - - Telemetry telemetry.Config `mapstructure:"telemetry"` - - TLS struct { - ACME struct { - Enabled bool `mapstructure:"enabled"` - Email string `mapstructure:"email"` - Challenge string `mapstructure:"challenge"` - ObtainBatchSize int `mapstructure:"obtain_batch_size"` - } `mapstructure:"acme"` - } `mapstructure:"tls"` - - DNS struct { - Resolvers []string `mapstructure:"resolvers"` - PropagationTimeout string `mapstructure:"propagation_timeout"` - PollingInterval string `mapstructure:"polling_interval"` - } `mapstructure:"dns"` -} - -// services holds all the services used by the application. -type services struct { - runtime *docker.Runtime - eventBus *eventbus.InMemory - blobStorage *filesystem.BlobStorage - manifestStorage *filesystem.ManifestStorage - backupStorage *filesystem.BackupStorage - volumeBackupStore out.VolumeBackupStorage - volumeBackupCfg domain.VolumeBackupConfig - envLoader out.EnvLoader - logWriter *logwriter.LogWriter - tokenStore out.TokenStore - configSvc *config.Service - secretSvc *secretsSvc.Service - containerSvc *container.Service - backupSvc *backup.Service - volumeBackupSvc *backup.VolumeService - registrySvc *registrySvc.Service - healthSvc *health.Service - logSvc *logs.Service - imageSvc *images.Service - volumeSvc *volumesSvc.Service - proxySvc *proxy.Service - standaloneServiceSvc in.StandaloneServiceService - serviceSecretProvider out.SecretProvider - authSvc *auth.Service - authHandler *authhandler.Handler - adminHandler *admin.Handler - httpProxyHandler http.Handler - httpsProxyHandler http.Handler - internalRegUser string - internalRegPass string - previewStore *filesystem.PreviewStore - previewService *preview.Service - envDir string - maxBlobChunkSize int64 - maxBlobSize int64 - caAdapter *pkiadapter.CA - pkiSvc *pkiusecase.Service - reloadCoordinator *reloadCoordinator - publicTLSSvc in.PublicTLSService - publicTLSRuntime publicTLSRuntime - trafficManager *trafficadapter.Manager - tlsHTTPEntryPoints map[string]struct{} - smartHTTPEntryPoints map[string]struct{} - registryHandler interface { - UpdateBlobLimits(maxBlobChunkSize, maxBlobSize int64) - } -} - // Run initializes and starts the Gordon application. func Run(ctx context.Context, configPath string) error { // Load configuration @@ -368,9 +86,10 @@ func Run(ctx context.Context, configPath string) error { if err != nil { return err } + svc.workloadLogs = workloadLogExporter(telProvider) // Register event handlers - cleanupHandlers, err := registerEventHandlers(ctx, svc, cfg) + cleanupHandlers, err := registerEventHandlers(ctx, svc) if err != nil { return err } @@ -392,50 +111,6 @@ func Run(ctx context.Context, configPath string) error { return runServers(ctx, v, cfg, svc, svc.reloadCoordinator, cleanupHandlers, log) } -func warnDeprecatedConfigKeys(v *viper.Viper, log zerowrap.Logger) { - for _, key := range []string{"server.tls_enabled", "server.force_hsts"} { - if v.IsSet(key) { - log.Warn().Str("key", key).Msg("deprecated config key — Gordon now uses an internal CA with automatic TLS; remove this from your config") - } - } -} - -// initConfig loads configuration from file. -func initConfig(configPath string) (*viper.Viper, Config, error) { - v := viper.New() - if err := loadConfig(v, configPath); err != nil { - return nil, Config{}, fmt.Errorf("failed to load config: %w", err) - } - - var cfg Config - if err := v.Unmarshal(&cfg); err != nil { - return nil, Config{}, fmt.Errorf("failed to unmarshal config: %w", err) - } - if err := validateEntrypointMigration(v, cfg); err != nil { - return nil, Config{}, err - } - - return v, cfg, nil -} - -func validateEntrypointMigration(v *viper.Viper, cfg Config) error { - if len(cfg.EntryPoints) > 0 { - return nil - } - - legacyKeys := make([]string, 0, 2) - for _, key := range []string{"server.port", "server.tls_port"} { - if v.IsSet(key) { - legacyKeys = append(legacyKeys, key) - } - } - if len(legacyKeys) == 0 { - return nil - } - - return fmt.Errorf("legacy %s configuration requires at least one [entrypoints] entry; see docs/upgrading.md", strings.Join(legacyKeys, " or ")) -} - // initLogger initializes the zerowrap logger. func initLogger(cfg Config) (zerowrap.Logger, func(), error) { logConfig := zerowrap.Config{ @@ -505,3534 +180,200 @@ func initAccessLog(cfg Config, log zerowrap.Logger) (*accesslog.Writer, error) { return writer, nil } -// serviceInit holds the shared context for service initialization helpers. -type serviceInit struct { - ctx context.Context - v *viper.Viper - cfg Config - log zerowrap.Logger - svc *services -} - -// createServices creates all the application services for server runtime. -// Public ACME reconciliation is started later, after the HTTP listener is bound. -func createServices(ctx context.Context, v *viper.Viper, cfg Config, log zerowrap.Logger) (_ *services, retErr error) { - return createServicesWithOptions(ctx, v, cfg, log) -} - -// createServicesWithOptions creates all the application services. -// ACME Reconcile and renewal loop are started later from runServers, -// after HTTP listeners are bound. -func createServicesWithOptions(ctx context.Context, v *viper.Viper, cfg Config, log zerowrap.Logger) (_ *services, retErr error) { - si := &serviceInit{ - ctx: ctx, - v: v, - cfg: cfg, - log: log, - svc: &services{}, - } - defer func() { - if retErr != nil { - if si.svc.pkiSvc != nil { - si.svc.pkiSvc.Stop() - } - if si.svc.publicTLSSvc != nil { - ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) - defer cancel() - if err := si.svc.publicTLSSvc.Stop(ctx); err != nil { - si.log.Warn().Err(err).Msg("failed to stop public TLS service during createServices cleanup") - } - } - } - }() - var err error - - // Create output adapters - runtimeSocket := resolveRuntimeConfig(v.GetString("server.runtime")) - if si.svc.runtime, si.svc.eventBus, err = createOutputAdapters(ctx, log, runtimeSocket); err != nil { - return nil, err - } - - // Create storage - if si.svc.blobStorage, si.svc.manifestStorage, err = createStorage(cfg, log); err != nil { - return nil, err - } - - // Create log writer - if si.svc.logWriter, err = createLogWriter(cfg, log); err != nil { - return nil, err - } - - // Create auth service (if enabled) - if si.svc.tokenStore, si.svc.authSvc, err = createAuthService(ctx, cfg, log); err != nil { - return nil, err - } - - if err := setupInternalRegistryAuth(si.svc, log); err != nil { - return nil, err - } - - // Create config service - si.svc.configSvc = config.NewService(v, si.svc.eventBus) - if err := si.svc.configSvc.Load(ctx); err != nil { - return nil, log.WrapErr(err, "failed to load configuration") - } - - if err := si.initPKI(); err != nil { - return nil, err - } - - if err := si.initSecrets(); err != nil { - return nil, err +// runServers starts the HTTP servers and waits for shutdown. +// Signal handling notes: +// - SIGINT/SIGTERM: Triggers graceful shutdown via signal.NotifyContext +// - SIGUSR1: Triggers config reload without restart +// (SIGUSR2 manual route deploy was removed with the declarative-apps +// cutover; app deploys are explicit via `gordon apps deploy`.) +// The deferred signal.Stop calls ensure signal handlers are properly +// cleaned up before program exit, preventing signal handler leaks. +func runServers(ctx context.Context, v *viper.Viper, cfg Config, svc *services, reload reloadTrigger, cleanupHandlers func(), log zerowrap.Logger) error { + // Initialize access log writer. Kept here (not in Run) to keep Run's cyclomatic + // complexity within the project limit of 15. + accessWriterConcrete, err := initAccessLog(cfg, log) + if err != nil { + return err } - - if err := si.initPublicTLS(); err != nil { - return nil, err + if accessWriterConcrete != nil { + defer accessWriterConcrete.Close() } - - if err := si.initRuntimeProxyAndTraffic(); err != nil { - return nil, err + // Convert to interface only when non-nil to avoid the Go nil-interface pitfall + // where a typed nil pointer becomes a non-nil interface value. + var accessWriter out.AccessLogWriter + if accessWriterConcrete != nil { + accessWriter = accessWriterConcrete } + accessWriter = withAccessLogExport(svc, accessWriter) + ctx, cancel := signal.NotifyContext(ctx, os.Interrupt, syscall.SIGTERM) + defer cancel() - si.svc.reloadCoordinator = newReloadCoordinator(v, si.svc.configSvc, si.svc.proxySvc, nil, si.svc.eventBus, si.svc.publicTLSSvc, log) - si.registerReloadCoordinatorHooks() + // Set up SIGUSR1 for reload. + // Note: signal.Stop must be called (via defer) to release the channel + // and prevent signal handler leaks when the function returns. + reloadChan := make(chan os.Signal, 1) + signal.Notify(reloadChan, syscall.SIGUSR1) + defer signal.Stop(reloadChan) - si.initHandlers() + errChan := make(chan error, 4) - return si.svc, nil -} + registryHandler, httpProxyHandler, httpsProxyHandler := createHTTPHandlers(svc, cfg, log, accessWriter) + svc.httpProxyHandler = httpProxyHandler + svc.httpsProxyHandler = httpsProxyHandler -// initPKI initialises the internal CA and PKI service when TLS is enabled. -func (si *serviceInit) initPKI() error { - if !hasTLSCapableEntrypoint(si.cfg) { - si.log.Info().Msg("internal CA disabled (no TLS-capable entrypoint configured)") - return nil - } + registrySrv, registryReady, internalRegistrySrv, internalRegistryReady := startRegistryServers(cfg, registryHandler, errChan, log) - if (si.cfg.Server.TLSCertFile == "") != (si.cfg.Server.TLSKeyFile == "") { - return fmt.Errorf("both tls_cert_file and tls_key_file must be set, or neither") + // closeStarted shuts down any servers that were started before an error occurred, + // preventing leaked listeners during partial startup failures. + closeStarted := func(servers ...*http.Server) { + shutdownCtx, shutdownCancel := context.WithTimeout(context.Background(), 5*time.Second) + defer shutdownCancel() + for _, srv := range servers { + if srv != nil { + if err := srv.Shutdown(shutdownCtx); err != nil { + log.Error().Err(err).Msg("failed to shut down server during startup cleanup") + } + } + } } - caAdapter, err := pkiadapter.NewCA(resolveDataDir(si.cfg.Server.DataDir), si.log) + proxySrv, proxyReady, tlsSrv, _, err := startProxyServers(cfg, httpProxyHandler, httpsProxyHandler, svc.pkiSvc, svc.publicTLSSvc, svc.trafficManager, log) if err != nil { - return si.log.WrapErr(err, "failed to initialize internal CA") - } - si.svc.caAdapter = caAdapter - si.svc.pkiSvc = pkiusecase.NewService(si.ctx, caAdapter, si.svc.configSvc, []string{si.cfg.Server.GordonDomain}, si.log) - return nil -} - -// initPublicTLS initializes the public ACME TLS service if enabled. -func (si *serviceInit) initPublicTLS() error { - if !si.cfg.TLS.ACME.Enabled { - return nil - } - - if err := validatePublicTLSReadiness(si.cfg); err != nil { + closeStarted(registrySrv, internalRegistrySrv) return err } - ctx := si.ctx - log := si.log - - dnsCfg, err := buildDNSConfig(si.cfg) - if err != nil { - return log.WrapErr(err, "invalid DNS configuration") - } - - publicTLSCfg := publictls.Config{ - Enabled: si.cfg.TLS.ACME.Enabled, - Email: si.cfg.TLS.ACME.Email, - Challenge: si.cfg.TLS.ACME.Challenge, - HTTPPort: effectiveHTTP01Port(si.cfg), - TLSPort: effectivePublicTLSPort(si.cfg), - DataDir: resolveDataDir(si.cfg.Server.DataDir), - ObtainBatchSize: si.cfg.TLS.ACME.ObtainBatchSize, - DNS: dnsCfg, - } - - tokenResolver := secrets.NewPublicTLSResolver(secrets.PublicTLSResolverConfig{}) - - effective, err := publictls.ResolveEffectiveChallenge(ctx, publicTLSCfg, tokenResolver) + // Owner-only local administration socket. Started before readiness is + // announced so the CLI can reach the daemon as soon as it is usable, + // and closed on every exit path. TCP /admin/* registration is unchanged. + localAdmin, err := startLocalAdminServer(svc, errChan, log) if err != nil { - return log.WrapErr(err, "resolve ACME challenge") - } - if err := validateEffectivePublicTLSReadiness(si.cfg, effective); err != nil { + closeStarted(registrySrv, internalRegistrySrv, proxySrv, tlsSrv) + shutdownTrafficManagerForStartupCleanup(svc.trafficManager, log) return err } - store, err := acmestore.New(filepath.Join(resolveDataDir(si.cfg.Server.DataDir), "acme")) - if err != nil { - return log.WrapErr(err, "create ACME store") - } - - challenges := publictls.NewHTTP01Challenges() - - var zoneResolver *acmelego.CloudflareZoneResolver - if effective.Mode == domain.ACMEChallengeCloudflareDNS01 { - zoneResolver = acmelego.NewCloudflareZoneResolver(effective.Token) - } - - issuer, err := acmelego.NewIssuer(acmelego.Config{ - Email: si.cfg.TLS.ACME.Email, - Challenge: effective.Mode, - Token: effective.Token, - Store: store, - HTTPChallengeSink: challenges, - DNSResolvers: publicTLSCfg.DNS.Resolvers, - DNSPropagationTimeout: publicTLSCfg.DNS.PropagationTimeout, - DNSPollingInterval: publicTLSCfg.DNS.PollingInterval, - }) - if err != nil { - return log.WrapErr(err, "create ACME issuer") - } - - svc := publictls.NewService(publicTLSCfg, publictls.ServiceDeps{ - Routes: si.svc.configSvc, - Issuer: issuer, - Store: store, - ZoneResolver: zoneResolver, - Challenges: challenges, - Effective: effective, - AdditionalHosts: []string{si.cfg.Server.GordonDomain}, - }) - - if err := svc.Load(ctx); err != nil { - log.Warn().Err(err).Msg("failed to load ACME certificates, continuing") + // cleanupAfterLocalAdmin releases the local admin socket and traffic + // manager in addition to the servers that already bound, so no startup + // failure after this point leaks a listening socket or a live manager. + cleanupAfterLocalAdmin := func(servers ...*http.Server) { + cleanupStartupResources(localAdmin, svc.trafficManager, log, servers...) } - log.Info(). - Str("email", si.cfg.TLS.ACME.Email). - Str("challenge", string(effective.Mode)). - Msg("public ACME TLS initialized (runtime start deferred)") - - si.svc.publicTLSSvc = svc - si.svc.publicTLSRuntime = svc - return nil -} - -// publicTLSRuntime is the subset of in.PublicTLSService needed at server runtime -// after the HTTP listener is bound: reconcile missing certs and start the renewal -// loop. It is nil when ACME is disabled. -type publicTLSRuntime interface { - Reconcile(context.Context) error - StartRenewalLoop(context.Context, time.Duration) <-chan struct{} -} - -func startPublicTLSRuntime(ctx context.Context, svc publicTLSRuntime, log zerowrap.Logger) error { - if svc == nil { - return nil - } - reconcileErr := svc.Reconcile(ctx) - svc.StartRenewalLoop(ctx, time.Hour) - log.Info().Msg("public ACME TLS runtime started") - return reconcileErr -} + svc.tlsHTTPEntryPoints = tlsMuxHTTPServerNames(cfg) + svc.smartHTTPEntryPoints = smartTCPHTTPServerNames(cfg) -// initSecrets creates the domain secret store, env loader, and secret service. -func (si *serviceInit) initSecrets() error { - envDir, backend, passStore, domainSecretStore, err := createDomainSecretStore(si.cfg, si.log) - if err != nil { + // Wait for both registry listeners and the HTTP proxy before applying traffic. + if err := waitForServerReady(ctx, internalRegistryReady, errChan); err != nil { + cleanupAfterLocalAdmin(registrySrv, internalRegistrySrv, proxySrv, tlsSrv) return err } - si.svc.envDir = envDir - - if si.svc.envLoader, err = createEnvLoader(backend, envDir, passStore, si.log); err != nil { + if err := waitForCoreProxyReady(ctx, registryReady, proxyReady, errChan); err != nil { + cleanupAfterLocalAdmin(registrySrv, internalRegistrySrv, proxySrv, tlsSrv) return err } - - si.svc.serviceSecretProvider = createStandaloneServiceSecretProvider(backend, resolveDataDir(si.cfg.Server.DataDir), si.log) - si.svc.secretSvc = secretsSvc.NewService(domainSecretStore, si.log, si.svc.eventBus) - return nil -} - -// initRuntimeAndProxy creates container, backup, registry, image, volume, and proxy services. -func (si *serviceInit) initRuntimeProxyAndTraffic() error { - if err := si.initRuntimeAndProxy(); err != nil { + // Apply only the installation graph fail-fast before slower app + // readiness checks. Persisted ACTIVE binds are intentionally excluded + // until boot reconciliation re-inspects them. + if err := applyTrafficRuntimeConfig(ctx, svc.trafficManager, cfg, svc.configSvc, nil); err != nil { + cleanupAfterLocalAdmin(registrySrv, internalRegistrySrv, proxySrv, tlsSrv) return err } - si.svc.trafficManager = trafficadapter.NewManager() - return nil -} - -func (si *serviceInit) initRuntimeAndProxy() error { - var err error - if si.svc.containerSvc, err = createContainerService(si.ctx, si.v, si.cfg, si.svc, si.log); err != nil { - return err - } + // Reconcile ACTIVE and rebuild its host index before announcing the + // daemon ready. Applying traffic first publishes an empty app graph and + // leaves healthy converged apps returning 404 until a later lifecycle + // operation happens to rebuild it. + reconcileAppsAtBoot(ctx, svc, log) + stopLogExport := startContainerLogExport(ctx, svc, log) + defer stopLogExport() - if si.svc.backupStorage, si.svc.backupSvc, err = createBackupService(si.cfg, si.svc, si.log); err != nil { - return err - } - if si.svc.volumeBackupStore, si.svc.volumeBackupSvc, si.svc.volumeBackupCfg, err = createVolumeBackupService(si.ctx, si.cfg, si.svc, si.log); err != nil { - return err + logEvent := log.Info(). + Int("proxy_port", cfg.Server.Port). + Int("registry_port", cfg.Server.RegistryPort) + if cfg.Server.TLSPort != 0 { + logEvent = logEvent.Int("tls_port", cfg.Server.TLSPort) } + logEvent.Msg("Gordon is running") - registryState := registrystate.New() - si.svc.registrySvc = registrySvc.NewService(si.svc.blobStorage, si.svc.manifestStorage, si.svc.eventBus, registryState) - si.svc.imageSvc = images.NewService(si.svc.runtime, si.svc.manifestStorage, si.svc.blobStorage, si.log, registryState) - si.svc.volumeSvc = volumesSvc.NewService(si.svc.runtime) - - injectTelemetryMetrics(si.cfg, si.svc, si.log) + startPublicTLSRuntimeWithWarning(ctx, svc.publicTLSRuntime, log) - proxyCfg, err := buildProxyConfig(si.cfg, si.log) + schedulerCleanup, err := startOptionalSchedulers(ctx, cfg, svc, log, v) if err != nil { + cleanupAfterLocalAdmin(registrySrv, internalRegistrySrv, proxySrv, tlsSrv) return err } - si.svc.maxBlobChunkSize = proxyCfg.maxBlobChunkSize - si.svc.maxBlobSize = proxyCfg.maxBlobSize - si.svc.proxySvc = proxy.NewService(si.svc.runtime, si.svc.containerSvc, si.svc.configSvc, proxyCfg.proxyConfig) - si.svc.standaloneServiceSvc = servicecfg.NewServiceWithSecretProvider(si.svc.runtime, si.svc.serviceSecretProvider) - - // Wire synchronous proxy cache invalidation for zero-downtime deployments. - // The proxy service implements out.ProxyCacheInvalidator via InvalidateTarget(). - si.svc.containerSvc.SetProxyCacheInvalidator(si.svc.proxySvc) - si.svc.containerSvc.SetProxyDrainWaiter(si.svc.proxySvc) - return nil -} - -// initHandlers creates the auth, health, log, preview, and admin handlers. -func (si *serviceInit) registerReloadCoordinatorHooks() { - if si.svc.reloadCoordinator == nil || si.svc.containerSvc == nil { - return + if schedulerCleanup != nil { + defer schedulerCleanup() } - si.svc.reloadCoordinator.SetContainerConfigApplier(func(reloadCtx context.Context, reloadCfg Config) error { - containerCfg, err := buildContainerServiceConfig(reloadCtx, si.v, reloadCfg, si.svc, si.log) - if err != nil { - return err - } - managementHosts := []string{reloadCfg.Server.GordonDomain} - if si.svc.pkiSvc != nil { - si.svc.pkiSvc.SetAdditionalDomains(managementHosts) - } - if si.svc.publicTLSSvc != nil { - si.svc.publicTLSSvc.SetAdditionalHosts(reloadCtx, managementHosts) - } - var tlsConfig *tls.Config - if hasTLSCapableEntrypoint(reloadCfg) && si.svc.httpsProxyHandler != nil { - var tlsErr error - tlsConfig, tlsErr = proxyTLSConfig(reloadCfg, si.svc.pkiSvc, si.svc.publicTLSSvc, si.log) - if tlsErr != nil { - return tlsErr - } - } - if err := applyTrafficRuntimeConfig(reloadCtx, si.svc.trafficManager, reloadCfg, si.svc.configSvc); err != nil { - return err - } - si.svc.tlsHTTPEntryPoints = registerTLSMuxHTTPServers(si.svc.trafficManager, reloadCfg, si.svc.httpsProxyHandler, tlsConfig, si.svc.tlsHTTPEntryPoints) - si.svc.smartHTTPEntryPoints = registerSmartTCPHTTPServers(si.svc.trafficManager, reloadCfg, si.svc.httpProxyHandler, si.svc.httpsProxyHandler, tlsConfig, si.svc.smartHTTPEntryPoints) - if err := reconcileStandaloneServices(reloadCtx, si.svc.standaloneServiceSvc, reloadCfg); err != nil { - return err - } - si.svc.containerSvc.UpdateConfig(containerCfg) + waitForShutdown(ctx, errChan, reloadChan, reload, svc.eventBus, log) + cleanupHandlers() // Stop debounce timers before draining containers + // Stop app administration before tearing down traffic/runtime dependencies, + // so no new mutation can begin during shutdown. + localAdmin.Close() + registrySrvs := []*http.Server{registrySrv, internalRegistrySrv} + return gracefulShutdown(registrySrvs, proxySrv, tlsSrv, svc.containerSvc, svc.proxySvc, svc.pkiSvc, svc.publicTLSSvc, svc.trafficManager, svc.appMonitor, svc.appSvcImpl, svc.appState, log) +} + +// workloadLogExporter returns the OTLP workload log exporter, or nil +// when telemetry log export is disabled. A typed nil must not leak into +// the out.LogExporter interface. +func workloadLogExporter(provider *telemetry.Provider) out.LogExporter { + if provider == nil || provider.WorkloadLogs == nil { return nil - }) -} - -func (si *serviceInit) initHandlers() { - if si.svc.authSvc != nil { - internalAuth := authhandler.InternalAuth{ - Username: si.svc.internalRegUser, - Password: si.svc.internalRegPass, - } - si.svc.authHandler = authhandler.NewHandler(si.svc.authSvc, internalAuth, si.log) - } - - prober := httpprober.New() - si.svc.healthSvc = health.NewService(si.svc.configSvc, si.svc.containerSvc, prober, si.log) - - si.svc.logSvc = logs.NewService(resolveLogFilePath(si.cfg), si.cfg.Logging.File.Enabled, si.svc.containerSvc, si.svc.runtime, si.log) - - initPreviewService(si.ctx, si.cfg, si.svc, si.log) - if si.svc.trafficManager == nil { - si.svc.trafficManager = trafficadapter.NewManager() - } - - si.svc.adminHandler = admin.NewHandler(admin.HandlerDeps{ - ConfigSvc: si.svc.configSvc, - AuthSvc: si.svc.authSvc, - ContainerSvc: si.svc.containerSvc, - HealthSvc: si.svc.healthSvc, - SecretSvc: si.svc.secretSvc, - LogSvc: si.svc.logSvc, - RegistrySvc: si.svc.registrySvc, - ReloadTrigger: si.svc.reloadCoordinator, - Log: si.log, - BackupSvc: si.svc.backupSvc, - VolumeBackupSvc: si.svc.volumeBackupSvc, - PreviewSvc: si.svc.previewService, - ImageSvc: si.svc.imageSvc, - VolumeSvc: si.svc.volumeSvc, - PublicTLSSvc: si.svc.publicTLSSvc, - TrafficSvc: si.svc.trafficManager, - }) -} - -// initPreviewService sets up the preview store, service, and TTL ticker. -func initPreviewService(ctx context.Context, cfg Config, svc *services, log zerowrap.Logger) { - previewStorePath := filepath.Join(resolveDataDir(cfg.Server.DataDir), "previews.json") - svc.previewStore = filesystem.NewPreviewStore(previewStorePath) - previewConfig := svc.configSvc.GetPreviewConfig() - svc.previewService = preview.NewService(svc.previewStore, previewConfig.TTL). - WithDeployer(svc.containerSvc). - WithRouteManager(svc.configSvc). - WithVolumeCloner(svc.runtime). - WithRegistryDomain(svc.configSvc.GetRegistryDomain()). - WithEnvLoader(svc.envLoader) - if err := svc.previewService.Load(ctx); err != nil { - log.Warn().Err(err).Msg("failed to load previews") - } - - // Derive sweep interval from TTL: half the TTL, capped at 1 hour, minimum 1 minute. - sweepInterval := max(min(previewConfig.TTL/2, time.Hour), time.Minute) - - svc.previewService.StartTicker(ctx, sweepInterval, func(ctx context.Context, p domain.PreviewRoute) { - teardownTrackedPreview(ctx, svc, p) - }, func(ctx context.Context) { - gcOrphanedPreviews(ctx, svc, previewConfig) - }) -} - -// teardownTrackedPreview removes all resources for a tracked preview that has expired. -func teardownTrackedPreview(ctx context.Context, svc *services, p domain.PreviewRoute) { - log := zerowrap.FromCtx(ctx) - for _, containerName := range p.Containers { - if err := svc.runtime.StopContainer(ctx, containerName); err != nil { - log.Warn().Err(err).Str("container", containerName).Str("preview", p.Domain).Msg("failed to stop preview container") - } - if err := svc.runtime.RemoveContainer(ctx, containerName, true); err != nil { - log.Warn().Err(err).Str("container", containerName).Str("preview", p.Domain).Msg("failed to remove preview container") - } - } - for _, volName := range p.Volumes { - if err := svc.runtime.RemoveVolume(ctx, volName, true); err != nil { - log.Warn().Err(err).Str("volume", volName).Str("preview", p.Domain).Msg("failed to remove preview volume") - } - } - // Remove network (naming convention: {networkPrefix}-{domain-sanitized}). - networkPrefix := svc.configSvc.GetNetworkPrefix() - networkName := networkPrefix + "-" + strings.ReplaceAll(p.Domain, ".", "-") - if err := svc.runtime.RemoveNetwork(ctx, networkName); err != nil { - log.Warn().Err(err).Str("network", networkName).Str("preview", p.Domain).Msg("failed to remove preview network") - } - // Remove route from config so proxy stops routing to this domain. - if err := svc.configSvc.RemoveRoute(ctx, p.Domain); err != nil { - if errors.Is(err, domain.ErrRouteNotFound) { - log.Debug().Str("domain", p.Domain).Msg("preview route already removed from config") - } else { - log.Warn().Err(err).Str("domain", p.Domain).Msg("failed to remove preview route from config") - } } - svc.proxySvc.InvalidateTarget(ctx, p.Domain) + return provider.WorkloadLogs } -// gcOrphanedPreviews finds and tears down untracked preview containers. -func gcOrphanedPreviews(ctx context.Context, svc *services, previewConfig domain.PreviewConfig) { - log := zerowrap.FromCtx(ctx) - orphans := svc.previewService.CollectOrphans(ctx, svc.runtime, previewConfig.TagPatterns, previewConfig.Separator) - for _, c := range orphans { - orphanDomain := c.Labels[domain.LabelDomain] - log.Warn().Str("container", c.Name).Str("image", c.Image).Str("domain", orphanDomain). - Time("created", c.Created).Msg("orphaned preview container detected, cleaning up") - - // Order: stop → remove volumes/route → remove container → remove network. - // Network removal must happen after container removal because the runtime - // refuses to delete a network that still has containers attached. - if err := svc.runtime.StopContainer(ctx, c.Name); err != nil { - log.Warn().Err(err).Str("container", c.Name).Msg("failed to stop orphan container") - } - - if orphanDomain != "" { - if err := cleanupOrphanDomainResources(ctx, svc, orphanDomain); err != nil { - log.Warn().Err(err).Str("container", c.Name).Str("domain", orphanDomain). - Msg("orphan resource cleanup failed, deferring container removal to next scan") - continue - } - } - - if err := svc.runtime.RemoveContainer(ctx, c.Name, true); err != nil { - log.Warn().Err(err).Str("container", c.Name).Msg("failed to remove orphan container") - } - - // Remove network after container is gone. - if orphanDomain != "" { - if err := removeOrphanNetwork(ctx, svc, strings.ReplaceAll(orphanDomain, ".", "-"), log); err != nil { - log.Warn().Err(err).Str("container", c.Name).Msg("failed to remove orphan network after container removal") - } - } - - log.Info().Str("container", c.Name).Str("domain", orphanDomain).Msg("orphaned preview container cleaned up") - } - - // Re-sync container state so stale routes disappear from routes list. - if len(orphans) > 0 { - if err := svc.containerSvc.SyncContainers(ctx); err != nil { - log.Warn().Err(err).Msg("failed to sync containers after orphan GC") - } +// withAccessLogExport also exports access entries over OTLP when +// workload log export is enabled. next may be nil (no local sink). +func withAccessLogExport(svc *services, next out.AccessLogWriter) out.AccessLogWriter { + if svc.workloadLogs == nil { + return next } + return logexport.NewAccessLogExporter(svc.appHostIndex, svc.workloadLogs, next) } -// cleanupOrphanDomainResources removes volumes and route for an orphaned -// preview domain. Network removal is handled separately in gcOrphanedPreviews -// after the container is removed. Returns an error if any step fails so the -// caller can defer container removal until the next scan. -func cleanupOrphanDomainResources(ctx context.Context, svc *services, orphanDomain string) error { - log := zerowrap.FromCtx(ctx) - domainSanitized := strings.ReplaceAll(orphanDomain, ".", "-") - var errs []error - - errs = append(errs, removeOrphanVolumes(ctx, svc, domainSanitized, log)...) - errs = append(errs, removeOrphanRoute(ctx, svc, orphanDomain, log)) - svc.proxySvc.InvalidateTarget(ctx, orphanDomain) - - return errors.Join(errs...) -} - -func removeOrphanVolumes(ctx context.Context, svc *services, domainSanitized string, log zerowrap.Logger) []error { - _, volPrefix, _ := svc.configSvc.GetVolumeConfig() - prefix := volPrefix + "-" + domainSanitized + "-" - - volumes, err := svc.runtime.ListVolumes(ctx) - if err != nil { - log.Warn().Err(err).Msg("failed to list volumes for orphan cleanup") - return []error{err} +// startContainerLogExport follows ACTIVE app containers and exports +// their output when telemetry log export is enabled. The returned stop +// function cancels every follower and waits for them, so buffered +// records reach the exporter before telemetry shuts down. +func startContainerLogExport(ctx context.Context, svc *services, log zerowrap.Logger) func() { + if svc.workloadLogs == nil || svc.appState == nil || svc.runtime == nil { + return func() {} } - - var errs []error - for _, v := range volumes { - if !strings.HasPrefix(v.Name, prefix) { - continue - } - if err := svc.runtime.RemoveVolume(ctx, v.Name, true); err != nil { - log.Warn().Err(err).Str("volume", v.Name).Msg("failed to remove orphan volume") - errs = append(errs, err) - } + ctx, cancel := context.WithCancel(ctx) + done := make(chan struct{}) + collector := logexport.NewCollector(svc.appState, svc.runtime, svc.workloadLogs) + go func() { + defer close(done) + collector.Run(ctx) + }() + log.Info().Msg("app container log export enabled") + return func() { + cancel() + <-done } - return errs } -func removeOrphanNetwork(ctx context.Context, svc *services, domainSanitized string, log zerowrap.Logger) error { - name := svc.configSvc.GetNetworkPrefix() + "-" + domainSanitized - if err := svc.runtime.RemoveNetwork(ctx, name); err != nil { - log.Warn().Err(err).Str("network", name).Msg("failed to remove orphan network") +func waitForCoreProxyReady(ctx context.Context, registryReady <-chan struct{}, proxyReady <-chan struct{}, errChan <-chan error) error { + if err := waitForServerReady(ctx, registryReady, errChan); err != nil { return err } - return nil + return waitForServerReady(ctx, proxyReady, errChan) } -func removeOrphanRoute(ctx context.Context, svc *services, orphanDomain string, log zerowrap.Logger) error { - err := svc.configSvc.RemoveRoute(ctx, orphanDomain) - if err == nil || errors.Is(err, domain.ErrRouteNotFound) { +func waitForServerReady(ctx context.Context, ready <-chan struct{}, errChan <-chan error) error { + if ready == nil { return nil } - log.Warn().Err(err).Str("domain", orphanDomain).Msg("failed to remove orphan route") - return err -} - -// injectTelemetryMetrics creates and injects OTel metrics into services when -// telemetry is enabled. Skipped otherwise to avoid unnecessary allocations. -func injectTelemetryMetrics(cfg Config, svc *services, log zerowrap.Logger) { - if !cfg.Telemetry.Enabled || !cfg.Telemetry.Metrics { - return - } - gordonMetrics, err := telemetry.NewMetrics() - if err != nil { - log.Warn().Err(err).Msg("failed to create telemetry metrics, continuing without metrics") - return + select { + case <-ready: + return nil + case err := <-errChan: + return err + case <-ctx.Done(): + return ctx.Err() } - svc.containerSvc.SetMetrics(gordonMetrics) - svc.registrySvc.SetMetrics(gordonMetrics) - svc.eventBus.SetMetrics(gordonMetrics) -} - -func setupInternalRegistryAuth(svc *services, log zerowrap.Logger) error { - var err error - svc.internalRegUser, svc.internalRegPass, err = generateInternalRegistryAuth() - if err != nil { - return log.WrapErr(err, "failed to generate internal registry credentials") - } - - // Persist credentials to file for CLI access (gordon auth internal) - if err := persistInternalCredentials(svc.internalRegUser, svc.internalRegPass); err != nil { - log.Warn().Err(err).Msg("failed to persist internal credentials for CLI access") - } - - log.Debug().Msg("internal registry auth generated for loopback pulls") - return nil -} - -func createDomainSecretStore(cfg Config, log zerowrap.Logger) (string, domain.SecretsBackend, *domainsecrets.PassStore, out.DomainSecretStore, error) { - envDir := resolveEnvDir(cfg) - backend, err := resolveSecretsBackend(cfg.Auth.SecretsBackend) - if err != nil { - return "", "", nil, nil, log.WrapErr(err, "failed to resolve secrets backend") - } - - switch backend { - case domain.SecretsBackendPass: - passStore, err := domainsecrets.NewPassStore(log) - if err != nil { - return "", backend, nil, nil, log.WrapErr(err, "failed to create pass domain secret store") - } - if err := migrateEnvFilesToPass(envDir, passStore, log); err != nil { - return "", backend, nil, nil, log.WrapErr(err, "failed to migrate env files to pass") - } - return envDir, backend, passStore, passStore, nil - default: - store, err := domainsecrets.NewFileStore(envDir, log) - if err != nil { - return "", backend, nil, nil, log.WrapErr(err, "failed to create domain secret store") - } - return envDir, backend, nil, store, nil - } -} - -// resolveLogFilePath returns the configured log file path or a default. -func resolveLogFilePath(cfg Config) string { - if cfg.Logging.File.Path != "" { - return cfg.Logging.File.Path - } - dataDir := cfg.Server.DataDir - if dataDir == "" { - dataDir = DefaultDataDir() - } - return filepath.Join(dataDir, "logs", "gordon.log") -} - -// resolveRuntimeConfig converts a server.runtime config value to a socket path. -// "auto" or "" means auto-detect. -// Named runtimes ("podman", "docker") are resolved to well-known socket paths. -// URI schemes (unix://) are stripped so callers receive a bare path. -func resolveRuntimeConfig(value string) string { - if value == "" || value == "auto" { - return "" - } - // Named runtimes: resolve to well-known socket paths. - switch value { - case "podman": - // Check XDG_RUNTIME_DIR first (rootless Podman). - if xdg := os.Getenv("XDG_RUNTIME_DIR"); xdg != "" { - candidate := filepath.Join(xdg, "podman", "podman.sock") - if _, err := os.Stat(candidate); err == nil { - return candidate - } - } - // Fallback to system-wide Podman socket. - return "/run/podman/podman.sock" - case "docker": - return "/var/run/docker.sock" - } - // Explicit socket path — strip URI scheme if present. - if socketPath, ok := strings.CutPrefix(value, "unix://"); ok { - return socketPath - } - return value -} - -// createOutputAdapters creates the container runtime and event bus. -func createOutputAdapters(ctx context.Context, log zerowrap.Logger, runtimeSocket string) (*docker.Runtime, *eventbus.InMemory, error) { - detection := docker.DetectRuntimeSocket(runtimeSocket) - - var runtime *docker.Runtime - var err error - - switch detection.Source { - case "none": - return nil, nil, fmt.Errorf("no container runtime found: checked Docker socket, Podman socket, DOCKER_HOST env var. Install Docker or Podman, or set server.runtime in config") - case "DOCKER_HOST_passthrough": - detection.RuntimeName = "docker" - runtime, err = docker.NewRuntime() - default: - if detection.SocketPath != "" { - runtime, err = docker.NewRuntimeWithSocket(detection.SocketPath) - } else { - runtime, err = docker.NewRuntime() - } - } - if err != nil { - return nil, nil, log.WrapErr(err, "failed to create container runtime") - } - - if err := runtime.Ping(ctx); err != nil { - return nil, nil, log.WrapErr(err, fmt.Sprintf("container runtime not available (detected: %s via %s)", detection.RuntimeName, detection.Source)) - } - - runtimeVersion, _ := runtime.Version(ctx) - log.Info(). - Str("runtime", detection.RuntimeName). - Str("version", runtimeVersion). - Str("source", detection.Source). - Msg("container runtime initialized") - - eventBus := eventbus.NewInMemory(100, log) - - return runtime, eventBus, nil -} - -// createStorage creates blob and manifest storage. -func createStorage(cfg Config, log zerowrap.Logger) (*filesystem.BlobStorage, *filesystem.ManifestStorage, error) { - dataDir := cfg.Server.DataDir - if dataDir == "" { - dataDir = DefaultDataDir() - } - - registryDir := filepath.Join(dataDir, "registry") - - blobStorage, err := filesystem.NewBlobStorage(registryDir, log) - if err != nil { - return nil, nil, log.WrapErr(err, "failed to create blob storage") - } - - manifestStorage, err := filesystem.NewManifestStorage(registryDir, log) - if err != nil { - return nil, nil, log.WrapErr(err, "failed to create manifest storage") - } - - return blobStorage, manifestStorage, nil -} - -// createEnvLoader creates the environment loader with secret providers. -func createEnvLoader(backend domain.SecretsBackend, envDir string, passStore *domainsecrets.PassStore, log zerowrap.Logger) (out.EnvLoader, error) { - switch backend { - case domain.SecretsBackendPass: - loader, err := envloader.NewPassLoader(passStore, log) - if err != nil { - return nil, log.WrapErr(err, "failed to create pass env loader") - } - return loader, nil - default: - loader, err := envloader.NewFileLoader(envDir, log) - if err != nil { - return nil, log.WrapErr(err, "failed to create env loader") - } - - // Register secret providers - passProvider := secrets.NewPassProvider(log) - if passProvider.IsAvailable() { - loader.RegisterSecretProvider(passProvider) - log.Debug().Msg("pass secret provider registered") - } - - sopsProvider := secrets.NewSopsProvider(log) - if sopsProvider.IsAvailable() { - loader.RegisterSecretProvider(sopsProvider) - log.Debug().Msg("sops secret provider registered") - } - - return loader, nil - } -} - -// createLogWriter creates the container log writer. -func createLogWriter(cfg Config, log zerowrap.Logger) (*logwriter.LogWriter, error) { - if !cfg.Logging.ContainerLogs.Enabled { - log.Debug().Msg("container log collection disabled") - return nil, nil - } - - // Determine log directory - logDir := cfg.Logging.ContainerLogs.Dir - if logDir == "" { - dataDir := cfg.Server.DataDir - if dataDir == "" { - dataDir = DefaultDataDir() - } - logDir = filepath.Join(dataDir, "logs", "containers") - } - - writer, err := logwriter.New(logwriter.Config{ - Dir: logDir, - MaxSize: cfg.Logging.ContainerLogs.MaxSize, - MaxBackups: cfg.Logging.ContainerLogs.MaxBackups, - MaxAge: cfg.Logging.ContainerLogs.MaxAge, - }) - if err != nil { - return nil, log.WrapErr(err, "failed to create container log writer") - } - - log.Info().Str("dir", logDir).Msg("container log collection enabled") - return writer, nil -} - -const ( - internalRegistryUsername = "gordon-internal" - serviceTokenSubject = "gordon-service" - serviceTokenDefaultTTL = 30 * 24 * time.Hour -) - -func generateInternalRegistryAuth() (string, string, error) { - password, err := randomTokenHex(32) - if err != nil { - return "", "", err - } - return internalRegistryUsername, password, nil -} - -func randomTokenHex(size int) (string, error) { - buf := make([]byte, size) - if _, err := rand.Read(buf); err != nil { - return "", err - } - return hex.EncodeToString(buf), nil -} - -// InternalCredentials holds the internal registry credentials for CLI access. -type InternalCredentials struct { - Username string `json:"username"` - Password string `json:"password"` -} - -// getSecureRuntimeDir returns a secure directory for runtime files. -// Priority: XDG_RUNTIME_DIR > ~/.gordon/run -func getSecureRuntimeDir() (string, error) { - // Try XDG_RUNTIME_DIR first (typically /run/user/ on Linux) - if runtimeDir := os.Getenv("XDG_RUNTIME_DIR"); runtimeDir != "" { - gordonDir := filepath.Join(runtimeDir, "gordon") - if err := os.MkdirAll(gordonDir, 0700); err == nil { - return gordonDir, nil - } - } - - // Fall back to ~/.gordon/run - homeDir, err := os.UserHomeDir() - if err != nil { - return "", fmt.Errorf("failed to get home directory: %w", err) - } - - gordonDir := filepath.Join(homeDir, ".gordon", "run") - if err := os.MkdirAll(gordonDir, 0700); err != nil { - return "", fmt.Errorf("failed to create runtime directory: %w", err) - } - - return gordonDir, nil -} - -// getInternalCredentialsFile returns the path to the internal credentials file. -// SECURITY: Credentials are stored in a secure location with restricted permissions. -func getInternalCredentialsFile() string { - runtimeDir, err := getSecureRuntimeDir() - if err != nil { - // Fall back to temp dir if we can't get secure dir (shouldn't happen) - return filepath.Join(os.TempDir(), "gordon-internal-creds.json") - } - return filepath.Join(runtimeDir, "internal-creds.json") -} - -// persistInternalCredentials saves the internal registry credentials to a secure file. -// SECURITY: Credentials are stored in XDG_RUNTIME_DIR or ~/.gordon/run with 0600 permissions. -// The file is cleaned up on graceful shutdown but may persist if Gordon crashes. -// These credentials are for internal loopback communication only and are regenerated on each start. -func persistInternalCredentials(username, password string) error { - creds := InternalCredentials{ - Username: username, - Password: password, - } - data, err := json.Marshal(creds) - if err != nil { - return fmt.Errorf("failed to marshal credentials: %w", err) - } - - credFile := getInternalCredentialsFile() - - // Ensure parent directory exists with secure permissions - if err := os.MkdirAll(filepath.Dir(credFile), 0700); err != nil { - return fmt.Errorf("failed to create credentials directory: %w", err) - } - - // Write file with restrictive permissions (owner read/write only) - if err := os.WriteFile(credFile, data, 0600); err != nil { - return fmt.Errorf("failed to write credentials file: %w", err) - } - return nil -} - -// cleanupInternalCredentials removes the internal credentials file. -func cleanupInternalCredentials() { - _ = os.Remove(getInternalCredentialsFile()) -} - -// getInternalCredentialsCandidates returns candidate file paths in priority order: -// 1. XDG_RUNTIME_DIR/gordon/ (set by systemd for the daemon) -// 2. /run/user//gordon/ (well-known systemd default, for CLI in shells without XDG_RUNTIME_DIR) -// 3. ~/.gordon/run/ (fallback for non-systemd environments) -// 4. os.TempDir() (last resort, matches getInternalCredentialsFile fallback path) -func getInternalCredentialsCandidates() []string { - var candidates []string - - // 1. XDG_RUNTIME_DIR (set in daemon's environment) - if runtimeDir := os.Getenv("XDG_RUNTIME_DIR"); runtimeDir != "" { - candidates = append(candidates, filepath.Join(runtimeDir, "gordon", "internal-creds.json")) - } - - // 2. /run/user//gordon/ (systemd default, may not be in CLI's env) - uid := os.Getuid() - sysRuntime := filepath.Join("/run/user", fmt.Sprintf("%d", uid), "gordon", "internal-creds.json") - // Avoid duplicate if XDG_RUNTIME_DIR already points here - if len(candidates) == 0 || candidates[0] != sysRuntime { - candidates = append(candidates, sysRuntime) - } - - // 3. ~/.gordon/run/ fallback - if homeDir, err := os.UserHomeDir(); err == nil { - candidates = append(candidates, filepath.Join(homeDir, ".gordon", "run", "internal-creds.json")) - } - - // 4. os.TempDir() last resort — matches the fallback path in getInternalCredentialsFile, - // ensuring GetInternalCredentials can find credentials even when getSecureRuntimeDir fails. - candidates = append(candidates, filepath.Join(os.TempDir(), "gordon-internal-creds.json")) - - return candidates -} - -// GetInternalCredentialsFromCandidates reads credentials from the first candidate file that exists. -// Exported for testing. -func GetInternalCredentialsFromCandidates(candidates []string) (*InternalCredentials, error) { - var lastErr error - for _, path := range candidates { - data, err := os.ReadFile(path) - if os.IsNotExist(err) { - continue - } - if err != nil { - // Non-permission errors (e.g. EACCES) may be transient or path-specific; - // record and try the next candidate rather than failing immediately. - lastErr = fmt.Errorf("failed to read credentials file %s: %w", path, err) - continue - } - var creds InternalCredentials - if err := json.Unmarshal(data, &creds); err != nil { - // Corrupt file — record and fall through to lower-priority candidates. - lastErr = fmt.Errorf("failed to parse credentials at %s: %w", path, err) - continue - } - return &creds, nil - } - if lastErr != nil { - return nil, lastErr - } - return nil, fmt.Errorf("no credentials file found (is Gordon running?): checked %v", candidates) -} - -// GetInternalCredentials reads the internal registry credentials from file. -// Probes all candidate runtime directories so CLI works regardless of whether -// XDG_RUNTIME_DIR is set in the current shell environment. -func GetInternalCredentials() (*InternalCredentials, error) { - return GetInternalCredentialsFromCandidates(getInternalCredentialsCandidates()) -} - -// createAuthService creates the authentication service and token store. -func createAuthService(ctx context.Context, cfg Config, log zerowrap.Logger) (out.TokenStore, *auth.Service, error) { - if !cfg.Auth.Enabled { - log.Warn().Msg("auth.enabled=false detected: running in local-only mode (registry loopback-only, admin API disabled)") - return nil, nil, nil - } - - authType, err := resolveAuthType(cfg) - if err != nil { - return nil, nil, fmt.Errorf("resolve auth type: %w", err) - } - backend, err := resolveSecretsBackend(cfg.Auth.SecretsBackend) - if err != nil { - return nil, nil, log.WrapErr(err, "failed to resolve secrets backend") - } - dataDir := resolveDataDir(cfg.Server.DataDir) - - store, err := createTokenStore(backend, dataDir, log) - if err != nil { - return nil, nil, err - } - - authConfig, err := buildAuthConfig(ctx, cfg, authType, backend, dataDir, log) - if err != nil { - return nil, nil, err - } - - authSvc := auth.NewService(authConfig, store, log) - - log.Info(). - Str("type", string(authType)). - Str("backend", string(backend)). - Msg("registry authentication enabled") - - return store, authSvc, nil -} - -// resolveAuthType determines the auth type from config. -// Token-only authentication is the only supported mode. -func resolveAuthType(cfg Config) (domain.AuthType, error) { - if cfg.Auth.Type != "" && cfg.Auth.Type != "token" { - return "", fmt.Errorf("unsupported auth.type %q; only \"token\" is supported", cfg.Auth.Type) - } - return domain.AuthTypeToken, nil -} - -func resolveSecretsBackend(backend string) (domain.SecretsBackend, error) { - switch backend { - case "pass": - return domain.SecretsBackendPass, nil - case "sops": - return domain.SecretsBackendSops, nil - case "unsafe": - return domain.SecretsBackendUnsafe, nil - case "": - return "", fmt.Errorf("auth.secrets_backend is required") - default: - return "", fmt.Errorf("unsupported auth.secrets_backend %q", backend) - } -} - -func resolveDataDir(dataDir string) string { - if dataDir == "" { - return DefaultDataDir() - } - return dataDir -} - -func resolveEnvDir(cfg Config) string { - dataDir := resolveDataDir(cfg.Server.DataDir) - envDir := cfg.Env.Dir - if envDir == "" { - envDir = filepath.Join(dataDir, "env") - } - return envDir -} - -func resolveRegistryDomains(cfg Config) (string, []string) { - registryDomain := cfg.Server.GordonDomain - if registryDomain == "" { - registryDomain = cfg.Server.RegistryDomain - } - return registryDomain, append([]string{}, cfg.Server.LegacyRegistryDomains...) -} - -func createTokenStore(backend domain.SecretsBackend, dataDir string, log zerowrap.Logger) (out.TokenStore, error) { - // Token store is always created since tokens work in both auth modes - store, err := tokenstore.NewStore(backend, dataDir, log) - if err != nil { - return nil, log.WrapErr(err, "failed to create token store") - } - return store, nil -} - -func buildAuthConfig(ctx context.Context, cfg Config, authType domain.AuthType, backend domain.SecretsBackend, dataDir string, log zerowrap.Logger) (auth.Config, error) { - authConfig := auth.Config{ - Enabled: cfg.Auth.Enabled, - AuthType: authType, - Username: cfg.Auth.Username, - } - - // Token config is always required (tokens work in all auth modes) - secret, expiry, err := loadTokenConfig(ctx, cfg, backend, dataDir, log) - if err != nil { - return auth.Config{}, err - } - authConfig.TokenSecret = secret - authConfig.TokenExpiry = expiry - - accessTokenTTL := 15 * time.Minute // default - if cfg.Auth.AccessTokenTTL != "" { - parsed, err := time.ParseDuration(cfg.Auth.AccessTokenTTL) - if err != nil { - return auth.Config{}, fmt.Errorf("invalid auth.access_token_ttl %q: %w", cfg.Auth.AccessTokenTTL, err) - } - if parsed <= 0 { - return auth.Config{}, fmt.Errorf("auth.access_token_ttl must be positive") - } - if parsed > auth.MaxAccessTokenLifetime { - return auth.Config{}, fmt.Errorf("auth.access_token_ttl must not exceed %v", auth.MaxAccessTokenLifetime) - } - accessTokenTTL = parsed - } - authConfig.AccessTokenTTL = accessTokenTTL - - return authConfig, nil -} - -func loadTokenConfig(ctx context.Context, cfg Config, backend domain.SecretsBackend, dataDir string, log zerowrap.Logger) ([]byte, time.Duration, error) { - secret, err := loadTokenSecret(ctx, cfg, backend, dataDir, log) - if err != nil { - return nil, 0, err - } - - expiry, err := parseTokenExpiry(cfg.Auth.TokenExpiry) - if err != nil { - return nil, 0, err - } - - return secret, expiry, nil -} - -// TokenSecretEnvVar is the environment variable for the JWT signing secret. -// SECURITY: This takes priority over config file to allow secure secret injection. -const TokenSecretEnvVar = "GORDON_AUTH_TOKEN_SECRET" //nolint:gosec // This is an env var name, not a credential - -func loadTokenSecret(ctx context.Context, cfg Config, backend domain.SecretsBackend, dataDir string, log zerowrap.Logger) ([]byte, error) { - // SECURITY: Priority order for token secret: - // 1. Environment variable (most secure - no disk exposure) - // 2. Secrets backend (pass/sops - encrypted) - // 3. Config file path (least preferred) - - const minTokenSecretLength = 32 - - // Check environment variable first - if envSecret := os.Getenv(TokenSecretEnvVar); envSecret != "" { - if len(envSecret) < minTokenSecretLength { - return nil, fmt.Errorf("token secret from %s must be at least %d bytes (got %d)", TokenSecretEnvVar, minTokenSecretLength, len(envSecret)) - } - log.Debug().Msg("using token secret from environment variable") - return []byte(envSecret), nil - } - - // Fall back to config-specified path via secrets backend - if cfg.Auth.TokenSecret == "" { - return nil, fmt.Errorf("token_secret is required for JWT token generation; set %s environment variable or configure auth.token_secret", TokenSecretEnvVar) - } - - secret, err := loadSecret(ctx, backend, cfg.Auth.TokenSecret, dataDir, log) - if err != nil { - return nil, log.WrapErr(err, "failed to load token secret") - } - - if len(secret) < minTokenSecretLength { - return nil, fmt.Errorf("token_secret must be at least %d bytes (got %d); use a strong random secret", minTokenSecretLength, len(secret)) - } - - return []byte(secret), nil -} - -func parseTokenExpiry(expiry string) (time.Duration, error) { - if expiry == "" { - return 0, nil - } - - parsed, err := duration.Parse(expiry) - if err != nil { - return 0, fmt.Errorf("invalid token_expiry: %w", err) - } - - return parsed, nil -} - -func resolveServiceTokenExpiry(cfg Config) (time.Duration, error) { - expiry, err := parseTokenExpiry(cfg.Auth.TokenExpiry) - if err != nil { - return 0, err - } - if expiry <= 0 { - return serviceTokenDefaultTTL, nil - } - return expiry, nil -} - -// loadSecret loads a secret from the configured backend. -func loadSecret(ctx context.Context, backend domain.SecretsBackend, path, dataDir string, log zerowrap.Logger) (string, error) { - switch backend { - case domain.SecretsBackendPass: - provider := secrets.NewPassProvider(log) - return provider.GetSecret(ctx, path) - case domain.SecretsBackendSops: - provider := secrets.NewSopsProvider(log) - return provider.GetSecret(ctx, path) - case domain.SecretsBackendUnsafe: - // For unsafe backend, path is relative to dataDir/secrets/. - return readUnsafeSecret(dataDir, path) - default: - return "", fmt.Errorf("unknown secrets backend: %s", backend) - } -} - -func readFileBeneath(root, cleanedRelPath string) ([]byte, error) { - rootFD, err := unix.Open(root, unix.O_RDONLY|unix.O_DIRECTORY|unix.O_CLOEXEC|unix.O_NOFOLLOW, 0) - if err != nil { - return nil, fmt.Errorf("failed to open secrets root: %w", err) - } - defer unix.Close(rootFD) - - parts := strings.Split(filepath.ToSlash(cleanedRelPath), "/") - dirFD := rootFD - var closeDirFDs []int - defer func() { - for i := len(closeDirFDs) - 1; i >= 0; i-- { - _ = unix.Close(closeDirFDs[i]) - } - }() - - for i, part := range parts { - if part == "" || part == "." || part == ".." { - return nil, fmt.Errorf("invalid secret path: path must stay under dataDir/secrets") - } - last := i == len(parts)-1 - if last { - fd, err := unix.Openat(dirFD, part, unix.O_RDONLY|unix.O_NONBLOCK|unix.O_CLOEXEC|unix.O_NOFOLLOW, 0) - if err != nil { - return nil, fmt.Errorf("failed to read secret file: %w", err) - } - defer unix.Close(fd) - var st unix.Stat_t - if err := unix.Fstat(fd, &st); err != nil { - return nil, fmt.Errorf("failed to stat secret file: %w", err) - } - if st.Mode&unix.S_IFMT != unix.S_IFREG { - return nil, fmt.Errorf("invalid secret path: secret must be a regular file") - } - data, err := os.ReadFile(fmt.Sprintf("/proc/self/fd/%d", fd)) - if err != nil { - return nil, fmt.Errorf("failed to read secret file: %w", err) - } - return data, nil - } - - nextFD, err := unix.Openat(dirFD, part, unix.O_RDONLY|unix.O_DIRECTORY|unix.O_CLOEXEC|unix.O_NOFOLLOW, 0) - if err != nil { - return nil, fmt.Errorf("failed to open secret path component: %w", err) - } - closeDirFDs = append(closeDirFDs, nextFD) - dirFD = nextFD - } - - return nil, fmt.Errorf("invalid secret path: empty path") -} - -type unsafeSecretProvider struct { - dataDir string -} - -func (p unsafeSecretProvider) Name() string { return string(domain.SecretsBackendUnsafe) } - -func (p unsafeSecretProvider) IsAvailable() bool { return true } - -func (p unsafeSecretProvider) GetSecret(_ context.Context, path string) (string, error) { - return readUnsafeSecret(p.dataDir, path) -} - -func createStandaloneServiceSecretProvider(backend domain.SecretsBackend, dataDir string, log zerowrap.Logger) out.SecretProvider { - switch backend { - case domain.SecretsBackendPass: - return secrets.NewPassProvider(log) - case domain.SecretsBackendSops: - return secrets.NewSopsProvider(log) - case domain.SecretsBackendUnsafe: - return unsafeSecretProvider{dataDir: dataDir} - default: - return nil - } -} - -func readUnsafeSecret(dataDir, secretPath string) (string, error) { - if filepath.IsAbs(secretPath) { - return "", fmt.Errorf("invalid secret path: absolute paths are not allowed") - } - cleaned := filepath.Clean(secretPath) - if cleaned == "." || cleaned == ".." || strings.HasPrefix(cleaned, ".."+string(filepath.Separator)) { - return "", fmt.Errorf("invalid secret path: path must stay under dataDir/secrets") - } - - root := filepath.Clean(filepath.Join(dataDir, "secrets")) - data, err := readFileBeneath(root, cleaned) - if err != nil { - return "", err - } - return string(data), nil -} - -// proxyConfigResult holds parsed proxy and blob chunk size config. -type proxyConfigResult struct { - proxyConfig proxy.Config - maxBlobChunkSize int64 - maxBlobSize int64 -} - -type configWatcher interface { - Watch(ctx context.Context, onChange func()) error -} - -// publicTLSReconciler is the interface for reconciling public TLS certificates. -type publicTLSReconciler interface { - Reconcile(context.Context) error -} - -type configReloader interface { - Reload(ctx context.Context) error -} - -type proxyConfigUpdater interface { - UpdateConfig(config proxy.Config) -} - -type reloadTrigger interface { - Trigger(ctx context.Context) error -} - -type loadedConfigApplier interface { - ApplyLoadedConfig(ctx context.Context) error -} - -type reloadCoordinator struct { - mu sync.Mutex - lastRun time.Time - debounce time.Duration - - configSvc configReloader - v *viper.Viper - proxySvc proxyConfigUpdater - applyContainerConfig func(context.Context, Config) error - registryLimits interface { - UpdateBlobLimits(maxBlobChunkSize, maxBlobSize int64) - } - eventBus out.EventPublisher - publicTLS publicTLSReconciler - log zerowrap.Logger -} - -func newReloadCoordinator(v *viper.Viper, configSvc configReloader, proxySvc proxyConfigUpdater, registryLimits interface { - UpdateBlobLimits(maxBlobChunkSize, maxBlobSize int64) -}, eventBus out.EventPublisher, publicTLS publicTLSReconciler, log zerowrap.Logger) *reloadCoordinator { - return &reloadCoordinator{ - debounce: 500 * time.Millisecond, - configSvc: configSvc, - v: v, - proxySvc: proxySvc, - registryLimits: registryLimits, - eventBus: eventBus, - publicTLS: publicTLS, - log: log, - } -} - -func (c *reloadCoordinator) SetRegistryLimits(limits interface { - UpdateBlobLimits(maxBlobChunkSize, maxBlobSize int64) -}) { - c.mu.Lock() - defer c.mu.Unlock() - c.registryLimits = limits -} - -func (c *reloadCoordinator) SetContainerConfigApplier(apply func(context.Context, Config) error) { - c.mu.Lock() - defer c.mu.Unlock() - c.applyContainerConfig = apply -} - -func (c *reloadCoordinator) Trigger(ctx context.Context) error { - c.mu.Lock() - defer c.mu.Unlock() - - return c.reloadLocked(ctx, true) -} - -func (c *reloadCoordinator) ApplyLoadedConfig(ctx context.Context) error { - c.mu.Lock() - defer c.mu.Unlock() - - return c.reloadLocked(ctx, false) -} - -func (c *reloadCoordinator) reloadLocked(ctx context.Context, loadConfig bool) error { - now := time.Now() - if !c.lastRun.IsZero() && now.Sub(c.lastRun) < c.debounce { - c.log.Debug().Dur("since_last_reload", now.Sub(c.lastRun)).Msg("skipping config reload trigger due to debounce") - return nil - } - - if loadConfig { - if err := c.configSvc.Reload(ctx); err != nil { - c.log.Error().Err(err).Msg("failed to reload config") - return fmt.Errorf("failed to reload config: %w", err) - } - } - - if err := c.applyLoadedConfig(ctx, now); err != nil { - return err - } - - return nil -} - -func (c *reloadCoordinator) applyLoadedConfig(ctx context.Context, now time.Time) error { - var reloadCfg Config - if err := c.v.Unmarshal(&reloadCfg); err != nil { - c.log.Error().Err(err).Msg("failed to unmarshal config on reload") - return fmt.Errorf("failed to unmarshal config on reload: %w", err) - } - if err := validateEntrypointMigration(c.v, reloadCfg); err != nil { - return err - } - - reloadedProxy, err := buildProxyConfig(reloadCfg, c.log) - if err != nil { - c.log.Error().Err(err).Msg("failed to parse proxy config on reload") - return fmt.Errorf("failed to parse proxy config on reload: %w", err) - } - - if c.applyContainerConfig != nil { - if err := c.applyContainerConfig(ctx, reloadCfg); err != nil { - c.log.Error().Err(err).Msg("failed to apply container config on reload") - return fmt.Errorf("failed to apply container config on reload: %w", err) - } - } - c.proxySvc.UpdateConfig(reloadedProxy.proxyConfig) - if c.registryLimits != nil { - c.registryLimits.UpdateBlobLimits(reloadedProxy.maxBlobChunkSize, reloadedProxy.maxBlobSize) - } - - // Reconcile public TLS before publishing reload events so certificate - // authorization reflects the loaded config even if event delivery fails. - // A transient ACME issue must not abort the rest of the reload. - if c.publicTLS != nil { - if err := c.publicTLS.Reconcile(ctx); err != nil { - c.log.Warn().Err(err).Msg("failed to reconcile public TLS certificates after reload, continuing") - } - } - - if c.eventBus != nil { - if err := c.eventBus.Publish(domain.EventConfigReload, nil); err != nil { - c.log.Error().Err(err).Msg("failed to publish config reload event") - return fmt.Errorf("failed to publish config reload event: %w", err) - } - } - - c.lastRun = now - - c.log.Debug().Msg("config hot reload complete") - return nil -} - -// buildProxyConfig parses size-related config fields and builds the proxy config. -func buildProxyConfig(cfg Config, log zerowrap.Logger) (*proxyConfigResult, error) { - maxProxyBodySize := int64(512 << 20) // 512MB default - if cfg.Server.MaxProxyBodySize != "" { - parsedSize, err := bytesize.Parse(cfg.Server.MaxProxyBodySize) - if err != nil { - return nil, log.WrapErrWithFields(err, "invalid server.max_proxy_body_size configuration", map[string]any{"value": cfg.Server.MaxProxyBodySize}) - } - maxProxyBodySize = parsedSize - } - - maxBlobChunkSize := int64(registry.DefaultMaxBlobChunkSize) - if cfg.Server.MaxBlobChunkSize != "" { - parsedSize, err := bytesize.Parse(cfg.Server.MaxBlobChunkSize) - if err != nil { - return nil, log.WrapErrWithFields(err, "invalid server.max_blob_chunk_size configuration", map[string]any{"value": cfg.Server.MaxBlobChunkSize}) - } - maxBlobChunkSize = parsedSize - } - - maxBlobSize := int64(registry.DefaultMaxBlobSize) - if cfg.Server.MaxBlobSize != "" { - parsedSize, err := bytesize.Parse(cfg.Server.MaxBlobSize) - if err != nil { - return nil, log.WrapErrWithFields(err, "invalid server.max_blob_size configuration", map[string]any{"value": cfg.Server.MaxBlobSize}) - } - maxBlobSize = parsedSize - } - - maxProxyResponseSize := int64(1 << 30) // 1GB default - if cfg.Server.MaxProxyResponseSize != "" { - parsedSize, err := bytesize.Parse(cfg.Server.MaxProxyResponseSize) - if err != nil { - return nil, log.WrapErrWithFields(err, "invalid server.max_proxy_response_size configuration", map[string]any{"value": cfg.Server.MaxProxyResponseSize}) - } - maxProxyResponseSize = parsedSize - } - - maxConcurrentConns := cfg.Server.MaxConcurrentConns - if maxConcurrentConns < 0 { - maxConcurrentConns = 10000 // default when explicitly set to -1 - } - // 0 means no limit (as documented in proxy.Config) - - registryDomain, _ := resolveRegistryDomains(cfg) - - return &proxyConfigResult{ - proxyConfig: proxy.Config{ - RegistryDomain: registryDomain, - RegistryPort: cfg.Server.RegistryPort, - MaxBodySize: maxProxyBodySize, - MaxResponseSize: maxProxyResponseSize, - MaxConcurrentConns: maxConcurrentConns, - }, - maxBlobChunkSize: maxBlobChunkSize, - maxBlobSize: maxBlobSize, - }, nil -} - -// buildDNSConfig parses the raw dns config section into a publictls.DNSConfig. -func buildDNSConfig(cfg Config) (publictls.DNSConfig, error) { - defaults := publictls.DefaultDNSConfig() - - resolvers := cfg.DNS.Resolvers - if len(resolvers) == 0 { - resolvers = defaults.Resolvers - } - - propagationTimeout := defaults.PropagationTimeout - if cfg.DNS.PropagationTimeout != "" { - parsed, err := time.ParseDuration(cfg.DNS.PropagationTimeout) - if err != nil { - return publictls.DNSConfig{}, fmt.Errorf("invalid dns.propagation_timeout: %w", err) - } - propagationTimeout = parsed - } - - pollingInterval := defaults.PollingInterval - if cfg.DNS.PollingInterval != "" { - parsed, err := time.ParseDuration(cfg.DNS.PollingInterval) - if err != nil { - return publictls.DNSConfig{}, fmt.Errorf("invalid dns.polling_interval: %w", err) - } - pollingInterval = parsed - } - - dnsCfg := publictls.DNSConfig{ - Resolvers: append([]string(nil), resolvers...), - PropagationTimeout: propagationTimeout, - PollingInterval: pollingInterval, - } - if err := dnsCfg.Validate(); err != nil { - return publictls.DNSConfig{}, err - } - return dnsCfg, nil -} - -func buildContainerServiceConfig(ctx context.Context, v *viper.Viper, cfg Config, svc *services, log zerowrap.Logger) (container.Config, error) { - if cfg.Containers.CPULimit < 0 { - return container.Config{}, fmt.Errorf("containers.cpu_limit must be >= 0 (got %f)", cfg.Containers.CPULimit) - } - if cfg.Containers.PidsLimit < 0 { - return container.Config{}, fmt.Errorf("containers.pids_limit must be >= 0 (got %d)", cfg.Containers.PidsLimit) - } - var defaultMemoryLimit int64 - if cfg.Containers.MemoryLimit != "" { - parsed, err := bytesize.Parse(cfg.Containers.MemoryLimit) - if err != nil { - return container.Config{}, fmt.Errorf("invalid containers.memory_limit %q: %w", cfg.Containers.MemoryLimit, err) - } - if parsed <= 0 { - return container.Config{}, fmt.Errorf("containers.memory_limit must be positive (got %q)", cfg.Containers.MemoryLimit) - } - defaultMemoryLimit = parsed - } - var defaultNanoCPUs int64 - if cfg.Containers.CPULimit > 0 { - defaultNanoCPUs = int64(cfg.Containers.CPULimit * 1e9) - } - - attachmentConfig := svc.configSvc.GetAttachmentConfig() - registryDomain, legacyRegistryDomains := resolveRegistryDomains(cfg) - - containerConfig := container.Config{ - RegistryAuthEnabled: cfg.Auth.Enabled, - RegistryDomain: registryDomain, - LegacyRegistryDomains: legacyRegistryDomains, - RegistryPort: cfg.Server.RegistryPort, - InternalRegistryUsername: svc.internalRegUser, - InternalRegistryPassword: svc.internalRegPass, - PullPolicy: v.GetString("deploy.pull_policy"), - VolumeAutoCreate: v.GetBool("volumes.auto_create"), - VolumePrefix: v.GetString("volumes.prefix"), - VolumePreserve: v.GetBool("volumes.preserve"), - NetworkIsolation: v.GetBool("network_isolation.enabled"), - NetworkPrefix: v.GetString("network_isolation.network_prefix"), - NetworkGroups: attachmentConfig.NetworkGroups, - NetworkInternal: v.GetBool("network_isolation.internal"), - Attachments: attachmentConfig.Attachments, - AllowedRegistries: cfg.Images.AllowedRegistries, - RequireImageDigest: cfg.Images.RequireDigest, - SecurityProfile: cfg.Containers.SecurityProfile, - ReadinessDelay: v.GetDuration("deploy.readiness_delay"), - ReadinessMode: v.GetString("deploy.readiness_mode"), - HealthTimeout: v.GetDuration("deploy.health_timeout"), - StabilizationDelay: v.GetDuration("deploy.stabilization_delay"), - TCPProbeTimeout: v.GetDuration("deploy.tcp_probe_timeout"), - HTTPProbeTimeout: v.GetDuration("deploy.http_probe_timeout"), - DrainDelay: v.GetDuration("deploy.drain_delay"), - DrainMode: v.GetString("deploy.drain_mode"), - DrainTimeout: v.GetDuration("deploy.drain_timeout"), - DefaultMemoryLimit: defaultMemoryLimit, - DefaultNanoCPUs: defaultNanoCPUs, - DefaultPidsLimit: cfg.Containers.PidsLimit, - AttachmentReadinessTimeout: v.GetDuration("deploy.attachment_readiness_timeout"), - } - if v.IsSet("deploy.drain_delay") { - containerConfig.DrainDelayConfigured = true - containerConfig.DrainDelay = v.GetDuration("deploy.drain_delay") - } - - if containerConfig.RegistryAuthEnabled { - if svc.authSvc == nil { - return container.Config{}, fmt.Errorf("authentication service unavailable: cannot generate registry service token") - } - expiry, err := resolveServiceTokenExpiry(cfg) - if err != nil { - return container.Config{}, log.WrapErr(err, "failed to resolve service token expiry") - } - serviceToken, err := svc.authSvc.GenerateToken(ctx, serviceTokenSubject, []string{"pull"}, expiry) - if err != nil { - return container.Config{}, log.WrapErr(err, "failed to generate registry service token") - } - log.Info(). - Str("subject", serviceTokenSubject). - Str("expiry", expiry.String()). - Msg("generated service token for container registry access") - containerConfig.ServiceTokenUsername = serviceTokenSubject - containerConfig.ServiceToken = serviceToken - } else { - log.Warn().Msg("registry auth disabled; container image pulls will use unauthenticated mode") - } - - return containerConfig, nil -} - -// createContainerService creates the container service with configuration. -func createContainerService(ctx context.Context, v *viper.Viper, cfg Config, svc *services, log zerowrap.Logger) (*container.Service, error) { - containerConfig, err := buildContainerServiceConfig(ctx, v, cfg, svc, log) - if err != nil { - return nil, err - } - return container.NewService(svc.runtime, svc.envLoader, svc.eventBus, svc.logWriter, containerConfig, svc.configSvc), nil -} - -type databaseBackupSettingsConfig struct { - Enabled bool - Schedule string - StorageDir string - Retention struct { - Hourly int - Daily int - Weekly int - Monthly int - } -} - -func databaseBackupSettings(cfg Config) databaseBackupSettingsConfig { - out := databaseBackupSettingsConfig{ - Enabled: cfg.Backups.Databases.Enabled, - Schedule: cfg.Backups.Databases.Schedule, - StorageDir: cfg.Backups.Databases.StorageDir, - } - out.Retention.Hourly = cfg.Backups.Databases.Retention.Hourly - out.Retention.Daily = cfg.Backups.Databases.Retention.Daily - out.Retention.Weekly = cfg.Backups.Databases.Retention.Weekly - out.Retention.Monthly = cfg.Backups.Databases.Retention.Monthly - // Legacy [backups] keys intentionally override new database defaults when - // backups.enabled is true, preserving existing working pg_dump schedules. - // Otherwise, prefer the already-populated backups.databases.* values. - if cfg.Backups.Enabled { - out.Enabled = true - if cfg.Backups.Schedule != "" { - out.Schedule = cfg.Backups.Schedule - } - if cfg.Backups.StorageDir != "" { - out.StorageDir = cfg.Backups.StorageDir - } - } else { - if out.Schedule == "" { - out.Schedule = cfg.Backups.Schedule - } - if out.StorageDir == "" { - out.StorageDir = cfg.Backups.StorageDir - } - } - if out.Retention.Hourly == 0 { - out.Retention.Hourly = cfg.Backups.Retention.Hourly - } - if out.Retention.Daily == 0 { - out.Retention.Daily = cfg.Backups.Retention.Daily - } - if out.Retention.Weekly == 0 { - out.Retention.Weekly = cfg.Backups.Retention.Weekly - } - if out.Retention.Monthly == 0 { - out.Retention.Monthly = cfg.Backups.Retention.Monthly - } - return out -} - -func createBackupService(cfg Config, svc *services, log zerowrap.Logger) (*filesystem.BackupStorage, *backup.Service, error) { - dbCfg := databaseBackupSettings(cfg) - if !dbCfg.Enabled { - return nil, nil, nil - } - - storageDir := dbCfg.StorageDir - if storageDir == "" { - dataDir := resolveDataDir(cfg.Server.DataDir) - storageDir = filepath.Join(dataDir, "backups") - } - - backupStorage, err := filesystem.NewBackupStorage(storageDir, log) - if err != nil { - return nil, nil, log.WrapErr(err, "failed to create backup storage") - } - - retention, err := validateBackupRetention(cfg) - if err != nil { - return nil, nil, log.WrapErr(err, "invalid backup retention policy") - } - - backupCfg := domain.BackupConfig{ - Enabled: dbCfg.Enabled, - StorageDir: storageDir, - Retention: retention, - } - - backupSvc := backup.NewService(svc.runtime, backupStorage, svc.containerSvc, backupCfg, log) - - log.Info(). - Str("storage_dir", storageDir). - Msg("backup service initialized") - - return backupStorage, backupSvc, nil -} - -func createVolumeBackupService(ctx context.Context, cfg Config, svc *services, log zerowrap.Logger) (out.VolumeBackupStorage, *backup.VolumeService, domain.VolumeBackupConfig, error) { - if !cfg.Backups.Volumes.Enabled { - return nil, nil, domain.VolumeBackupConfig{}, nil - } - volumeCfg, err := validateVolumeBackupConfig(cfg) - if err != nil { - return nil, nil, domain.VolumeBackupConfig{}, log.WrapErr(err, "invalid volume backup configuration") - } - - storage, err := s3storage.NewVolumeBackupStorage(ctx, volumeCfg) - if err != nil { - return nil, nil, domain.VolumeBackupConfig{}, log.WrapErr(err, "failed to create volume backup storage") - } - - volumeSvc := backup.NewVolumeService(svc.runtime, svc.runtime, storage, volumeCfg, log) - log.Info(). - Str("bucket", volumeCfg.S3Bucket). - Str("prefix", volumeCfg.S3Prefix). - Msg("volume backup service initialized") - - return storage, volumeSvc, volumeCfg, nil -} - -func validateBackupRetention(cfg Config) (domain.RetentionPolicy, error) { - dbCfg := databaseBackupSettings(cfg) - if dbCfg.Retention.Hourly < 0 { - return domain.RetentionPolicy{}, fmt.Errorf("backups.databases.retention.hourly cannot be negative") - } - if dbCfg.Retention.Daily < 0 { - return domain.RetentionPolicy{}, fmt.Errorf("backups.databases.retention.daily cannot be negative") - } - if dbCfg.Retention.Weekly < 0 { - return domain.RetentionPolicy{}, fmt.Errorf("backups.databases.retention.weekly cannot be negative") - } - if dbCfg.Retention.Monthly < 0 { - return domain.RetentionPolicy{}, fmt.Errorf("backups.databases.retention.monthly cannot be negative") - } - - return domain.RetentionPolicy{ - Hourly: dbCfg.Retention.Hourly, - Daily: dbCfg.Retention.Daily, - Weekly: dbCfg.Retention.Weekly, - Monthly: dbCfg.Retention.Monthly, - }, nil -} - -func validateVolumeBackupConfig(cfg Config) (domain.VolumeBackupConfig, error) { - volumeCfg := cfg.Backups.Volumes - interval, err := parsePositiveDurationDefault(volumeCfg.Interval, "24h", "backups.volumes.interval") - if err != nil { - return domain.VolumeBackupConfig{}, err - } - timeout, err := parsePositiveDurationDefault(volumeCfg.Timeout, "2h", "backups.volumes.timeout") - if err != nil { - return domain.VolumeBackupConfig{}, err - } - compression, err := parseVolumeBackupCompression(volumeCfg.Compression) - if err != nil { - return domain.VolumeBackupConfig{}, err - } - maxConcurrency := volumeCfg.MaxConcurrency - if maxConcurrency == 0 { - maxConcurrency = 2 - } - helperImage := strings.TrimSpace(volumeCfg.HelperImage) - if helperImage == "" { - helperImage = "alpine:3.20" - } - volumePrefix := strings.TrimSpace(cfg.Volumes.Prefix) - if volumePrefix == "" { - volumePrefix = "gordon" - } - if compression == domain.VolumeBackupCompressionZstd && helperImage == "alpine:3.20" { - return domain.VolumeBackupConfig{}, fmt.Errorf("backups.volumes.compression zstd requires a helper_image that provides zstd") - } - if err := validateVolumeBackupS3Settings(volumeCfg.Enabled, volumeCfg.Retention.Keep, maxConcurrency, volumeCfg.S3.Bucket, volumeCfg.S3.Region); err != nil { - return domain.VolumeBackupConfig{}, err - } - return domain.VolumeBackupConfig{ - Enabled: volumeCfg.Enabled, - Interval: interval, - Compression: compression, - Retention: domain.VolumeBackupRetentionPolicy{Keep: volumeCfg.Retention.Keep}, - Timeout: timeout, - MaxConcurrency: maxConcurrency, - HelperImage: helperImage, - VolumePrefix: volumePrefix, - S3Bucket: strings.TrimSpace(volumeCfg.S3.Bucket), - S3Region: strings.TrimSpace(volumeCfg.S3.Region), - S3Prefix: strings.TrimSpace(volumeCfg.S3.Prefix), - S3Endpoint: strings.TrimSpace(volumeCfg.S3.Endpoint), - S3PathStyle: volumeCfg.S3.PathStyle, - S3SSEAlgorithm: strings.TrimSpace(volumeCfg.S3.SSEAlgorithm), - S3SSEKMSKeyID: strings.TrimSpace(volumeCfg.S3.SSEKMSKeyID), - }, nil -} - -func parsePositiveDurationDefault(raw, defaultValue, field string) (time.Duration, error) { - if strings.TrimSpace(raw) == "" { - raw = defaultValue - } - d, err := time.ParseDuration(strings.TrimSpace(raw)) - if err != nil || d <= 0 { - return 0, fmt.Errorf("%s must be a positive duration", field) - } - return d, nil -} - -func parseVolumeBackupCompression(raw string) (domain.VolumeBackupCompression, error) { - if strings.TrimSpace(raw) == "" { - raw = string(domain.VolumeBackupCompressionGzip) - } - compression := domain.VolumeBackupCompression(strings.ToLower(strings.TrimSpace(raw))) - switch compression { - case domain.VolumeBackupCompressionGzip, domain.VolumeBackupCompressionZstd: - return compression, nil - default: - return "", fmt.Errorf("backups.volumes.compression must be one of: gzip, zstd") - } -} - -func validateVolumeBackupS3Settings(enabled bool, keep, maxConcurrency int, bucket, region string) error { - if keep < 0 { - return fmt.Errorf("backups.volumes.retention.keep cannot be negative") - } - if enabled && keep == 0 { - return fmt.Errorf("backups.volumes.retention.keep must be positive when volume backups are enabled") - } - if maxConcurrency < 1 { - return fmt.Errorf("backups.volumes.max_concurrency must be at least 1") - } - if enabled && strings.TrimSpace(bucket) == "" { - return fmt.Errorf("backups.volumes.s3.bucket is required when volume backups are enabled") - } - if enabled && strings.TrimSpace(region) == "" { - return fmt.Errorf("backups.volumes.s3.region is required when volume backups are enabled") - } - return nil -} - -// registerEventHandlers registers all event handlers. -func registerEventHandlers(ctx context.Context, svc *services, cfg Config) (func(), error) { - imagePushedHandler := container.NewImagePushedHandler(ctx, svc.containerSvc, svc.configSvc) - if err := svc.eventBus.Subscribe(imagePushedHandler); err != nil { - return nil, fmt.Errorf("failed to subscribe image pushed handler: %w", err) - } - - // Auto-route handler for creating routes from image labels - registryDomain, legacyRegistryDomains := resolveRegistryDomains(cfg) - autoRouteHandler := container.NewAutoRouteHandler(ctx, svc.configSvc, svc.containerSvc, svc.blobStorage, registryDomain, legacyRegistryDomains...). - WithEnvExtractor(svc.runtime, svc.envDir) - - // Preview handler for creating preview environments from tagged images - autoPreviewHandler := preview.NewAutoPreviewHandler( - ctx, - svc.configSvc, - svc.previewService, - ) - - // Dispatcher routes image push events to either auto-route or preview handler - dispatcher := auto.NewImagePushDispatcher(svc.configSvc, autoRouteHandler, autoPreviewHandler) - if err := svc.eventBus.Subscribe(dispatcher); err != nil { - return nil, fmt.Errorf("subscribe image push dispatcher: %w", err) - } - - configReloadHandler := container.NewConfigReloadHandler(ctx, svc.containerSvc, svc.configSvc) - if err := svc.eventBus.Subscribe(configReloadHandler); err != nil { - return nil, fmt.Errorf("failed to subscribe config reload handler: %w", err) - } - - manualDeployHandler := container.NewManualDeployHandler(ctx, svc.containerSvc, svc.configSvc) - if err := svc.eventBus.Subscribe(manualDeployHandler); err != nil { - return nil, fmt.Errorf("failed to subscribe manual deploy handler: %w", err) - } - - secretsChangedHandler := container.NewSecretsChangedHandler(ctx, svc.containerSvc, svc.configSvc, container.DefaultSecretsDebounce) - if err := svc.eventBus.Subscribe(secretsChangedHandler); err != nil { - return nil, fmt.Errorf("failed to subscribe secrets changed handler: %w", err) - } - - // Proxy cache invalidation on config reload (clears stale targets for removed routes) - configReloadProxyHandler := proxy.NewConfigReloadProxyHandler(ctx, svc.proxySvc) - if err := svc.eventBus.Subscribe(configReloadProxyHandler); err != nil { - return nil, fmt.Errorf("failed to subscribe config reload proxy handler: %w", err) - } - - cleanup := func() { - secretsChangedHandler.Stop() - } - - return cleanup, nil -} - -// setupConfigHotReload sets up config hot reload. -func setupConfigHotReload(ctx context.Context, configSvc configWatcher, coordinator loadedConfigApplier) error { - if err := configSvc.Watch(ctx, func() { - _ = coordinator.ApplyLoadedConfig(ctx) - }); err != nil { - return fmt.Errorf("failed to watch config: %w", err) - } - - return nil -} - -func loopbackOnly(next http.Handler, log zerowrap.Logger) http.Handler { - return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - host, _, err := net.SplitHostPort(r.RemoteAddr) - if err != nil { - host = r.RemoteAddr - } - - ip := net.ParseIP(host) - if ip == nil || !ip.IsLoopback() { - log.Warn(). - Str("path", r.URL.Path). - Str("remote_addr", r.RemoteAddr). - Msg("blocked non-loopback access on internal admin route") - w.Header().Set("Content-Type", "application/json") - w.WriteHeader(http.StatusForbidden) - _ = json.NewEncoder(w).Encode(dto.ErrorResponse{Error: "Forbidden"}) - return - } - - next.ServeHTTP(w, r) - }) -} - -// createHTTPHandlers creates HTTP handlers with middleware. -// Returns three handlers: registry, HTTP proxy (with CIDR + onboarding), and HTTPS proxy. -func createHTTPHandlers(svc *services, cfg Config, log zerowrap.Logger, accessWriter out.AccessLogWriter) (http.Handler, http.Handler, http.Handler) { - // Parse trusted proxies once for all middleware chains. - // This ensures consistent IP extraction across logging, rate limiting, and auth. - trustedNets := httphelper.ParseTrustedProxies(cfg.API.RateLimit.TrustedProxies) - - // Registry handler - registryHandler := registry.NewHandler(svc.registrySvc, log, svc.maxBlobChunkSize, svc.maxBlobSize) - svc.registryHandler = registryHandler - if svc.reloadCoordinator != nil { - svc.reloadCoordinator.SetRegistryLimits(registryHandler) - } - registryWithMiddleware, cidrAllowlistMiddleware, rateLimitMiddleware := buildRegistryHandlerWithMiddleware( - svc, - cfg, - trustedNets, - registryHandler, - log, - ) - - registryMux := http.NewServeMux() - registerAuthRoutes(registryMux, svc, trustedNets, cidrAllowlistMiddleware, rateLimitMiddleware, cfg, log) - registryMux.Handle("/v2/", wrapRegistryForLocalMode(registryWithMiddleware, cfg, log)) - registerAdminRoutes(registryMux, svc, cfg, trustedNets, log) - - // Proxy handler - proxyHandler := proxyadapter.NewHandler(svc.proxySvc, trustedNets, log) - - // HTTP proxy handler chain: HTTPS redirect for non-proxy clients, then CIDR allowlist - proxyAllowedNets, proxyCIDRMiddleware := buildProxyCIDRAllowlistMiddleware(cfg, trustedNets, log) - - httpProxyMiddlewares := []func(http.Handler) http.Handler{ - middleware.PanicRecovery(log), - middleware.RequestLogger(log, trustedNets), - middleware.SecurityHeaders, - middleware.HTTPSRedirect(proxyAllowedNets, effectiveProxyHTTPPort(cfg), effectiveProxyTLSPort(cfg), cfg.Server.ForceHTTPSRedirect, log, func(host string) bool { - return svc.proxySvc.IsKnownHost(context.Background(), host) - }), - } - if proxyCIDRMiddleware != nil { - httpProxyMiddlewares = append(httpProxyMiddlewares, proxyCIDRMiddleware) - } - - httpProxyWithMiddleware := otelhttp.NewHandler( - middleware.Chain(httpProxyMiddlewares...)(proxyHandler), - "gordon.proxy", - ) - - // Build the onboarding handler once if internal CA is available and TLS is enabled. - var obHandler *onboarding.Handler - if svc.caAdapter != nil && effectiveProxyTLSPort(cfg) != 0 { - mobileconfigBytes := pkiadapter.GenerateMobileconfig( - svc.caAdapter.RootCertificateDER(), - svc.caAdapter.RootCommonName(), - ) - obHandler = onboarding.NewHandler( - svc.caAdapter.RootCertificate(), - mobileconfigBytes, - svc.caAdapter.RootFingerprint(), - effectiveProxyHTTPPort(cfg), - effectiveProxyTLSPort(cfg), - ) - } - - // HTTP proxyMux: trusted proxy traffic flows through the normal proxy chain. - // Direct clients get an onboarding gate (when CA is available) placed BEFORE - // HTTPSRedirect so force_https_redirect cannot bypass onboarding. - // ACME HTTP-01 challenge handler is registered before the catch-all "/" so - // it gets first chance regardless of source IP. - proxyMux := http.NewServeMux() - - // Register ACME HTTP-01 challenge handler before all other routes so - // Let's Encrypt validation always succeeds, even for onboarding clients. - if svc.publicTLSSvc != nil { - proxyMux.Handle(acmehttp.Prefix, acmehttp.NewHandler(svc.publicTLSSvc)) - } - - if proxyCIDRMiddleware != nil && proxyAllowedNets == nil { - // Invalid proxy_allowed_ips: deny all traffic (fail-closed). - proxyMux.Handle("/", proxyCIDRMiddleware(httpProxyWithMiddleware)) - } else if obHandler != nil { - proxyMux.Handle("/", directHTTPOnboardingGate(obHandler, proxyAllowedNets, httpProxyWithMiddleware, log)) - } else { - proxyMux.Handle("/", httpProxyWithMiddleware) - } - - // HTTPS proxy handler chain: security headers + proxy + CA onboarding - // Onboarding routes live on the TLS port so Tailnet / direct clients - // can click through the initial cert warning, install the CA, and - // then trust all subsequent connections. - // The middleware chain wraps the entire mux so onboarding routes also - // get PanicRecovery, RequestLogger, and SecurityHeaders. - httpsProxyMiddlewares := []func(http.Handler) http.Handler{ - middleware.PanicRecovery(log), - middleware.RequestLogger(log, trustedNets), - middleware.SecurityHeaders, - } - - httpsMux := http.NewServeMux() - if obHandler != nil && cfg.Server.GordonDomain != "" { - gordonDomain := strings.TrimSuffix(strings.ToLower(strings.TrimSpace(cfg.Server.GordonDomain)), ".") - onboardingMux := http.NewServeMux() - registerOnboardingRoutes(onboardingMux, obHandler) - // Register onboarding paths host-gated so normal traffic hits - // proxyHandler directly through the catch-all / pattern. - httpsMux.Handle("GET /.well-known/gordon/", gordonDomainOnboardingGate(gordonDomain, onboardingMux, proxyHandler)) - httpsMux.Handle("GET /.well-known/gordon/ca", gordonDomainOnboardingGate(gordonDomain, onboardingMux, proxyHandler)) - httpsMux.Handle("GET /.well-known/gordon/ca.crt", gordonDomainOnboardingGate(gordonDomain, onboardingMux, proxyHandler)) - httpsMux.Handle("GET /.well-known/gordon/ca.mobileconfig", gordonDomainOnboardingGate(gordonDomain, onboardingMux, proxyHandler)) - httpsMux.Handle("/", proxyHandler) - } else { - httpsMux.Handle("/", proxyHandler) - } - - httpsHandler := otelhttp.NewHandler(middleware.Chain(httpsProxyMiddlewares...)(httpsMux), "gordon.proxy.tls") - - // Wrap top-level handlers with access logging outside all gates - // (loopbackOnly, denyAllHandler, CIDR allowlist) so every request — - // including rejected probes — produces exactly one access-log line. - var registryOut, proxyOut, httpsOut http.Handler = registryMux, proxyMux, httpsHandler - if accessWriter != nil { - excludeHC := cfg.Logging.AccessLog.ExcludeHealthChecks - registryOut = middleware.AccessLogger(accessWriter, excludeHC, log, trustedNets)(registryOut) - proxyOut = middleware.AccessLogger(accessWriter, excludeHC, log, trustedNets)(proxyOut) - httpsOut = middleware.AccessLogger(accessWriter, excludeHC, log, trustedNets)(httpsOut) - } - svc.httpsProxyHandler = httpsOut - - return registryOut, proxyOut, httpsOut -} - -// registerOnboardingRoutes registers CA onboarding well-known HTTP routes on -// the given mux. Both direct-HTTP and Gordon-domain HTTPS onboarding use this. -func registerOnboardingRoutes(mux *http.ServeMux, ob *onboarding.Handler) { - mux.HandleFunc("GET /.well-known/gordon/", ob.ServeOnboardingPage) - mux.HandleFunc("GET /.well-known/gordon/ca", ob.ServeOnboardingPage) - mux.HandleFunc("GET /.well-known/gordon/ca.crt", ob.ServeCACert) - mux.HandleFunc("GET /.well-known/gordon/ca.mobileconfig", ob.ServeMobileconfig) -} - -// directHTTPOnboardingGate returns an http.Handler that splits HTTP traffic -// by source IP. Trusted proxy IPs flow through to the normal proxy chain. -// Direct clients are served the CA onboarding flow on allowed paths and -// receive 403 on everything else. This gate runs BEFORE HTTPSRedirect so -// force_https_redirect cannot bypass onboarding for direct clients. -func directHTTPOnboardingGate(ob *onboarding.Handler, proxyNets []*net.IPNet, proxyChain http.Handler, log zerowrap.Logger) http.Handler { - // Build a small mux for direct-client onboarding paths. - onboardingMux := http.NewServeMux() - registerOnboardingRoutes(onboardingMux, ob) - - // Reserve ACME challenge path for future use. - onboardingMux.HandleFunc("/.well-known/acme-challenge/", func(w http.ResponseWriter, _ *http.Request) { - http.Error(w, "not found", http.StatusNotFound) - }) - - // Catch-all: reject any other direct HTTP request. - // Uses a method-aware split: GET writes a body, HEAD gets an empty 403. - onboardingMux.HandleFunc("/", directHTTPForbidden) - - onboardingWithMiddleware := middleware.Chain( - middleware.PanicRecovery(log), - middleware.RequestLogger(log), // Intentionally omit trusted proxy nets so direct onboarding logs use RemoteAddr only. - middleware.SecurityHeaders, - )(onboardingMux) - - return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - remoteIP := httphelper.ExtractRemoteIP(r.RemoteAddr) - if httphelper.IsTrustedOrLocal(remoteIP, proxyNets) { - proxyChain.ServeHTTP(w, r) - return - } - onboardingWithMiddleware.ServeHTTP(w, r) - }) -} - -// canonicalHostsEqual compares two hosts after normalising both: stripping -// port, trimming spaces, lowercasing, and removing trailing dot. -func canonicalHostsEqual(host, expected string) bool { - host = strings.TrimSpace(host) - if h, _, err := net.SplitHostPort(host); err == nil { - host = h - } - host = strings.TrimSuffix(strings.ToLower(host), ".") - expected = strings.TrimSuffix(strings.ToLower(strings.TrimSpace(expected)), ".") - return host == expected -} - -// gordonDomainOnboardingGate returns a handler that serves onboarding routes -// only when the request host matches gordonDomain. For mismatched hosts it -// delegates to proxyHandler. gordonDomain must already be canonicalised -// (trimmed, lowered, trailing dot removed). -func gordonDomainOnboardingGate(gordonDomain string, onboardingMux, proxyHandler http.Handler) http.Handler { - return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - if gordonDomain != "" && canonicalHostsEqual(r.Host, gordonDomain) { - onboardingMux.ServeHTTP(w, r) - return - } - proxyHandler.ServeHTTP(w, r) - }) -} - -// directHTTPForbidden responds with 403 for non-onboarding HTTP paths. -// HEAD requests get an empty body per HTTP semantics. -func directHTTPForbidden(w http.ResponseWriter, r *http.Request) { - w.Header().Set("Content-Type", "text/plain; charset=utf-8") - w.WriteHeader(http.StatusForbidden) - if r.Method != http.MethodHead { - _, _ = w.Write([]byte("Only certificate onboarding is available over HTTP.\n")) - } -} - -func buildRegistryHandlerWithMiddleware( - svc *services, - cfg Config, - trustedNets []*net.IPNet, - registryHandler http.Handler, - log zerowrap.Logger, -) (http.Handler, func(http.Handler) http.Handler, func(http.Handler) http.Handler) { - registryMiddlewares := []func(http.Handler) http.Handler{ - middleware.PanicRecovery(log), - middleware.RequestLogger(log, trustedNets), - middleware.SecurityHeaders, - } - - cidrAllowlistMiddleware := buildRegistryCIDRAllowlistMiddleware(cfg, trustedNets, log) - if cidrAllowlistMiddleware != nil { - registryMiddlewares = append(registryMiddlewares, cidrAllowlistMiddleware) - } - - rateLimitMiddleware := buildRegistryRateLimitMiddleware(cfg, log) - registryMiddlewares = append(registryMiddlewares, rateLimitMiddleware) - - appendRegistryAuthMiddleware(®istryMiddlewares, svc, cfg, trustedNets, log) - - registryWithOtel := otelhttp.NewHandler( - middleware.Chain(registryMiddlewares...)(registryHandler), - "gordon.registry", - ) - return registryWithOtel, cidrAllowlistMiddleware, rateLimitMiddleware -} - -// parseCIDRAllowlist parses a list of IPs/CIDRs, logs warnings for invalid entries, -// and returns the parsed nets. label is used in log messages (e.g. "registry_allowed_ips"). -func parseCIDRAllowlist(ips []string, label string, log zerowrap.Logger) ([]*net.IPNet, bool) { - if len(ips) == 0 { - return nil, false - } - - allowedNets := httphelper.ParseTrustedProxies(ips) - if len(allowedNets) != len(ips) { - for _, entry := range ips { - if nets := httphelper.ParseTrustedProxies([]string{entry}); len(nets) == 0 { - log.Warn().Str("entry", entry).Msgf("ignoring invalid %s entry", label) - } - } - } - - if len(allowedNets) == 0 { - log.Error(). - Strs(label, ips). - Msgf("%s is set but no valid entries were parsed; will deny all traffic (fail-closed)", label) - return nil, true // allInvalid - } - - return allowedNets, false -} - -// denyAllHandler returns a middleware that rejects every request with 403 Forbidden. -func denyAllHandler(label string, trustedNets []*net.IPNet, log zerowrap.Logger) func(http.Handler) http.Handler { - return func(next http.Handler) http.Handler { - return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - log.Warn(). - Str(zerowrap.FieldClientIP, middleware.GetClientIP(r, trustedNets)). - Msgf("access denied due to invalid %s configuration", label) - w.Header().Set("Content-Type", "application/json") - w.WriteHeader(http.StatusForbidden) - _ = json.NewEncoder(w).Encode(dto.ErrorResponse{Error: "Forbidden"}) - }) - } -} - -func buildRegistryCIDRAllowlistMiddleware(cfg Config, trustedNets []*net.IPNet, log zerowrap.Logger) func(http.Handler) http.Handler { - allowedNets, allInvalid := parseCIDRAllowlist(cfg.Server.RegistryAllowedIPs, "registry_allowed_ips", log) - if allInvalid { - return denyAllHandler("registry_allowed_ips", trustedNets, log) - } - if allowedNets == nil { - return nil - } - return middleware.RegistryCIDRAllowlist(allowedNets, trustedNets, log) -} - -func buildProxyCIDRAllowlistMiddleware(cfg Config, trustedNets []*net.IPNet, log zerowrap.Logger) ([]*net.IPNet, func(http.Handler) http.Handler) { - allowedNets, allInvalid := parseCIDRAllowlist(cfg.Server.ProxyAllowedIPs, "proxy_allowed_ips", log) - if allInvalid { - return nil, denyAllHandler("proxy_allowed_ips", trustedNets, log) - } - if allowedNets == nil { - return nil, nil - } - - log.Info(). - Strs("proxy_allowed_ips", cfg.Server.ProxyAllowedIPs). - Msg("proxy origin IP allowlist enabled") - - return allowedNets, middleware.ProxyCIDRAllowlist(allowedNets, log) -} - -func buildRegistryRateLimitMiddleware(cfg Config, log zerowrap.Logger) func(http.Handler) http.Handler { - if cfg.API.RateLimit.Enabled { - globalLimiter := ratelimit.NewMemoryStore(cfg.API.RateLimit.GlobalRPS, cfg.API.RateLimit.Burst, log) - ipLimiter := ratelimit.NewMemoryStore(cfg.API.RateLimit.PerIPRPS, cfg.API.RateLimit.Burst, log) - return registry.RateLimitMiddleware( - globalLimiter, - ipLimiter, - cfg.API.RateLimit.TrustedProxies, - log, - ) - } - - return registry.RateLimitMiddleware(nil, nil, nil, log) -} - -func appendRegistryAuthMiddleware(registryMiddlewares *[]func(http.Handler) http.Handler, svc *services, cfg Config, trustedNets []*net.IPNet, log zerowrap.Logger) { - if svc.authSvc != nil { - internalAuth := middleware.InternalRegistryAuth{ - Username: svc.internalRegUser, - Password: svc.internalRegPass, - } - *registryMiddlewares = append(*registryMiddlewares, middleware.RegistryAuthV2(svc.authSvc, internalAuth, trustedNets, log)) - return - } - - if cfg.Auth.Enabled { - log.Error().Msg("authentication service unavailable; registry requests will be denied") - *registryMiddlewares = append(*registryMiddlewares, func(next http.Handler) http.Handler { - return http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { - w.Header().Set("Content-Type", "application/json") - w.WriteHeader(http.StatusServiceUnavailable) - _ = json.NewEncoder(w).Encode(dto.ErrorResponse{Error: "authentication service unavailable"}) - }) - }) - } -} - -func registerAuthRoutes( - registryMux *http.ServeMux, - svc *services, - trustedNets []*net.IPNet, - cidrAllowlistMiddleware func(http.Handler) http.Handler, - rateLimitMiddleware func(http.Handler) http.Handler, - cfg Config, - log zerowrap.Logger, -) { - if svc.authHandler == nil { - return - } - - // Auth endpoints always get rate limiting, even if global rate limiting is disabled. - // This prevents brute-force attacks against password/token endpoints. - authRateLimitMiddleware := rateLimitMiddleware - if !cfg.API.RateLimit.Enabled { - authGlobalLimiter := ratelimit.NewMemoryStore(50, 100, log) - authIPLimiter := ratelimit.NewMemoryStore(5, 10, log) - authRateLimitMiddleware = registry.RateLimitMiddleware(authGlobalLimiter, authIPLimiter, cfg.API.RateLimit.TrustedProxies, log) - } - - // Auth endpoints are NOT protected by auth - they're where clients authenticate - // but still need rate limiting to prevent brute force attacks. - authMiddlewares := []func(http.Handler) http.Handler{ - middleware.PanicRecovery(log), - middleware.RequestLogger(log, trustedNets), - middleware.SecurityHeaders, - } - if cidrAllowlistMiddleware != nil { - authMiddlewares = append(authMiddlewares, cidrAllowlistMiddleware) - } - authMiddlewares = append(authMiddlewares, authRateLimitMiddleware) - authWithMiddleware := otelhttp.NewHandler( - middleware.Chain(authMiddlewares...)(svc.authHandler), - "gordon.auth", - ) - registryMux.Handle("/auth/", authWithMiddleware) -} - -func wrapRegistryForLocalMode(registryWithMiddleware http.Handler, cfg Config, log zerowrap.Logger) http.Handler { - if !cfg.Auth.Enabled { - return loopbackOnly(registryWithMiddleware, log) - } - return registryWithMiddleware -} - -func registerAdminRoutes(registryMux *http.ServeMux, svc *services, cfg Config, trustedNets []*net.IPNet, log zerowrap.Logger) { - if svc.adminHandler == nil { - return - } - - if !cfg.Auth.Enabled { - log.Warn().Msg("auth disabled: admin API endpoints are not registered") - return - } - - adminMiddlewares := []func(http.Handler) http.Handler{ - middleware.PanicRecovery(log), - middleware.RequestLogger(log, trustedNets), - middleware.SecurityHeaders, - } - - if svc.authSvc != nil { - // Create rate limiters for admin API - uses same config as registry. - var globalLimiter, ipLimiter out.RateLimiter - if cfg.API.RateLimit.Enabled { - globalLimiter = ratelimit.NewMemoryStore(cfg.API.RateLimit.GlobalRPS, cfg.API.RateLimit.Burst, log) - ipLimiter = ratelimit.NewMemoryStore(cfg.API.RateLimit.PerIPRPS, cfg.API.RateLimit.Burst, log) - } - adminMiddlewares = append(adminMiddlewares, admin.AuthMiddleware(svc.authSvc, globalLimiter, ipLimiter, trustedNets, log)) - } else { - adminMiddlewares = append(adminMiddlewares, func(next http.Handler) http.Handler { - return http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { - w.Header().Set("Content-Type", "application/json") - w.WriteHeader(http.StatusServiceUnavailable) - _ = json.NewEncoder(w).Encode(dto.ErrorResponse{Error: "authentication service unavailable"}) - }) - }) - } - - adminWithMiddleware := otelhttp.NewHandler( - middleware.Chain(adminMiddlewares...)(svc.adminHandler), - "gordon.admin", - ) - registryMux.Handle("/admin/", loopbackOnly(adminWithMiddleware, log)) -} - -// runServers starts the HTTP servers and waits for shutdown. -// Signal handling notes: -// - SIGINT/SIGTERM: Triggers graceful shutdown via signal.NotifyContext -// - SIGUSR1: Triggers config reload without restart -// - SIGUSR2: Triggers manual deploy for a specific route -// The deferred signal.Stop calls ensure signal handlers are properly -// cleaned up before program exit, preventing signal handler leaks. -func runServers(ctx context.Context, v *viper.Viper, cfg Config, svc *services, reload reloadTrigger, cleanupHandlers func(), log zerowrap.Logger) error { - // Initialize access log writer. Kept here (not in Run) to keep Run's cyclomatic - // complexity within the project limit of 15. - accessWriterConcrete, err := initAccessLog(cfg, log) - if err != nil { - return err - } - if accessWriterConcrete != nil { - defer accessWriterConcrete.Close() - } - // Convert to interface only when non-nil to avoid the Go nil-interface pitfall - // where a typed nil pointer becomes a non-nil interface value. - var accessWriter out.AccessLogWriter - if accessWriterConcrete != nil { - accessWriter = accessWriterConcrete - } - ctx, cancel := signal.NotifyContext(ctx, os.Interrupt, syscall.SIGTERM) - defer cancel() - - // Set up SIGUSR1 for reload. - // Note: signal.Stop must be called (via defer) to release the channel - // and prevent signal handler leaks when the function returns. - reloadChan := make(chan os.Signal, 1) - signal.Notify(reloadChan, syscall.SIGUSR1) - defer signal.Stop(reloadChan) - - // Set up SIGUSR2 for manual deploy. - deployChan := make(chan os.Signal, 1) - signal.Notify(deployChan, syscall.SIGUSR2) - defer signal.Stop(deployChan) - - errChan := make(chan error, 3) - - registryHandler, httpProxyHandler, httpsProxyHandler := createHTTPHandlers(svc, cfg, log, accessWriter) - svc.httpProxyHandler = httpProxyHandler - svc.httpsProxyHandler = httpsProxyHandler - - registryAddr := net.JoinHostPort(cfg.Server.RegistryListenAddr, strconv.Itoa(cfg.Server.RegistryPort)) - registrySrv, registryReady := startServer(registryAddr, registryHandler, "registry", nil, errChan, log) - - // closeStarted shuts down any servers that were started before an error occurred, - // preventing leaked listeners during partial startup failures. - closeStarted := func(servers ...*http.Server) { - shutdownCtx, shutdownCancel := context.WithTimeout(context.Background(), 5*time.Second) - defer shutdownCancel() - for _, srv := range servers { - if srv != nil { - if err := srv.Shutdown(shutdownCtx); err != nil { - log.Error().Err(err).Msg("failed to shut down server during startup cleanup") - } - } - } - } - - proxySrv, proxyReady, tlsSrv, _, err := startProxyServers(cfg, httpProxyHandler, httpsProxyHandler, svc.pkiSvc, svc.publicTLSSvc, svc.trafficManager, log) - if err != nil { - closeStarted(registrySrv) - return err - } - svc.tlsHTTPEntryPoints = tlsMuxHTTPServerNames(cfg) - svc.smartHTTPEntryPoints = smartTCPHTTPServerNames(cfg) - - // Wait for the registry and HTTP proxy to bind before applying the traffic graph. - // This prevents auto-start races while keeping the TLS mux under one owner. - if err := waitForCoreProxyReadyAndApplyTraffic(ctx, cfg, svc, registryReady, proxyReady, errChan); err != nil { - closeStarted(registrySrv, proxySrv, tlsSrv) - return err - } - - logEvent := log.Info(). - Int("proxy_port", cfg.Server.Port). - Int("registry_port", cfg.Server.RegistryPort) - if cfg.Server.TLSPort != 0 { - logEvent = logEvent.Int("tls_port", cfg.Server.TLSPort) - } - logEvent.Msg("Gordon is running") - - startPublicTLSRuntimeWithWarning(ctx, svc.publicTLSRuntime, log) - - schedulerCleanup, err := startOptionalSchedulers(ctx, cfg, svc, log, v) - if err != nil { - shutdownTrafficManagerForStartupCleanup(svc.trafficManager, log) - closeStarted(registrySrv, proxySrv, tlsSrv) - return err - } - if schedulerCleanup != nil { - defer schedulerCleanup() - } - - // Recover configured routes after servers are listening (registry port is now bound). - syncAndRecoverConfiguredRoutes(ctx, svc.configSvc, svc.containerSvc, log) - - waitForShutdown(ctx, errChan, reloadChan, deployChan, reload, svc.eventBus, log) - cleanupHandlers() // Stop debounce timers before draining containers - gracefulShutdown(registrySrv, proxySrv, tlsSrv, svc.containerSvc, svc.proxySvc, svc.pkiSvc, svc.publicTLSSvc, svc.trafficManager, log) - return nil -} - -func startPublicTLSRuntimeWithWarning(ctx context.Context, svc publicTLSRuntime, log zerowrap.Logger) { - if err := startPublicTLSRuntime(ctx, svc, log); err != nil { - log.Warn().Err(err).Msg("initial public ACME reconcile failed, continuing with renewal loop") - } -} - -func waitForCoreProxyReadyAndApplyTraffic(ctx context.Context, cfg Config, svc *services, registryReady <-chan struct{}, proxyReady <-chan struct{}, errChan <-chan error) error { - if err := waitForServerReady(registryReady, errChan); err != nil { - return err - } - if err := waitForServerReady(proxyReady, errChan); err != nil { - return err - } - if err := applyTrafficRuntimeConfig(ctx, svc.trafficManager, cfg, svc.configSvc); err != nil { - return err - } - return reconcileStandaloneServices(ctx, svc.standaloneServiceSvc, cfg) -} - -func reconcileStandaloneServices(ctx context.Context, serviceSvc in.StandaloneServiceService, cfg Config) error { - if serviceSvc == nil { - return nil - } - standaloneServices, err := servicecfg.ToDomain(cfg.Services) - if err != nil { - return fmt.Errorf("convert standalone service config: %w", err) - } - if err := serviceSvc.Reconcile(ctx, standaloneServices); err != nil { - return fmt.Errorf("reconcile standalone services: %w", err) - } - return nil -} - -func shutdownTrafficManagerForStartupCleanup(manager *trafficadapter.Manager, log zerowrap.Logger) { - if manager == nil { - return - } - shutdownCtx, shutdownCancel := context.WithTimeout(context.Background(), 5*time.Second) - defer shutdownCancel() - if err := manager.Shutdown(shutdownCtx); err != nil { - log.Error().Err(err).Msg("failed to shut down traffic manager during startup cleanup") - } -} - -func waitForServerReady(ready <-chan struct{}, errChan <-chan error) error { - if ready == nil { - return nil - } - select { - case <-ready: - return nil - case err := <-errChan: - return err - } -} - -func startOptionalSchedulers(ctx context.Context, cfg Config, svc *services, log zerowrap.Logger, v *viper.Viper) (func(), error) { - schedulers := make([]*cronSvc.Scheduler, 0, 3) - - backupScheduler, err := startBackupScheduler(ctx, cfg, svc, log) - if err != nil { - return nil, err - } - if backupScheduler != nil { - schedulers = append(schedulers, backupScheduler) - } - - volumeBackupScheduler, err := startVolumeBackupScheduler(ctx, cfg, svc, log) - if err != nil { - return nil, err - } - if volumeBackupScheduler != nil { - schedulers = append(schedulers, volumeBackupScheduler) - } - - imageScheduler, err := startImagePruneScheduler(ctx, cfg, svc, log, func() int { - return v.GetInt("images.prune.keep_last") - }) - if err != nil { - return nil, err - } - if imageScheduler != nil { - schedulers = append(schedulers, imageScheduler) - } - - if len(schedulers) == 0 { - return nil, nil - } - - return func() { - for i := len(schedulers) - 1; i >= 0; i-- { - schedulers[i].Stop() - } - }, nil -} - -func startBackupScheduler(ctx context.Context, cfg Config, svc *services, log zerowrap.Logger) (*cronSvc.Scheduler, error) { - dbCfg := databaseBackupSettings(cfg) - if !dbCfg.Enabled || svc == nil || svc.backupSvc == nil { - return nil, nil - } - - preset, err := resolveBackupSchedule(dbCfg.Schedule) - if err != nil { - return nil, err - } - - scheduler := cronSvc.NewScheduler(log) - err = scheduler.Add( - "backup-scheduler", - "Backups", - domain.CronSchedule{Preset: preset}, - func(jobCtx context.Context) error { - if err := svc.backupSvc.RunForSchedule(jobCtx, preset); err != nil { - return err - } - log.Info(). - Str("schedule", string(preset)). - Msg("scheduled backup run complete") - return nil - }, - ) - if err != nil { - return nil, log.WrapErr(err, "failed to register backup schedule") - } - - scheduler.Start(ctx) - log.Info(). - Str("schedule", string(preset)). - Msg("backup scheduler enabled") - - return scheduler, nil -} - -func resolveBackupSchedule(raw string) (domain.BackupSchedule, error) { - return resolveSchedulePreset(raw, "backups.databases.schedule", domain.ScheduleDaily) -} - -func startVolumeBackupScheduler(ctx context.Context, cfg Config, svc *services, log zerowrap.Logger) (*cronSvc.Scheduler, error) { - if !cfg.Backups.Volumes.Enabled || svc == nil || svc.volumeBackupSvc == nil { - return nil, nil - } - - volumeCfg, err := validateVolumeBackupConfig(cfg) - if err != nil { - return nil, err - } - - scheduler := cronSvc.NewScheduler(log) - err = scheduler.Add( - "volume-backup-scheduler", - "Volume Backups", - domain.CronSchedule{Interval: volumeCfg.Interval}, - func(jobCtx context.Context) error { - if _, err := svc.volumeBackupSvc.RunVolumeBackups(jobCtx, "", ""); err != nil { - return err - } - log.Info(). - Dur("interval", volumeCfg.Interval). - Msg("scheduled volume backup run complete") - return nil - }, - ) - if err != nil { - return nil, log.WrapErr(err, "failed to register volume backup schedule") - } - - scheduler.Start(ctx) - log.Info(). - Dur("interval", volumeCfg.Interval). - Msg("volume backup scheduler enabled") - - return scheduler, nil -} - -func startImagePruneScheduler(ctx context.Context, cfg Config, svc *services, log zerowrap.Logger, keepLastGetter func() int) (*cronSvc.Scheduler, error) { - if !cfg.Images.Prune.Enabled || svc == nil || svc.imageSvc == nil { - return nil, nil - } - if keepLastGetter == nil { - keepLastGetter = func() int { return cfg.Images.Prune.KeepLast } - } - if keepLastGetter() < 0 { - return nil, fmt.Errorf("images.prune.keep_last must be >= 0") - } - - preset, err := resolveImagePruneSchedule(cfg.Images.Prune.Schedule) - if err != nil { - return nil, err - } - - scheduler := cronSvc.NewScheduler(log) - err = scheduler.Add( - "image-prune", - "Image prune", - domain.CronSchedule{Preset: preset}, - func(jobCtx context.Context) error { - keepLast := keepLastGetter() - if keepLast < 0 { - log.Warn(). - Int("configured_keep_last", keepLast). - Int("fallback_keep_last", domain.DefaultImagePruneKeepLast). - Msg("invalid images.prune.keep_last; using default") - keepLast = domain.DefaultImagePruneKeepLast - } - - report, err := svc.imageSvc.Prune(jobCtx, domain.ImagePruneOptions{ - KeepLast: keepLast, - PruneDangling: true, - PruneRegistry: true, - }) - if err != nil { - return err - } - - log.Info(). - Int("keep_last", keepLast). - Int("runtime_deleted", report.Runtime.DeletedCount). - Int64("runtime_reclaimed_bytes", report.Runtime.SpaceReclaimed). - Int("registry_tags_removed", report.Registry.TagsRemoved). - Int("registry_blobs_removed", report.Registry.BlobsRemoved). - Int("registry_uploads_removed", report.Registry.UploadsRemoved). - Int64("registry_upload_bytes_reclaimed", report.Registry.UploadSpaceReclaimed). - Msg("scheduled image prune complete") - return nil - }, - ) - if err != nil { - return nil, log.WrapErr(err, "failed to register image prune schedule") - } - - scheduler.Start(ctx) - log.Info(). - Str("schedule", string(preset)). - Int("keep_last", keepLastGetter()). - Msg("image prune scheduler enabled") - - return scheduler, nil -} - -func resolveImagePruneSchedule(raw string) (domain.BackupSchedule, error) { - return resolveSchedulePreset(raw, "images.prune.schedule", domain.ScheduleDaily) -} - -func resolveSchedulePreset(raw, name string, defaultVal domain.BackupSchedule) (domain.BackupSchedule, error) { - schedule := domain.BackupSchedule(strings.ToLower(strings.TrimSpace(raw))) - if schedule == "" { - schedule = defaultVal - } - - switch schedule { - case domain.ScheduleHourly, domain.ScheduleDaily, domain.ScheduleWeekly, domain.ScheduleMonthly: - return schedule, nil - default: - return "", fmt.Errorf("%s must be one of: hourly, daily, weekly, monthly", name) - } -} - -// waitForShutdown blocks on the event loop, handling server errors and -// Unix signals (reload, deploy, shutdown) until the context is cancelled. -func waitForShutdown(ctx context.Context, errChan <-chan error, reloadChan, deployChan <-chan os.Signal, reload reloadTrigger, eventBus out.EventBus, log zerowrap.Logger) { - for { - select { - case err := <-errChan: - log.Error().Err(err).Msg("server error") - return - case <-reloadChan: - log.Info().Msg("reload signal received (SIGUSR1)") - _ = reload.Trigger(ctx) - case <-deployChan: - log.Info().Msg("deploy signal received (SIGUSR2)") - domainName, err := readDeployRequest() - if err != nil { - log.Error().Err(err).Msg("failed to read deploy request") - continue - } - payload := &domain.ManualDeployPayload{Domain: domainName} - if err := eventBus.Publish(domain.EventManualDeploy, payload); err != nil { - log.Error().Err(err).Str("domain", domainName).Msg("failed to publish manual deploy event") - } - case <-ctx.Done(): - log.Info().Msg("shutdown signal received") - return - } - } -} - -// gracefulShutdown stops HTTP servers with a 30s timeout, then shuts down -// the container service and cleans up runtime files. -func gracefulShutdown(registrySrv, proxySrv, tlsSrv *http.Server, containerSvc *container.Service, proxySvc *proxy.Service, pkiSvc *pkiusecase.Service, publicTLS in.PublicTLSService, trafficManager *trafficadapter.Manager, log zerowrap.Logger) { - log.Info().Msg("shutting down Gordon...") - - shutdownCtx, shutdownCancel := context.WithTimeout(context.Background(), 30*time.Second) - defer shutdownCancel() - - // Phase 1: Stop ingress frontends (TLS, then proxy) — no new traffic accepted - for _, srv := range []*http.Server{tlsSrv, proxySrv} { - if srv == nil { - continue - } - if err := srv.Shutdown(shutdownCtx); err != nil { - log.Warn().Err(err).Str("addr", srv.Addr).Msg("server shutdown error") - } - } - - if trafficManager != nil { - if err := trafficManager.Shutdown(shutdownCtx); err != nil { - log.Warn().Err(err).Msg("traffic manager shutdown error") - } - } - - // Stop PKI maintenance goroutines - if pkiSvc != nil { - pkiSvc.Stop() - } - - // Stop public ACME TLS renewal loop - if publicTLS != nil { - if err := publicTLS.Stop(shutdownCtx); err != nil { - log.Warn().Err(err).Msg("public TLS stop error") - } - } - - // Phase 2: Drain in-flight registry push sessions before stopping the backend - if proxySvc != nil { - log.Info().Msg("draining in-flight registry requests...") - if drained := proxySvc.DrainRegistryInFlight(25 * time.Second); !drained { - log.Warn().Int64("in_flight", proxySvc.RegistryInFlight()).Msg("registry drain timed out; some in-flight pushes may be interrupted") - } - } - - // Phase 3: Stop the registry backend - if registrySrv != nil { - if err := registrySrv.Shutdown(shutdownCtx); err != nil { - log.Warn().Err(err).Str("addr", registrySrv.Addr).Msg("server shutdown error") - } - } - - containerSvc.StopMonitor() - - if err := containerSvc.Shutdown(shutdownCtx); err != nil { - log.Warn().Err(err).Msg("error during container shutdown") - } - - cleanupInternalCredentials() - log.Info().Msg("Gordon stopped") -} - -// startProxyServers sets up the HTTP proxy server and, when tls_port != 0, -// an HTTPS proxy server with on-demand TLS certificates from the internal CA. -// certificateSelector implements a multi-source TLS certificate lookup. -// Priority: static certs → public ACME TLS → local PKI (internal CA). -type certificateSelector struct { - staticCerts []staticTLSCertificate - publicTLS in.PublicTLSService - localPKI *pkiusecase.Service -} - -type staticTLSCertificate struct { - cert tls.Certificate - leaf *x509.Certificate -} - -// GetCertificate selects a TLS certificate based on the ClientHello SNI. -// -// Priority: -// 1. Static certs — exact SNI match (leaf VerifyHostname) -// 2. Public ACME TLS — if the host requires ACME coverage -// 3. Local PKI (internal CA) — fallback for all other hosts -// 4. nil, nil — if no source can serve the host -func (s *certificateSelector) GetCertificate(hello *tls.ClientHelloInfo) (*tls.Certificate, error) { - // 1. Static certs — exact match via leaf VerifyHostname. - if cert := matchingPreparedStaticCert(s.staticCerts, hello.ServerName); cert != nil { - return cert, nil - } - - // 2. Public ACME TLS. - if s.publicTLS != nil { - cert, err := s.publicTLS.GetCertificateForHost(hello.ServerName) - if err == nil && cert != nil { - return cert, nil - } - // nil, nil means this host is not an ACME-required route. Errors mean - // public ACME cannot currently serve this host. In both cases, fall - // through to local PKI instead of aborting the TLS handshake. - } - - // 3. Local PKI (internal CA). - if s.localPKI != nil { - return s.localPKI.GetCertificate(hello) - } - - return nil, nil -} - -func prepareStaticTLSCertificates(certs []tls.Certificate) []staticTLSCertificate { - prepared := make([]staticTLSCertificate, 0, len(certs)) - for _, cert := range certs { - if cert.Leaf == nil && len(cert.Certificate) > 0 { - leaf, err := x509.ParseCertificate(cert.Certificate[0]) - if err == nil { - cert.Leaf = leaf - } - } - prepared = append(prepared, staticTLSCertificate{cert: cert, leaf: cert.Leaf}) - } - return prepared -} - -func matchingPreparedStaticCert(certs []staticTLSCertificate, serverName string) *tls.Certificate { - if serverName == "" { - if len(certs) == 0 { - return nil - } - return &certs[0].cert - } - for i := range certs { - if certs[i].leaf == nil { - continue - } - if err := certs[i].leaf.VerifyHostname(serverName); err == nil { - return &certs[i].cert - } - } - return nil -} - -// matchingStaticCert returns a pointer to the first static certificate whose -// leaf verifies the given serverName. Returns nil if no match is found. -func matchingStaticCert(certs []tls.Certificate, serverName string) *tls.Certificate { - return matchingPreparedStaticCert(prepareStaticTLSCertificates(certs), serverName) -} - -func startProxyServers(cfg Config, httpHandler, httpsHandler http.Handler, pkiSvc *pkiusecase.Service, publicTLS in.PublicTLSService, trafficManager *trafficadapter.Manager, log zerowrap.Logger) (*http.Server, <-chan struct{}, *http.Server, <-chan struct{}, error) { - var httpSrv *http.Server - var httpReady <-chan struct{} - - needsTLS := hasTLSCapableEntrypoint(cfg) - if !hasSmartTCPEntrypoint(cfg) && !needsTLS { - return httpSrv, httpReady, nil, nil, nil - } - cleanupHTTP := func(err error) (*http.Server, <-chan struct{}, *http.Server, <-chan struct{}, error) { - return httpSrv, httpReady, nil, nil, err - } - - var tlsConfig *tls.Config - if needsTLS { - var err error - tlsConfig, err = proxyTLSConfig(cfg, pkiSvc, publicTLS, log) - if err != nil { - return cleanupHTTP(err) - } - } - if trafficManager == nil { - return cleanupHTTP(fmt.Errorf("traffic manager is required when traffic entrypoints are enabled")) - } - registerTLSMuxHTTPServers(trafficManager, cfg, httpsHandler, tlsConfig, nil) - registerSmartTCPHTTPServers(trafficManager, cfg, httpHandler, httpsHandler, tlsConfig, nil) - - return httpSrv, httpReady, nil, nil, nil -} - -func hasSmartTCPEntrypoint(cfg Config) bool { - for _, entryPoint := range cfg.EntryPoints { - if entryPoint.Protocol == domain.EntryPointProtocolSmartTCP { - return true - } - } - return false -} - -func hasTLSCapableEntrypoint(cfg Config) bool { - for _, entryPoint := range cfg.EntryPoints { - switch entryPoint.Protocol { - case domain.EntryPointProtocolSmartTCP, domain.EntryPointProtocolTLSMux: - return true - } - } - return false -} - -func effectivePublicTLSPort(cfg Config) int { - return effectiveEntrypointPort(cfg, tlsCapableEntryPoint) -} - -func effectiveProxyTLSPort(cfg Config) int { - return effectivePublicTLSPort(cfg) -} - -func effectiveProxyHTTPPort(cfg Config) int { - return effectiveEntrypointPort(cfg, func(protocol domain.EntryPointProtocol) bool { - return protocol == domain.EntryPointProtocolSmartTCP - }) -} - -func effectiveEntrypointPort(cfg Config, match func(domain.EntryPointProtocol) bool) int { - if entryPoint, ok := cfg.EntryPoints[traffic.DefaultEdgeEntryPointName]; ok && match(entryPoint.Protocol) { - if port := portFromAddress(entryPoint.Address); port > 0 { - return port - } - } - - var candidatePort int - candidates := 0 - for name, entryPoint := range cfg.EntryPoints { - if name == traffic.DefaultEdgeEntryPointName || !match(entryPoint.Protocol) { - continue - } - port := portFromAddress(entryPoint.Address) - if port == 0 { - continue - } - candidatePort = port - candidates++ - } - if candidates == 1 { - return candidatePort - } - return 0 -} - -func tlsCapableEntryPoint(protocol domain.EntryPointProtocol) bool { - switch protocol { - case domain.EntryPointProtocolSmartTCP, domain.EntryPointProtocolTLSMux: - return true - default: - return false - } -} - -func effectiveHTTP01Port(cfg Config) int { - if hasSmartTCPHTTP01Entrypoint(cfg) { - return 80 - } - return 0 -} - -func validatePublicTLSReadiness(cfg Config) error { - mode, err := domain.ParseACMEChallengeMode(cfg.TLS.ACME.Challenge) - if err != nil { - return err - } - switch mode { - case domain.ACMEChallengeCloudflareDNS01, domain.ACMEChallengeAuto: - return nil - case domain.ACMEChallengeHTTP01: - return validateHTTP01ChallengeReadiness(cfg) - default: - return fmt.Errorf("%w: %q", domain.ErrACMEChallengeInvalid, mode) - } -} - -func validateEffectivePublicTLSReadiness(cfg Config, effective publictls.EffectiveChallenge) error { - if effective.Mode != domain.ACMEChallengeHTTP01 { - return nil - } - return validateHTTP01ChallengeReadiness(cfg) -} - -func validateHTTP01ChallengeReadiness(cfg Config) error { - if hasBoundHTTP01ChallengeListener(cfg) { - return nil - } - return fmt.Errorf("%w: http-01 requires an actually bound HTTP-01 challenge listener on external :80", domain.ErrACMEChallengeInvalid) -} - -func hasBoundHTTP01ChallengeListener(cfg Config) bool { - return hasSmartTCPHTTP01Entrypoint(cfg) -} - -func hasSmartTCPHTTP01Entrypoint(cfg Config) bool { - for _, entryPoint := range cfg.EntryPoints { - if entryPoint.Protocol == domain.EntryPointProtocolSmartTCP && portFromAddress(entryPoint.Address) == 80 { - return true - } - } - return false -} - -func portFromAddress(address string) int { - _, portText, err := net.SplitHostPort(address) - if err != nil { - return 0 - } - port, err := strconv.Atoi(portText) - if err != nil { - return 0 - } - return port -} - -func proxyTLSConfig(cfg Config, pkiSvc *pkiusecase.Service, publicTLS in.PublicTLSService, log zerowrap.Logger) (*tls.Config, error) { - var staticCerts []tls.Certificate - if cfg.Server.TLSCertFile != "" { - staticCert, err := tls.LoadX509KeyPair(cfg.Server.TLSCertFile, cfg.Server.TLSKeyFile) - if err != nil { - return nil, fmt.Errorf("load TLS keypair: %w", err) - } - staticCerts = []tls.Certificate{staticCert} - log.Info(). - Str("cert", cfg.Server.TLSCertFile). - Str("key", cfg.Server.TLSKeyFile). - Msg("loaded static TLS certificate (public ACME and internal CA handle remaining domains)") - } - selector := &certificateSelector{ - staticCerts: prepareStaticTLSCertificates(staticCerts), - publicTLS: publicTLS, - localPKI: pkiSvc, - } - return &tls.Config{ - MinVersion: tls.VersionTLS12, - GetCertificate: selector.GetCertificate, - NextProtos: []string{"h2", "http/1.1"}, - }, nil -} - -func registerSmartTCPHTTPServers(manager *trafficadapter.Manager, cfg Config, httpHandler, httpsHandler http.Handler, tlsConfig *tls.Config, previous map[string]struct{}) map[string]struct{} { - if manager == nil { - return previous - } - next := smartTCPHTTPServerNames(cfg) - for name := range previous { - if _, ok := next[name]; !ok { - manager.SetSmartTCPHTTPServer(name, nil, nil) - manager.SetSmartTCPTLSServer(name, nil, nil) - } - } - var httpProtos http.Protocols - httpProtos.SetHTTP1(true) - httpProtos.SetUnencryptedHTTP2(true) - for name := range next { - if httpHandler != nil { - manager.SetSmartTCPHTTPServer(name, httpHandler, &httpProtos) - } - if httpsHandler != nil && tlsConfig != nil { - manager.SetSmartTCPTLSServer(name, httpsHandler, tlsConfig) - } - } - return next -} - -func smartTCPHTTPServerNames(cfg Config) map[string]struct{} { - names := map[string]struct{}{} - for name, entryPoint := range cfg.EntryPoints { - if entryPoint.Protocol == domain.EntryPointProtocolSmartTCP { - names[name] = struct{}{} - } - } - return names -} - -func registerTLSMuxHTTPServers(manager *trafficadapter.Manager, cfg Config, httpsHandler http.Handler, tlsConfig *tls.Config, previous map[string]struct{}) map[string]struct{} { - if manager == nil { - return previous - } - next := tlsMuxHTTPServerNames(cfg) - for name := range previous { - if _, ok := next[name]; !ok { - manager.SetTLSHTTPServer(name, nil, nil) - } - } - if httpsHandler == nil || tlsConfig == nil { - return next - } - for name := range next { - manager.SetTLSHTTPServer(name, httpsHandler, tlsConfig) - } - return next -} - -func tlsMuxHTTPServerNames(cfg Config) map[string]struct{} { - names := map[string]struct{}{} - for name, entryPoint := range cfg.EntryPoints { - domainEntryPoint := domain.EntryPoint{Name: name, Protocol: entryPoint.Protocol} - if trafficManagerOwnsEntryPoint(domainEntryPoint) && domainEntryPoint.Protocol == domain.EntryPointProtocolTLSMux { - names[name] = struct{}{} - } - } - return names -} - -// startServer starts an HTTP server, returning the server instance and a channel -// that closes once the listening socket is bound. This lets callers wait for the -// port to be ready before taking actions that depend on it (e.g. auto-start -// pulling from the local registry). The returned *http.Server can be used for -// graceful shutdown. -func startServer(addr string, handler http.Handler, name string, protocols *http.Protocols, errChan chan<- error, log zerowrap.Logger) (*http.Server, <-chan struct{}) { - ready := make(chan struct{}) - - server := &http.Server{ - Addr: addr, - Handler: handler, - Protocols: protocols, - ReadHeaderTimeout: 10 * time.Second, - ReadTimeout: 5 * time.Minute, - WriteTimeout: 5 * time.Minute, - IdleTimeout: 120 * time.Second, - MaxHeaderBytes: 1 << 20, - } - - go func() { - log.Info().Str("address", addr).Msgf("%s server starting", name) - - ln, err := net.Listen("tcp", addr) - if err != nil { - errChan <- fmt.Errorf("%s server error: %w", name, err) - return - } - close(ready) // signal: port is bound and accepting connections - - if err := server.Serve(ln); err != nil && err != http.ErrServerClosed { - errChan <- fmt.Errorf("%s server error: %w", name, err) - } - }() - - return server, ready -} - -// SendReloadSignal sends SIGUSR1 to the running Gordon process. -func SendReloadSignal() error { - process, _, err := findRunningProcess() - if err != nil { - return err - } - - if err := process.Signal(syscall.SIGUSR1); err != nil { - return fmt.Errorf("failed to send reload signal: %w", err) - } - - return nil -} - -// getDeployRequestFile returns the path to the deploy request file. -func getDeployRequestFile() string { - return filepath.Join(os.TempDir(), "gordon-deploy-request") -} - -// writeDeployRequestFile creates the deploy request file exclusively with retry. -// This prevents race conditions when multiple deploy commands run simultaneously. -func writeDeployRequestFile(path string, data []byte, timeout time.Duration) error { - deadline := time.Now().Add(timeout) - - for { - f, err := os.OpenFile(path, os.O_WRONLY|os.O_CREATE|os.O_EXCL, 0600) - if err == nil { - _, writeErr := f.Write(data) - f.Close() - if writeErr != nil { - _ = os.Remove(path) - return writeErr - } - return nil - } - - if os.IsExist(err) { - if time.Now().After(deadline) { - return fmt.Errorf("deploy request file still present after timeout; another deploy may be in progress") - } - time.Sleep(100 * time.Millisecond) - continue - } - - return err - } -} - -// SendDeploySignal triggers a manual deploy for a specific route via SIGUSR2. -// Returns the domain name on success for the caller to display. -func SendDeploySignal(domain string) (string, error) { - deployFile := getDeployRequestFile() - if err := writeDeployRequestFile(deployFile, []byte(domain), 5*time.Second); err != nil { - return "", fmt.Errorf("failed to write deploy request: %w", err) - } - - // Find PID and send SIGUSR2 - process, _, err := findRunningProcess() - if err != nil { - _ = os.Remove(deployFile) - return "", err - } - - if err := process.Signal(syscall.SIGUSR2); err != nil { - _ = os.Remove(deployFile) - return "", fmt.Errorf("failed to send deploy signal: %w", err) - } - - return domain, nil -} - -// readDeployRequest reads and removes the deploy request file atomically. -// Returns empty string if file doesn't exist (may have been consumed by another handler). -func readDeployRequest() (string, error) { - deployFile := getDeployRequestFile() - - // Rename to a temp file first to make the read-and-delete atomic - tmpFile := deployFile + ".processing" - if err := os.Rename(deployFile, tmpFile); err != nil { - if os.IsNotExist(err) { - return "", fmt.Errorf("deploy request file not found (may have been processed already)") - } - return "", fmt.Errorf("failed to acquire deploy request: %w", err) - } - - data, err := os.ReadFile(tmpFile) - _ = os.Remove(tmpFile) // Always clean up - if err != nil { - return "", fmt.Errorf("failed to read deploy request: %w", err) - } - - return string(data), nil -} - -// createPidFile creates a PID file for the Gordon process. -// SECURITY: Prefers secure locations (XDG_RUNTIME_DIR, ~/.gordon/run) over /tmp -// to prevent symlink attacks and unauthorized access. -func createPidFile(log zerowrap.Logger) string { - pid := os.Getpid() - - // SECURITY: Prioritize secure locations over /tmp - var locations []string - - // Try secure runtime directory first - if runtimeDir, err := getSecureRuntimeDir(); err == nil { - locations = append(locations, filepath.Join(runtimeDir, "gordon.pid")) - } - - // Fall back to home directory - if homeDir, err := os.UserHomeDir(); err == nil { - locations = append(locations, filepath.Join(homeDir, ".gordon", "gordon.pid")) - } - - // Last resort: /tmp (least secure due to world-writable) - locations = append(locations, filepath.Join(os.TempDir(), "gordon.pid")) - - for _, location := range locations { - // Ensure parent directory exists with secure permissions - if err := os.MkdirAll(filepath.Dir(location), 0700); err != nil { - continue - } - if err := os.WriteFile(location, fmt.Appendf(nil, "%d", pid), 0600); err == nil { - log.Debug().Str("pid_file", location).Int("pid", pid).Msg("created PID file") - return location - } - } - - log.Warn().Int("pid", pid).Msg("failed to create PID file in any location") - return "" -} - -// removePidFile removes the PID file. -func removePidFile(pidFile string, log zerowrap.Logger) { - if err := os.Remove(pidFile); err != nil { - log.Warn().Err(err).Str("pid_file", pidFile).Msg("failed to remove PID file") - } else { - log.Debug().Str("pid_file", pidFile).Msg("removed PID file") - } -} - -func pidFileLocations() []string { - var locations []string - seen := make(map[string]struct{}) - add := func(path string) { - if path == "" { - return - } - if _, ok := seen[path]; ok { - return - } - seen[path] = struct{}{} - locations = append(locations, path) - } - - // Check secure runtime directory first - if runtimeDir, err := getSecureRuntimeDir(); err == nil { - add(filepath.Join(runtimeDir, "gordon.pid")) - } - - // Also check canonical /run/user/ runtime path. This handles cases where - // Gordon started under systemd user services with runtime dir available, but - // CLI invocations (e.g. non-interactive SSH) don't have XDG_RUNTIME_DIR set. - runtimeByUID := filepath.Join("/run/user", strconv.Itoa(os.Getuid()), "gordon", "gordon.pid") - add(runtimeByUID) - - // Check explicit XDG_RUNTIME_DIR if present in this process env. - if runtimeDir := os.Getenv("XDG_RUNTIME_DIR"); runtimeDir != "" { - add(filepath.Join(runtimeDir, "gordon.pid")) - } - - // Check home directory - if homeDir, err := os.UserHomeDir(); err == nil { - add(filepath.Join(homeDir, ".gordon", "gordon.pid")) - // Legacy location for backward compatibility - add(filepath.Join(homeDir, ".gordon.pid")) - } - - // Legacy /tmp locations for backward compatibility - add(filepath.Join(os.TempDir(), "gordon.pid")) - add("/tmp/gordon.pid") - - return locations -} - -// findRunningPidFile returns the first PID file whose PID belongs to a live process. -// Stale/invalid PID files are ignored and removed when possible. -func findRunningPidFile() (string, int, error) { - return findRunningPidFileInLocations(pidFileLocations()) -} - -func findRunningPidFileInLocations(locations []string) (string, int, error) { - foundAny := false - - for _, location := range locations { - pidBytes, err := os.ReadFile(location) - if err != nil { - continue - } - - foundAny = true - - var pid int - if _, err := fmt.Sscanf(string(pidBytes), "%d", &pid); err != nil || pid <= 0 { - _ = os.Remove(location) - continue - } - - if isProcessAlive(pid) { - return location, pid, nil - } - - _ = os.Remove(location) - } - - if foundAny { - return "", 0, fmt.Errorf("found stale gordon PID file(s), is Gordon running?") - } - - return "", 0, fmt.Errorf("gordon PID file not found, is Gordon running?") -} - -func findRunningProcess() (*os.Process, int, error) { - _, pid, err := findRunningPidFile() - if err != nil { - return nil, 0, err - } - - process, err := os.FindProcess(pid) - if err != nil { - return nil, 0, fmt.Errorf("failed to find process: %w", err) - } - - return process, pid, nil -} - -func isProcessAlive(pid int) bool { - process, err := os.FindProcess(pid) - if err != nil { - return false - } - - err = process.Signal(syscall.Signal(0)) - if err == nil { - return true - } - - return errors.Is(err, syscall.EPERM) -} - -// loadConfig loads configuration from file and sets defaults. -func loadConfig(v *viper.Viper, configPath string) error { - v.SetDefault("server.registry_port", 5000) - v.SetDefault("server.legacy_registry_domains", []string{}) - v.SetDefault("server.tls_cert_file", "") - v.SetDefault("server.tls_key_file", "") - v.SetDefault("tls.acme.enabled", false) - v.SetDefault("tls.acme.email", "") - v.SetDefault("tls.acme.challenge", "auto") - v.SetDefault("tls.acme.obtain_batch_size", 1) - v.SetDefault("dns.resolvers", publictls.DefaultDNSResolvers) - v.SetDefault("dns.propagation_timeout", "5m") - v.SetDefault("dns.polling_interval", "5s") - v.SetDefault("server.force_https_redirect", false) - v.SetDefault("server.data_dir", DefaultDataDir()) - v.SetDefault("server.runtime", "auto") - v.SetDefault("logging.level", "info") - v.SetDefault("logging.format", "console") - v.SetDefault("logging.file.enabled", false) - v.SetDefault("logging.file.max_size", 100) - v.SetDefault("logging.file.max_backups", 3) - v.SetDefault("logging.file.max_age", 28) - v.SetDefault("logging.container_logs.enabled", true) - v.SetDefault("logging.container_logs.dir", "") - v.SetDefault("logging.container_logs.max_size", 100) - v.SetDefault("logging.container_logs.max_backups", 3) - v.SetDefault("logging.container_logs.max_age", 28) - v.SetDefault("logging.access_log.enabled", false) - v.SetDefault("logging.access_log.format", "json") - v.SetDefault("logging.access_log.output", "stdout") - v.SetDefault("logging.access_log.file_path", "") - v.SetDefault("logging.access_log.max_size", 100) - v.SetDefault("logging.access_log.max_backups", 3) - v.SetDefault("logging.access_log.max_age", 28) - v.SetDefault("logging.access_log.exclude_health_checks", true) - v.SetDefault("logging.access_log.syslog_identifier", "gordon-access") - v.SetDefault("env.dir", "") // defaults to {data_dir}/env when empty - v.SetDefault("auth.enabled", true) - // Note: auth.type defaults to "token" (the only supported mode) - v.SetDefault("auth.secrets_backend", "") - v.SetDefault("auth.token_expiry", "720h") - v.SetDefault("api.rate_limit.enabled", true) - v.SetDefault("api.rate_limit.global_rps", 500) - v.SetDefault("api.rate_limit.per_ip_rps", 50) - v.SetDefault("api.rate_limit.burst", 100) - v.SetDefault("auto_route.enabled", false) - v.SetDefault("network_isolation.enabled", true) - v.SetDefault("network_isolation.network_prefix", "gordon") - v.SetDefault("network_isolation.internal", false) - v.SetDefault("volumes.auto_create", true) - v.SetDefault("volumes.prefix", "gordon") - v.SetDefault("volumes.preserve", true) - v.SetDefault("deploy.pull_policy", container.PullPolicyIfTagChanged) - v.SetDefault("backups.databases.enabled", false) - v.SetDefault("backups.databases.schedule", string(domain.ScheduleDaily)) - v.SetDefault("backups.databases.storage_dir", "") - v.SetDefault("backups.databases.retention.hourly", 0) - v.SetDefault("backups.databases.retention.daily", 0) - v.SetDefault("backups.databases.retention.weekly", 0) - v.SetDefault("backups.databases.retention.monthly", 0) - v.SetDefault("backups.volumes.enabled", false) - v.SetDefault("backups.volumes.interval", "24h") - v.SetDefault("backups.volumes.compression", string(domain.VolumeBackupCompressionGzip)) - v.SetDefault("backups.volumes.timeout", "2h") - v.SetDefault("backups.volumes.max_concurrency", 2) - v.SetDefault("backups.volumes.helper_image", "alpine:3.20") - v.SetDefault("backups.volumes.s3.bucket", "") - v.SetDefault("backups.volumes.s3.region", "") - v.SetDefault("backups.volumes.s3.prefix", "") - v.SetDefault("backups.volumes.s3.endpoint", "") - v.SetDefault("backups.volumes.s3.path_style", false) - v.SetDefault("backups.volumes.s3.sse_algorithm", "") - v.SetDefault("backups.volumes.s3.sse_kms_key_id", "") - v.SetDefault("backups.volumes.retention.keep", 14) - v.SetDefault("images.allowed_registries", []string{}) - v.SetDefault("images.require_digest", false) - v.SetDefault("images.prune.enabled", false) - v.SetDefault("images.prune.schedule", string(domain.ScheduleDaily)) - v.SetDefault("images.prune.keep_last", domain.DefaultImagePruneKeepLast) - v.SetDefault("containers.security_profile", "compat") - v.SetDefault("telemetry.enabled", false) - v.SetDefault("telemetry.endpoint", "") - v.SetDefault("telemetry.auth_token", "") - v.SetDefault("telemetry.traces", true) - v.SetDefault("telemetry.metrics", true) - v.SetDefault("telemetry.logs", true) - v.SetDefault("telemetry.trace_sample_rate", 1.0) - - v.SetDefault("server.max_concurrent_connections", -1) // -1 = use default (10000), 0 = no limit - v.SetDefault("server.registry_allowed_ips", []string{}) - v.SetDefault("server.proxy_allowed_ips", []string{}) - v.SetDefault("server.registry_listen_address", "") - v.SetDefault("deploy.readiness_delay", "5s") - v.SetDefault("deploy.readiness_mode", "auto") - v.SetDefault("deploy.health_timeout", "90s") - v.SetDefault("deploy.stabilization_delay", "2s") - v.SetDefault("deploy.tcp_probe_timeout", "30s") - v.SetDefault("deploy.http_probe_timeout", "60s") - v.SetDefault("deploy.attachment_readiness_timeout", "30s") - v.SetDefault("deploy.drain_mode", "auto") - v.SetDefault("deploy.drain_timeout", "30s") - - ConfigureViper(v, configPath) - - if err := v.ReadInConfig(); err != nil { - if _, ok := err.(viper.ConfigFileNotFoundError); !ok { - return fmt.Errorf("failed to read config file: %w", err) - } - } - - v.SetEnvPrefix("GORDON") - v.SetEnvKeyReplacer(strings.NewReplacer(".", "_")) - v.AutomaticEnv() - - return nil } diff --git a/internal/app/run_app_lifecycle_test.go b/internal/app/run_app_lifecycle_test.go new file mode 100644 index 000000000..3edd109be --- /dev/null +++ b/internal/app/run_app_lifecycle_test.go @@ -0,0 +1,172 @@ +package app + +import ( + "context" + "errors" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/container" + "github.com/bnema/gordon/internal/usecase/deployment" +) + +// daemonCtxKey tags the daemon/supervisor lifecycle context so a test can +// prove the execution context descends from it and not from a request. +type daemonCtxKey struct{} + +// fakeAppDeployEngine implements the deployment-engine port used by +// newAppDaemonService. ExecuteDeploy reports the context it received and +// blocks until that context ends, so cancellation is observable. +type fakeAppDeployEngine struct { + executeStarted chan context.Context +} + +func (f *fakeAppDeployEngine) StartDeploy(context.Context, deployment.DeployInput) (*deployment.StartDeployResult, error) { + return &deployment.StartDeployResult{ + Owned: true, + Claim: deployment.DeployClaim{ + App: "blog", Op: "op-1", Service: "web", Revision: "rev-1", + Journal: domain.AppOperation{Op: "op-1", Kind: "deploy", App: "blog", InputRevision: "rev-1"}, + }, + }, nil +} + +func (f *fakeAppDeployEngine) ExecuteDeploy(ctx context.Context, _ deployment.DeployClaim) (*deployment.DeployResult, error) { + f.executeStarted <- ctx + <-ctx.Done() + return nil, ctx.Err() +} + +func (f *fakeAppDeployEngine) AbandonDeploy(context.Context, deployment.DeployClaim) error { + return nil +} + +func (f *fakeAppDeployEngine) Stop(context.Context, string, string) (*deployment.LifecycleResult, error) { + return nil, nil +} + +func (f *fakeAppDeployEngine) Start(context.Context, string, string) (*deployment.LifecycleResult, error) { + return nil, nil +} + +func (f *fakeAppDeployEngine) Restart(context.Context, string, string, string) (*deployment.LifecycleResult, error) { + return nil, nil +} + +func (f *fakeAppDeployEngine) Remove(context.Context, string, string) (*deployment.LifecycleResult, error) { + return nil, nil +} + +// TestNewAppDaemonService_DaemonCancellationReachesInFlightDeploy proves the +// bootstrap wiring used by initApps: the app service derives background +// executions from the daemon/supervisor context, that context — not the +// request context — owns the in-flight replacement, and daemon cancellation +// reaches it. +func TestNewAppDaemonService_DaemonCancellationReachesInFlightDeploy(t *testing.T) { + store := outmocks.NewMockAppState(t) + store.EXPECT().LoadOperation(mock.Anything, "blog", "op-1"). + Return(domain.AppOperation{Op: "op-1", Kind: "deploy", App: "blog", InputRevision: "rev-1"}, nil).Maybe() + + engine := &fakeAppDeployEngine{executeStarted: make(chan context.Context, 1)} + daemonCtx, cancelDaemon := context.WithCancel(context.WithValue(context.Background(), daemonCtxKey{}, "daemon")) + svc := newAppDaemonService(daemonCtx, store, engine, nil, zerowrap.Default()) + t.Cleanup(func() { require.NoError(t, svc.Shutdown(context.Background())) }) + + requestCtx, cancelRequest := context.WithCancel(context.Background()) + defer cancelRequest() + + deployDone := make(chan struct{}) + go func() { + defer close(deployDone) + _, _ = svc.Deploy(requestCtx, "blog", "rev-1", "web", false, "key-1") + }() + + var execCtx context.Context + select { + case execCtx = <-engine.executeStarted: + case <-time.After(3 * time.Second): + t.Fatal("background deploy execution never started") + } + assert.Equal(t, "daemon", execCtx.Value(daemonCtxKey{}), "execution must run on the daemon context") + + // The request ending is unrelated to the daemon-owned execution. + cancelRequest() + select { + case <-execCtx.Done(): + t.Fatal("request cancellation aborted the daemon-owned execution") + case <-time.After(50 * time.Millisecond): + } + + cancelDaemon() + select { + case <-execCtx.Done(): + case <-time.After(3 * time.Second): + t.Fatal("daemon context cancellation did not reach the in-flight execution") + } + <-deployDone +} + +// orderRecordingAdmin records when app administration quiescence runs and +// on which context, so shutdown ordering is observable. +type orderRecordingAdmin struct { + order *[]string + deadline bool + err error +} + +func (a *orderRecordingAdmin) Shutdown(ctx context.Context) error { + *a.order = append(*a.order, "app-shutdown") + _, a.deadline = ctx.Deadline() + return a.err +} + +// TestGracefulShutdown_FailsClosedWhenAppAdministrationDoesNotQuiesce proves +// the fail-closed shutdown ordering: when app administration cannot unwind on +// the bounded shutdown context, the remaining state and runtime teardown is +// skipped and the error is returned for the process to exit non-zero, so an +// in-flight execution can never write to a closed store. +func TestGracefulShutdown_FailsClosedWhenAppAdministrationDoesNotQuiesce(t *testing.T) { + // Keep internal-credential cleanup inside the test temp dir. + t.Setenv("XDG_RUNTIME_DIR", t.TempDir()) + + order := []string{} + admin := &orderRecordingAdmin{order: &order, err: errors.New("in-flight execution did not unwind")} + + store := outmocks.NewMockAppState(t) + + containerSvc := container.NewService(nil, nil, nil, container.Config{}) + err := gracefulShutdown(nil, nil, nil, containerSvc, nil, nil, nil, nil, nil, admin, store, zerowrap.Default()) + + require.Error(t, err) + assert.Contains(t, err.Error(), "quiescence") + require.Equal(t, []string{"app-shutdown"}, order, "state teardown must not run under an unfinished execution") + store.AssertNotCalled(t, "Close") + assert.True(t, admin.deadline, "app administration shutdown must run on the bounded shutdown context") +} + +// TestGracefulShutdown_QuiescesAppAdministrationBeforeClosingState keeps the +// normal ordering: app administration is cancelled and joined on the bounded +// shutdown context before the app state store closes. +func TestGracefulShutdown_QuiescesAppAdministrationBeforeClosingState(t *testing.T) { + // Keep internal-credential cleanup inside the test temp dir. + t.Setenv("XDG_RUNTIME_DIR", t.TempDir()) + + order := []string{} + admin := &orderRecordingAdmin{order: &order} + + store := outmocks.NewMockAppState(t) + store.EXPECT().Close().Run(func() { order = append(order, "state-close") }).Return(nil) + + containerSvc := container.NewService(nil, nil, nil, container.Config{}) + require.NoError(t, gracefulShutdown(nil, nil, nil, containerSvc, nil, nil, nil, nil, nil, admin, store, zerowrap.Default())) + + require.Equal(t, []string{"app-shutdown", "state-close"}, order) + assert.True(t, admin.deadline, "app administration shutdown must run on the bounded shutdown context") +} diff --git a/internal/app/run_entrypoint_migration_test.go b/internal/app/run_entrypoint_migration_test.go index d39da3c91..2e1b5f3bc 100644 --- a/internal/app/run_entrypoint_migration_test.go +++ b/internal/app/run_entrypoint_migration_test.go @@ -26,3 +26,23 @@ func TestInitConfigAcceptsLegacyPortsWithEntrypoint(t *testing.T) { require.NoError(t, err) require.Contains(t, cfg.EntryPoints, "edge") } + +func TestInitConfigRejectsRetiredAppKeys(t *testing.T) { + configPath := filepath.Join(t.TempDir(), "gordon.toml") + require.NoError(t, os.WriteFile(configPath, []byte("[routes]\n[routes.blog]\nimage = \"reg/blog:latest\"\n"), 0o600)) + + _, _, err := initConfig(configPath) + + require.ErrorContains(t, err, "config-retired") + require.ErrorContains(t, err, `"routes"`) +} + +func TestInitConfigRejectsRetiredEnvSection(t *testing.T) { + configPath := filepath.Join(t.TempDir(), "gordon.toml") + require.NoError(t, os.WriteFile(configPath, []byte("[env]\ndir = \"/tmp/env\"\n"), 0o600)) + + _, _, err := initConfig(configPath) + + require.ErrorContains(t, err, "config-retired") + require.ErrorContains(t, err, `"env"`) +} diff --git a/internal/app/run_example_config_test.go b/internal/app/run_example_config_test.go new file mode 100644 index 000000000..d93e1d660 --- /dev/null +++ b/internal/app/run_example_config_test.go @@ -0,0 +1,21 @@ +package app + +import ( + "path/filepath" + "testing" + + "github.com/stretchr/testify/require" +) + +// TestInitConfigAcceptsShippedExampleConfig guards the example shipped in +// the image: it must load cleanly, so no retired key may reappear as an +// active table. Viper reports a table that only holds comments as present +// in the config, so an empty [routes]-style header fails startup. +func TestInitConfigAcceptsShippedExampleConfig(t *testing.T) { + path, err := filepath.Abs(filepath.Join("..", "..", "gordon.toml.example")) + require.NoError(t, err) + + _, _, err = initConfig(path) + + require.NoError(t, err) +} diff --git a/internal/app/run_http_handlers_test.go b/internal/app/run_http_handlers_test.go index 155d73e34..b41489deb 100644 --- a/internal/app/run_http_handlers_test.go +++ b/internal/app/run_http_handlers_test.go @@ -14,7 +14,6 @@ import ( "github.com/bnema/zerowrap" "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/mock" "github.com/stretchr/testify/require" adminhttp "github.com/bnema/gordon/internal/adapters/in/http/admin" @@ -30,10 +29,7 @@ func newNotFoundProxyService(t *testing.T) *proxyusecase.Service { t.Helper() configSvc := inmocks.NewMockConfigService(t) configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{}).Maybe() - configSvc.EXPECT().GetRoutes(mock.Anything).Return(nil).Maybe() - containerSvc := inmocks.NewMockContainerService(t) - containerSvc.EXPECT().Get(mock.Anything, mock.Anything).Return(nil, false).Maybe() - return proxyusecase.NewService(nil, containerSvc, configSvc, proxyusecase.Config{}) + return proxyusecase.NewService(configSvc, proxyusecase.Config{}) } // testAccessLogWriter is a thread-safe mock AccessLogWriter for tests. @@ -171,7 +167,7 @@ func TestCreateHTTPHandlers_DirectHTTPOnboarding_AllPaths(t *testing.T) { req := httptest.NewRequest(http.MethodGet, tc.path, nil) req.RemoteAddr = directAddr - req.Host = "o2.bnema.dev" + req.Host = "app.example.com" rec := httptest.NewRecorder() httpHandler.ServeHTTP(rec, req) @@ -236,7 +232,7 @@ func TestCreateHTTPHandlers_ForceHTTPSRedirect_DoesNotBypassDirectOnboarding(t * req := httptest.NewRequest(http.MethodGet, "/.well-known/gordon/", nil) req.RemoteAddr = directAddr - req.Host = "o2.bnema.dev" + req.Host = "app.example.com" rec := httptest.NewRecorder() httpHandler.ServeHTTP(rec, req) @@ -245,6 +241,14 @@ func TestCreateHTTPHandlers_ForceHTTPSRedirect_DoesNotBypassDirectOnboarding(t * assert.Contains(t, rec.Body.String(), "Trust CA Certificate") } +// stubAppTargets is a fixed proxy.TargetProvider for handler tests. +type stubAppTargets map[string]domain.AppBackend + +func (s stubAppTargets) LookupHost(host string) (domain.AppBackend, bool) { + backend, ok := s[host] + return backend, ok +} + func TestCreateHTTPHandlers_RedirectUsesSelectedEntrypointPorts(t *testing.T) { t.Parallel() cfg := Config{} @@ -255,8 +259,13 @@ func TestCreateHTTPHandlers_RedirectUsesSelectedEntrypointPorts(t *testing.T) { traffic.DefaultEdgeEntryPointName: {Address: ":9443", Protocol: domain.EntryPointProtocolSmartTCP}, } configSvc := inmocks.NewMockConfigService(t) - configSvc.EXPECT().GetRoute(mock.Anything, "app.example.com").Return(&domain.Route{Domain: "app.example.com"}, nil) - svc := &services{proxySvc: proxyusecase.NewService(nil, nil, configSvc, proxyusecase.Config{})} + configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{}).Maybe() + proxySvc := proxyusecase.NewService(configSvc, proxyusecase.Config{}) + // app.example.com serves from the ACTIVE-derived index in this test. + proxySvc.WithAppTargets(stubAppTargets(map[string]domain.AppBackend{ + "app.example.com": {Host: "127.0.0.1", Port: 18080, ContainerPort: 8080, ContainerID: "c-app"}, + })) + svc := &services{proxySvc: proxySvc} _, httpHandler, _ := createHTTPHandlers(svc, cfg, zerowrap.Default(), nil) @@ -283,12 +292,12 @@ func TestCreateHTTPHandlers_OnboardingUsesSelectedEntrypointPorts(t *testing.T) req := httptest.NewRequest(http.MethodGet, "/.well-known/gordon/", nil) req.RemoteAddr = directAddr - req.Host = "o2.bnema.dev:9443" + req.Host = "app.example.com:9443" rec := httptest.NewRecorder() httpHandler.ServeHTTP(rec, req) assert.Equal(t, http.StatusOK, rec.Code) - assert.Contains(t, rec.Body.String(), "https://o2.bnema.dev:9443/") + assert.Contains(t, rec.Body.String(), "https://app.example.com:9443/") } func TestCreateHTTPHandlers_HTTPSOnboarding_RemainsAvailableOnGordonDomain(t *testing.T) { @@ -380,7 +389,7 @@ func TestCreateHTTPHandlers_DirectHTTPOnboarding_HEADRoutesRemainAvailable(t *te ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { r.RemoteAddr = directAddr - r.Host = "o2.bnema.dev" + r.Host = "app.example.com" httpHandler.ServeHTTP(w, r) })) defer ts.Close() @@ -415,7 +424,7 @@ func TestCreateHTTPHandlers_DirectHTTPForbiddenHEAD_ReturnsForbidden(t *testing. ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { r.RemoteAddr = directAddr - r.Host = "o2.bnema.dev" + r.Host = "app.example.com" httpHandler.ServeHTTP(w, r) })) defer ts.Close() @@ -441,7 +450,7 @@ func TestCreateHTTPHandlers_DirectHTTPACMEChallenge_Returns404(t *testing.T) { req := httptest.NewRequest(http.MethodGet, "/.well-known/acme-challenge/test-token", nil) req.RemoteAddr = directAddr - req.Host = "o2.bnema.dev" + req.Host = "app.example.com" rec := httptest.NewRecorder() httpHandler.ServeHTTP(rec, req) @@ -459,7 +468,7 @@ func TestCreateHTTPHandlers_TLSDisabled_DoesNotServeHTTPOnboarding(t *testing.T) req := httptest.NewRequest(http.MethodGet, "/", nil) req.RemoteAddr = directAddr - req.Host = "o2.bnema.dev" + req.Host = "app.example.com" rec := httptest.NewRecorder() httpHandler.ServeHTTP(rec, req) @@ -563,32 +572,29 @@ func TestBuildRegistryCIDRAllowlistMiddleware_InvalidEntries_DenyAll(t *testing. } } -// TestAccessLog_AdminRejectedByLoopbackOnly_IsLogged verifies that requests -// to /admin/ blocked by loopbackOnly() still produce an access-log entry. -// This exercises the outer-deny path: AccessLogger wraps the top-level -// handler so even gates that run before any inner middleware are logged. -func TestAccessLog_AdminRejectedByLoopbackOnly_IsLogged(t *testing.T) { +// TestAccessLog_AdminRemoteRequest_ReachesAuth verifies that authenticated-mode +// admin routes are reachable remotely and delegated to the auth middleware. +func TestAccessLog_AdminRemoteRequest_ReachesAuth(t *testing.T) { t.Parallel() cfg := Config{} - cfg.Auth.Enabled = true // required for admin routes to be registered + cfg.Auth.Enabled = true svc := &services{adminHandler: &adminhttp.Handler{}} aw := &testAccessLogWriter{} registryHandler, _, _ := createHTTPHandlers(svc, cfg, zerowrap.Default(), aw) - // Non-loopback IP — loopbackOnly() will reject this with 403 before any - // inner middleware runs. req := httptest.NewRequest(http.MethodGet, "/admin/status", nil) req.RemoteAddr = "192.0.2.10:12345" rec := httptest.NewRecorder() registryHandler.ServeHTTP(rec, req) - require.Equal(t, http.StatusForbidden, rec.Code) + require.Equal(t, http.StatusServiceUnavailable, rec.Code) + assert.JSONEq(t, `{"error":"authentication service unavailable"}`, rec.Body.String()) entries := aw.snapshot() - require.Len(t, entries, 1, "access log must capture the loopback-rejected admin request") - assert.Equal(t, http.StatusForbidden, entries[0].Status) + require.Len(t, entries, 1) + assert.Equal(t, http.StatusServiceUnavailable, entries[0].Status) assert.Equal(t, "/admin/status", entries[0].Path) } diff --git a/internal/app/run_registry_endpoints_test.go b/internal/app/run_registry_endpoints_test.go new file mode 100644 index 000000000..d66240051 --- /dev/null +++ b/internal/app/run_registry_endpoints_test.go @@ -0,0 +1,26 @@ +package app + +import "testing" + +func TestNeedsInternalRegistryListener(t *testing.T) { + tests := []struct { + name string + addr string + want bool + }{ + {name: "specific IPv4", addr: "100.64.0.10", want: true}, + {name: "specific IPv6", addr: "fd00::1", want: true}, + {name: "IPv4 wildcard covers loopback", addr: "0.0.0.0", want: false}, + {name: "IPv6 wildcard", addr: "::", want: false}, + {name: "IPv4 loopback", addr: "127.0.0.1", want: false}, + {name: "IPv6 loopback", addr: "::1", want: false}, + {name: "empty wildcard", addr: "", want: false}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + if got := needsInternalRegistryListener(tt.addr); got != tt.want { + t.Fatalf("needsInternalRegistryListener(%q) = %v, want %v", tt.addr, got, tt.want) + } + }) + } +} diff --git a/internal/app/run_reload_test.go b/internal/app/run_reload_test.go index c0235a4d4..d16d58de6 100644 --- a/internal/app/run_reload_test.go +++ b/internal/app/run_reload_test.go @@ -6,10 +6,8 @@ import ( "errors" "net" "net/http" - "os" "path/filepath" "strconv" - "strings" "sync" "testing" "time" @@ -25,10 +23,10 @@ import ( inmocks "github.com/bnema/gordon/internal/boundaries/in/mocks" outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/apps" cfgusecase "github.com/bnema/gordon/internal/usecase/config" "github.com/bnema/gordon/internal/usecase/container" "github.com/bnema/gordon/internal/usecase/proxy" - servicecfg "github.com/bnema/gordon/internal/usecase/services" traffic "github.com/bnema/gordon/internal/usecase/traffic" ) @@ -106,27 +104,57 @@ func (r *reloadRecorder) Calls() int { } type proxyRecorder struct { + mu sync.Mutex calls int config proxy.Config } func (p *proxyRecorder) UpdateConfig(config proxy.Config) { + p.mu.Lock() + defer p.mu.Unlock() p.calls++ p.config = config } +func (p *proxyRecorder) Calls() int { + p.mu.Lock() + defer p.mu.Unlock() + return p.calls +} + +func (p *proxyRecorder) Config() proxy.Config { + p.mu.Lock() + defer p.mu.Unlock() + return p.config +} + type registryLimitsRecorder struct { + mu sync.Mutex calls int maxBlobChunkSize int64 maxBlobSize int64 } func (r *registryLimitsRecorder) UpdateBlobLimits(maxBlobChunkSize, maxBlobSize int64) { + r.mu.Lock() + defer r.mu.Unlock() r.calls++ r.maxBlobChunkSize = maxBlobChunkSize r.maxBlobSize = maxBlobSize } +func (r *registryLimitsRecorder) Calls() int { + r.mu.Lock() + defer r.mu.Unlock() + return r.calls +} + +func (r *registryLimitsRecorder) Limits() (chunk, size int64) { + r.mu.Lock() + defer r.mu.Unlock() + return r.maxBlobChunkSize, r.maxBlobSize +} + type containerConfigApplyRecorder struct { calls int ctx context.Context @@ -140,37 +168,6 @@ func (r *containerConfigApplyRecorder) Apply(ctx context.Context, cfg Config) er return nil } -type standaloneServiceRecorder struct { - mu sync.Mutex - calls int - services []domain.StandaloneService - err error - onReconcile func() -} - -func (r *standaloneServiceRecorder) Reconcile(_ context.Context, services []domain.StandaloneService) error { - r.mu.Lock() - r.calls++ - r.services = append([]domain.StandaloneService(nil), services...) - err := r.err - onReconcile := r.onReconcile - r.mu.Unlock() - if onReconcile != nil { - onReconcile() - } - return err -} - -func (r *standaloneServiceRecorder) Status(context.Context) ([]domain.StandaloneServiceStatus, error) { - return nil, nil -} - -func (r *standaloneServiceRecorder) Calls() int { - r.mu.Lock() - defer r.mu.Unlock() - return r.calls -} - type publicTLSReconcileRecorder struct { calls int } @@ -204,6 +201,12 @@ func (e *eventBusRecorder) Calls() int { return e.calls } +func (e *eventBusRecorder) Type() domain.EventType { + e.mu.Lock() + defer e.mu.Unlock() + return e.eventType +} + func TestSetupConfigHotReload_WatchCallbackInvokesCoordinator(t *testing.T) { ctx := context.Background() configSvc := &watchRecorder{} @@ -256,18 +259,18 @@ func TestReloadCoordinator_ApplyLoadedConfig_RebuildsProxyConfigAndPublishesEven containerCfg := &containerConfigApplyRecorder{} registryLimits := ®istryLimitsRecorder{} - coord := newReloadCoordinator(v, reloadSvc, proxySvc, registryLimits, events, nil, zerowrap.Default()) - coord.SetContainerConfigApplier(containerCfg.Apply) + coord := newReloadCoordinator(v, reloadSvc, proxySvc, registryLimits, events, nil, containerCfg.Apply, zerowrap.Default()) require.NoError(t, coord.ApplyLoadedConfig(ctx)) require.Equal(t, 0, reloadSvc.Calls()) - require.Equal(t, 1, proxySvc.calls) + require.Equal(t, 1, proxySvc.Calls()) require.Equal(t, 1, events.Calls()) - require.Equal(t, 1, registryLimits.calls) + require.Equal(t, 1, registryLimits.Calls()) require.Equal(t, 1, containerCfg.calls) - require.Equal(t, int64(6<<20), registryLimits.maxBlobChunkSize) - require.Equal(t, int64(8<<20), registryLimits.maxBlobSize) - require.Equal(t, domain.EventConfigReload, events.eventType) + rlChunk, rlSize := registryLimits.Limits() + require.Equal(t, int64(6<<20), rlChunk) + require.Equal(t, int64(8<<20), rlSize) + require.Equal(t, domain.EventConfigReload, events.Type()) require.Equal(t, "reload.example.com", containerCfg.cfg.Server.GordonDomain) require.Equal(t, []string{"old.example.com"}, containerCfg.cfg.Server.LegacyRegistryDomains) require.Equal(t, []string{"docker.io"}, containerCfg.cfg.Images.AllowedRegistries) @@ -277,10 +280,17 @@ func TestReloadCoordinator_ApplyLoadedConfig_RebuildsProxyConfigAndPublishesEven MaxBodySize: 5 << 20, MaxResponseSize: 7 << 20, MaxConcurrentConns: 99, - }, proxySvc.config) + }, proxySvc.Config()) +} + +// newTestServices wires the serialized traffic publisher the reload hook +// requires. Tests that omit app state still exercise the graph apply. +func newTestServices(svc *services) *services { + svc.appTrafficPublisher = newAppTrafficPublisher(svc, Config{}) + return svc } -func TestServiceInit_ReloadRegistersCustomTLSMuxHTTPSFallback(t *testing.T) { +func TestReloadRuntime_RegistersCustomTLSMuxHTTPSFallback(t *testing.T) { ctx := context.Background() v := viper.New() v.Set("server.gordon_domain", "reload.example.com") @@ -301,18 +311,17 @@ func TestServiceInit_ReloadRegistersCustomTLSMuxHTTPSFallback(t *testing.T) { v: v, cfg: Config{}, log: zerowrap.Default(), - svc: &services{ + svc: newTestServices(&services{ configSvc: configSvc, - containerSvc: container.NewService(nil, nil, nil, nil, container.Config{}, configSvc), + containerSvc: container.NewService(nil, nil, nil, container.Config{}), internalRegUser: "gordon", internalRegPass: "secret", - reloadCoordinator: newReloadCoordinator(v, &reloadRecorder{}, &proxyRecorder{}, nil, nil, nil, zerowrap.Default()), + reloadCoordinator: newReloadCoordinator(v, &reloadRecorder{}, &proxyRecorder{}, nil, nil, nil, nil, zerowrap.Default()), trafficManager: manager, httpsProxyHandler: httpsHandler, pkiSvc: pkiSvc, - }, + }), } - si.registerReloadCoordinatorHooks() reloadCfg := Config{} reloadCfg.Server.GordonDomain = "reload.example.com" @@ -322,7 +331,7 @@ func TestServiceInit_ReloadRegistersCustomTLSMuxHTTPSFallback(t *testing.T) { "custom-secure": {Address: customAddress, Protocol: domain.EntryPointProtocolTLSMux}, } - require.NoError(t, si.svc.reloadCoordinator.applyContainerConfig(ctx, reloadCfg)) + require.NoError(t, (&reloadRuntime{v: si.v, svc: si.svc, log: si.log}).Apply(ctx, reloadCfg)) managementCert, err := pkiSvc.GetCertificate(&tls.ClientHelloInfo{ServerName: "reload.example.com"}) require.NoError(t, err) require.NotNil(t, managementCert) @@ -335,7 +344,7 @@ func TestServiceInit_ReloadRegistersCustomTLSMuxHTTPSFallback(t *testing.T) { require.Contains(t, body, "reload secure") } -func TestServiceInit_ReloadUpdatesManagementHostsBeforeLaterFailure(t *testing.T) { +func TestReloadRuntime_UpdatesManagementHosts(t *testing.T) { ctx := t.Context() v := viper.New() v.Set("server.gordon_domain", "old.example.com") @@ -345,35 +354,30 @@ func TestServiceInit_ReloadUpdatesManagementHostsBeforeLaterFailure(t *testing.T require.NoError(t, configSvc.Load(ctx)) publicTLS := inmocks.NewMockPublicTLSService(t) publicTLS.EXPECT().SetAdditionalHosts(mock.Anything, []string{"new.example.com"}).Once() - reconcileErr := errors.New("standalone reconcile failed") si := &serviceInit{ ctx: ctx, v: v, cfg: Config{}, log: zerowrap.Default(), - svc: &services{ - configSvc: configSvc, - containerSvc: container.NewService(nil, nil, nil, nil, container.Config{}, configSvc), - internalRegUser: "gordon", - internalRegPass: "secret", - reloadCoordinator: newReloadCoordinator(v, &reloadRecorder{}, &proxyRecorder{}, nil, nil, publicTLS, zerowrap.Default()), - publicTLSSvc: publicTLS, - standaloneServiceSvc: &standaloneServiceRecorder{err: reconcileErr}, - }, + svc: newTestServices(&services{ + configSvc: configSvc, + containerSvc: container.NewService(nil, nil, nil, container.Config{}), + internalRegUser: "gordon", + internalRegPass: "secret", + reloadCoordinator: newReloadCoordinator(v, &reloadRecorder{}, &proxyRecorder{}, nil, nil, publicTLS, nil, zerowrap.Default()), + publicTLSSvc: publicTLS, + }), } - si.registerReloadCoordinatorHooks() reloadCfg := Config{} reloadCfg.Server.GordonDomain = "new.example.com" reloadCfg.Server.RegistryPort = 5000 - reloadCfg.Services = []servicecfg.Config{{Name: "game", Image: "game:latest", Enabled: true}} - err := si.svc.reloadCoordinator.applyContainerConfig(ctx, reloadCfg) - require.ErrorIs(t, err, reconcileErr) + require.NoError(t, (&reloadRuntime{v: si.v, svc: si.svc, log: si.log}).Apply(ctx, reloadCfg)) } -func TestServiceInit_RegisterReloadCoordinatorHooks_WiresContainerConfigApplier(t *testing.T) { +func TestReloadRuntime_AppliesContainerConfig(t *testing.T) { ctx := context.Background() v := viper.New() v.Set("server.gordon_domain", "reload.example.com") @@ -387,55 +391,24 @@ func TestServiceInit_RegisterReloadCoordinatorHooks_WiresContainerConfigApplier( v: v, cfg: Config{}, log: zerowrap.Default(), - svc: &services{ + svc: newTestServices(&services{ configSvc: configSvc, - containerSvc: container.NewService(nil, nil, nil, nil, container.Config{}, configSvc), + containerSvc: container.NewService(nil, nil, nil, container.Config{}), internalRegUser: "gordon", internalRegPass: "secret", - reloadCoordinator: newReloadCoordinator(v, &reloadRecorder{}, &proxyRecorder{}, nil, nil, nil, zerowrap.Default()), - }, + reloadCoordinator: newReloadCoordinator(v, &reloadRecorder{}, &proxyRecorder{}, nil, nil, nil, nil, zerowrap.Default()), + }), } - si.registerReloadCoordinatorHooks() - require.NotNil(t, si.svc.reloadCoordinator.applyContainerConfig) - reloadCfg := Config{} reloadCfg.Server.GordonDomain = "reload.example.com" reloadCfg.Server.RegistryPort = 5000 reloadCfg.Images.AllowedRegistries = []string{"docker.io"} - require.NoError(t, si.svc.reloadCoordinator.applyContainerConfig(ctx, reloadCfg)) + require.NoError(t, (&reloadRuntime{v: si.v, svc: si.svc, log: si.log}).Apply(ctx, reloadCfg)) } -func TestStandaloneServiceSecretProviderUnsafeBackendReconcilesServiceSecrets(t *testing.T) { - ctx := context.Background() - dataDir := t.TempDir() - secretPath := "service/game/rcon" - require.NoError(t, os.MkdirAll(filepath.Join(dataDir, "secrets", "service", "game"), 0700)) - require.NoError(t, os.WriteFile(filepath.Join(dataDir, "secrets", secretPath), []byte("test-secret-value"), 0600)) - - rt := outmocks.NewMockContainerRuntime(t) - var sawSecretEnv bool - rt.On("ListContainers", mock.Anything, true).Return([]*domain.Container{}, nil).Once() - rt.On("InspectImageVolumes", mock.Anything, "game:latest").Return([]string(nil), nil).Once() - rt.On("CreateContainer", mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")).Run(func(args mock.Arguments) { - config := args.Get(1).(*domain.ContainerConfig) - for _, entry := range config.Env { - if strings.HasPrefix(entry, "RCON_PASSWORD=") { - sawSecretEnv = true - } - } - }).Return(&domain.Container{ID: "created-1"}, nil).Once() - rt.On("StartContainer", mock.Anything, "created-1").Return(nil).Once() - - provider := createStandaloneServiceSecretProvider(domain.SecretsBackendUnsafe, dataDir, zerowrap.Default()) - reconciler := servicecfg.NewServiceWithSecretProvider(rt, provider) - err := reconciler.Reconcile(ctx, []domain.StandaloneService{{Name: "game", Image: "game:latest", Enabled: true, Secrets: []domain.StandaloneServiceSecretRef{{Name: "rcon", Key: "RCON_PASSWORD"}}}}) - require.NoError(t, err) - assert.True(t, sawSecretEnv) -} - -func TestServiceInit_RegisterReloadCoordinatorHooks_AppliesTrafficBeforeStandaloneServices(t *testing.T) { +func TestReloadRuntime_AppliesTraffic(t *testing.T) { ctx := context.Background() v := viper.New() v.Set("server.gordon_domain", "reload.example.com") @@ -446,72 +419,123 @@ func TestServiceInit_RegisterReloadCoordinatorHooks_AppliesTrafficBeforeStandalo manager := trafficadapter.NewManager() defer func() { require.NoError(t, manager.Shutdown(ctx)) }() - reconciler := &standaloneServiceRecorder{onReconcile: func() { - assert.NotEmpty(t, manager.Status().EntryPoints, "traffic graph should apply before standalone services reconcile") - }} si := &serviceInit{ ctx: ctx, v: v, cfg: Config{}, log: zerowrap.Default(), - svc: &services{ - configSvc: configSvc, - containerSvc: container.NewService(nil, nil, nil, nil, container.Config{}, configSvc), - internalRegUser: "gordon", - internalRegPass: "secret", - reloadCoordinator: newReloadCoordinator(v, &reloadRecorder{}, &proxyRecorder{}, nil, nil, nil, zerowrap.Default()), - trafficManager: manager, - standaloneServiceSvc: reconciler, - }, + svc: newTestServices(&services{ + configSvc: configSvc, + containerSvc: container.NewService(nil, nil, nil, container.Config{}), + internalRegUser: "gordon", + internalRegPass: "secret", + reloadCoordinator: newReloadCoordinator(v, &reloadRecorder{}, &proxyRecorder{}, nil, nil, nil, nil, zerowrap.Default()), + trafficManager: manager, + }), } - si.registerReloadCoordinatorHooks() reloadCfg := Config{} reloadCfg.Server.GordonDomain = "reload.example.com" reloadCfg.Server.RegistryPort = 5000 - reloadCfg.Services = []servicecfg.Config{{Name: "game", Image: "game:latest", Enabled: true}} reloadCfg.EntryPoints = map[string]traffic.EntryPointConfig{"game": {Address: freeTCPAddress(t), Protocol: domain.EntryPointProtocolTCP}} reloadCfg.NetworkServices = []traffic.NetworkServiceConfig{{Name: "game", Ports: []traffic.PortConfig{{Name: "game", Container: 28015, Protocol: domain.NetworkProtocolTCP}}}} reloadCfg.Traffic.TCP.Routers = []traffic.RouterConfig{{Name: "game", EntryPoint: "game", Service: "network_service:game:game"}} - require.NoError(t, si.svc.reloadCoordinator.applyContainerConfig(ctx, reloadCfg)) - require.Equal(t, 1, reconciler.Calls()) - require.Len(t, reconciler.services, 1) - assert.Equal(t, "game", reconciler.services[0].Name) + require.NoError(t, (&reloadRuntime{v: si.v, svc: si.svc, log: si.log}).Apply(ctx, reloadCfg)) assert.NotEmpty(t, manager.Status().EntryPoints) } -func TestWaitForCoreProxyReadyAndApplyTraffic_AppliesTrafficBeforeStandaloneServices(t *testing.T) { - ctx := context.Background() +// TestReloadRuntime_InvalidPolicyAppliesNothing proves the runtime step +// prepares every derived value first: an invalid app mount fails the reload +// before management hosts, traffic, or container config change. +func TestReloadRuntime_InvalidPolicyAppliesNothing(t *testing.T) { + ctx := t.Context() v := viper.New() - configSvc := cfgusecase.NewService(v, nil) - require.NoError(t, configSvc.Load(ctx)) + publicTLS := inmocks.NewMockPublicTLSService(t) // no expectations: any call fails + store := outmocks.NewMockAppState(t) + + svc := newTestServices(&services{ + containerSvc: container.NewService(nil, nil, nil, container.Config{}), + publicTLSSvc: publicTLS, + appSvcImpl: apps.NewAppServiceImpl(store, nil, nil, zerowrap.Default()), + }) + reloadCfg := Config{} + reloadCfg.Server.GordonDomain = "new.example.com" + reloadCfg.AppMounts = map[string]AppMountPolicy{"config": {Source: "relative"}} + + err := (&reloadRuntime{v: v, svc: svc, log: zerowrap.Default()}).Apply(ctx, reloadCfg) + require.ErrorIs(t, err, domain.ErrBindPolicy) +} + +// TestReloadRuntime_TLSLoadFailureAppliesNothing proves an unreadable static +// TLS keypair fails the reload before management hosts change. +func TestReloadRuntime_TLSLoadFailureAppliesNothing(t *testing.T) { + ctx := t.Context() + publicTLS := inmocks.NewMockPublicTLSService(t) // no expectations: any call fails + + svc := newTestServices(&services{ + containerSvc: container.NewService(nil, nil, nil, container.Config{}), + publicTLSSvc: publicTLS, + httpsProxyHandler: http.NotFoundHandler(), + }) + reloadCfg := Config{} + reloadCfg.Server.GordonDomain = "new.example.com" + reloadCfg.Server.TLSCertFile = filepath.Join(t.TempDir(), "missing.crt") + reloadCfg.Server.TLSKeyFile = filepath.Join(t.TempDir(), "missing.key") + reloadCfg.EntryPoints = map[string]traffic.EntryPointConfig{ + "secure": {Address: freeTCPAddress(t), Protocol: domain.EntryPointProtocolTLSMux}, + } + + err := (&reloadRuntime{v: viper.New(), svc: svc, log: zerowrap.Default()}).Apply(ctx, reloadCfg) + require.ErrorContains(t, err, "load TLS keypair") +} + +func TestWaitForCoreProxyReady_WaitsWithoutPublishingEmptyTraffic(t *testing.T) { + ctx := context.Background() manager := trafficadapter.NewManager() defer func() { require.NoError(t, manager.Shutdown(ctx)) }() - reconciler := &standaloneServiceRecorder{onReconcile: func() { - assert.NotEmpty(t, manager.Status().EntryPoints, "traffic graph should apply before standalone services reconcile") - }} - cfg := Config{} - cfg.Services = []servicecfg.Config{{Name: "game", Image: "game:latest", Enabled: true}} - cfg.EntryPoints = map[string]traffic.EntryPointConfig{"game": {Address: freeTCPAddress(t), Protocol: domain.EntryPointProtocolTCP}} - cfg.NetworkServices = []traffic.NetworkServiceConfig{{Name: "game", Ports: []traffic.PortConfig{{Name: "game", Container: 28015, Protocol: domain.NetworkProtocolTCP}}}} - cfg.Traffic.TCP.Routers = []traffic.RouterConfig{{Name: "game", EntryPoint: "game", Service: "network_service:game:game"}} - ready := make(chan struct{}) close(ready) - svc := &services{configSvc: configSvc, trafficManager: manager, standaloneServiceSvc: reconciler} - require.NoError(t, waitForCoreProxyReadyAndApplyTraffic(ctx, cfg, svc, ready, ready, nil)) - require.Equal(t, 1, reconciler.Calls()) - assert.NotEmpty(t, manager.Status().EntryPoints) + require.NoError(t, waitForCoreProxyReady(ctx, ready, ready, nil)) + assert.Empty(t, manager.Status().EntryPoints) } -func TestReconcileStandaloneServices_ConfigConversionErrorIsClear(t *testing.T) { - reconciler := &standaloneServiceRecorder{} - err := reconcileStandaloneServices(context.Background(), reconciler, Config{Services: []servicecfg.Config{{Name: "broken", Image: "", Enabled: true}}}) - require.Error(t, err) - assert.ErrorContains(t, err, "convert standalone service config") - assert.Equal(t, 0, reconciler.Calls()) +func TestWaitForCoreProxyReady_ObservesContextCancellation(t *testing.T) { + ctx, cancel := context.WithCancel(context.Background()) + cancel() + + // Neither readiness channel ever closes: the canceled startup context must + // unblock the wait rather than hang on a listener that will never bind. + err := waitForCoreProxyReady(ctx, make(chan struct{}), make(chan struct{}), make(chan error)) + require.ErrorIs(t, err, context.Canceled) +} + +func TestReloadCoordinator_StopCancelsOwnedTrailingReloadLifecycle(t *testing.T) { + ctx := context.Background() + v := viper.New() + v.Set("server.gordon_domain", "lifecycle.example.com") + v.Set("server.registry_port", 5000) + + reloadSvc := &reloadRecorder{} + proxySvc := &proxyRecorder{} + coord := newReloadCoordinator(v, reloadSvc, proxySvc, nil, nil, nil, nil, zerowrap.Default()) + // Keep the trailing timer pending so Stop is the only thing that can tear + // the coordinator-owned lifecycle down. + coord.debounce = time.Hour + + require.NoError(t, coord.Trigger(ctx)) // immediate apply records lastRun + require.NoError(t, coord.Trigger(ctx)) // coalesced into the trailing timer + + coord.mu.Lock() + lifecycleCtx := coord.lifecycleCtx + coord.mu.Unlock() + require.NotNil(t, lifecycleCtx, "trailing reload must own a lifecycle context") + require.NoError(t, lifecycleCtx.Err()) + + coord.Stop() + require.ErrorIs(t, lifecycleCtx.Err(), context.Canceled) + require.Equal(t, 1, proxySvc.Calls(), "stopped coordinator must not apply the trailing reload") } func TestReloadCoordinator_DebouncesRepeatedWatchCallbacks(t *testing.T) { @@ -530,24 +554,116 @@ func TestReloadCoordinator_DebouncesRepeatedWatchCallbacks(t *testing.T) { events := &eventBusRecorder{} registryLimits := ®istryLimitsRecorder{} - coord := newReloadCoordinator(v, reloadSvc, proxySvc, registryLimits, events, nil, zerowrap.Default()) + coord := newReloadCoordinator(v, reloadSvc, proxySvc, registryLimits, events, nil, nil, zerowrap.Default()) + coord.debounce = 20 * time.Millisecond + require.NoError(t, coord.Trigger(ctx)) require.NoError(t, coord.Trigger(ctx)) require.NoError(t, coord.Trigger(ctx)) - require.Equal(t, 1, reloadSvc.Calls()) - require.Equal(t, 1, proxySvc.calls) - require.Equal(t, 1, events.Calls()) - require.Equal(t, 1, registryLimits.calls) - require.Equal(t, int64(6<<20), registryLimits.maxBlobChunkSize) - require.Equal(t, int64(8<<20), registryLimits.maxBlobSize) - require.Equal(t, domain.EventConfigReload, events.eventType) + // Observe the trailing reload through the recorders' own locks instead of + // locking the coordinator: the reload event is published last, so waiting + // for all recorders proves the trailing reload finished applying. + require.Eventually(t, func() bool { + return reloadSvc.Calls() == 2 && proxySvc.Calls() == 2 && + events.Calls() == 2 && registryLimits.Calls() == 2 + }, time.Second, 5*time.Millisecond) + // No further reload may fire: the debounce window is over. + require.Never(t, func() bool { + return proxySvc.Calls() > 2 || events.Calls() > 2 + }, 50*time.Millisecond, 5*time.Millisecond) + require.Equal(t, 2, proxySvc.Calls()) + require.Equal(t, 2, events.Calls()) + require.Equal(t, 2, registryLimits.Calls()) + rlChunk, rlSize := registryLimits.Limits() + require.Equal(t, int64(6<<20), rlChunk) + require.Equal(t, int64(8<<20), rlSize) + require.Equal(t, domain.EventConfigReload, events.Type()) require.Equal(t, proxy.Config{ RegistryDomain: "new.example.com", RegistryPort: 5000, MaxBodySize: 5 << 20, MaxResponseSize: 7 << 20, MaxConcurrentConns: 99, - }, proxySvc.config) + }, proxySvc.Config()) +} + +func TestSetupConfigHotReload_CoalescesRepeatedWatcherCallbacks(t *testing.T) { + ctx := context.Background() + v := viper.New() + v.Set("server.gordon_domain", "watch.example.com") + v.Set("server.registry_port", 5000) + + configSvc := &watchRecorder{} + reloadSvc := &reloadRecorder{} + proxySvc := &proxyRecorder{} + coord := newReloadCoordinator(v, reloadSvc, proxySvc, nil, nil, nil, nil, zerowrap.Default()) + coord.debounce = 20 * time.Millisecond + defer coord.Stop() + + require.NoError(t, setupConfigHotReload(ctx, configSvc, coord)) + require.NotNil(t, configSvc.onChange) + + // fsnotify delivers a burst of events for a single edit; each callback + // must not apply the config on its own. + configSvc.onChange() + configSvc.onChange() + configSvc.onChange() + + require.Eventually(t, func() bool { return proxySvc.Calls() == 2 }, time.Second, 5*time.Millisecond) + require.Never(t, func() bool { return proxySvc.Calls() > 2 }, 50*time.Millisecond, 5*time.Millisecond) + // ApplyLoadedConfig applies what the watcher already loaded: the trailing + // apply must not read the config again. + require.Zero(t, reloadSvc.Calls()) + require.Equal(t, "watch.example.com", proxySvc.Config().RegistryDomain) +} + +func TestReloadCoordinator_TrailingReloadAppliesFinalState(t *testing.T) { + ctx := context.Background() + v := viper.New() + v.Set("server.gordon_domain", "initial.example.com") + v.Set("server.registry_port", 5000) + + reloadSvc := &reloadRecorder{} + proxySvc := &proxyRecorder{} + coord := newReloadCoordinator(v, reloadSvc, proxySvc, nil, nil, nil, nil, zerowrap.Default()) + coord.debounce = 20 * time.Millisecond + + require.NoError(t, coord.Trigger(ctx)) + v.Set("server.gordon_domain", "intermediate.example.com") + require.NoError(t, coord.Trigger(ctx)) + v.Set("server.gordon_domain", "final.example.com") + require.NoError(t, coord.Trigger(ctx)) + + // The trailing reload is the only writer of the final state: wait for + // the observable outcome instead of locking the coordinator. + require.Eventually(t, func() bool { + return proxySvc.Config().RegistryDomain == "final.example.com" + }, time.Second, 5*time.Millisecond) +} + +func TestReloadCoordinator_InvalidReloadKeepsActiveConfigThenAcceptsValidChange(t *testing.T) { + ctx := context.Background() + v := viper.New() + v.Set("server.gordon_domain", "active.example.com") + v.Set("server.registry_port", 5000) + + reloadSvc := &reloadRecorder{} + proxySvc := &proxyRecorder{} + coord := newReloadCoordinator(v, reloadSvc, proxySvc, nil, nil, nil, nil, zerowrap.Default()) + coord.debounce = 10 * time.Millisecond + + require.NoError(t, coord.Trigger(ctx)) + require.Equal(t, "active.example.com", proxySvc.Config().RegistryDomain) + + time.Sleep(coord.debounce) + v.Set("server.max_proxy_body_size", "invalid") + require.Error(t, coord.Trigger(ctx)) + require.Equal(t, "active.example.com", proxySvc.Config().RegistryDomain) + + v.Set("server.max_proxy_body_size", "5MB") + v.Set("server.gordon_domain", "recovered.example.com") + require.NoError(t, coord.Trigger(ctx)) + require.Equal(t, "recovered.example.com", proxySvc.Config().RegistryDomain) } func TestReloadCoordinator_RetriesImmediatelyAfterFailedReload(t *testing.T) { @@ -563,7 +679,7 @@ func TestReloadCoordinator_RetriesImmediatelyAfterFailedReload(t *testing.T) { reloadErr := errors.New("reload failed") reloadSvc := &reloadRecorder{err: reloadErr} proxySvc := &proxyRecorder{} - coord := newReloadCoordinator(v, reloadSvc, proxySvc, nil, nil, nil, zerowrap.Default()) + coord := newReloadCoordinator(v, reloadSvc, proxySvc, nil, nil, nil, nil, zerowrap.Default()) err := coord.Trigger(ctx) require.ErrorIs(t, err, reloadErr) @@ -574,7 +690,7 @@ func TestReloadCoordinator_RetriesImmediatelyAfterFailedReload(t *testing.T) { require.NoError(t, coord.Trigger(ctx)) require.Equal(t, 2, reloadSvc.Calls()) - assert.Equal(t, 1, proxySvc.calls) + assert.Equal(t, 1, proxySvc.Calls()) } func TestReloadCoordinator_PublishErrorDoesNotAdvanceDebounceState(t *testing.T) { @@ -593,7 +709,7 @@ func TestReloadCoordinator_PublishErrorDoesNotAdvanceDebounceState(t *testing.T) events := &eventBusRecorder{err: publishErr} publicTLS := &publicTLSReconcileRecorder{} - coord := newReloadCoordinator(v, reloadSvc, proxySvc, nil, events, publicTLS, zerowrap.Default()) + coord := newReloadCoordinator(v, reloadSvc, proxySvc, nil, events, publicTLS, nil, zerowrap.Default()) err := coord.Trigger(ctx) require.ErrorIs(t, err, publishErr) @@ -602,10 +718,10 @@ func TestReloadCoordinator_PublishErrorDoesNotAdvanceDebounceState(t *testing.T) require.ErrorIs(t, err, publishErr) require.Equal(t, 2, reloadSvc.Calls()) - require.Equal(t, 2, proxySvc.calls) + require.Equal(t, 2, proxySvc.Calls()) require.Equal(t, 2, events.Calls()) require.Equal(t, 2, publicTLS.calls) - require.Equal(t, domain.EventConfigReload, events.eventType) + require.Equal(t, domain.EventConfigReload, events.Type()) } func TestReloadCoordinator_SerializesOverlappingReloadRequests(t *testing.T) { @@ -626,7 +742,8 @@ func TestReloadCoordinator_SerializesOverlappingReloadRequests(t *testing.T) { <-release }} proxySvc := &proxyRecorder{} - coord := newReloadCoordinator(v, reloadSvc, proxySvc, nil, nil, nil, zerowrap.Default()) + coord := newReloadCoordinator(v, reloadSvc, proxySvc, nil, nil, nil, nil, zerowrap.Default()) + coord.debounce = 20 * time.Millisecond firstDone := make(chan struct{}) go func() { @@ -667,6 +784,7 @@ func TestReloadCoordinator_SerializesOverlappingReloadRequests(t *testing.T) { t.Fatal("second reload did not finish") } - require.Equal(t, 1, reloadSvc.Calls()) - assert.Equal(t, 1, proxySvc.calls) + require.Eventually(t, func() bool { return proxySvc.Calls() == 2 }, time.Second, 5*time.Millisecond) + require.Never(t, func() bool { return proxySvc.Calls() > 2 }, 50*time.Millisecond, 5*time.Millisecond) + assert.Equal(t, 2, proxySvc.Calls()) } diff --git a/internal/app/run_standalone_services_test.go b/internal/app/run_standalone_services_test.go new file mode 100644 index 000000000..97bc3e2dc --- /dev/null +++ b/internal/app/run_standalone_services_test.go @@ -0,0 +1,40 @@ +package app + +import ( + "os" + "path/filepath" + "testing" + + "github.com/stretchr/testify/require" +) + +// TestInitConfigAcceptsStandaloneServices proves the installation-level +// [[services]] L4 standalone workload key is still live config and must +// not be rejected as a retired app key. App workloads now live in +// standalone app files, but [[services]] in gordon.toml keeps managing +// Gordon-owned L4 containers. +func TestInitConfigAcceptsStandaloneServices(t *testing.T) { + configPath := filepath.Join(t.TempDir(), "gordon.toml") + doc := ` +[[services]] +name = "rust" +image = "registry.example.com:5000/rust:latest" +enabled = true + +[[services.ports]] +name = "game" +container = 28015 +protocol = "udp" +publish = "127.0.0.1:38015" +` + require.NoError(t, os.WriteFile(configPath, []byte(doc), 0o600)) + + _, cfg, err := initConfig(configPath) + + require.NoError(t, err) + require.Len(t, cfg.Services, 1) + require.Equal(t, "rust", cfg.Services[0].Name) + require.True(t, cfg.Services[0].Enabled) + require.Len(t, cfg.Services[0].Ports, 1) + require.Equal(t, 28015, cfg.Services[0].Ports[0].Container) +} diff --git a/internal/app/schedulers.go b/internal/app/schedulers.go new file mode 100644 index 000000000..d6bdbec95 --- /dev/null +++ b/internal/app/schedulers.go @@ -0,0 +1,224 @@ +package app + +import ( + "context" + "fmt" + "strings" + + "github.com/bnema/zerowrap" + "github.com/spf13/viper" + + "github.com/bnema/gordon/internal/domain" + cronSvc "github.com/bnema/gordon/internal/usecase/cron" +) + +func startOptionalSchedulers(ctx context.Context, cfg Config, svc *services, log zerowrap.Logger, v *viper.Viper) (func(), error) { + schedulers := make([]*cronSvc.Scheduler, 0, 3) + stopAll := func() { + for i := len(schedulers) - 1; i >= 0; i-- { + schedulers[i].Stop() + } + } + + backupScheduler, err := startBackupScheduler(ctx, cfg, svc, log) + if err != nil { + return nil, err + } + if backupScheduler != nil { + schedulers = append(schedulers, backupScheduler) + } + + volumeBackupScheduler, err := startVolumeBackupScheduler(ctx, cfg, svc, log) + if err != nil { + stopAll() + return nil, err + } + if volumeBackupScheduler != nil { + schedulers = append(schedulers, volumeBackupScheduler) + } + + imageScheduler, err := startImagePruneScheduler(ctx, cfg, svc, log, func() int { + return v.GetInt("images.prune.keep_last") + }) + if err != nil { + stopAll() + return nil, err + } + if imageScheduler != nil { + schedulers = append(schedulers, imageScheduler) + } + + if len(schedulers) == 0 { + return nil, nil + } + + return stopAll, nil +} + +func startBackupScheduler(ctx context.Context, cfg Config, svc *services, log zerowrap.Logger) (*cronSvc.Scheduler, error) { + dbCfg := databaseBackupSettings(cfg) + if !dbCfg.Enabled || svc == nil || svc.backupSvc == nil { + return nil, nil + } + + preset, err := resolveBackupSchedule(dbCfg.Schedule) + if err != nil { + return nil, err + } + + scheduler := cronSvc.NewScheduler(log) + err = scheduler.Add( + "backup-scheduler", + "Backups", + domain.CronSchedule{Preset: preset}, + func(jobCtx context.Context) error { + if err := svc.backupSvc.RunForSchedule(jobCtx, preset); err != nil { + return err + } + log.Info(). + Str("schedule", string(preset)). + Msg("scheduled backup run complete") + return nil + }, + ) + if err != nil { + return nil, log.WrapErr(err, "failed to register backup schedule") + } + + scheduler.Start(ctx) + log.Info(). + Str("schedule", string(preset)). + Msg("backup scheduler enabled") + + return scheduler, nil +} + +func resolveBackupSchedule(raw string) (domain.BackupSchedule, error) { + return resolveSchedulePreset(raw, "backups.databases.schedule", domain.ScheduleDaily) +} + +func startVolumeBackupScheduler(ctx context.Context, cfg Config, svc *services, log zerowrap.Logger) (*cronSvc.Scheduler, error) { + if !cfg.Backups.Volumes.Enabled || svc == nil || svc.volumeBackupSvc == nil { + return nil, nil + } + + volumeCfg, err := validateVolumeBackupConfig(cfg) + if err != nil { + return nil, err + } + + scheduler := cronSvc.NewScheduler(log) + err = scheduler.Add( + "volume-backup-scheduler", + "Volume Backups", + domain.CronSchedule{Interval: volumeCfg.Interval}, + func(jobCtx context.Context) error { + if err := svc.volumeBackupSvc.RunVolumeBackupsForSchedule(jobCtx, ""); err != nil { + return err + } + log.Info(). + Dur("interval", volumeCfg.Interval). + Msg("scheduled volume backup run complete") + return nil + }, + ) + if err != nil { + return nil, log.WrapErr(err, "failed to register volume backup schedule") + } + + scheduler.Start(ctx) + log.Info(). + Dur("interval", volumeCfg.Interval). + Msg("volume backup scheduler enabled") + + return scheduler, nil +} + +func startImagePruneScheduler(ctx context.Context, cfg Config, svc *services, log zerowrap.Logger, keepLastGetter func() int) (*cronSvc.Scheduler, error) { + if !cfg.Images.Prune.Enabled || svc == nil || svc.imageSvc == nil { + return nil, nil + } + // The scheduler runs the same use case and planner as manual + // execution. App presence is normal: apps exist on every real + // server, and prune decides per candidate. + if keepLastGetter == nil { + keepLastGetter = func() int { return cfg.Images.Prune.KeepLast } + } + if keepLastGetter() < 0 { + return nil, fmt.Errorf("images.prune.keep_last must be >= 0") + } + + preset, err := resolveImagePruneSchedule(cfg.Images.Prune.Schedule) + if err != nil { + return nil, err + } + + scheduler := cronSvc.NewScheduler(log) + err = scheduler.Add( + "image-prune", + "Image prune", + domain.CronSchedule{Preset: preset}, + func(jobCtx context.Context) error { + keepLast := keepLastGetter() + if keepLast < 0 { + log.Warn(). + Int("configured_keep_last", keepLast). + Int("fallback_keep_last", domain.DefaultImagePruneKeepLast). + Msg("invalid images.prune.keep_last; using default") + keepLast = domain.DefaultImagePruneKeepLast + } + + report, err := svc.imageSvc.Prune(jobCtx, domain.ImagePruneOptions{ + KeepLast: keepLast, + PruneDangling: true, + PruneRegistry: true, + }) + if err != nil { + return err + } + + log.Info(). + Int("keep_last", keepLast). + Int("eligible", report.Plan.CountByVerdict(domain.PruneVerdictEligible)). + Int("protected", report.Plan.CountByVerdict(domain.PruneVerdictProtected)). + Int("unknown", report.Plan.CountByVerdict(domain.PruneVerdictUnknown)). + Int("deleted", len(report.Plan.Deleted)). + Int("failures", len(report.Plan.Failures)). + Int("inventory_gaps", len(report.Plan.Gaps)). + Int("runtime_deleted", report.Runtime.DeletedCount). + Int("registry_tags_removed", report.Registry.TagsRemoved). + Int("registry_blobs_removed", report.Registry.BlobsRemoved). + Msg("scheduled image prune complete") + return nil + }, + ) + if err != nil { + return nil, log.WrapErr(err, "failed to register image prune schedule") + } + + scheduler.Start(ctx) + log.Info(). + Str("schedule", string(preset)). + Int("keep_last", keepLastGetter()). + Msg("image prune scheduler enabled") + + return scheduler, nil +} + +func resolveImagePruneSchedule(raw string) (domain.BackupSchedule, error) { + return resolveSchedulePreset(raw, "images.prune.schedule", domain.ScheduleDaily) +} + +func resolveSchedulePreset(raw, name string, defaultVal domain.BackupSchedule) (domain.BackupSchedule, error) { + schedule := domain.BackupSchedule(strings.ToLower(strings.TrimSpace(raw))) + if schedule == "" { + schedule = defaultVal + } + + switch schedule { + case domain.ScheduleHourly, domain.ScheduleDaily, domain.ScheduleWeekly, domain.ScheduleMonthly: + return schedule, nil + default: + return "", fmt.Errorf("%s must be one of: hourly, daily, weekly, monthly", name) + } +} diff --git a/internal/app/services.go b/internal/app/services.go new file mode 100644 index 000000000..873b74951 --- /dev/null +++ b/internal/app/services.go @@ -0,0 +1,707 @@ +package app + +import ( + "context" + "fmt" + "net/http" + "path/filepath" + "time" + + "github.com/bnema/zerowrap" + "github.com/spf13/viper" + + "github.com/bnema/gordon/internal/adapters/in/http/admin" + authhandler "github.com/bnema/gordon/internal/adapters/in/http/auth" + "github.com/bnema/gordon/internal/adapters/in/http/registry" + trafficadapter "github.com/bnema/gordon/internal/adapters/in/traffic" + acmelego "github.com/bnema/gordon/internal/adapters/out/acmelego" + acmestore "github.com/bnema/gordon/internal/adapters/out/acmestore" + "github.com/bnema/gordon/internal/adapters/out/docker" + "github.com/bnema/gordon/internal/adapters/out/eventbus" + "github.com/bnema/gordon/internal/adapters/out/filesystem" + "github.com/bnema/gordon/internal/adapters/out/logwriter" + pkiadapter "github.com/bnema/gordon/internal/adapters/out/pki" + "github.com/bnema/gordon/internal/adapters/out/secrets" + "github.com/bnema/gordon/internal/adapters/out/telemetry" + "github.com/bnema/gordon/internal/boundaries/in" + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/apps" + "github.com/bnema/gordon/internal/usecase/apptraffic" + "github.com/bnema/gordon/internal/usecase/auth" + "github.com/bnema/gordon/internal/usecase/backup" + "github.com/bnema/gordon/internal/usecase/config" + "github.com/bnema/gordon/internal/usecase/container" + "github.com/bnema/gordon/internal/usecase/deployment" + "github.com/bnema/gordon/internal/usecase/health" + "github.com/bnema/gordon/internal/usecase/images" + "github.com/bnema/gordon/internal/usecase/logs" + pkiusecase "github.com/bnema/gordon/internal/usecase/pki" + "github.com/bnema/gordon/internal/usecase/proxy" + "github.com/bnema/gordon/internal/usecase/publictls" + registrySvc "github.com/bnema/gordon/internal/usecase/registry" + "github.com/bnema/gordon/internal/usecase/registrystate" + volumesSvc "github.com/bnema/gordon/internal/usecase/volumes" + "github.com/bnema/gordon/pkg/bytesize" +) + +// services holds all the services used by the application. +type services struct { + runtime *docker.Runtime + eventBus *eventbus.InMemory + metrics *telemetry.Metrics + blobStorage *filesystem.BlobStorage + manifestStorage *filesystem.ManifestStorage + backupStorage *filesystem.BackupStorage + volumeBackupStore out.VolumeBackupStorage + volumeBackupCfg domain.VolumeBackupConfig + logWriter *logwriter.LogWriter + tokenStore out.TokenStore + configSvc *config.Service + containerSvc *container.Service + backupSvc *backup.Service + volumeBackupSvc *backup.VolumeService + registrySvc *registrySvc.Service + healthSvc *health.Service + logSvc *logs.Service + imageSvc *images.Service + volumeSvc *volumesSvc.Service + proxySvc *proxy.Service + serviceSecretProvider out.SecretProvider + authSvc *auth.Service + authHandler *authhandler.Handler + adminHandler *admin.Handler + httpProxyHandler http.Handler + httpsProxyHandler http.Handler + internalRegUser string + internalRegPass string + maxBlobChunkSize int64 + maxBlobSize int64 + caAdapter *pkiadapter.CA + pkiSvc *pkiusecase.Service + appState out.AppState + // gcBarrier serializes resource acquisition and its durable + // protection publication against destructive prune. + gcBarrier out.GCBarrier + appDeploySvc *deployment.Service + appSvc in.AppService + appSvcImpl *apps.AppServiceImpl + appActivator *apptraffic.Activator + appHostIndex *apptraffic.HostIndex + appTrafficPublisher *appTrafficPublisher + appMonitor *appMonitor + reloadCoordinator *reloadCoordinator + publicTLSSvc in.PublicTLSService + publicTLSRuntime publicTLSRuntime + trafficManager *trafficadapter.Manager + tlsHTTPEntryPoints map[string]struct{} + smartHTTPEntryPoints map[string]struct{} + registryHandler interface { + UpdateBlobLimits(maxBlobChunkSize, maxBlobSize int64) + } + // workloadLogs exports proxy access and app container logs over + // OTLP. Nil when telemetry log export is disabled. + workloadLogs out.LogExporter +} + +// serviceInit holds the shared context for service initialization helpers. +type serviceInit struct { + ctx context.Context + v *viper.Viper + cfg Config + log zerowrap.Logger + svc *services +} + +// createServices creates all the application services for server runtime. +// Public ACME reconciliation is started later, after the HTTP listener is bound. +func createServices(ctx context.Context, v *viper.Viper, cfg Config, log zerowrap.Logger) (_ *services, retErr error) { + return createServicesWithOptions(ctx, v, cfg, log) +} + +// createServicesWithOptions creates all the application services. +// ACME Reconcile and renewal loop are started later from runServers, +// after HTTP listeners are bound. +func createServicesWithOptions(ctx context.Context, v *viper.Viper, cfg Config, log zerowrap.Logger) (_ *services, retErr error) { + si := &serviceInit{ + ctx: ctx, + v: v, + cfg: cfg, + log: log, + svc: &services{}, + } + defer func() { + if retErr != nil { + if si.svc.pkiSvc != nil { + si.svc.pkiSvc.Stop() + } + if si.svc.publicTLSSvc != nil { + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + if err := si.svc.publicTLSSvc.Stop(ctx); err != nil { + si.log.Warn().Err(err).Msg("failed to stop public TLS service during createServices cleanup") + } + } + } + }() + var err error + + // Create output adapters + runtimeSocket := resolveRuntimeConfig(v.GetString("server.runtime")) + if si.svc.runtime, si.svc.eventBus, err = createOutputAdapters(ctx, log, runtimeSocket); err != nil { + return nil, err + } + + // Create storage + if si.svc.blobStorage, si.svc.manifestStorage, err = createStorage(cfg, log); err != nil { + return nil, err + } + + // Create log writer + if si.svc.logWriter, err = createLogWriter(cfg, log); err != nil { + return nil, err + } + + // Create auth service (if enabled) + if si.svc.tokenStore, si.svc.authSvc, err = createAuthService(ctx, cfg, log); err != nil { + return nil, err + } + + if err := setupInternalRegistryAuth(si.svc, log); err != nil { + return nil, err + } + + // Create config service + si.svc.configSvc = config.NewService(v, si.svc.eventBus) + if err := si.svc.configSvc.Load(ctx); err != nil { + return nil, log.WrapErr(err, "failed to load configuration") + } + + if err := si.initPKI(); err != nil { + return nil, err + } + + if err := si.initSecrets(); err != nil { + return nil, err + } + + if err := si.initPublicTLS(); err != nil { + return nil, err + } + + if err := si.initRuntimeProxyAndTraffic(); err != nil { + return nil, err + } + + runtime := &reloadRuntime{v: v, svc: si.svc, log: log} + si.svc.reloadCoordinator = newReloadCoordinator(v, si.svc.configSvc, si.svc.proxySvc, nil, si.svc.eventBus, si.svc.publicTLSSvc, runtime.Apply, log) + + si.initHandlers() + + return si.svc, nil +} + +// initPKI initialises the internal CA and PKI service when TLS is enabled. +func (si *serviceInit) initPKI() error { + if !hasTLSCapableEntrypoint(si.cfg) { + si.log.Info().Msg("internal CA disabled (no TLS-capable entrypoint configured)") + return nil + } + + if (si.cfg.Server.TLSCertFile == "") != (si.cfg.Server.TLSKeyFile == "") { + return fmt.Errorf("both tls_cert_file and tls_key_file must be set, or neither") + } + + caAdapter, err := pkiadapter.NewCA(resolveDataDir(si.cfg.Server.DataDir), si.log) + if err != nil { + return si.log.WrapErr(err, "failed to initialize internal CA") + } + si.svc.caAdapter = caAdapter + si.svc.pkiSvc = pkiusecase.NewService(si.ctx, caAdapter, si.appRoutes(), []string{si.cfg.Server.GordonDomain}, si.log) + return nil +} + +// initPublicTLS initializes the public ACME TLS service if enabled. +func (si *serviceInit) initPublicTLS() error { + if !si.cfg.TLS.ACME.Enabled { + return nil + } + + if err := validatePublicTLSReadiness(si.cfg); err != nil { + return err + } + + ctx := si.ctx + log := si.log + + dnsCfg, err := buildDNSConfig(si.cfg) + if err != nil { + return log.WrapErr(err, "invalid DNS configuration") + } + + publicTLSCfg := publictls.Config{ + Enabled: si.cfg.TLS.ACME.Enabled, + Email: si.cfg.TLS.ACME.Email, + Challenge: si.cfg.TLS.ACME.Challenge, + HTTPPort: effectiveHTTP01Port(si.cfg), + TLSPort: effectivePublicTLSPort(si.cfg), + DataDir: resolveDataDir(si.cfg.Server.DataDir), + ObtainBatchSize: si.cfg.TLS.ACME.ObtainBatchSize, + DNS: dnsCfg, + } + + tokenResolver := secrets.NewPublicTLSResolver(secrets.PublicTLSResolverConfig{}) + + effective, err := publictls.ResolveEffectiveChallenge(ctx, publicTLSCfg, tokenResolver) + if err != nil { + return log.WrapErr(err, "resolve ACME challenge") + } + if err := validateEffectivePublicTLSReadiness(si.cfg, effective); err != nil { + return err + } + + store, err := acmestore.New(filepath.Join(resolveDataDir(si.cfg.Server.DataDir), "acme")) + if err != nil { + return log.WrapErr(err, "create ACME store") + } + + challenges := publictls.NewHTTP01Challenges() + + var zoneResolver *acmelego.CloudflareZoneResolver + if effective.Mode == domain.ACMEChallengeCloudflareDNS01 { + zoneResolver = acmelego.NewCloudflareZoneResolver(effective.Token) + } + + issuer, err := acmelego.NewIssuer(acmelego.Config{ + Email: si.cfg.TLS.ACME.Email, + Challenge: effective.Mode, + Token: effective.Token, + Store: store, + HTTPChallengeSink: challenges, + DNSResolvers: publicTLSCfg.DNS.Resolvers, + DNSPropagationTimeout: publicTLSCfg.DNS.PropagationTimeout, + DNSPollingInterval: publicTLSCfg.DNS.PollingInterval, + }) + if err != nil { + return log.WrapErr(err, "create ACME issuer") + } + + svc := publictls.NewService(publicTLSCfg, publictls.ServiceDeps{ + Routes: si.appRoutes(), + Issuer: issuer, + Store: store, + ZoneResolver: zoneResolver, + Challenges: challenges, + Effective: effective, + AdditionalHosts: []string{si.cfg.Server.GordonDomain}, + }) + + if err := svc.Load(ctx); err != nil { + log.Warn().Err(err).Msg("failed to load ACME certificates, continuing") + } + + log.Info(). + Str("email", si.cfg.TLS.ACME.Email). + Str("challenge", string(effective.Mode)). + Msg("public ACME TLS initialized (runtime start deferred)") + + si.svc.publicTLSSvc = svc + si.svc.publicTLSRuntime = svc + return nil +} + +// publicTLSRuntime is the subset of in.PublicTLSService needed at server runtime +// after the HTTP listener is bound: reconcile missing certs and start the renewal +// loop. It is nil when ACME is disabled. +type publicTLSRuntime interface { + Reconcile(context.Context) error + StartRenewalLoop(context.Context, time.Duration) <-chan struct{} +} + +func startPublicTLSRuntime(ctx context.Context, svc publicTLSRuntime, log zerowrap.Logger) error { + if svc == nil { + return nil + } + reconcileErr := svc.Reconcile(ctx) + svc.StartRenewalLoop(ctx, time.Hour) + log.Info().Msg("public ACME TLS runtime started") + return reconcileErr +} + +func startPublicTLSRuntimeWithWarning(ctx context.Context, svc publicTLSRuntime, log zerowrap.Logger) { + if err := startPublicTLSRuntime(ctx, svc, log); err != nil { + log.Warn().Err(err).Msg("initial public ACME reconcile failed, continuing with renewal loop") + } +} + +// initSecrets validates the secrets backend and creates the app service +// secret provider. +func (si *serviceInit) initSecrets() error { + backend, err := resolveSecretsBackend(si.cfg.Auth.SecretsBackend) + if err != nil { + return si.log.WrapErr(err, "failed to resolve secrets backend") + } + si.svc.serviceSecretProvider = createStandaloneServiceSecretProvider(backend, resolveDataDir(si.cfg.Server.DataDir), si.log) + return nil +} + +// initRuntimeAndProxy creates container, backup, registry, image, volume, and proxy services. +func (si *serviceInit) initRuntimeProxyAndTraffic() error { + if err := si.initRuntimeAndProxy(); err != nil { + return err + } + si.svc.trafficManager = trafficadapter.NewManager() + return si.initApps() +} + +func containerResourceLimits(cfg Config) (deployment.ResourceLimits, error) { + limits := deployment.ResourceLimits{PidsLimit: cfg.Containers.PidsLimit} + if cfg.Containers.MemoryLimit != "" { + parsed, err := bytesize.Parse(cfg.Containers.MemoryLimit) + if err != nil { + return limits, fmt.Errorf("invalid containers.memory_limit %q: %w", cfg.Containers.MemoryLimit, err) + } + limits.MemoryBytes = parsed + } + if cfg.Containers.CPULimit > 0 { + limits.NanoCPUs = int64(cfg.Containers.CPULimit * float64(time.Second)) + } + return limits, nil +} + +func (si *serviceInit) initRuntimeAndProxy() error { + var err error + + if si.svc.containerSvc, err = createContainerService(si.ctx, si.v, si.cfg, si.svc, si.log); err != nil { + return err + } + + if si.svc.backupStorage, si.svc.backupSvc, err = createBackupService(si.cfg, si.svc, si.log); err != nil { + return err + } + if si.svc.volumeBackupStore, si.svc.volumeBackupSvc, si.svc.volumeBackupCfg, err = createVolumeBackupService(si.ctx, si.cfg, si.svc, si.log); err != nil { + return err + } + + registryState := registrystate.New() + si.svc.registrySvc = registrySvc.NewService(si.svc.blobStorage, si.svc.manifestStorage, si.svc.eventBus, registryState) + si.svc.imageSvc = images.NewService(si.svc.runtime, si.svc.manifestStorage, si.svc.blobStorage, si.log, registryState) + si.svc.volumeSvc = volumesSvc.NewService(si.svc.runtime) + + injectTelemetryMetrics(si.cfg, si.svc, si.log) + + proxyCfg, err := buildProxyConfig(si.cfg, si.log) + if err != nil { + return err + } + si.svc.maxBlobChunkSize = proxyCfg.maxBlobChunkSize + si.svc.maxBlobSize = proxyCfg.maxBlobSize + si.svc.proxySvc = proxy.NewService(si.svc.configSvc, proxyCfg.proxyConfig) + + // Wire synchronous proxy cache invalidation for zero-downtime deployments. + return nil +} + +func (si *serviceInit) initHandlers() { + if si.svc.authSvc != nil { + internalAuth := authhandler.InternalAuth{ + Username: si.svc.internalRegUser, + Password: si.svc.internalRegPass, + } + si.svc.authHandler = authhandler.NewHandler(si.svc.authSvc, internalAuth, si.log) + } + + si.svc.logSvc = logs.NewService(resolveLogFilePath(si.cfg), si.cfg.Logging.File.Enabled, si.svc.runtime, si.log) + // initApps (via initRuntimeProxyAndTraffic) runs before initHandlers, + // so durable ACTIVE state is already open here. + if si.svc.appState != nil { + si.svc.logSvc.WithAppState(si.svc.appState) + } + + if si.svc.trafficManager == nil { + si.svc.trafficManager = trafficadapter.NewManager() + } + + si.svc.adminHandler = admin.NewHandler(admin.HandlerDeps{ + ConfigSvc: si.svc.configSvc, + AuthSvc: si.svc.authSvc, + ContainerSvc: si.svc.containerSvc, + HealthSvc: si.svc.healthSvc, + LogSvc: si.svc.logSvc, + RegistrySvc: si.svc.registrySvc, + ReloadTrigger: si.svc.reloadCoordinator, + Log: si.log, + BackupSvc: si.svc.backupSvc, + VolumeBackupSvc: si.svc.volumeBackupSvc, + ImageSvc: si.svc.imageSvc, + VolumeSvc: si.svc.volumeSvc, + PublicTLSSvc: si.svc.publicTLSSvc, + TrafficSvc: si.svc.trafficManager, + AppSvc: si.svc.appSvc, + }) +} + +// injectTelemetryMetrics creates and injects OTel metrics into services when +// telemetry is enabled. Skipped otherwise to avoid unnecessary allocations. +func injectTelemetryMetrics(cfg Config, svc *services, log zerowrap.Logger) { + if !cfg.Telemetry.Enabled || !cfg.Telemetry.Metrics { + return + } + gordonMetrics, err := telemetry.NewMetrics() + if err != nil { + log.Warn().Err(err).Msg("failed to create telemetry metrics, continuing without metrics") + return + } + svc.registrySvc.SetMetrics(gordonMetrics) + svc.eventBus.SetMetrics(gordonMetrics) + svc.metrics = gordonMetrics +} + +// observeManagedContainers exports the managed-container gauge from app +// state. No-op when metrics are disabled. +func observeManagedContainers(metrics *telemetry.Metrics, state out.AppStateReader, log zerowrap.Logger) { + if metrics == nil { + return + } + if err := metrics.ObserveManagedContainers(func(ctx context.Context) (int64, error) { + return apps.CountManagedContainers(ctx, state) + }); err != nil { + log.Warn().Err(err).Msg("failed to register managed container gauge") + } +} + +// createOutputAdapters creates the container runtime and event bus. +func createOutputAdapters(ctx context.Context, log zerowrap.Logger, runtimeSocket string) (*docker.Runtime, *eventbus.InMemory, error) { + detection := docker.DetectRuntimeSocket(runtimeSocket) + + var runtime *docker.Runtime + var err error + + switch detection.Source { + case "none": + return nil, nil, fmt.Errorf("no container runtime found: checked Docker socket, Podman socket, DOCKER_HOST env var. Install Docker or Podman, or set server.runtime in config") + case "DOCKER_HOST_passthrough": + detection.RuntimeName = "docker" + runtime, err = docker.NewRuntime() + default: + if detection.SocketPath != "" { + runtime, err = docker.NewRuntimeWithSocket(detection.SocketPath) + } else { + runtime, err = docker.NewRuntime() + } + } + if err != nil { + return nil, nil, log.WrapErr(err, "failed to create container runtime") + } + + if err := runtime.Ping(ctx); err != nil { + return nil, nil, log.WrapErr(err, fmt.Sprintf("container runtime not available (detected: %s via %s)", detection.RuntimeName, detection.Source)) + } + + runtimeVersion, _ := runtime.Version(ctx) + log.Info(). + Str("runtime", detection.RuntimeName). + Str("version", runtimeVersion). + Str("source", detection.Source). + Msg("container runtime initialized") + + eventBus := eventbus.NewInMemory(100, log) + + return runtime, eventBus, nil +} + +// createStorage creates blob and manifest storage. +func createStorage(cfg Config, log zerowrap.Logger) (*filesystem.BlobStorage, *filesystem.ManifestStorage, error) { + dataDir := cfg.Server.DataDir + if dataDir == "" { + dataDir = DefaultDataDir() + } + + registryDir := filepath.Join(dataDir, "registry") + + blobStorage, err := filesystem.NewBlobStorage(registryDir, log) + if err != nil { + return nil, nil, log.WrapErr(err, "failed to create blob storage") + } + + manifestStorage, err := filesystem.NewManifestStorage(registryDir, log) + if err != nil { + return nil, nil, log.WrapErr(err, "failed to create manifest storage") + } + + return blobStorage, manifestStorage, nil +} + +// createLogWriter creates the container log writer. +func createLogWriter(cfg Config, log zerowrap.Logger) (*logwriter.LogWriter, error) { + if !cfg.Logging.ContainerLogs.Enabled { + log.Debug().Msg("container log collection disabled") + return nil, nil + } + + // Determine log directory + logDir := cfg.Logging.ContainerLogs.Dir + if logDir == "" { + dataDir := cfg.Server.DataDir + if dataDir == "" { + dataDir = DefaultDataDir() + } + logDir = filepath.Join(dataDir, "logs", "containers") + } + + writer, err := logwriter.New(logwriter.Config{ + Dir: logDir, + MaxSize: cfg.Logging.ContainerLogs.MaxSize, + MaxBackups: cfg.Logging.ContainerLogs.MaxBackups, + MaxAge: cfg.Logging.ContainerLogs.MaxAge, + }) + if err != nil { + return nil, log.WrapErr(err, "failed to create container log writer") + } + + log.Info().Str("dir", logDir).Msg("container log collection enabled") + return writer, nil +} + +// proxyConfigResult holds parsed proxy and blob chunk size config. +type proxyConfigResult struct { + proxyConfig proxy.Config + maxBlobChunkSize int64 + maxBlobSize int64 +} + +// buildProxyConfig parses size-related config fields and builds the proxy config. +func buildProxyConfig(cfg Config, log zerowrap.Logger) (*proxyConfigResult, error) { + maxProxyBodySize := int64(512 << 20) // 512MB default + if cfg.Server.MaxProxyBodySize != "" { + parsedSize, err := bytesize.Parse(cfg.Server.MaxProxyBodySize) + if err != nil { + return nil, log.WrapErrWithFields(err, "invalid server.max_proxy_body_size configuration", map[string]any{"value": cfg.Server.MaxProxyBodySize}) + } + maxProxyBodySize = parsedSize + } + + maxBlobChunkSize := int64(registry.DefaultMaxBlobChunkSize) + if cfg.Server.MaxBlobChunkSize != "" { + parsedSize, err := bytesize.Parse(cfg.Server.MaxBlobChunkSize) + if err != nil { + return nil, log.WrapErrWithFields(err, "invalid server.max_blob_chunk_size configuration", map[string]any{"value": cfg.Server.MaxBlobChunkSize}) + } + maxBlobChunkSize = parsedSize + } + + maxBlobSize := int64(registry.DefaultMaxBlobSize) + if cfg.Server.MaxBlobSize != "" { + parsedSize, err := bytesize.Parse(cfg.Server.MaxBlobSize) + if err != nil { + return nil, log.WrapErrWithFields(err, "invalid server.max_blob_size configuration", map[string]any{"value": cfg.Server.MaxBlobSize}) + } + maxBlobSize = parsedSize + } + + maxProxyResponseSize := int64(1 << 30) // 1GB default + if cfg.Server.MaxProxyResponseSize != "" { + parsedSize, err := bytesize.Parse(cfg.Server.MaxProxyResponseSize) + if err != nil { + return nil, log.WrapErrWithFields(err, "invalid server.max_proxy_response_size configuration", map[string]any{"value": cfg.Server.MaxProxyResponseSize}) + } + maxProxyResponseSize = parsedSize + } + + maxConcurrentConns := cfg.Server.MaxConcurrentConns + if maxConcurrentConns < 0 { + maxConcurrentConns = 10000 // default when explicitly set to -1 + } + // 0 means no limit (as documented in proxy.Config) + + registryDomain, _ := resolveRegistryDomains(cfg) + + return &proxyConfigResult{ + proxyConfig: proxy.Config{ + RegistryDomain: registryDomain, + RegistryPort: cfg.Server.RegistryPort, + MaxBodySize: maxProxyBodySize, + MaxResponseSize: maxProxyResponseSize, + MaxConcurrentConns: maxConcurrentConns, + }, + maxBlobChunkSize: maxBlobChunkSize, + maxBlobSize: maxBlobSize, + }, nil +} + +// buildDNSConfig parses the raw dns config section into a publictls.DNSConfig. +func buildDNSConfig(cfg Config) (publictls.DNSConfig, error) { + defaults := publictls.DefaultDNSConfig() + + resolvers := cfg.DNS.Resolvers + if len(resolvers) == 0 { + resolvers = defaults.Resolvers + } + + propagationTimeout := defaults.PropagationTimeout + if cfg.DNS.PropagationTimeout != "" { + parsed, err := time.ParseDuration(cfg.DNS.PropagationTimeout) + if err != nil { + return publictls.DNSConfig{}, fmt.Errorf("invalid dns.propagation_timeout: %w", err) + } + propagationTimeout = parsed + } + + pollingInterval := defaults.PollingInterval + if cfg.DNS.PollingInterval != "" { + parsed, err := time.ParseDuration(cfg.DNS.PollingInterval) + if err != nil { + return publictls.DNSConfig{}, fmt.Errorf("invalid dns.polling_interval: %w", err) + } + pollingInterval = parsed + } + + dnsCfg := publictls.DNSConfig{ + Resolvers: append([]string(nil), resolvers...), + PropagationTimeout: propagationTimeout, + PollingInterval: pollingInterval, + } + if err := dnsCfg.Validate(); err != nil { + return publictls.DNSConfig{}, err + } + return dnsCfg, nil +} + +func buildContainerServiceConfig(_ context.Context, v *viper.Viper, _ Config, _ *services, _ zerowrap.Logger) (container.Config, error) { + return container.Config{ + NetworkPrefix: v.GetString("network_isolation.network_prefix"), + }, nil +} + +// createContainerService creates the container service with configuration. +func createContainerService(ctx context.Context, v *viper.Viper, cfg Config, svc *services, log zerowrap.Logger) (*container.Service, error) { + containerConfig, err := buildContainerServiceConfig(ctx, v, cfg, svc, log) + if err != nil { + return nil, err + } + return container.NewService(svc.runtime, svc.eventBus, svc.logWriter, containerConfig), nil +} + +// registerEventHandlers registers event handlers. Push-triggered +// route creation (auto-route, previews, image-pushed deploy) is +// retired: pushes transfer OCI content only and apps deploy +// explicitly via `gordon apps deploy`. The pre-v3 route-engine +// deploy arms (config-reload redeploy, manual SIGUSR2 deploy, +// secrets-changed redeploy) are removed with the declarative-apps +// cutover: reload never activates app state, and app secret changes +// take effect on explicit deploy. +func registerEventHandlers(ctx context.Context, svc *services) (func(), error) { + // Proxy cache invalidation on config reload (clears stale targets for removed routes) + configReloadProxyHandler := proxy.NewConfigReloadProxyHandler(ctx, svc.proxySvc) + if err := svc.eventBus.Subscribe(configReloadProxyHandler); err != nil { + return nil, fmt.Errorf("failed to subscribe config reload proxy handler: %w", err) + } + + cleanup := func() { + if svc.reloadCoordinator != nil { + svc.reloadCoordinator.Stop() + } + } + + return cleanup, nil +} diff --git a/internal/app/shutdown.go b/internal/app/shutdown.go new file mode 100644 index 000000000..46c86afbd --- /dev/null +++ b/internal/app/shutdown.go @@ -0,0 +1,187 @@ +package app + +import ( + "context" + "fmt" + "net/http" + "os" + "time" + + "github.com/bnema/zerowrap" + + trafficadapter "github.com/bnema/gordon/internal/adapters/in/traffic" + "github.com/bnema/gordon/internal/boundaries/in" + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/usecase/container" + pkiusecase "github.com/bnema/gordon/internal/usecase/pki" + "github.com/bnema/gordon/internal/usecase/proxy" +) + +// waitForShutdown blocks on the event loop, handling server errors and +// Unix signals (reload, deploy, shutdown) until the context is cancelled. +func waitForShutdown(ctx context.Context, errChan <-chan error, reloadChan <-chan os.Signal, reload reloadTrigger, eventBus out.EventBus, log zerowrap.Logger) { + for { + select { + case err := <-errChan: + log.Error().Err(err).Msg("server error") + return + case <-reloadChan: + log.Info().Msg("reload signal received (SIGUSR1)") + _ = reload.Trigger(ctx) + case <-ctx.Done(): + log.Info().Msg("shutdown signal received") + return + } + } +} + +// appAdministration is the daemon-owned application lifecycle torn down +// during graceful shutdown: it refuses new background deploys, cancels +// in-flight executions on the daemon context, and joins them. +type appAdministration interface { + Shutdown(context.Context) error +} + +// quiesceAppAdministration cancels and joins daemon-owned app work on the +// bounded shutdown context. A non-nil error means the bound expired while an +// execution was still unwinding: the caller must stop tearing down the state +// and runtime that execution still uses. +func quiesceAppAdministration(ctx context.Context, appAdmin appAdministration, log zerowrap.Logger) error { + if appAdmin == nil { + return nil + } + if err := appAdmin.Shutdown(ctx); err != nil { + log.Error().Err(err).Msg("app administration did not quiesce before the shutdown deadline") + return err + } + return nil +} + +// gracefulShutdown stops HTTP servers with a 30s timeout, then shuts down +// the container service and cleans up runtime files. It returns an error when +// app administration could not quiesce: in that case the remaining teardown +// is skipped and the error is propagated to the process entry point, which +// exits non-zero instead of closing state or runtime under an in-flight +// ExecuteDeploy. +func gracefulShutdown(registrySrvs []*http.Server, proxySrv, tlsSrv *http.Server, containerSvc *container.Service, proxySvc *proxy.Service, pkiSvc *pkiusecase.Service, publicTLS in.PublicTLSService, trafficManager *trafficadapter.Manager, monitor *appMonitor, appAdmin appAdministration, appState out.AppState, log zerowrap.Logger) error { + log.Info().Msg("shutting down Gordon...") + + // Phase 0: stop periodic app recovery first and wait for any in-flight + // bounded pass, so no reconcile can race runtime, traffic, or state + // teardown (and no goroutine is left behind). + monitor.Stop() + + shutdownCtx, shutdownCancel := context.WithTimeout(context.Background(), 30*time.Second) + defer shutdownCancel() + + // Phase 0.5: quiesce daemon-owned app administration. New background + // deploys are refused and in-flight executions are cancelled on the + // daemon context and joined here — before traffic, runtime, or the app + // state store is torn down, so no execution can write to closed state. + // A timeout is fail-closed: the unfinished execution still owns the + // state and runtime it uses, so teardown stops here and the error makes + // the process exit non-zero rather than close resources underneath it. + if err := quiesceAppAdministration(shutdownCtx, appAdmin, log); err != nil { + return fmt.Errorf("app administration quiescence: %w", err) + } + + // Phase 1: Stop ingress frontends (TLS, then proxy) — no new traffic accepted + shutdownHTTPServers(shutdownCtx, log, tlsSrv, proxySrv) + + if trafficManager != nil { + if err := trafficManager.Shutdown(shutdownCtx); err != nil { + log.Warn().Err(err).Msg("traffic manager shutdown error") + } + } + + // Stop PKI maintenance goroutines + if pkiSvc != nil { + pkiSvc.Stop() + } + + // Stop public ACME TLS renewal loop + if publicTLS != nil { + if err := publicTLS.Stop(shutdownCtx); err != nil { + log.Warn().Err(err).Msg("public TLS stop error") + } + } + + // Phase 2: Drain in-flight registry push sessions before stopping the backend + if proxySvc != nil { + log.Info().Msg("draining in-flight registry requests...") + if drained := proxySvc.DrainRegistryInFlight(25 * time.Second); !drained { + log.Warn().Int64("in_flight", proxySvc.RegistryInFlight()).Msg("registry drain timed out; some in-flight pushes may be interrupted") + } + } + + // Phase 3: Stop the registry backends. + shutdownHTTPServers(shutdownCtx, log, registrySrvs...) + + if err := containerSvc.Shutdown(shutdownCtx); err != nil { + log.Warn().Err(err).Msg("error during container shutdown") + } + closeAppState(appState, log) + + cleanupInternalCredentials() + log.Info().Msg("Gordon stopped") + return nil +} + +// shutdownHTTPServers gracefully stops each non-nil server on the shutdown +// context, logging per-server failures so one listener cannot mask another. +func shutdownHTTPServers(ctx context.Context, log zerowrap.Logger, servers ...*http.Server) { + for _, srv := range servers { + if srv == nil { + continue + } + if err := srv.Shutdown(ctx); err != nil { + log.Warn().Err(err).Str("addr", srv.Addr).Msg("server shutdown error") + } + } +} + +func closeAppState(state out.AppState, log zerowrap.Logger) { + if state == nil { + return + } + if err := state.Close(); err != nil { + log.Warn().Err(err).Msg("app state close error") + } +} + +// shutdownStartedServers gracefully shuts down listeners that bound before a +// partial startup failure so an error return does not leak them. +func shutdownStartedServers(servers []*http.Server, log zerowrap.Logger) { + shutdownCtx, shutdownCancel := context.WithTimeout(context.Background(), 5*time.Second) + defer shutdownCancel() + for _, srv := range servers { + if srv != nil { + if err := srv.Shutdown(shutdownCtx); err != nil { + log.Error().Err(err).Msg("failed to shut down server during startup cleanup") + } + } + } +} + +// cleanupStartupResources releases every resource a partial startup may have +// acquired once the local admin socket is serving: the admin socket first (no +// further mutations can begin), then the bound servers, then the traffic +// manager that startProxyServers may already have populated. +func cleanupStartupResources(localAdmin *localAdminServer, trafficManager *trafficadapter.Manager, log zerowrap.Logger, servers ...*http.Server) { + if localAdmin != nil { + localAdmin.Close() + } + shutdownStartedServers(servers, log) + shutdownTrafficManagerForStartupCleanup(trafficManager, log) +} + +func shutdownTrafficManagerForStartupCleanup(manager *trafficadapter.Manager, log zerowrap.Logger) { + if manager == nil { + return + } + shutdownCtx, shutdownCancel := context.WithTimeout(context.Background(), 5*time.Second) + defer shutdownCancel() + if err := manager.Shutdown(shutdownCtx); err != nil { + log.Error().Err(err).Msg("failed to shut down traffic manager during startup cleanup") + } +} diff --git a/internal/app/start_proxy_servers_test.go b/internal/app/start_proxy_servers_test.go index 80b5f6eff..2872315bf 100644 --- a/internal/app/start_proxy_servers_test.go +++ b/internal/app/start_proxy_servers_test.go @@ -8,18 +8,18 @@ import ( "io" "net" "net/http" + "path/filepath" "strconv" "testing" "time" "github.com/bnema/zerowrap" - "github.com/stretchr/testify/mock" "github.com/stretchr/testify/require" trafficadapter "github.com/bnema/gordon/internal/adapters/in/traffic" + "github.com/bnema/gordon/internal/adapters/localadmin" pkiadapter "github.com/bnema/gordon/internal/adapters/out/pki" inmocks "github.com/bnema/gordon/internal/boundaries/in/mocks" - outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" "github.com/bnema/gordon/internal/domain" pkiusecase "github.com/bnema/gordon/internal/usecase/pki" traffic "github.com/bnema/gordon/internal/usecase/traffic" @@ -30,7 +30,7 @@ func TestWaitForServerReady_NilReadyReturnsImmediately(t *testing.T) { done := make(chan error, 1) go func() { - done <- waitForServerReady(nil, errChan) + done <- waitForServerReady(context.Background(), nil, errChan) }() select { @@ -46,10 +46,50 @@ func TestWaitForServerReady_NonNilReadyPreservesErrorBehavior(t *testing.T) { errChan := make(chan error, 1) errChan <- expected - err := waitForServerReady(make(chan struct{}), errChan) + err := waitForServerReady(context.Background(), make(chan struct{}), errChan) require.ErrorIs(t, err, expected) } +func TestWaitForServerReady_ReturnsOnContextCancellation(t *testing.T) { + ctx, cancel := context.WithCancel(context.Background()) + cancel() + + // A canceled startup context must unblock readiness instead of waiting on a + // server that will never signal ready. + err := waitForServerReady(ctx, make(chan struct{}), make(chan error)) + require.ErrorIs(t, err, context.Canceled) +} + +func TestCleanupStartupResources_ClosesAdminSocketsServersAndTrafficManager(t *testing.T) { + xdg := t.TempDir() + t.Setenv("XDG_RUNTIME_DIR", xdg) + + localAdmin, err := startLocalAdminServer(localAdminTestServices(t, inmocks.NewMockAppService(t)), make(chan error, 4), zerowrap.Default()) + require.NoError(t, err) + require.NotNil(t, localAdmin) + socketPath := localadmin.SocketPath(filepath.Join(xdg, "gordon")) + require.NoError(t, localadmin.ValidateSocket(socketPath)) + + ln, err := net.Listen("tcp", "127.0.0.1:0") + require.NoError(t, err) + started := make(chan struct{}) + srv := &http.Server{Handler: http.HandlerFunc(func(http.ResponseWriter, *http.Request) {})} + go func() { + close(started) + _ = srv.Serve(ln) + }() + <-started + addr := ln.Addr().String() + + manager := trafficadapter.NewManager() + cleanupStartupResources(localAdmin, manager, zerowrap.Default(), srv) + + require.NoFileExists(t, socketPath, "local admin socket must be closed") + _, err = net.DialTimeout("tcp", addr, 200*time.Millisecond) + require.Error(t, err, "bound server must be shut down") + require.Empty(t, manager.Status().EntryPoints) +} + func TestStartProxyServers_DoesNotStartLegacyHTTPListenerFromServerPort(t *testing.T) { cfg := Config{} cfg.Server.Port = freeTCPPort(t) @@ -129,9 +169,8 @@ func TestStartProxyServers_ConfiguresSmartTCPHTTPAndHTTPSRoutesWithoutPublicHTTP assertTCPPortClosed(t, fmt.Sprintf("127.0.0.1:%d", cfg.Server.Port)) configSvc := inmocks.NewMockConfigService(t) - configSvc.EXPECT().GetRoutes(context.Background()).Return([]domain.Route{{Domain: "app.example.com", HTTPS: true}}) configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{}) - require.NoError(t, applyTrafficRuntimeConfig(context.Background(), manager, cfg, configSvc)) + require.NoError(t, applyTrafficRuntimeConfig(context.Background(), manager, cfg, configSvc, stubHosts("app.example.com"))) _, smartPortText, err := net.SplitHostPort(cfg.EntryPoints[traffic.DefaultEdgeEntryPointName].Address) require.NoError(t, err) @@ -170,9 +209,8 @@ func TestStartProxyServers_ConfiguresTrafficManagerHTTPSRoute(t *testing.T) { require.Nil(t, tlsReady) configSvc := inmocks.NewMockConfigService(t) - configSvc.EXPECT().GetRoutes(context.Background()).Return([]domain.Route{{Domain: "app.example.com", HTTPS: true}}) configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{}) - require.NoError(t, applyTrafficRuntimeConfig(context.Background(), manager, cfg, configSvc)) + require.NoError(t, applyTrafficRuntimeConfig(context.Background(), manager, cfg, configSvc, stubHosts("app.example.com"))) _, portText, err := net.SplitHostPort(cfg.EntryPoints[traffic.DefaultEdgeEntryPointName].Address) require.NoError(t, err) @@ -212,9 +250,8 @@ func TestStartProxyServers_ConfiguresCustomTLSMuxHTTPSRoute(t *testing.T) { require.Nil(t, tlsReady) configSvc := inmocks.NewMockConfigService(t) - configSvc.EXPECT().GetRoutes(context.Background()).Return(nil) configSvc.EXPECT().GetExternalRoutes().Return(nil) - require.NoError(t, applyTrafficRuntimeConfig(context.Background(), manager, cfg, configSvc)) + require.NoError(t, applyTrafficRuntimeConfig(context.Background(), manager, cfg, configSvc, stubHosts())) _, customPort, err := net.SplitHostPort(cfg.EntryPoints["custom-secure"].Address) require.NoError(t, err) @@ -282,9 +319,7 @@ func TestStartProxyServers_TLSRequiresTrafficManager(t *testing.T) { func newTestPKIService(t *testing.T) *pkiusecase.Service { t.Helper() - routes := outmocks.NewMockRouteChecker(t) - routes.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{{Domain: "app.example.com", HTTPS: true}}).Maybe() - routes.EXPECT().GetExternalRoutes().Return(map[string]string{}).Maybe() + routes := newRoutesMock(t, "app.example.com") ca, err := pkiadapter.NewCA(t.TempDir(), zerowrap.Default()) require.NoError(t, err) pkiSvc := pkiusecase.NewService(context.Background(), ca, routes, nil, zerowrap.Default()) diff --git a/internal/app/startup_recovery.go b/internal/app/startup_recovery.go deleted file mode 100644 index 8425a9470..000000000 --- a/internal/app/startup_recovery.go +++ /dev/null @@ -1,44 +0,0 @@ -package app - -import ( - "context" - - "github.com/bnema/zerowrap" - - "github.com/bnema/gordon/internal/domain" -) - -// startupConfigService defines the route configuration needed during startup -// recovery. It intentionally excludes auto-route settings because reboot -// recovery always works from configured routes. -type startupConfigService interface { - GetRoutes(ctx context.Context) []domain.Route -} - -// startupContainerService defines the container lifecycle operations needed -// during startup recovery. -type startupContainerService interface { - SyncContainers(ctx context.Context) error - AutoStart(ctx context.Context, routes []domain.Route) error - StartMonitor(ctx context.Context) -} - -// syncAndRecoverConfiguredRoutes performs best-effort startup recovery for -// configured routes after listeners are ready. -func syncAndRecoverConfiguredRoutes( - ctx context.Context, - configSvc startupConfigService, - containerSvc startupContainerService, - log zerowrap.Logger, -) { - defer containerSvc.StartMonitor(ctx) - - if err := containerSvc.SyncContainers(ctx); err != nil { - log.Warn().Err(err).Msg("failed to sync existing containers") - } - - routes := configSvc.GetRoutes(ctx) - if err := containerSvc.AutoStart(domain.WithInternalDeploy(ctx), routes); err != nil { - log.Warn().Err(err).Msg("failed to auto-start configured routes") - } -} diff --git a/internal/app/startup_recovery_test.go b/internal/app/startup_recovery_test.go deleted file mode 100644 index 812ce5ad4..000000000 --- a/internal/app/startup_recovery_test.go +++ /dev/null @@ -1,119 +0,0 @@ -package app - -import ( - "context" - "testing" - - "github.com/bnema/zerowrap" - "github.com/stretchr/testify/assert" - - "github.com/bnema/gordon/internal/domain" -) - -type startupRecoveryFakeConfigService struct { - calls *[]string - routes []domain.Route -} - -var _ startupConfigService = (*startupRecoveryFakeConfigService)(nil) - -func (f *startupRecoveryFakeConfigService) GetRoutes(_ context.Context) []domain.Route { - *f.calls = append(*f.calls, "routes") - return append([]domain.Route(nil), f.routes...) -} - -type startupRecoveryFakeContainerService struct { - calls *[]string - syncErr error - autoStartErr error - autoStartCtx context.Context - autoStartRoutes []domain.Route - startMonitorCtx context.Context -} - -var _ startupContainerService = (*startupRecoveryFakeContainerService)(nil) - -func (f *startupRecoveryFakeContainerService) SyncContainers(_ context.Context) error { - *f.calls = append(*f.calls, "sync") - return f.syncErr -} - -func (f *startupRecoveryFakeContainerService) AutoStart(ctx context.Context, routes []domain.Route) error { - *f.calls = append(*f.calls, "autostart") - f.autoStartCtx = ctx - f.autoStartRoutes = append([]domain.Route(nil), routes...) - return f.autoStartErr -} - -func (f *startupRecoveryFakeContainerService) StartMonitor(ctx context.Context) { - *f.calls = append(*f.calls, "monitor") - f.startMonitorCtx = ctx -} - -func TestSyncAndRecoverConfiguredRoutes_HappyPath(t *testing.T) { - ctx := context.Background() - routes := []domain.Route{{Domain: "app.example.com", Image: "reg.example.com/app:latest", HTTPS: true}} - calls := make([]string, 0, 4) - - configSvc := &startupRecoveryFakeConfigService{ - calls: &calls, - routes: routes, - } - containerSvc := &startupRecoveryFakeContainerService{calls: &calls} - - syncAndRecoverConfiguredRoutes(ctx, configSvc, containerSvc, zerowrap.Default()) - - assert.Equal(t, []string{"sync", "routes", "autostart", "monitor"}, calls) - assert.Equal(t, routes, containerSvc.autoStartRoutes) - assert.True(t, domain.IsInternalDeploy(containerSvc.autoStartCtx)) - assert.False(t, domain.IsInternalDeploy(ctx)) - assert.False(t, domain.IsInternalDeploy(containerSvc.startMonitorCtx)) -} - -func TestSyncAndRecoverConfiguredRoutes_SyncFailureStillRecoversAndStartsMonitor(t *testing.T) { - ctx := context.Background() - routes := []domain.Route{{Domain: "app.example.com", Image: "reg.example.com/app:latest", HTTPS: true}} - calls := make([]string, 0, 4) - - configSvc := &startupRecoveryFakeConfigService{ - calls: &calls, - routes: routes, - } - containerSvc := &startupRecoveryFakeContainerService{ - calls: &calls, - syncErr: assert.AnError, - } - - assert.NotPanics(t, func() { - syncAndRecoverConfiguredRoutes(ctx, configSvc, containerSvc, zerowrap.Default()) - }) - - assert.Equal(t, []string{"sync", "routes", "autostart", "monitor"}, calls) - assert.Equal(t, routes, containerSvc.autoStartRoutes) - assert.True(t, domain.IsInternalDeploy(containerSvc.autoStartCtx)) - assert.False(t, domain.IsInternalDeploy(containerSvc.startMonitorCtx)) -} - -func TestSyncAndRecoverConfiguredRoutes_AutoStartFailureStillStartsMonitor(t *testing.T) { - ctx := context.Background() - routes := []domain.Route{{Domain: "app.example.com", Image: "reg.example.com/app:latest", HTTPS: true}} - calls := make([]string, 0, 4) - - configSvc := &startupRecoveryFakeConfigService{ - calls: &calls, - routes: routes, - } - containerSvc := &startupRecoveryFakeContainerService{ - calls: &calls, - autoStartErr: assert.AnError, - } - - assert.NotPanics(t, func() { - syncAndRecoverConfiguredRoutes(ctx, configSvc, containerSvc, zerowrap.Default()) - }) - - assert.Equal(t, []string{"sync", "routes", "autostart", "monitor"}, calls) - assert.Equal(t, routes, containerSvc.autoStartRoutes) - assert.True(t, domain.IsInternalDeploy(containerSvc.autoStartCtx)) - assert.False(t, domain.IsInternalDeploy(containerSvc.startMonitorCtx)) -} diff --git a/internal/app/traffic.go b/internal/app/traffic.go index 66d942a05..1d14eeeed 100644 --- a/internal/app/traffic.go +++ b/internal/app/traffic.go @@ -6,12 +6,22 @@ import ( trafficadapter "github.com/bnema/gordon/internal/adapters/in/traffic" "github.com/bnema/gordon/internal/boundaries/in" + "github.com/bnema/gordon/internal/boundaries/out" "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/apptraffic" servicecfg "github.com/bnema/gordon/internal/usecase/services" trafficbuilder "github.com/bnema/gordon/internal/usecase/traffic" ) -func applyTrafficRuntimeConfig(ctx context.Context, manager *trafficadapter.Manager, cfg Config, configSvc in.ConfigService) error { +// appHostRoutes adapts ACTIVE-derived hosts to the traffic builder's +// route input: host-only entries, Image explicitly unused (no app +// meaning). Stopped-intent and unresolved hosts never reach the graph. +type appHostRoutes interface { + AppHosts() []out.AppHost + L4Entries() []apptraffic.RouteEntry +} + +func applyTrafficRuntimeConfig(ctx context.Context, manager *trafficadapter.Manager, cfg Config, configSvc in.ConfigService, hosts appHostRoutes) error { if manager == nil || configSvc == nil { return nil } @@ -22,7 +32,7 @@ func applyTrafficRuntimeConfig(ctx context.Context, manager *trafficadapter.Mana graph, err := trafficbuilder.Build(trafficbuilder.Input{ EntryPoints: cfg.EntryPoints, Traffic: cfg.Traffic, - Routes: configSvc.GetRoutes(ctx), + Routes: appHostDomainRoutes(hosts), ExternalRoutes: configSvc.GetExternalRoutes(), NetworkServices: cfg.NetworkServices, Services: standaloneServices, @@ -30,6 +40,9 @@ func applyTrafficRuntimeConfig(ctx context.Context, manager *trafficadapter.Mana if err != nil { return fmt.Errorf("build traffic graph: %w", err) } + if err := addAppL4Routes(&graph, hosts); err != nil { + return fmt.Errorf("add app L4 routes: %w", err) + } owned, err := trafficRuntimeGraph(graph) if err != nil { return fmt.Errorf("filter traffic graph for runtime ownership: %w", err) @@ -40,6 +53,91 @@ func applyTrafficRuntimeConfig(ctx context.Context, manager *trafficadapter.Mana return nil } +func addAppL4Routes(graph *domain.TrafficGraph, routes appHostRoutes) error { + if routes == nil { + return nil + } + existing := map[string]struct{}{} + for _, router := range graph.Routers { + existing["router:"+router.Name] = struct{}{} + } + for _, service := range graph.Services { + existing["service:"+service.Name] = struct{}{} + } + entrypointAddress := make(map[string]string, len(graph.EntryPoints)) + for _, entryPoint := range graph.EntryPoints { + entrypointAddress[entryPoint.Name] = entryPoint.Address + } + for _, entry := range routes.L4Entries() { + protocol := domain.RouterProtocolTCP + backendProtocol := domain.NetworkProtocolTCP + if entry.Kind == "udp" { + protocol = domain.RouterProtocolUDP + backendProtocol = domain.NetworkProtocolUDP + } + if !entry.Backend.Resolved() { + return fmt.Errorf("app L4 route %q has no resolved backend", entry.RouterName) + } + // The declared bind must be the listener this router actually + // attaches to: the runtime binds the entrypoint address, so a + // narrower declaration must have been rejected earlier, never + // silently widened here. + address, ok := entrypointAddress[entry.Entrypoint] + if !ok { + return fmt.Errorf("app L4 route %q references entrypoint %q absent from the traffic graph", entry.RouterName, entry.Entrypoint) + } + listenerHost, listenerPort, err := domain.SplitListenerAddress(address) + if err != nil { + return fmt.Errorf("app L4 entrypoint %q has invalid listener address %q: %w", entry.Entrypoint, address, err) + } + if listenerPort != entry.BindPort || domain.CanonicalBindHost(listenerHost) != domain.CanonicalBindHost(entry.BindIP) { + return fmt.Errorf( + "app L4 route %q declares bind %s:%d but entrypoint %q binds %s", + entry.RouterName, entry.BindIP, entry.BindPort, entry.Entrypoint, address, + ) + } + // service:--:- satisfies the + // service-ref grammar (no colon in the port name) and stays + // unique per interface port. Historical builder services + // never collide: they use route:/external_route:/ + // network_service: kinds or standalone names. + serviceName := fmt.Sprintf("service:%s--%s:%s-%d", entry.App, entry.Service, entry.Kind, entry.Backend.ContainerPort) + if _, ok := existing["router:"+entry.RouterName]; ok { + return fmt.Errorf("app L4 router %q collides with a configured route", entry.RouterName) + } + if _, ok := existing["service:"+serviceName]; ok { + return fmt.Errorf("app L4 service %q collides with a configured service", serviceName) + } + existing["router:"+entry.RouterName] = struct{}{} + existing["service:"+serviceName] = struct{}{} + graph.Routers = append(graph.Routers, domain.TrafficRouter{ + Name: entry.RouterName, EntryPoint: entry.Entrypoint, + Protocol: protocol, Service: serviceName, + }) + graph.Services = append(graph.Services, domain.TrafficService{ + Name: serviceName, + Backends: []domain.TrafficBackend{{ + Name: entry.App + "--" + entry.Service, + Host: entry.Backend.Host, Port: entry.Backend.Port, Protocol: backendProtocol, + }}, + }) + } + return graph.Validate() +} + +// appHostDomainRoutes maps served hosts to builder route inputs. +func appHostDomainRoutes(hosts appHostRoutes) []domain.Route { + if hosts == nil { + return nil + } + appHosts := hosts.AppHosts() + routes := make([]domain.Route, 0, len(appHosts)) + for _, h := range appHosts { + routes = append(routes, domain.Route{Domain: h.Host}) + } + return routes +} + func trafficRuntimeGraph(graph domain.TrafficGraph) (domain.TrafficGraph, error) { ownedEntryPoints := map[string]struct{}{} filtered := domain.TrafficGraph{Options: graph.Options} diff --git a/internal/app/traffic_test.go b/internal/app/traffic_test.go index a1a387170..c94376569 100644 --- a/internal/app/traffic_test.go +++ b/internal/app/traffic_test.go @@ -10,7 +10,9 @@ import ( trafficadapter "github.com/bnema/gordon/internal/adapters/in/traffic" inmocks "github.com/bnema/gordon/internal/boundaries/in/mocks" + "github.com/bnema/gordon/internal/boundaries/out" "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/apptraffic" servicecfg "github.com/bnema/gordon/internal/usecase/services" "github.com/bnema/gordon/internal/usecase/traffic" ) @@ -87,6 +89,15 @@ func TestTrafficRuntimeGraphAllowsTLSPassthroughOnDefaultTLS(t *testing.T) { assert.Equal(t, "raw", filtered.Routers[0].Name) } +// stubHosts builds an ACTIVE-derived host source for traffic tests. +func stubHosts(domains ...string) *stubAppRoutes { + hosts := make([]out.AppHost, 0, len(domains)) + for _, d := range domains { + hosts = append(hosts, out.AppHost{Host: d}) + } + return &stubAppRoutes{hosts: hosts} +} + func TestApplyTrafficRuntimeConfigAppliesSmartTCPEntrypointWithRawFallbackPolicy(t *testing.T) { manager := trafficadapter.NewManager() defer func() { require.NoError(t, manager.Shutdown(context.Background())) }() @@ -107,10 +118,9 @@ func TestApplyTrafficRuntimeConfigAppliesSmartTCPEntrypointWithRawFallbackPolicy cfg.Traffic.TCP.Routers = []traffic.RouterConfig{{Name: "ssh", EntryPoint: traffic.DefaultEdgeEntryPointName, Service: "network_service:ssh:ssh"}} configSvc := inmocks.NewMockConfigService(t) - configSvc.EXPECT().GetRoutes(context.Background()).Return([]domain.Route{{Domain: "app.example.com", HTTPS: true}}) configSvc.EXPECT().GetExternalRoutes().Return(nil) - require.NoError(t, applyTrafficRuntimeConfig(context.Background(), manager, cfg, configSvc)) + require.NoError(t, applyTrafficRuntimeConfig(context.Background(), manager, cfg, configSvc, stubHosts("app.example.com"))) status := manager.Status() require.Len(t, status.EntryPoints, 1) assert.Equal(t, traffic.DefaultEdgeEntryPointName, status.EntryPoints[0].Name) @@ -135,10 +145,9 @@ func TestApplyTrafficRuntimeConfigPassesStandaloneServicesToBuilder(t *testing.T cfg.Traffic.UDP.Routers = []traffic.RouterConfig{{Name: "rust-game", EntryPoint: "rust", Service: "service:rust:game"}} configSvc := inmocks.NewMockConfigService(t) - configSvc.EXPECT().GetRoutes(context.Background()).Return(nil) configSvc.EXPECT().GetExternalRoutes().Return(nil) - require.NoError(t, applyTrafficRuntimeConfig(context.Background(), manager, cfg, configSvc)) + require.NoError(t, applyTrafficRuntimeConfig(context.Background(), manager, cfg, configSvc, stubHosts())) status := manager.Status() require.Len(t, status.EntryPoints, 1) assert.Equal(t, "rust", status.EntryPoints[0].Name) @@ -160,16 +169,119 @@ func TestApplyTrafficRuntimeConfigAppliesCustomL4Entrypoint(t *testing.T) { cfg.Traffic.TCP.Routers = []traffic.RouterConfig{{Name: "postgres", EntryPoint: "postgres", Service: "network_service:postgres:db"}} configSvc := inmocks.NewMockConfigService(t) - configSvc.EXPECT().GetRoutes(context.Background()).Return(nil) configSvc.EXPECT().GetExternalRoutes().Return(nil) - require.NoError(t, applyTrafficRuntimeConfig(context.Background(), manager, cfg, configSvc)) + require.NoError(t, applyTrafficRuntimeConfig(context.Background(), manager, cfg, configSvc, stubHosts())) status := manager.Status() require.Len(t, status.EntryPoints, 1) assert.Equal(t, "postgres", status.EntryPoints[0].Name) assert.True(t, status.EntryPoints[0].Active) } +// stubL4Hosts is an ACTIVE-derived source with resolved TCP+UDP entries. +type stubL4Hosts struct { + stubAppRoutes + entries []apptraffic.RouteEntry +} + +func (s *stubL4Hosts) L4Entries() []apptraffic.RouteEntry { return s.entries } + +// TestApplyTrafficRuntimeConfigFusesAppL4Routes proves ACTIVE TCP+UDP +// projections join the applied graph as validated router/service pairs +// alongside historical routes, with exact-protocol loopback backends. +func TestApplyTrafficRuntimeConfigFusesAppL4Routes(t *testing.T) { + manager := trafficadapter.NewManager() + defer func() { require.NoError(t, manager.Shutdown(context.Background())) }() + tcpAddress := freeTCPAddress(t) + udpAddress := freeUDPAddress(t) + tcpHost, tcpPort, err := domain.SplitListenerAddress(tcpAddress) + require.NoError(t, err) + udpHost, udpPort, err := domain.SplitListenerAddress(udpAddress) + require.NoError(t, err) + cfg := Config{ + EntryPoints: map[string]traffic.EntryPointConfig{ + "tcp": {Address: tcpAddress, Protocol: domain.EntryPointProtocolTCP}, + "udp": {Address: udpAddress, Protocol: domain.EntryPointProtocolUDP}, + }, + } + hosts := &stubL4Hosts{ + entries: []apptraffic.RouteEntry{ + { + Kind: "tcp", RouterName: "app-game--server--tcp-9000", Entrypoint: "tcp", + BindIP: tcpHost, BindPort: tcpPort, + App: "game", Service: "server", + Backend: domain.AppBackend{Host: "127.0.0.1", Port: 19000, ContainerPort: 9000, ContainerID: "c-game"}, + }, + { + Kind: "udp", RouterName: "app-game--server--udp-9000", Entrypoint: "udp", + BindIP: udpHost, BindPort: udpPort, + App: "game", Service: "server", + Backend: domain.AppBackend{Host: "127.0.0.1", Port: 19001, ContainerPort: 9000, ContainerID: "c-game"}, + }, + }, + } + + configSvc := inmocks.NewMockConfigService(t) + configSvc.EXPECT().GetExternalRoutes().Return(nil) + + require.NoError(t, applyTrafficRuntimeConfig(context.Background(), manager, cfg, configSvc, hosts)) + status := manager.Status() + assert.Equal(t, "ok", status.LastReloadStatus) + routers := map[string]domain.TrafficRouterStatus{} + for _, router := range status.Routers { + routers[router.Name] = router + } + require.Contains(t, routers, "app-game--server--tcp-9000") + require.Contains(t, routers, "app-game--server--udp-9000") + assert.Equal(t, domain.RouterProtocolTCP, routers["app-game--server--tcp-9000"].Protocol) + assert.Equal(t, domain.RouterProtocolUDP, routers["app-game--server--udp-9000"].Protocol) + assert.True(t, routers["app-game--server--tcp-9000"].Active) + assert.True(t, routers["app-game--server--udp-9000"].Active) + services := map[string]domain.TrafficServiceStatus{} + for _, service := range status.Services { + services[service.Name] = service + } + tcpSvc, ok := services["service:game--server:tcp-9000"] + require.True(t, ok) + require.Len(t, tcpSvc.Backends, 1) + assert.Equal(t, "127.0.0.1", tcpSvc.Backends[0].Host) + assert.Equal(t, 19000, tcpSvc.Backends[0].Port) + assert.Equal(t, domain.NetworkProtocolTCP, tcpSvc.Backends[0].Protocol) + udpSvc, ok := services["service:game--server:udp-9000"] + require.True(t, ok) + require.Len(t, udpSvc.Backends, 1) + assert.Equal(t, "127.0.0.1", udpSvc.Backends[0].Host) + assert.Equal(t, 19001, udpSvc.Backends[0].Port) + assert.Equal(t, domain.NetworkProtocolUDP, udpSvc.Backends[0].Protocol) +} + +// TestApplyTrafficRuntimeConfigRejectsMismatchedAppL4Bind proves the graph +// stage refuses a projected bind that differs from the entrypoint listener, +// so a mismatched declaration can never be silently widened. +func TestApplyTrafficRuntimeConfigRejectsMismatchedAppL4Bind(t *testing.T) { + manager := trafficadapter.NewManager() + defer func() { require.NoError(t, manager.Shutdown(context.Background())) }() + tcpAddress := freeTCPAddress(t) + cfg := Config{ + EntryPoints: map[string]traffic.EntryPointConfig{ + "tcp": {Address: tcpAddress, Protocol: domain.EntryPointProtocolTCP}, + }, + } + hosts := &stubL4Hosts{ + entries: []apptraffic.RouteEntry{{ + Kind: "tcp", RouterName: "app-game--server--tcp-9000", Entrypoint: "tcp", + BindIP: "127.0.0.1", BindPort: 1, + App: "game", Service: "server", + Backend: domain.AppBackend{Host: "127.0.0.1", Port: 19000, ContainerPort: 9000, ContainerID: "c-game"}, + }}, + } + configSvc := inmocks.NewMockConfigService(t) + configSvc.EXPECT().GetExternalRoutes().Return(nil) + + err := applyTrafficRuntimeConfig(context.Background(), manager, cfg, configSvc, hosts) + require.Error(t, err) +} + func entryPointNames(entries []domain.EntryPoint) []string { names := make([]string, 0, len(entries)) for _, entry := range entries { diff --git a/internal/boundaries/in/app_reconciler.go b/internal/boundaries/in/app_reconciler.go new file mode 100644 index 000000000..f53fb1d18 --- /dev/null +++ b/internal/boundaries/in/app_reconciler.go @@ -0,0 +1,17 @@ +package in + +import "context" + +// AppReconciler is the daemon-only driving port for periodic app +// recovery. It is deliberately absent from AppService and from any +// HTTP or CLI surface: only the daemon-owned monitor calls it. +// +// ReconcileRunning converges ACTIVE workloads to durable intent. It +// never selects content from DESIRED, never changes intent, and never +// pulls, creates, or removes a container. +type AppReconciler interface { + // ReconcileRunning runs one deterministic recovery pass over every + // known app. Busy apps are skipped, independent per-service + // deadlines bound the work, and per-app errors are aggregated. + ReconcileRunning(ctx context.Context) error +} diff --git a/internal/boundaries/in/apps.go b/internal/boundaries/in/apps.go new file mode 100644 index 000000000..afefe4bd6 --- /dev/null +++ b/internal/boundaries/in/apps.go @@ -0,0 +1,151 @@ +package in + +import ( + "context" + "time" + + "github.com/bnema/gordon/internal/domain" +) + +// AppService is the driving port for the v3 app lifecycle: +// apply + reads (list/show/diff), lifecycle verbs +// (deploy/stop/start/restart/remove), op recovery by idempotency key, +// and service-scoped secret value writes. +// +// The port owns its view structs and speaks domain types only — never +// adapter DTOs. Secret VALUES cross SetSecrets by necessity (that is the +// write path); every other method carries names, digests, and ids only. +// +// This is preparation: the interface is frozen here, unreachable. +// The admin handler and the service implementation wire in at cutover; +// until then no caller exists. +type AppService interface { + // Apply validates a manifest spec and persists it as desired state, + // or validates without persistence when dryRun is set. source is the + // raw manifest bytes (hashed for audit, never re-read). + Apply(ctx context.Context, spec domain.AppSpec, source []byte, dryRun bool) (*AppApplyResult, *AppDryRunResult, error) + + // List returns one summary per known app. + List(ctx context.Context) ([]AppSummary, error) + + // Show inspects desired + active + intent + last op for one app. + Show(ctx context.Context, app string) (*AppDetail, error) + + // Diff returns the normalized desired-vs-active diff for one app. + Diff(ctx context.Context, app string) (domain.AppDiff, error) + + // Deploy activates a captured revision (empty revision means the + // desired head). An empty service targets every service, which a + // multi-service app accepts only with all set (domain.ErrAppServiceScope). + Deploy(ctx context.Context, app, revision, service string, all bool, idempotencyKey string) (*domain.AppOperation, error) + + // Stop persists the durable stopped intent and stops exact containers. + Stop(ctx context.Context, app, idempotencyKey string) (*domain.AppOperation, error) + + // Start clears the stopped intent and ensures running from active. + Start(ctx context.Context, app, idempotencyKey string) (*domain.AppOperation, error) + + // Restart restarts from pinned digests without re-resolution. + // An empty service targets every service, which a multi-service app + // accepts only with all set (domain.ErrAppServiceScope). + Restart(ctx context.Context, app, service string, all bool, idempotencyKey string) (*domain.AppOperation, error) + + // Remove withdraws workloads; volumes and secrets are retained as + // owned orphans and the name stays reserved. + Remove(ctx context.Context, app, idempotencyKey string) (*domain.AppOperation, error) + + // OperationByKey recovers an ambiguous mutation outcome by the + // client-supplied idempotency key. + OperationByKey(ctx context.Context, app, key string) (*domain.AppOperation, error) + + // ListSecrets returns metadata-only desired and active registrations. + ListSecrets(ctx context.Context, app, service string) ([]AppSecretMetadata, error) + + // SetSecrets writes secret values for pre-registered names in one + // service. service is required; every key must already exist in + // desired or active state. + SetSecrets(ctx context.Context, app, service string, values map[string]string) error + + // DeleteSecret removes one secret value. Refused when referenced. + DeleteSecret(ctx context.Context, app, service, key string) error +} + +// AppApplyResult describes one accepted (or no-op) apply. +// AppSecretMetadata is a secret registration without its value. +type AppSecretMetadata struct { + Service string + Key string + Name string + Source string + Presence string +} + +type AppApplyResult struct { + App string + FormerRevision string + ResultingRevision string + Noop bool + Pending bool + Diff domain.AppDiff + IntentID string +} + +// AppDryRunResult describes validation without persistence. +// The resulting revision is a preview only and is never stored. +type AppDryRunResult struct { + App string + Valid bool + Diff domain.AppDiff +} + +// AppSummary is one row of the app list: revisions, acceptance status, +// and the latest recorded outcome. +type AppSummary struct { + App string + Desired string + DesiredStatus string + Active string + Converged bool + // Pending reports desired state that the active revision has not + // reached yet. It is derived from desired vs ACTIVE, never from a + // journal. + Pending bool + Stopped bool + LastOutcome string +} + +// AppServiceView is one service's effective state for inspection. +// Digest and container are ids, never secret values. +type AppServiceView struct { + EffectiveRevision string + Digest string + Container string + RestartUnsafe bool +} + +// AppRetainedView lists the resources the app owns and would retain on +// removal: runtime volume names, secret paths, and pinned image +// references. It never carries secret values. +type AppRetainedView struct { + Volumes []string + Secrets []string + Images []string +} + +// AppDetail inspects desired + active + intent + ownership + last op for +// one app. +type AppDetail struct { + App string + DesiredRevision string + DesiredStatus string + Pending bool + Converged bool + ConvergedRevision string + Services map[string]AppServiceView + Stopped bool + Retained AppRetainedView + LastOp string + LastOpKind string + LastOutcome string + LastOpStartedAt time.Time +} diff --git a/internal/boundaries/in/backup.go b/internal/boundaries/in/backup.go index 2b2732617..11b1fc881 100644 --- a/internal/boundaries/in/backup.go +++ b/internal/boundaries/in/backup.go @@ -7,22 +7,33 @@ import ( "github.com/bnema/gordon/internal/domain" ) -// DatabaseBackupService defines database backup orchestration use cases. +// DatabaseBackupService defines declarative app database backup use +// cases. App names are the backup identity: a domain is never accepted. type DatabaseBackupService interface { - ListBackups(ctx context.Context, domainName string) ([]domain.DatabaseBackupJob, error) - RunBackup(ctx context.Context, domainName, dbName string) (*domain.DatabaseBackupResult, error) - Restore(ctx context.Context, domainName, backupID string) error - RestorePITR(ctx context.Context, domainName string, targetTime time.Time) error - Status(ctx context.Context) ([]domain.DatabaseBackupJob, error) - DetectDatabases(ctx context.Context, domainName string) ([]domain.DBInfo, error) + // ListBackups lists stored backups of one app, or of every app when + // app is empty. + ListBackups(ctx context.Context, app string) ([]domain.BackupJob, error) + // RunBackup runs one declared database backup. service and database + // are explicit selectors; an omitted selector succeeds only when + // exactly one compatible target exists. + RunBackup(ctx context.Context, app, service, database string) (*domain.BackupResult, error) + Restore(ctx context.Context, app, backupID string) error + RestorePITR(ctx context.Context, app string, targetTime time.Time) error + // Status reports stored backups plus declared targets without a + // completed backup. + Status(ctx context.Context) ([]domain.BackupJob, error) } -// BackupService is kept as a compatibility alias for the existing database backup feature. +// BackupService is the database backup service. type BackupService = DatabaseBackupService -// VolumeBackupService defines volume archive backup orchestration use cases. +// VolumeBackupService defines declarative app volume archive backup use +// cases. type VolumeBackupService interface { - ListVolumeBackups(ctx context.Context, domainName string) ([]domain.VolumeBackupJob, error) - RunVolumeBackups(ctx context.Context, domainName string, volumeName string) ([]domain.VolumeBackupJob, error) + ListVolumeBackups(ctx context.Context, app string) ([]domain.VolumeBackupJob, error) + // RunVolumeBackups runs the declared volume backups of one app. + // service and volume are explicit selectors; an omitted selector + // succeeds only when exactly one compatible target exists. + RunVolumeBackups(ctx context.Context, app, service, volume string) ([]domain.VolumeBackupJob, error) VolumeBackupStatus(ctx context.Context) ([]domain.VolumeBackupJob, error) } diff --git a/internal/boundaries/in/config.go b/internal/boundaries/in/config.go index 60085d108..6265984e3 100644 --- a/internal/boundaries/in/config.go +++ b/internal/boundaries/in/config.go @@ -2,11 +2,13 @@ package in import ( "context" - - "github.com/bnema/gordon/internal/domain" ) -// ConfigService defines the contract for configuration management. +// ConfigService defines the contract for installation configuration. +// Application state (routes, attachments, networks, auto, preview) was +// removed with the declarative-apps cutover: presence fails startup and +// reload closed with a config-retired diagnostic, and app data flows +// through ACTIVE state, never here. type ConfigService interface { // Load loads the configuration from the configured source. Load(ctx context.Context) error @@ -15,28 +17,6 @@ type ConfigService interface { // This is different from Load() which only loads from the cached viper values. Reload(ctx context.Context) error - // GetRoutes returns all configured routes. - GetRoutes(ctx context.Context) []domain.Route - - // GetRoute returns a single route by domain. - GetRoute(ctx context.Context, domain string) (*domain.Route, error) - - // FindRoutesByImage returns all routes that match the given image name. - // Image names are normalized by stripping the registry domain prefix before comparison. - FindRoutesByImage(ctx context.Context, imageName string) []domain.Route - - // AddRoute adds a new route to the configuration. - AddRoute(ctx context.Context, route domain.Route) error - - // UpdateRoute updates an existing route. - UpdateRoute(ctx context.Context, route domain.Route) error - - // RemoveRoute removes a route from the configuration. - RemoveRoute(ctx context.Context, domain string) error - - // Save persists the current configuration to disk. - Save(ctx context.Context) error - // Watch starts watching for configuration changes. // The onChange callback is called when configuration changes are detected. Watch(ctx context.Context, onChange func()) error @@ -53,9 +33,6 @@ type ConfigService interface { // GetDataDir returns the configured data directory. GetDataDir() string - // IsAutoRouteEnabled returns whether auto-route is enabled. - IsAutoRouteEnabled() bool - // IsNetworkIsolationEnabled returns whether network isolation is enabled. IsNetworkIsolationEnabled() bool @@ -64,31 +41,4 @@ type ConfigService interface { // GetExternalRoutes returns all configured external routes. GetExternalRoutes() map[string]string - - // GetAllAttachments returns all configured attachments. - GetAllAttachments(ctx context.Context) map[string][]string - - // GetAttachmentsFor returns attachments for a specific domain or network group. - GetAttachmentsFor(ctx context.Context, domainOrGroup string) ([]string, error) - - // FindAttachmentTargetsByImage returns all attachment targets that match the given image name. - FindAttachmentTargetsByImage(ctx context.Context, imageName string) []string - - // AddAttachment adds an image to a domain/group's attachments. - AddAttachment(ctx context.Context, domainOrGroup, image string) error - - // RemoveAttachment removes an image from a domain/group's attachments. - RemoveAttachment(ctx context.Context, domainOrGroup, image string) error - - // Auto-route domain allowlist - GetAutoRouteAllowedDomains(ctx context.Context) ([]string, error) - AddAutoRouteAllowedDomain(ctx context.Context, pattern string) error - RemoveAutoRouteAllowedDomain(ctx context.Context, pattern string) error - - // Preview and auto config - GetPreviewConfig() domain.PreviewConfig - IsPreviewEnabled() bool - IsAutoEnabled() bool - GetPreviewTagPatterns() []string - GetAllowedDomains() []string } diff --git a/internal/boundaries/in/container.go b/internal/boundaries/in/container.go index a34730e5d..2bcf162cf 100644 --- a/internal/boundaries/in/container.go +++ b/internal/boundaries/in/container.go @@ -9,60 +9,15 @@ import ( "github.com/bnema/gordon/internal/domain" ) -// ContainerService defines the contract for container management operations. +// ContainerService defines the contract for container runtime reads and +// lifecycle retained by the v3 cutover. The route-container engine +// (deploy/restart/remove/reconcile/attachments/sync/autostart) was +// removed with the declarative-apps cutover; the deployment engine owns +// workload effects through out.ContainerRuntime directly. type ContainerService interface { - // Deploy creates and starts a container for the given route. - Deploy(ctx context.Context, route domain.Route) (*domain.Container, error) - - // Stop stops a running container. - Stop(ctx context.Context, containerID string) error - - // Remove removes a container, optionally forcing removal. - Remove(ctx context.Context, containerID string, force bool) error - - // ReconcileRemovedRoute removes active route runtime containers after the - // route has been removed from configuration, preserving stateful resources. - ReconcileRemovedRoute(ctx context.Context, domain string) (*domain.CleanupReport, error) - - // Get retrieves a container by domain name. - Get(ctx context.Context, domain string) (*domain.Container, bool) - - // Restart restarts a running container for the given domain. - // If withAttachments is true, also restarts attachment containers. - Restart(ctx context.Context, domain string, withAttachments bool) error - - // List returns all managed containers. - List(ctx context.Context) map[string]*domain.Container - - // ListRoutesWithDetails returns routes with network and attachment info. - ListRoutesWithDetails(ctx context.Context) []domain.RouteInfo - - // ListAttachments returns attachments for a domain. - ListAttachments(ctx context.Context, domain string) []domain.Attachment - - // ListOrphanedAttachments returns running attachment containers no longer configured. - ListOrphanedAttachments(ctx context.Context) ([]domain.CleanupAttachment, error) - - // CleanupOrphanedAttachments optionally stops/removes orphaned attachment containers. - // When owner is non-empty, cleanup is scoped to that attachment owner. - CleanupOrphanedAttachments(ctx context.Context, owner string, stop bool) (*domain.CleanupReport, error) - // ListNetworks returns Gordon-managed networks. ListNetworks(ctx context.Context) ([]*domain.NetworkInfo, error) - // HealthCheck performs health checks on all containers. - HealthCheck(ctx context.Context) map[string]bool - - // SyncContainers synchronizes containers with configured routes. - SyncContainers(ctx context.Context) error - - // UpdateAttachments updates the attachment configuration in the container service. - // This is called after a config reload to propagate attachment changes without restart. - UpdateAttachments(attachments map[string][]string) - - // AutoStart starts containers for the provided routes that aren't running. - AutoStart(ctx context.Context, routes []domain.Route) error - // Shutdown gracefully shuts down all managed containers. Shutdown(ctx context.Context) error } diff --git a/internal/boundaries/in/deploy.go b/internal/boundaries/in/deploy.go deleted file mode 100644 index d75c4966a..000000000 --- a/internal/boundaries/in/deploy.go +++ /dev/null @@ -1,10 +0,0 @@ -package in - -// DeployCoordinator defines deploy coordination operations. -// -// These methods let the CLI/admin API suppress image-pushed deploy events -// while an explicit deploy flow is in progress. -type DeployCoordinator interface { - SuppressDeployEvent(imageName string) - ClearDeployEventSuppression(imageName string) -} diff --git a/internal/boundaries/in/health.go b/internal/boundaries/in/health.go index fbbb38ef6..19dfd4935 100644 --- a/internal/boundaries/in/health.go +++ b/internal/boundaries/in/health.go @@ -9,14 +9,10 @@ import ( "github.com/bnema/gordon/internal/domain" ) -// HealthService defines the contract for route health checking operations. +// HealthService defines the contract for app health checking operations. type HealthService interface { - // CheckRoute performs a health check on a single route. - // It checks both container status and HTTP reachability. - CheckRoute(ctx context.Context, route domain.Route) *domain.RouteHealth - - // CheckAllRoutes performs health checks on all configured routes. - // Returns a map of domain to health status. + // CheckAllRoutes performs health checks on all app-served HTTP hosts. + // Returns a map of canonical host to health status. CheckAllRoutes(ctx context.Context) map[string]*domain.RouteHealth } diff --git a/internal/boundaries/in/mocks/mock_app_reconciler.go b/internal/boundaries/in/mocks/mock_app_reconciler.go new file mode 100644 index 000000000..87d3f1fce --- /dev/null +++ b/internal/boundaries/in/mocks/mock_app_reconciler.go @@ -0,0 +1,98 @@ +// Code generated by mockery; DO NOT EDIT. +// github.com/vektra/mockery +// template: testify + +package mocks + +import ( + "context" + + mock "github.com/stretchr/testify/mock" +) + +// NewMockAppReconciler creates a new instance of MockAppReconciler. It also registers a testing interface on the mock and a cleanup function to assert the mocks expectations. +// The first argument is typically a *testing.T value. +func NewMockAppReconciler(t interface { + mock.TestingT + Cleanup(func()) +}) *MockAppReconciler { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + + mock := &MockAppReconciler{} + mock.Mock.Test(t) + + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) + + return mock +} + +// MockAppReconciler is an autogenerated mock type for the AppReconciler type +type MockAppReconciler struct { + mock.Mock +} + +type MockAppReconciler_Expecter struct { + mock *mock.Mock +} + +func (_m *MockAppReconciler) EXPECT() *MockAppReconciler_Expecter { + return &MockAppReconciler_Expecter{mock: &_m.Mock} +} + +// ReconcileRunning provides a mock function for the type MockAppReconciler +func (_mock *MockAppReconciler) ReconcileRunning(ctx context.Context) error { + ret := _mock.Called(ctx) + + if len(ret) == 0 { + panic("no return value specified for ReconcileRunning") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context) error); ok { + r0 = returnFunc(ctx) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockAppReconciler_ReconcileRunning_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ReconcileRunning' +type MockAppReconciler_ReconcileRunning_Call struct { + *mock.Call +} + +// ReconcileRunning is a helper method to define mock.On call +// - ctx context.Context +func (_e *MockAppReconciler_Expecter) ReconcileRunning(ctx any) *MockAppReconciler_ReconcileRunning_Call { + return &MockAppReconciler_ReconcileRunning_Call{Call: _e.mock.On("ReconcileRunning", ctx)} +} + +func (_c *MockAppReconciler_ReconcileRunning_Call) Run(run func(ctx context.Context)) *MockAppReconciler_ReconcileRunning_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + run( + arg0, + ) + }) + return _c +} + +func (_c *MockAppReconciler_ReconcileRunning_Call) Return(err error) *MockAppReconciler_ReconcileRunning_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockAppReconciler_ReconcileRunning_Call) RunAndReturn(run func(ctx context.Context) error) *MockAppReconciler_ReconcileRunning_Call { + _c.Call.Return(run) + return _c +} diff --git a/internal/boundaries/in/mocks/mock_app_service.go b/internal/boundaries/in/mocks/mock_app_service.go new file mode 100644 index 000000000..3df8e7d7f --- /dev/null +++ b/internal/boundaries/in/mocks/mock_app_service.go @@ -0,0 +1,1019 @@ +// Code generated by mockery; DO NOT EDIT. +// github.com/vektra/mockery +// template: testify + +package mocks + +import ( + "context" + + "github.com/bnema/gordon/internal/boundaries/in" + "github.com/bnema/gordon/internal/domain" + mock "github.com/stretchr/testify/mock" +) + +// NewMockAppService creates a new instance of MockAppService. It also registers a testing interface on the mock and a cleanup function to assert the mocks expectations. +// The first argument is typically a *testing.T value. +func NewMockAppService(t interface { + mock.TestingT + Cleanup(func()) +}) *MockAppService { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + + mock := &MockAppService{} + mock.Mock.Test(t) + + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) + + return mock +} + +// MockAppService is an autogenerated mock type for the AppService type +type MockAppService struct { + mock.Mock +} + +type MockAppService_Expecter struct { + mock *mock.Mock +} + +func (_m *MockAppService) EXPECT() *MockAppService_Expecter { + return &MockAppService_Expecter{mock: &_m.Mock} +} + +// Apply provides a mock function for the type MockAppService +func (_mock *MockAppService) Apply(ctx context.Context, spec domain.AppSpec, source []byte, dryRun bool) (*in.AppApplyResult, *in.AppDryRunResult, error) { + ret := _mock.Called(ctx, spec, source, dryRun) + + if len(ret) == 0 { + panic("no return value specified for Apply") + } + + var r0 *in.AppApplyResult + var r1 *in.AppDryRunResult + var r2 error + if returnFunc, ok := ret.Get(0).(func(context.Context, domain.AppSpec, []byte, bool) (*in.AppApplyResult, *in.AppDryRunResult, error)); ok { + return returnFunc(ctx, spec, source, dryRun) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, domain.AppSpec, []byte, bool) *in.AppApplyResult); ok { + r0 = returnFunc(ctx, spec, source, dryRun) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).(*in.AppApplyResult) + } + } + if returnFunc, ok := ret.Get(1).(func(context.Context, domain.AppSpec, []byte, bool) *in.AppDryRunResult); ok { + r1 = returnFunc(ctx, spec, source, dryRun) + } else { + if ret.Get(1) != nil { + r1 = ret.Get(1).(*in.AppDryRunResult) + } + } + if returnFunc, ok := ret.Get(2).(func(context.Context, domain.AppSpec, []byte, bool) error); ok { + r2 = returnFunc(ctx, spec, source, dryRun) + } else { + r2 = ret.Error(2) + } + return r0, r1, r2 +} + +// MockAppService_Apply_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Apply' +type MockAppService_Apply_Call struct { + *mock.Call +} + +// Apply is a helper method to define mock.On call +// - ctx context.Context +// - spec domain.AppSpec +// - source []byte +// - dryRun bool +func (_e *MockAppService_Expecter) Apply(ctx any, spec any, source any, dryRun any) *MockAppService_Apply_Call { + return &MockAppService_Apply_Call{Call: _e.mock.On("Apply", ctx, spec, source, dryRun)} +} + +func (_c *MockAppService_Apply_Call) Run(run func(ctx context.Context, spec domain.AppSpec, source []byte, dryRun bool)) *MockAppService_Apply_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 domain.AppSpec + if args[1] != nil { + arg1 = args[1].(domain.AppSpec) + } + var arg2 []byte + if args[2] != nil { + arg2 = args[2].([]byte) + } + var arg3 bool + if args[3] != nil { + arg3 = args[3].(bool) + } + run( + arg0, + arg1, + arg2, + arg3, + ) + }) + return _c +} + +func (_c *MockAppService_Apply_Call) Return(appApplyResult *in.AppApplyResult, appDryRunResult *in.AppDryRunResult, err error) *MockAppService_Apply_Call { + _c.Call.Return(appApplyResult, appDryRunResult, err) + return _c +} + +func (_c *MockAppService_Apply_Call) RunAndReturn(run func(ctx context.Context, spec domain.AppSpec, source []byte, dryRun bool) (*in.AppApplyResult, *in.AppDryRunResult, error)) *MockAppService_Apply_Call { + _c.Call.Return(run) + return _c +} + +// DeleteSecret provides a mock function for the type MockAppService +func (_mock *MockAppService) DeleteSecret(ctx context.Context, app string, service string, key string) error { + ret := _mock.Called(ctx, app, service, key) + + if len(ret) == 0 { + panic("no return value specified for DeleteSecret") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, string) error); ok { + r0 = returnFunc(ctx, app, service, key) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockAppService_DeleteSecret_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'DeleteSecret' +type MockAppService_DeleteSecret_Call struct { + *mock.Call +} + +// DeleteSecret is a helper method to define mock.On call +// - ctx context.Context +// - app string +// - service string +// - key string +func (_e *MockAppService_Expecter) DeleteSecret(ctx any, app any, service any, key any) *MockAppService_DeleteSecret_Call { + return &MockAppService_DeleteSecret_Call{Call: _e.mock.On("DeleteSecret", ctx, app, service, key)} +} + +func (_c *MockAppService_DeleteSecret_Call) Run(run func(ctx context.Context, app string, service string, key string)) *MockAppService_DeleteSecret_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } + var arg3 string + if args[3] != nil { + arg3 = args[3].(string) + } + run( + arg0, + arg1, + arg2, + arg3, + ) + }) + return _c +} + +func (_c *MockAppService_DeleteSecret_Call) Return(err error) *MockAppService_DeleteSecret_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockAppService_DeleteSecret_Call) RunAndReturn(run func(ctx context.Context, app string, service string, key string) error) *MockAppService_DeleteSecret_Call { + _c.Call.Return(run) + return _c +} + +// Deploy provides a mock function for the type MockAppService +func (_mock *MockAppService) Deploy(ctx context.Context, app string, revision string, service string, all bool, idempotencyKey string) (*domain.AppOperation, error) { + ret := _mock.Called(ctx, app, revision, service, all, idempotencyKey) + + if len(ret) == 0 { + panic("no return value specified for Deploy") + } + + var r0 *domain.AppOperation + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, string, bool, string) (*domain.AppOperation, error)); ok { + return returnFunc(ctx, app, revision, service, all, idempotencyKey) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, string, bool, string) *domain.AppOperation); ok { + r0 = returnFunc(ctx, app, revision, service, all, idempotencyKey) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).(*domain.AppOperation) + } + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string, string, string, bool, string) error); ok { + r1 = returnFunc(ctx, app, revision, service, all, idempotencyKey) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockAppService_Deploy_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Deploy' +type MockAppService_Deploy_Call struct { + *mock.Call +} + +// Deploy is a helper method to define mock.On call +// - ctx context.Context +// - app string +// - revision string +// - service string +// - all bool +// - idempotencyKey string +func (_e *MockAppService_Expecter) Deploy(ctx any, app any, revision any, service any, all any, idempotencyKey any) *MockAppService_Deploy_Call { + return &MockAppService_Deploy_Call{Call: _e.mock.On("Deploy", ctx, app, revision, service, all, idempotencyKey)} +} + +func (_c *MockAppService_Deploy_Call) Run(run func(ctx context.Context, app string, revision string, service string, all bool, idempotencyKey string)) *MockAppService_Deploy_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } + var arg3 string + if args[3] != nil { + arg3 = args[3].(string) + } + var arg4 bool + if args[4] != nil { + arg4 = args[4].(bool) + } + var arg5 string + if args[5] != nil { + arg5 = args[5].(string) + } + run( + arg0, + arg1, + arg2, + arg3, + arg4, + arg5, + ) + }) + return _c +} + +func (_c *MockAppService_Deploy_Call) Return(appOperation *domain.AppOperation, err error) *MockAppService_Deploy_Call { + _c.Call.Return(appOperation, err) + return _c +} + +func (_c *MockAppService_Deploy_Call) RunAndReturn(run func(ctx context.Context, app string, revision string, service string, all bool, idempotencyKey string) (*domain.AppOperation, error)) *MockAppService_Deploy_Call { + _c.Call.Return(run) + return _c +} + +// Diff provides a mock function for the type MockAppService +func (_mock *MockAppService) Diff(ctx context.Context, app string) (domain.AppDiff, error) { + ret := _mock.Called(ctx, app) + + if len(ret) == 0 { + panic("no return value specified for Diff") + } + + var r0 domain.AppDiff + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string) (domain.AppDiff, error)); ok { + return returnFunc(ctx, app) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string) domain.AppDiff); ok { + r0 = returnFunc(ctx, app) + } else { + r0 = ret.Get(0).(domain.AppDiff) + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { + r1 = returnFunc(ctx, app) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockAppService_Diff_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Diff' +type MockAppService_Diff_Call struct { + *mock.Call +} + +// Diff is a helper method to define mock.On call +// - ctx context.Context +// - app string +func (_e *MockAppService_Expecter) Diff(ctx any, app any) *MockAppService_Diff_Call { + return &MockAppService_Diff_Call{Call: _e.mock.On("Diff", ctx, app)} +} + +func (_c *MockAppService_Diff_Call) Run(run func(ctx context.Context, app string)) *MockAppService_Diff_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockAppService_Diff_Call) Return(appDiff domain.AppDiff, err error) *MockAppService_Diff_Call { + _c.Call.Return(appDiff, err) + return _c +} + +func (_c *MockAppService_Diff_Call) RunAndReturn(run func(ctx context.Context, app string) (domain.AppDiff, error)) *MockAppService_Diff_Call { + _c.Call.Return(run) + return _c +} + +// List provides a mock function for the type MockAppService +func (_mock *MockAppService) List(ctx context.Context) ([]in.AppSummary, error) { + ret := _mock.Called(ctx) + + if len(ret) == 0 { + panic("no return value specified for List") + } + + var r0 []in.AppSummary + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context) ([]in.AppSummary, error)); ok { + return returnFunc(ctx) + } + if returnFunc, ok := ret.Get(0).(func(context.Context) []in.AppSummary); ok { + r0 = returnFunc(ctx) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).([]in.AppSummary) + } + } + if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { + r1 = returnFunc(ctx) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockAppService_List_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'List' +type MockAppService_List_Call struct { + *mock.Call +} + +// List is a helper method to define mock.On call +// - ctx context.Context +func (_e *MockAppService_Expecter) List(ctx any) *MockAppService_List_Call { + return &MockAppService_List_Call{Call: _e.mock.On("List", ctx)} +} + +func (_c *MockAppService_List_Call) Run(run func(ctx context.Context)) *MockAppService_List_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + run( + arg0, + ) + }) + return _c +} + +func (_c *MockAppService_List_Call) Return(appSummarys []in.AppSummary, err error) *MockAppService_List_Call { + _c.Call.Return(appSummarys, err) + return _c +} + +func (_c *MockAppService_List_Call) RunAndReturn(run func(ctx context.Context) ([]in.AppSummary, error)) *MockAppService_List_Call { + _c.Call.Return(run) + return _c +} + +// ListSecrets provides a mock function for the type MockAppService +func (_mock *MockAppService) ListSecrets(ctx context.Context, app string, service string) ([]in.AppSecretMetadata, error) { + ret := _mock.Called(ctx, app, service) + + if len(ret) == 0 { + panic("no return value specified for ListSecrets") + } + + var r0 []in.AppSecretMetadata + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) ([]in.AppSecretMetadata, error)); ok { + return returnFunc(ctx, app, service) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) []in.AppSecretMetadata); ok { + r0 = returnFunc(ctx, app, service) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).([]in.AppSecretMetadata) + } + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string, string) error); ok { + r1 = returnFunc(ctx, app, service) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockAppService_ListSecrets_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListSecrets' +type MockAppService_ListSecrets_Call struct { + *mock.Call +} + +// ListSecrets is a helper method to define mock.On call +// - ctx context.Context +// - app string +// - service string +func (_e *MockAppService_Expecter) ListSecrets(ctx any, app any, service any) *MockAppService_ListSecrets_Call { + return &MockAppService_ListSecrets_Call{Call: _e.mock.On("ListSecrets", ctx, app, service)} +} + +func (_c *MockAppService_ListSecrets_Call) Run(run func(ctx context.Context, app string, service string)) *MockAppService_ListSecrets_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } + run( + arg0, + arg1, + arg2, + ) + }) + return _c +} + +func (_c *MockAppService_ListSecrets_Call) Return(appSecretMetadatas []in.AppSecretMetadata, err error) *MockAppService_ListSecrets_Call { + _c.Call.Return(appSecretMetadatas, err) + return _c +} + +func (_c *MockAppService_ListSecrets_Call) RunAndReturn(run func(ctx context.Context, app string, service string) ([]in.AppSecretMetadata, error)) *MockAppService_ListSecrets_Call { + _c.Call.Return(run) + return _c +} + +// OperationByKey provides a mock function for the type MockAppService +func (_mock *MockAppService) OperationByKey(ctx context.Context, app string, key string) (*domain.AppOperation, error) { + ret := _mock.Called(ctx, app, key) + + if len(ret) == 0 { + panic("no return value specified for OperationByKey") + } + + var r0 *domain.AppOperation + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) (*domain.AppOperation, error)); ok { + return returnFunc(ctx, app, key) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) *domain.AppOperation); ok { + r0 = returnFunc(ctx, app, key) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).(*domain.AppOperation) + } + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string, string) error); ok { + r1 = returnFunc(ctx, app, key) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockAppService_OperationByKey_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'OperationByKey' +type MockAppService_OperationByKey_Call struct { + *mock.Call +} + +// OperationByKey is a helper method to define mock.On call +// - ctx context.Context +// - app string +// - key string +func (_e *MockAppService_Expecter) OperationByKey(ctx any, app any, key any) *MockAppService_OperationByKey_Call { + return &MockAppService_OperationByKey_Call{Call: _e.mock.On("OperationByKey", ctx, app, key)} +} + +func (_c *MockAppService_OperationByKey_Call) Run(run func(ctx context.Context, app string, key string)) *MockAppService_OperationByKey_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } + run( + arg0, + arg1, + arg2, + ) + }) + return _c +} + +func (_c *MockAppService_OperationByKey_Call) Return(appOperation *domain.AppOperation, err error) *MockAppService_OperationByKey_Call { + _c.Call.Return(appOperation, err) + return _c +} + +func (_c *MockAppService_OperationByKey_Call) RunAndReturn(run func(ctx context.Context, app string, key string) (*domain.AppOperation, error)) *MockAppService_OperationByKey_Call { + _c.Call.Return(run) + return _c +} + +// Remove provides a mock function for the type MockAppService +func (_mock *MockAppService) Remove(ctx context.Context, app string, idempotencyKey string) (*domain.AppOperation, error) { + ret := _mock.Called(ctx, app, idempotencyKey) + + if len(ret) == 0 { + panic("no return value specified for Remove") + } + + var r0 *domain.AppOperation + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) (*domain.AppOperation, error)); ok { + return returnFunc(ctx, app, idempotencyKey) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) *domain.AppOperation); ok { + r0 = returnFunc(ctx, app, idempotencyKey) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).(*domain.AppOperation) + } + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string, string) error); ok { + r1 = returnFunc(ctx, app, idempotencyKey) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockAppService_Remove_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Remove' +type MockAppService_Remove_Call struct { + *mock.Call +} + +// Remove is a helper method to define mock.On call +// - ctx context.Context +// - app string +// - idempotencyKey string +func (_e *MockAppService_Expecter) Remove(ctx any, app any, idempotencyKey any) *MockAppService_Remove_Call { + return &MockAppService_Remove_Call{Call: _e.mock.On("Remove", ctx, app, idempotencyKey)} +} + +func (_c *MockAppService_Remove_Call) Run(run func(ctx context.Context, app string, idempotencyKey string)) *MockAppService_Remove_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } + run( + arg0, + arg1, + arg2, + ) + }) + return _c +} + +func (_c *MockAppService_Remove_Call) Return(appOperation *domain.AppOperation, err error) *MockAppService_Remove_Call { + _c.Call.Return(appOperation, err) + return _c +} + +func (_c *MockAppService_Remove_Call) RunAndReturn(run func(ctx context.Context, app string, idempotencyKey string) (*domain.AppOperation, error)) *MockAppService_Remove_Call { + _c.Call.Return(run) + return _c +} + +// Restart provides a mock function for the type MockAppService +func (_mock *MockAppService) Restart(ctx context.Context, app string, service string, all bool, idempotencyKey string) (*domain.AppOperation, error) { + ret := _mock.Called(ctx, app, service, all, idempotencyKey) + + if len(ret) == 0 { + panic("no return value specified for Restart") + } + + var r0 *domain.AppOperation + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, bool, string) (*domain.AppOperation, error)); ok { + return returnFunc(ctx, app, service, all, idempotencyKey) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, bool, string) *domain.AppOperation); ok { + r0 = returnFunc(ctx, app, service, all, idempotencyKey) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).(*domain.AppOperation) + } + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string, string, bool, string) error); ok { + r1 = returnFunc(ctx, app, service, all, idempotencyKey) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockAppService_Restart_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Restart' +type MockAppService_Restart_Call struct { + *mock.Call +} + +// Restart is a helper method to define mock.On call +// - ctx context.Context +// - app string +// - service string +// - all bool +// - idempotencyKey string +func (_e *MockAppService_Expecter) Restart(ctx any, app any, service any, all any, idempotencyKey any) *MockAppService_Restart_Call { + return &MockAppService_Restart_Call{Call: _e.mock.On("Restart", ctx, app, service, all, idempotencyKey)} +} + +func (_c *MockAppService_Restart_Call) Run(run func(ctx context.Context, app string, service string, all bool, idempotencyKey string)) *MockAppService_Restart_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } + var arg3 bool + if args[3] != nil { + arg3 = args[3].(bool) + } + var arg4 string + if args[4] != nil { + arg4 = args[4].(string) + } + run( + arg0, + arg1, + arg2, + arg3, + arg4, + ) + }) + return _c +} + +func (_c *MockAppService_Restart_Call) Return(appOperation *domain.AppOperation, err error) *MockAppService_Restart_Call { + _c.Call.Return(appOperation, err) + return _c +} + +func (_c *MockAppService_Restart_Call) RunAndReturn(run func(ctx context.Context, app string, service string, all bool, idempotencyKey string) (*domain.AppOperation, error)) *MockAppService_Restart_Call { + _c.Call.Return(run) + return _c +} + +// SetSecrets provides a mock function for the type MockAppService +func (_mock *MockAppService) SetSecrets(ctx context.Context, app string, service string, values map[string]string) error { + ret := _mock.Called(ctx, app, service, values) + + if len(ret) == 0 { + panic("no return value specified for SetSecrets") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, map[string]string) error); ok { + r0 = returnFunc(ctx, app, service, values) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockAppService_SetSecrets_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'SetSecrets' +type MockAppService_SetSecrets_Call struct { + *mock.Call +} + +// SetSecrets is a helper method to define mock.On call +// - ctx context.Context +// - app string +// - service string +// - values map[string]string +func (_e *MockAppService_Expecter) SetSecrets(ctx any, app any, service any, values any) *MockAppService_SetSecrets_Call { + return &MockAppService_SetSecrets_Call{Call: _e.mock.On("SetSecrets", ctx, app, service, values)} +} + +func (_c *MockAppService_SetSecrets_Call) Run(run func(ctx context.Context, app string, service string, values map[string]string)) *MockAppService_SetSecrets_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } + var arg3 map[string]string + if args[3] != nil { + arg3 = args[3].(map[string]string) + } + run( + arg0, + arg1, + arg2, + arg3, + ) + }) + return _c +} + +func (_c *MockAppService_SetSecrets_Call) Return(err error) *MockAppService_SetSecrets_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockAppService_SetSecrets_Call) RunAndReturn(run func(ctx context.Context, app string, service string, values map[string]string) error) *MockAppService_SetSecrets_Call { + _c.Call.Return(run) + return _c +} + +// Show provides a mock function for the type MockAppService +func (_mock *MockAppService) Show(ctx context.Context, app string) (*in.AppDetail, error) { + ret := _mock.Called(ctx, app) + + if len(ret) == 0 { + panic("no return value specified for Show") + } + + var r0 *in.AppDetail + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string) (*in.AppDetail, error)); ok { + return returnFunc(ctx, app) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string) *in.AppDetail); ok { + r0 = returnFunc(ctx, app) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).(*in.AppDetail) + } + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { + r1 = returnFunc(ctx, app) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockAppService_Show_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Show' +type MockAppService_Show_Call struct { + *mock.Call +} + +// Show is a helper method to define mock.On call +// - ctx context.Context +// - app string +func (_e *MockAppService_Expecter) Show(ctx any, app any) *MockAppService_Show_Call { + return &MockAppService_Show_Call{Call: _e.mock.On("Show", ctx, app)} +} + +func (_c *MockAppService_Show_Call) Run(run func(ctx context.Context, app string)) *MockAppService_Show_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockAppService_Show_Call) Return(appDetail *in.AppDetail, err error) *MockAppService_Show_Call { + _c.Call.Return(appDetail, err) + return _c +} + +func (_c *MockAppService_Show_Call) RunAndReturn(run func(ctx context.Context, app string) (*in.AppDetail, error)) *MockAppService_Show_Call { + _c.Call.Return(run) + return _c +} + +// Start provides a mock function for the type MockAppService +func (_mock *MockAppService) Start(ctx context.Context, app string, idempotencyKey string) (*domain.AppOperation, error) { + ret := _mock.Called(ctx, app, idempotencyKey) + + if len(ret) == 0 { + panic("no return value specified for Start") + } + + var r0 *domain.AppOperation + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) (*domain.AppOperation, error)); ok { + return returnFunc(ctx, app, idempotencyKey) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) *domain.AppOperation); ok { + r0 = returnFunc(ctx, app, idempotencyKey) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).(*domain.AppOperation) + } + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string, string) error); ok { + r1 = returnFunc(ctx, app, idempotencyKey) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockAppService_Start_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Start' +type MockAppService_Start_Call struct { + *mock.Call +} + +// Start is a helper method to define mock.On call +// - ctx context.Context +// - app string +// - idempotencyKey string +func (_e *MockAppService_Expecter) Start(ctx any, app any, idempotencyKey any) *MockAppService_Start_Call { + return &MockAppService_Start_Call{Call: _e.mock.On("Start", ctx, app, idempotencyKey)} +} + +func (_c *MockAppService_Start_Call) Run(run func(ctx context.Context, app string, idempotencyKey string)) *MockAppService_Start_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } + run( + arg0, + arg1, + arg2, + ) + }) + return _c +} + +func (_c *MockAppService_Start_Call) Return(appOperation *domain.AppOperation, err error) *MockAppService_Start_Call { + _c.Call.Return(appOperation, err) + return _c +} + +func (_c *MockAppService_Start_Call) RunAndReturn(run func(ctx context.Context, app string, idempotencyKey string) (*domain.AppOperation, error)) *MockAppService_Start_Call { + _c.Call.Return(run) + return _c +} + +// Stop provides a mock function for the type MockAppService +func (_mock *MockAppService) Stop(ctx context.Context, app string, idempotencyKey string) (*domain.AppOperation, error) { + ret := _mock.Called(ctx, app, idempotencyKey) + + if len(ret) == 0 { + panic("no return value specified for Stop") + } + + var r0 *domain.AppOperation + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) (*domain.AppOperation, error)); ok { + return returnFunc(ctx, app, idempotencyKey) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) *domain.AppOperation); ok { + r0 = returnFunc(ctx, app, idempotencyKey) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).(*domain.AppOperation) + } + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string, string) error); ok { + r1 = returnFunc(ctx, app, idempotencyKey) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockAppService_Stop_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Stop' +type MockAppService_Stop_Call struct { + *mock.Call +} + +// Stop is a helper method to define mock.On call +// - ctx context.Context +// - app string +// - idempotencyKey string +func (_e *MockAppService_Expecter) Stop(ctx any, app any, idempotencyKey any) *MockAppService_Stop_Call { + return &MockAppService_Stop_Call{Call: _e.mock.On("Stop", ctx, app, idempotencyKey)} +} + +func (_c *MockAppService_Stop_Call) Run(run func(ctx context.Context, app string, idempotencyKey string)) *MockAppService_Stop_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } + run( + arg0, + arg1, + arg2, + ) + }) + return _c +} + +func (_c *MockAppService_Stop_Call) Return(appOperation *domain.AppOperation, err error) *MockAppService_Stop_Call { + _c.Call.Return(appOperation, err) + return _c +} + +func (_c *MockAppService_Stop_Call) RunAndReturn(run func(ctx context.Context, app string, idempotencyKey string) (*domain.AppOperation, error)) *MockAppService_Stop_Call { + _c.Call.Return(run) + return _c +} diff --git a/internal/boundaries/in/mocks/mock_auth_service.go b/internal/boundaries/in/mocks/mock_auth_service.go index df8006af2..fd2048589 100644 --- a/internal/boundaries/in/mocks/mock_auth_service.go +++ b/internal/boundaries/in/mocks/mock_auth_service.go @@ -18,10 +18,19 @@ func NewMockAuthService(t interface { mock.TestingT Cleanup(func()) }) *MockAuthService { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockAuthService{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/in/mocks/mock_backup_service.go b/internal/boundaries/in/mocks/mock_backup_service.go index c8e522547..10b63d5b9 100644 --- a/internal/boundaries/in/mocks/mock_backup_service.go +++ b/internal/boundaries/in/mocks/mock_backup_service.go @@ -18,10 +18,19 @@ func NewMockBackupService(t interface { mock.TestingT Cleanup(func()) }) *MockBackupService { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockBackupService{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } @@ -39,96 +48,28 @@ func (_m *MockBackupService) EXPECT() *MockBackupService_Expecter { return &MockBackupService_Expecter{mock: &_m.Mock} } -// DetectDatabases provides a mock function for the type MockBackupService -func (_mock *MockBackupService) DetectDatabases(ctx context.Context, domainName string) ([]domain.DBInfo, error) { - ret := _mock.Called(ctx, domainName) - - if len(ret) == 0 { - panic("no return value specified for DetectDatabases") - } - - var r0 []domain.DBInfo - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) ([]domain.DBInfo, error)); ok { - return returnFunc(ctx, domainName) - } - if returnFunc, ok := ret.Get(0).(func(context.Context, string) []domain.DBInfo); ok { - r0 = returnFunc(ctx, domainName) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]domain.DBInfo) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { - r1 = returnFunc(ctx, domainName) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockBackupService_DetectDatabases_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'DetectDatabases' -type MockBackupService_DetectDatabases_Call struct { - *mock.Call -} - -// DetectDatabases is a helper method to define mock.On call -// - ctx context.Context -// - domainName string -func (_e *MockBackupService_Expecter) DetectDatabases(ctx any, domainName any) *MockBackupService_DetectDatabases_Call { - return &MockBackupService_DetectDatabases_Call{Call: _e.mock.On("DetectDatabases", ctx, domainName)} -} - -func (_c *MockBackupService_DetectDatabases_Call) Run(run func(ctx context.Context, domainName string)) *MockBackupService_DetectDatabases_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockBackupService_DetectDatabases_Call) Return(dBInfos []domain.DBInfo, err error) *MockBackupService_DetectDatabases_Call { - _c.Call.Return(dBInfos, err) - return _c -} - -func (_c *MockBackupService_DetectDatabases_Call) RunAndReturn(run func(ctx context.Context, domainName string) ([]domain.DBInfo, error)) *MockBackupService_DetectDatabases_Call { - _c.Call.Return(run) - return _c -} - // ListBackups provides a mock function for the type MockBackupService -func (_mock *MockBackupService) ListBackups(ctx context.Context, domainName string) ([]domain.DatabaseBackupJob, error) { - ret := _mock.Called(ctx, domainName) +func (_mock *MockBackupService) ListBackups(ctx context.Context, app string) ([]domain.BackupJob, error) { + ret := _mock.Called(ctx, app) if len(ret) == 0 { panic("no return value specified for ListBackups") } - var r0 []domain.DatabaseBackupJob + var r0 []domain.BackupJob var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) ([]domain.DatabaseBackupJob, error)); ok { - return returnFunc(ctx, domainName) + if returnFunc, ok := ret.Get(0).(func(context.Context, string) ([]domain.BackupJob, error)); ok { + return returnFunc(ctx, app) } - if returnFunc, ok := ret.Get(0).(func(context.Context, string) []domain.DatabaseBackupJob); ok { - r0 = returnFunc(ctx, domainName) + if returnFunc, ok := ret.Get(0).(func(context.Context, string) []domain.BackupJob); ok { + r0 = returnFunc(ctx, app) } else { if ret.Get(0) != nil { - r0 = ret.Get(0).([]domain.DatabaseBackupJob) + r0 = ret.Get(0).([]domain.BackupJob) } } if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { - r1 = returnFunc(ctx, domainName) + r1 = returnFunc(ctx, app) } else { r1 = ret.Error(1) } @@ -142,12 +83,12 @@ type MockBackupService_ListBackups_Call struct { // ListBackups is a helper method to define mock.On call // - ctx context.Context -// - domainName string -func (_e *MockBackupService_Expecter) ListBackups(ctx any, domainName any) *MockBackupService_ListBackups_Call { - return &MockBackupService_ListBackups_Call{Call: _e.mock.On("ListBackups", ctx, domainName)} +// - app string +func (_e *MockBackupService_Expecter) ListBackups(ctx any, app any) *MockBackupService_ListBackups_Call { + return &MockBackupService_ListBackups_Call{Call: _e.mock.On("ListBackups", ctx, app)} } -func (_c *MockBackupService_ListBackups_Call) Run(run func(ctx context.Context, domainName string)) *MockBackupService_ListBackups_Call { +func (_c *MockBackupService_ListBackups_Call) Run(run func(ctx context.Context, app string)) *MockBackupService_ListBackups_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -165,19 +106,19 @@ func (_c *MockBackupService_ListBackups_Call) Run(run func(ctx context.Context, return _c } -func (_c *MockBackupService_ListBackups_Call) Return(vs []domain.DatabaseBackupJob, err error) *MockBackupService_ListBackups_Call { - _c.Call.Return(vs, err) +func (_c *MockBackupService_ListBackups_Call) Return(backupJobs []domain.BackupJob, err error) *MockBackupService_ListBackups_Call { + _c.Call.Return(backupJobs, err) return _c } -func (_c *MockBackupService_ListBackups_Call) RunAndReturn(run func(ctx context.Context, domainName string) ([]domain.DatabaseBackupJob, error)) *MockBackupService_ListBackups_Call { +func (_c *MockBackupService_ListBackups_Call) RunAndReturn(run func(ctx context.Context, app string) ([]domain.BackupJob, error)) *MockBackupService_ListBackups_Call { _c.Call.Return(run) return _c } // Restore provides a mock function for the type MockBackupService -func (_mock *MockBackupService) Restore(ctx context.Context, domainName string, backupID string) error { - ret := _mock.Called(ctx, domainName, backupID) +func (_mock *MockBackupService) Restore(ctx context.Context, app string, backupID string) error { + ret := _mock.Called(ctx, app, backupID) if len(ret) == 0 { panic("no return value specified for Restore") @@ -185,7 +126,7 @@ func (_mock *MockBackupService) Restore(ctx context.Context, domainName string, var r0 error if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) error); ok { - r0 = returnFunc(ctx, domainName, backupID) + r0 = returnFunc(ctx, app, backupID) } else { r0 = ret.Error(0) } @@ -199,13 +140,13 @@ type MockBackupService_Restore_Call struct { // Restore is a helper method to define mock.On call // - ctx context.Context -// - domainName string +// - app string // - backupID string -func (_e *MockBackupService_Expecter) Restore(ctx any, domainName any, backupID any) *MockBackupService_Restore_Call { - return &MockBackupService_Restore_Call{Call: _e.mock.On("Restore", ctx, domainName, backupID)} +func (_e *MockBackupService_Expecter) Restore(ctx any, app any, backupID any) *MockBackupService_Restore_Call { + return &MockBackupService_Restore_Call{Call: _e.mock.On("Restore", ctx, app, backupID)} } -func (_c *MockBackupService_Restore_Call) Run(run func(ctx context.Context, domainName string, backupID string)) *MockBackupService_Restore_Call { +func (_c *MockBackupService_Restore_Call) Run(run func(ctx context.Context, app string, backupID string)) *MockBackupService_Restore_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -233,14 +174,14 @@ func (_c *MockBackupService_Restore_Call) Return(err error) *MockBackupService_R return _c } -func (_c *MockBackupService_Restore_Call) RunAndReturn(run func(ctx context.Context, domainName string, backupID string) error) *MockBackupService_Restore_Call { +func (_c *MockBackupService_Restore_Call) RunAndReturn(run func(ctx context.Context, app string, backupID string) error) *MockBackupService_Restore_Call { _c.Call.Return(run) return _c } // RestorePITR provides a mock function for the type MockBackupService -func (_mock *MockBackupService) RestorePITR(ctx context.Context, domainName string, targetTime time.Time) error { - ret := _mock.Called(ctx, domainName, targetTime) +func (_mock *MockBackupService) RestorePITR(ctx context.Context, app string, targetTime time.Time) error { + ret := _mock.Called(ctx, app, targetTime) if len(ret) == 0 { panic("no return value specified for RestorePITR") @@ -248,7 +189,7 @@ func (_mock *MockBackupService) RestorePITR(ctx context.Context, domainName stri var r0 error if returnFunc, ok := ret.Get(0).(func(context.Context, string, time.Time) error); ok { - r0 = returnFunc(ctx, domainName, targetTime) + r0 = returnFunc(ctx, app, targetTime) } else { r0 = ret.Error(0) } @@ -262,13 +203,13 @@ type MockBackupService_RestorePITR_Call struct { // RestorePITR is a helper method to define mock.On call // - ctx context.Context -// - domainName string +// - app string // - targetTime time.Time -func (_e *MockBackupService_Expecter) RestorePITR(ctx any, domainName any, targetTime any) *MockBackupService_RestorePITR_Call { - return &MockBackupService_RestorePITR_Call{Call: _e.mock.On("RestorePITR", ctx, domainName, targetTime)} +func (_e *MockBackupService_Expecter) RestorePITR(ctx any, app any, targetTime any) *MockBackupService_RestorePITR_Call { + return &MockBackupService_RestorePITR_Call{Call: _e.mock.On("RestorePITR", ctx, app, targetTime)} } -func (_c *MockBackupService_RestorePITR_Call) Run(run func(ctx context.Context, domainName string, targetTime time.Time)) *MockBackupService_RestorePITR_Call { +func (_c *MockBackupService_RestorePITR_Call) Run(run func(ctx context.Context, app string, targetTime time.Time)) *MockBackupService_RestorePITR_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -296,33 +237,33 @@ func (_c *MockBackupService_RestorePITR_Call) Return(err error) *MockBackupServi return _c } -func (_c *MockBackupService_RestorePITR_Call) RunAndReturn(run func(ctx context.Context, domainName string, targetTime time.Time) error) *MockBackupService_RestorePITR_Call { +func (_c *MockBackupService_RestorePITR_Call) RunAndReturn(run func(ctx context.Context, app string, targetTime time.Time) error) *MockBackupService_RestorePITR_Call { _c.Call.Return(run) return _c } // RunBackup provides a mock function for the type MockBackupService -func (_mock *MockBackupService) RunBackup(ctx context.Context, domainName string, dbName string) (*domain.DatabaseBackupResult, error) { - ret := _mock.Called(ctx, domainName, dbName) +func (_mock *MockBackupService) RunBackup(ctx context.Context, app string, service string, database string) (*domain.BackupResult, error) { + ret := _mock.Called(ctx, app, service, database) if len(ret) == 0 { panic("no return value specified for RunBackup") } - var r0 *domain.DatabaseBackupResult + var r0 *domain.BackupResult var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) (*domain.DatabaseBackupResult, error)); ok { - return returnFunc(ctx, domainName, dbName) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, string) (*domain.BackupResult, error)); ok { + return returnFunc(ctx, app, service, database) } - if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) *domain.DatabaseBackupResult); ok { - r0 = returnFunc(ctx, domainName, dbName) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, string) *domain.BackupResult); ok { + r0 = returnFunc(ctx, app, service, database) } else { if ret.Get(0) != nil { - r0 = ret.Get(0).(*domain.DatabaseBackupResult) + r0 = ret.Get(0).(*domain.BackupResult) } } - if returnFunc, ok := ret.Get(1).(func(context.Context, string, string) error); ok { - r1 = returnFunc(ctx, domainName, dbName) + if returnFunc, ok := ret.Get(1).(func(context.Context, string, string, string) error); ok { + r1 = returnFunc(ctx, app, service, database) } else { r1 = ret.Error(1) } @@ -336,13 +277,14 @@ type MockBackupService_RunBackup_Call struct { // RunBackup is a helper method to define mock.On call // - ctx context.Context -// - domainName string -// - dbName string -func (_e *MockBackupService_Expecter) RunBackup(ctx any, domainName any, dbName any) *MockBackupService_RunBackup_Call { - return &MockBackupService_RunBackup_Call{Call: _e.mock.On("RunBackup", ctx, domainName, dbName)} +// - app string +// - service string +// - database string +func (_e *MockBackupService_Expecter) RunBackup(ctx any, app any, service any, database any) *MockBackupService_RunBackup_Call { + return &MockBackupService_RunBackup_Call{Call: _e.mock.On("RunBackup", ctx, app, service, database)} } -func (_c *MockBackupService_RunBackup_Call) Run(run func(ctx context.Context, domainName string, dbName string)) *MockBackupService_RunBackup_Call { +func (_c *MockBackupService_RunBackup_Call) Run(run func(ctx context.Context, app string, service string, database string)) *MockBackupService_RunBackup_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -356,43 +298,48 @@ func (_c *MockBackupService_RunBackup_Call) Run(run func(ctx context.Context, do if args[2] != nil { arg2 = args[2].(string) } + var arg3 string + if args[3] != nil { + arg3 = args[3].(string) + } run( arg0, arg1, arg2, + arg3, ) }) return _c } -func (_c *MockBackupService_RunBackup_Call) Return(v *domain.DatabaseBackupResult, err error) *MockBackupService_RunBackup_Call { - _c.Call.Return(v, err) +func (_c *MockBackupService_RunBackup_Call) Return(backupResult *domain.BackupResult, err error) *MockBackupService_RunBackup_Call { + _c.Call.Return(backupResult, err) return _c } -func (_c *MockBackupService_RunBackup_Call) RunAndReturn(run func(ctx context.Context, domainName string, dbName string) (*domain.DatabaseBackupResult, error)) *MockBackupService_RunBackup_Call { +func (_c *MockBackupService_RunBackup_Call) RunAndReturn(run func(ctx context.Context, app string, service string, database string) (*domain.BackupResult, error)) *MockBackupService_RunBackup_Call { _c.Call.Return(run) return _c } // Status provides a mock function for the type MockBackupService -func (_mock *MockBackupService) Status(ctx context.Context) ([]domain.DatabaseBackupJob, error) { +func (_mock *MockBackupService) Status(ctx context.Context) ([]domain.BackupJob, error) { ret := _mock.Called(ctx) if len(ret) == 0 { panic("no return value specified for Status") } - var r0 []domain.DatabaseBackupJob + var r0 []domain.BackupJob var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context) ([]domain.DatabaseBackupJob, error)); ok { + if returnFunc, ok := ret.Get(0).(func(context.Context) ([]domain.BackupJob, error)); ok { return returnFunc(ctx) } - if returnFunc, ok := ret.Get(0).(func(context.Context) []domain.DatabaseBackupJob); ok { + if returnFunc, ok := ret.Get(0).(func(context.Context) []domain.BackupJob); ok { r0 = returnFunc(ctx) } else { if ret.Get(0) != nil { - r0 = ret.Get(0).([]domain.DatabaseBackupJob) + r0 = ret.Get(0).([]domain.BackupJob) } } if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { @@ -427,12 +374,12 @@ func (_c *MockBackupService_Status_Call) Run(run func(ctx context.Context)) *Moc return _c } -func (_c *MockBackupService_Status_Call) Return(vs []domain.DatabaseBackupJob, err error) *MockBackupService_Status_Call { - _c.Call.Return(vs, err) +func (_c *MockBackupService_Status_Call) Return(backupJobs []domain.BackupJob, err error) *MockBackupService_Status_Call { + _c.Call.Return(backupJobs, err) return _c } -func (_c *MockBackupService_Status_Call) RunAndReturn(run func(ctx context.Context) ([]domain.DatabaseBackupJob, error)) *MockBackupService_Status_Call { +func (_c *MockBackupService_Status_Call) RunAndReturn(run func(ctx context.Context) ([]domain.BackupJob, error)) *MockBackupService_Status_Call { _c.Call.Return(run) return _c } diff --git a/internal/boundaries/in/mocks/mock_config_service.go b/internal/boundaries/in/mocks/mock_config_service.go index d881f7ec2..dd2d134a3 100644 --- a/internal/boundaries/in/mocks/mock_config_service.go +++ b/internal/boundaries/in/mocks/mock_config_service.go @@ -7,7 +7,6 @@ package mocks import ( "context" - "github.com/bnema/gordon/internal/domain" mock "github.com/stretchr/testify/mock" ) @@ -17,10 +16,19 @@ func NewMockConfigService(t interface { mock.TestingT Cleanup(func()) }) *MockConfigService { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockConfigService{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } @@ -38,1263 +46,396 @@ func (_m *MockConfigService) EXPECT() *MockConfigService_Expecter { return &MockConfigService_Expecter{mock: &_m.Mock} } -// AddAttachment provides a mock function for the type MockConfigService -func (_mock *MockConfigService) AddAttachment(ctx context.Context, domainOrGroup string, image string) error { - ret := _mock.Called(ctx, domainOrGroup, image) +// GetDataDir provides a mock function for the type MockConfigService +func (_mock *MockConfigService) GetDataDir() string { + ret := _mock.Called() if len(ret) == 0 { - panic("no return value specified for AddAttachment") + panic("no return value specified for GetDataDir") } - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) error); ok { - r0 = returnFunc(ctx, domainOrGroup, image) + var r0 string + if returnFunc, ok := ret.Get(0).(func() string); ok { + r0 = returnFunc() } else { - r0 = ret.Error(0) + r0 = ret.Get(0).(string) } return r0 } -// MockConfigService_AddAttachment_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'AddAttachment' -type MockConfigService_AddAttachment_Call struct { +// MockConfigService_GetDataDir_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetDataDir' +type MockConfigService_GetDataDir_Call struct { *mock.Call } -// AddAttachment is a helper method to define mock.On call -// - ctx context.Context -// - domainOrGroup string -// - image string -func (_e *MockConfigService_Expecter) AddAttachment(ctx any, domainOrGroup any, image any) *MockConfigService_AddAttachment_Call { - return &MockConfigService_AddAttachment_Call{Call: _e.mock.On("AddAttachment", ctx, domainOrGroup, image)} +// GetDataDir is a helper method to define mock.On call +func (_e *MockConfigService_Expecter) GetDataDir() *MockConfigService_GetDataDir_Call { + return &MockConfigService_GetDataDir_Call{Call: _e.mock.On("GetDataDir")} } -func (_c *MockConfigService_AddAttachment_Call) Run(run func(ctx context.Context, domainOrGroup string, image string)) *MockConfigService_AddAttachment_Call { +func (_c *MockConfigService_GetDataDir_Call) Run(run func()) *MockConfigService_GetDataDir_Call { _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - var arg2 string - if args[2] != nil { - arg2 = args[2].(string) - } - run( - arg0, - arg1, - arg2, - ) + run() }) return _c } -func (_c *MockConfigService_AddAttachment_Call) Return(err error) *MockConfigService_AddAttachment_Call { - _c.Call.Return(err) +func (_c *MockConfigService_GetDataDir_Call) Return(s string) *MockConfigService_GetDataDir_Call { + _c.Call.Return(s) return _c } -func (_c *MockConfigService_AddAttachment_Call) RunAndReturn(run func(ctx context.Context, domainOrGroup string, image string) error) *MockConfigService_AddAttachment_Call { +func (_c *MockConfigService_GetDataDir_Call) RunAndReturn(run func() string) *MockConfigService_GetDataDir_Call { _c.Call.Return(run) return _c } -// AddAutoRouteAllowedDomain provides a mock function for the type MockConfigService -func (_mock *MockConfigService) AddAutoRouteAllowedDomain(ctx context.Context, pattern string) error { - ret := _mock.Called(ctx, pattern) +// GetExternalRoutes provides a mock function for the type MockConfigService +func (_mock *MockConfigService) GetExternalRoutes() map[string]string { + ret := _mock.Called() if len(ret) == 0 { - panic("no return value specified for AddAutoRouteAllowedDomain") + panic("no return value specified for GetExternalRoutes") } - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) error); ok { - r0 = returnFunc(ctx, pattern) + var r0 map[string]string + if returnFunc, ok := ret.Get(0).(func() map[string]string); ok { + r0 = returnFunc() } else { - r0 = ret.Error(0) + if ret.Get(0) != nil { + r0 = ret.Get(0).(map[string]string) + } } return r0 } -// MockConfigService_AddAutoRouteAllowedDomain_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'AddAutoRouteAllowedDomain' -type MockConfigService_AddAutoRouteAllowedDomain_Call struct { +// MockConfigService_GetExternalRoutes_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetExternalRoutes' +type MockConfigService_GetExternalRoutes_Call struct { *mock.Call } -// AddAutoRouteAllowedDomain is a helper method to define mock.On call -// - ctx context.Context -// - pattern string -func (_e *MockConfigService_Expecter) AddAutoRouteAllowedDomain(ctx any, pattern any) *MockConfigService_AddAutoRouteAllowedDomain_Call { - return &MockConfigService_AddAutoRouteAllowedDomain_Call{Call: _e.mock.On("AddAutoRouteAllowedDomain", ctx, pattern)} +// GetExternalRoutes is a helper method to define mock.On call +func (_e *MockConfigService_Expecter) GetExternalRoutes() *MockConfigService_GetExternalRoutes_Call { + return &MockConfigService_GetExternalRoutes_Call{Call: _e.mock.On("GetExternalRoutes")} } -func (_c *MockConfigService_AddAutoRouteAllowedDomain_Call) Run(run func(ctx context.Context, pattern string)) *MockConfigService_AddAutoRouteAllowedDomain_Call { +func (_c *MockConfigService_GetExternalRoutes_Call) Run(run func()) *MockConfigService_GetExternalRoutes_Call { _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) + run() }) return _c } -func (_c *MockConfigService_AddAutoRouteAllowedDomain_Call) Return(err error) *MockConfigService_AddAutoRouteAllowedDomain_Call { - _c.Call.Return(err) +func (_c *MockConfigService_GetExternalRoutes_Call) Return(stringToString map[string]string) *MockConfigService_GetExternalRoutes_Call { + _c.Call.Return(stringToString) return _c } -func (_c *MockConfigService_AddAutoRouteAllowedDomain_Call) RunAndReturn(run func(ctx context.Context, pattern string) error) *MockConfigService_AddAutoRouteAllowedDomain_Call { +func (_c *MockConfigService_GetExternalRoutes_Call) RunAndReturn(run func() map[string]string) *MockConfigService_GetExternalRoutes_Call { _c.Call.Return(run) return _c } -// AddRoute provides a mock function for the type MockConfigService -func (_mock *MockConfigService) AddRoute(ctx context.Context, route domain.Route) error { - ret := _mock.Called(ctx, route) +// GetNetworkPrefix provides a mock function for the type MockConfigService +func (_mock *MockConfigService) GetNetworkPrefix() string { + ret := _mock.Called() if len(ret) == 0 { - panic("no return value specified for AddRoute") + panic("no return value specified for GetNetworkPrefix") } - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, domain.Route) error); ok { - r0 = returnFunc(ctx, route) + var r0 string + if returnFunc, ok := ret.Get(0).(func() string); ok { + r0 = returnFunc() } else { - r0 = ret.Error(0) + r0 = ret.Get(0).(string) } return r0 } -// MockConfigService_AddRoute_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'AddRoute' -type MockConfigService_AddRoute_Call struct { +// MockConfigService_GetNetworkPrefix_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetNetworkPrefix' +type MockConfigService_GetNetworkPrefix_Call struct { *mock.Call } -// AddRoute is a helper method to define mock.On call -// - ctx context.Context -// - route domain.Route -func (_e *MockConfigService_Expecter) AddRoute(ctx any, route any) *MockConfigService_AddRoute_Call { - return &MockConfigService_AddRoute_Call{Call: _e.mock.On("AddRoute", ctx, route)} +// GetNetworkPrefix is a helper method to define mock.On call +func (_e *MockConfigService_Expecter) GetNetworkPrefix() *MockConfigService_GetNetworkPrefix_Call { + return &MockConfigService_GetNetworkPrefix_Call{Call: _e.mock.On("GetNetworkPrefix")} } -func (_c *MockConfigService_AddRoute_Call) Run(run func(ctx context.Context, route domain.Route)) *MockConfigService_AddRoute_Call { +func (_c *MockConfigService_GetNetworkPrefix_Call) Run(run func()) *MockConfigService_GetNetworkPrefix_Call { _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 domain.Route - if args[1] != nil { - arg1 = args[1].(domain.Route) - } - run( - arg0, - arg1, - ) + run() }) return _c } -func (_c *MockConfigService_AddRoute_Call) Return(err error) *MockConfigService_AddRoute_Call { - _c.Call.Return(err) +func (_c *MockConfigService_GetNetworkPrefix_Call) Return(s string) *MockConfigService_GetNetworkPrefix_Call { + _c.Call.Return(s) return _c } -func (_c *MockConfigService_AddRoute_Call) RunAndReturn(run func(ctx context.Context, route domain.Route) error) *MockConfigService_AddRoute_Call { +func (_c *MockConfigService_GetNetworkPrefix_Call) RunAndReturn(run func() string) *MockConfigService_GetNetworkPrefix_Call { _c.Call.Return(run) return _c } -// FindAttachmentTargetsByImage provides a mock function for the type MockConfigService -func (_mock *MockConfigService) FindAttachmentTargetsByImage(ctx context.Context, imageName string) []string { - ret := _mock.Called(ctx, imageName) +// GetRegistryDomain provides a mock function for the type MockConfigService +func (_mock *MockConfigService) GetRegistryDomain() string { + ret := _mock.Called() if len(ret) == 0 { - panic("no return value specified for FindAttachmentTargetsByImage") + panic("no return value specified for GetRegistryDomain") } - var r0 []string - if returnFunc, ok := ret.Get(0).(func(context.Context, string) []string); ok { - r0 = returnFunc(ctx, imageName) + var r0 string + if returnFunc, ok := ret.Get(0).(func() string); ok { + r0 = returnFunc() } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]string) - } + r0 = ret.Get(0).(string) } return r0 } -// MockConfigService_FindAttachmentTargetsByImage_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'FindAttachmentTargetsByImage' -type MockConfigService_FindAttachmentTargetsByImage_Call struct { +// MockConfigService_GetRegistryDomain_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetRegistryDomain' +type MockConfigService_GetRegistryDomain_Call struct { *mock.Call } -// FindAttachmentTargetsByImage is a helper method to define mock.On call -// - ctx context.Context -// - imageName string -func (_e *MockConfigService_Expecter) FindAttachmentTargetsByImage(ctx any, imageName any) *MockConfigService_FindAttachmentTargetsByImage_Call { - return &MockConfigService_FindAttachmentTargetsByImage_Call{Call: _e.mock.On("FindAttachmentTargetsByImage", ctx, imageName)} +// GetRegistryDomain is a helper method to define mock.On call +func (_e *MockConfigService_Expecter) GetRegistryDomain() *MockConfigService_GetRegistryDomain_Call { + return &MockConfigService_GetRegistryDomain_Call{Call: _e.mock.On("GetRegistryDomain")} } -func (_c *MockConfigService_FindAttachmentTargetsByImage_Call) Run(run func(ctx context.Context, imageName string)) *MockConfigService_FindAttachmentTargetsByImage_Call { +func (_c *MockConfigService_GetRegistryDomain_Call) Run(run func()) *MockConfigService_GetRegistryDomain_Call { _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) + run() }) return _c } -func (_c *MockConfigService_FindAttachmentTargetsByImage_Call) Return(strings []string) *MockConfigService_FindAttachmentTargetsByImage_Call { - _c.Call.Return(strings) +func (_c *MockConfigService_GetRegistryDomain_Call) Return(s string) *MockConfigService_GetRegistryDomain_Call { + _c.Call.Return(s) return _c } -func (_c *MockConfigService_FindAttachmentTargetsByImage_Call) RunAndReturn(run func(ctx context.Context, imageName string) []string) *MockConfigService_FindAttachmentTargetsByImage_Call { +func (_c *MockConfigService_GetRegistryDomain_Call) RunAndReturn(run func() string) *MockConfigService_GetRegistryDomain_Call { _c.Call.Return(run) return _c } -// FindRoutesByImage provides a mock function for the type MockConfigService -func (_mock *MockConfigService) FindRoutesByImage(ctx context.Context, imageName string) []domain.Route { - ret := _mock.Called(ctx, imageName) +// GetRegistryPort provides a mock function for the type MockConfigService +func (_mock *MockConfigService) GetRegistryPort() int { + ret := _mock.Called() if len(ret) == 0 { - panic("no return value specified for FindRoutesByImage") + panic("no return value specified for GetRegistryPort") } - var r0 []domain.Route - if returnFunc, ok := ret.Get(0).(func(context.Context, string) []domain.Route); ok { - r0 = returnFunc(ctx, imageName) + var r0 int + if returnFunc, ok := ret.Get(0).(func() int); ok { + r0 = returnFunc() } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]domain.Route) - } + r0 = ret.Get(0).(int) } return r0 } -// MockConfigService_FindRoutesByImage_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'FindRoutesByImage' -type MockConfigService_FindRoutesByImage_Call struct { +// MockConfigService_GetRegistryPort_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetRegistryPort' +type MockConfigService_GetRegistryPort_Call struct { *mock.Call } -// FindRoutesByImage is a helper method to define mock.On call -// - ctx context.Context -// - imageName string -func (_e *MockConfigService_Expecter) FindRoutesByImage(ctx any, imageName any) *MockConfigService_FindRoutesByImage_Call { - return &MockConfigService_FindRoutesByImage_Call{Call: _e.mock.On("FindRoutesByImage", ctx, imageName)} +// GetRegistryPort is a helper method to define mock.On call +func (_e *MockConfigService_Expecter) GetRegistryPort() *MockConfigService_GetRegistryPort_Call { + return &MockConfigService_GetRegistryPort_Call{Call: _e.mock.On("GetRegistryPort")} } -func (_c *MockConfigService_FindRoutesByImage_Call) Run(run func(ctx context.Context, imageName string)) *MockConfigService_FindRoutesByImage_Call { +func (_c *MockConfigService_GetRegistryPort_Call) Run(run func()) *MockConfigService_GetRegistryPort_Call { _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) + run() }) return _c } -func (_c *MockConfigService_FindRoutesByImage_Call) Return(routes []domain.Route) *MockConfigService_FindRoutesByImage_Call { - _c.Call.Return(routes) +func (_c *MockConfigService_GetRegistryPort_Call) Return(n int) *MockConfigService_GetRegistryPort_Call { + _c.Call.Return(n) return _c } -func (_c *MockConfigService_FindRoutesByImage_Call) RunAndReturn(run func(ctx context.Context, imageName string) []domain.Route) *MockConfigService_FindRoutesByImage_Call { +func (_c *MockConfigService_GetRegistryPort_Call) RunAndReturn(run func() int) *MockConfigService_GetRegistryPort_Call { _c.Call.Return(run) return _c } -// GetAllAttachments provides a mock function for the type MockConfigService -func (_mock *MockConfigService) GetAllAttachments(ctx context.Context) map[string][]string { - ret := _mock.Called(ctx) +// GetServerPort provides a mock function for the type MockConfigService +func (_mock *MockConfigService) GetServerPort() int { + ret := _mock.Called() if len(ret) == 0 { - panic("no return value specified for GetAllAttachments") + panic("no return value specified for GetServerPort") } - var r0 map[string][]string - if returnFunc, ok := ret.Get(0).(func(context.Context) map[string][]string); ok { - r0 = returnFunc(ctx) + var r0 int + if returnFunc, ok := ret.Get(0).(func() int); ok { + r0 = returnFunc() } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).(map[string][]string) - } + r0 = ret.Get(0).(int) } return r0 } -// MockConfigService_GetAllAttachments_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetAllAttachments' -type MockConfigService_GetAllAttachments_Call struct { +// MockConfigService_GetServerPort_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetServerPort' +type MockConfigService_GetServerPort_Call struct { *mock.Call } -// GetAllAttachments is a helper method to define mock.On call -// - ctx context.Context -func (_e *MockConfigService_Expecter) GetAllAttachments(ctx any) *MockConfigService_GetAllAttachments_Call { - return &MockConfigService_GetAllAttachments_Call{Call: _e.mock.On("GetAllAttachments", ctx)} +// GetServerPort is a helper method to define mock.On call +func (_e *MockConfigService_Expecter) GetServerPort() *MockConfigService_GetServerPort_Call { + return &MockConfigService_GetServerPort_Call{Call: _e.mock.On("GetServerPort")} } -func (_c *MockConfigService_GetAllAttachments_Call) Run(run func(ctx context.Context)) *MockConfigService_GetAllAttachments_Call { +func (_c *MockConfigService_GetServerPort_Call) Run(run func()) *MockConfigService_GetServerPort_Call { _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - run( - arg0, - ) + run() }) return _c } -func (_c *MockConfigService_GetAllAttachments_Call) Return(stringToStrings map[string][]string) *MockConfigService_GetAllAttachments_Call { - _c.Call.Return(stringToStrings) +func (_c *MockConfigService_GetServerPort_Call) Return(n int) *MockConfigService_GetServerPort_Call { + _c.Call.Return(n) return _c } -func (_c *MockConfigService_GetAllAttachments_Call) RunAndReturn(run func(ctx context.Context) map[string][]string) *MockConfigService_GetAllAttachments_Call { +func (_c *MockConfigService_GetServerPort_Call) RunAndReturn(run func() int) *MockConfigService_GetServerPort_Call { _c.Call.Return(run) return _c } -// GetAllowedDomains provides a mock function for the type MockConfigService -func (_mock *MockConfigService) GetAllowedDomains() []string { +// IsNetworkIsolationEnabled provides a mock function for the type MockConfigService +func (_mock *MockConfigService) IsNetworkIsolationEnabled() bool { ret := _mock.Called() if len(ret) == 0 { - panic("no return value specified for GetAllowedDomains") + panic("no return value specified for IsNetworkIsolationEnabled") } - var r0 []string - if returnFunc, ok := ret.Get(0).(func() []string); ok { + var r0 bool + if returnFunc, ok := ret.Get(0).(func() bool); ok { r0 = returnFunc() } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]string) - } + r0 = ret.Get(0).(bool) } return r0 } -// MockConfigService_GetAllowedDomains_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetAllowedDomains' -type MockConfigService_GetAllowedDomains_Call struct { +// MockConfigService_IsNetworkIsolationEnabled_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'IsNetworkIsolationEnabled' +type MockConfigService_IsNetworkIsolationEnabled_Call struct { *mock.Call } -// GetAllowedDomains is a helper method to define mock.On call -func (_e *MockConfigService_Expecter) GetAllowedDomains() *MockConfigService_GetAllowedDomains_Call { - return &MockConfigService_GetAllowedDomains_Call{Call: _e.mock.On("GetAllowedDomains")} +// IsNetworkIsolationEnabled is a helper method to define mock.On call +func (_e *MockConfigService_Expecter) IsNetworkIsolationEnabled() *MockConfigService_IsNetworkIsolationEnabled_Call { + return &MockConfigService_IsNetworkIsolationEnabled_Call{Call: _e.mock.On("IsNetworkIsolationEnabled")} } -func (_c *MockConfigService_GetAllowedDomains_Call) Run(run func()) *MockConfigService_GetAllowedDomains_Call { +func (_c *MockConfigService_IsNetworkIsolationEnabled_Call) Run(run func()) *MockConfigService_IsNetworkIsolationEnabled_Call { _c.Call.Run(func(args mock.Arguments) { run() }) return _c } -func (_c *MockConfigService_GetAllowedDomains_Call) Return(strings []string) *MockConfigService_GetAllowedDomains_Call { - _c.Call.Return(strings) +func (_c *MockConfigService_IsNetworkIsolationEnabled_Call) Return(b bool) *MockConfigService_IsNetworkIsolationEnabled_Call { + _c.Call.Return(b) return _c } -func (_c *MockConfigService_GetAllowedDomains_Call) RunAndReturn(run func() []string) *MockConfigService_GetAllowedDomains_Call { +func (_c *MockConfigService_IsNetworkIsolationEnabled_Call) RunAndReturn(run func() bool) *MockConfigService_IsNetworkIsolationEnabled_Call { _c.Call.Return(run) return _c } -// GetAttachmentsFor provides a mock function for the type MockConfigService -func (_mock *MockConfigService) GetAttachmentsFor(ctx context.Context, domainOrGroup string) ([]string, error) { - ret := _mock.Called(ctx, domainOrGroup) +// Load provides a mock function for the type MockConfigService +func (_mock *MockConfigService) Load(ctx context.Context) error { + ret := _mock.Called(ctx) if len(ret) == 0 { - panic("no return value specified for GetAttachmentsFor") + panic("no return value specified for Load") } - var r0 []string - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) ([]string, error)); ok { - return returnFunc(ctx, domainOrGroup) - } - if returnFunc, ok := ret.Get(0).(func(context.Context, string) []string); ok { - r0 = returnFunc(ctx, domainOrGroup) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]string) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { - r1 = returnFunc(ctx, domainOrGroup) + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context) error); ok { + r0 = returnFunc(ctx) } else { - r1 = ret.Error(1) + r0 = ret.Error(0) } - return r0, r1 + return r0 } -// MockConfigService_GetAttachmentsFor_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetAttachmentsFor' -type MockConfigService_GetAttachmentsFor_Call struct { +// MockConfigService_Load_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Load' +type MockConfigService_Load_Call struct { *mock.Call } -// GetAttachmentsFor is a helper method to define mock.On call +// Load is a helper method to define mock.On call // - ctx context.Context -// - domainOrGroup string -func (_e *MockConfigService_Expecter) GetAttachmentsFor(ctx any, domainOrGroup any) *MockConfigService_GetAttachmentsFor_Call { - return &MockConfigService_GetAttachmentsFor_Call{Call: _e.mock.On("GetAttachmentsFor", ctx, domainOrGroup)} +func (_e *MockConfigService_Expecter) Load(ctx any) *MockConfigService_Load_Call { + return &MockConfigService_Load_Call{Call: _e.mock.On("Load", ctx)} } -func (_c *MockConfigService_GetAttachmentsFor_Call) Run(run func(ctx context.Context, domainOrGroup string)) *MockConfigService_GetAttachmentsFor_Call { +func (_c *MockConfigService_Load_Call) Run(run func(ctx context.Context)) *MockConfigService_Load_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { arg0 = args[0].(context.Context) } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } run( arg0, - arg1, ) }) return _c } -func (_c *MockConfigService_GetAttachmentsFor_Call) Return(strings []string, err error) *MockConfigService_GetAttachmentsFor_Call { - _c.Call.Return(strings, err) +func (_c *MockConfigService_Load_Call) Return(err error) *MockConfigService_Load_Call { + _c.Call.Return(err) return _c } -func (_c *MockConfigService_GetAttachmentsFor_Call) RunAndReturn(run func(ctx context.Context, domainOrGroup string) ([]string, error)) *MockConfigService_GetAttachmentsFor_Call { +func (_c *MockConfigService_Load_Call) RunAndReturn(run func(ctx context.Context) error) *MockConfigService_Load_Call { _c.Call.Return(run) return _c } -// GetAutoRouteAllowedDomains provides a mock function for the type MockConfigService -func (_mock *MockConfigService) GetAutoRouteAllowedDomains(ctx context.Context) ([]string, error) { +// Reload provides a mock function for the type MockConfigService +func (_mock *MockConfigService) Reload(ctx context.Context) error { ret := _mock.Called(ctx) if len(ret) == 0 { - panic("no return value specified for GetAutoRouteAllowedDomains") + panic("no return value specified for Reload") } - var r0 []string - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context) ([]string, error)); ok { - return returnFunc(ctx) - } - if returnFunc, ok := ret.Get(0).(func(context.Context) []string); ok { + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context) error); ok { r0 = returnFunc(ctx) } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]string) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { - r1 = returnFunc(ctx) - } else { - r1 = ret.Error(1) + r0 = ret.Error(0) } - return r0, r1 + return r0 } -// MockConfigService_GetAutoRouteAllowedDomains_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetAutoRouteAllowedDomains' -type MockConfigService_GetAutoRouteAllowedDomains_Call struct { +// MockConfigService_Reload_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Reload' +type MockConfigService_Reload_Call struct { *mock.Call } -// GetAutoRouteAllowedDomains is a helper method to define mock.On call +// Reload is a helper method to define mock.On call // - ctx context.Context -func (_e *MockConfigService_Expecter) GetAutoRouteAllowedDomains(ctx any) *MockConfigService_GetAutoRouteAllowedDomains_Call { - return &MockConfigService_GetAutoRouteAllowedDomains_Call{Call: _e.mock.On("GetAutoRouteAllowedDomains", ctx)} +func (_e *MockConfigService_Expecter) Reload(ctx any) *MockConfigService_Reload_Call { + return &MockConfigService_Reload_Call{Call: _e.mock.On("Reload", ctx)} } -func (_c *MockConfigService_GetAutoRouteAllowedDomains_Call) Run(run func(ctx context.Context)) *MockConfigService_GetAutoRouteAllowedDomains_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - run( - arg0, - ) - }) - return _c -} - -func (_c *MockConfigService_GetAutoRouteAllowedDomains_Call) Return(strings []string, err error) *MockConfigService_GetAutoRouteAllowedDomains_Call { - _c.Call.Return(strings, err) - return _c -} - -func (_c *MockConfigService_GetAutoRouteAllowedDomains_Call) RunAndReturn(run func(ctx context.Context) ([]string, error)) *MockConfigService_GetAutoRouteAllowedDomains_Call { - _c.Call.Return(run) - return _c -} - -// GetDataDir provides a mock function for the type MockConfigService -func (_mock *MockConfigService) GetDataDir() string { - ret := _mock.Called() - - if len(ret) == 0 { - panic("no return value specified for GetDataDir") - } - - var r0 string - if returnFunc, ok := ret.Get(0).(func() string); ok { - r0 = returnFunc() - } else { - r0 = ret.Get(0).(string) - } - return r0 -} - -// MockConfigService_GetDataDir_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetDataDir' -type MockConfigService_GetDataDir_Call struct { - *mock.Call -} - -// GetDataDir is a helper method to define mock.On call -func (_e *MockConfigService_Expecter) GetDataDir() *MockConfigService_GetDataDir_Call { - return &MockConfigService_GetDataDir_Call{Call: _e.mock.On("GetDataDir")} -} - -func (_c *MockConfigService_GetDataDir_Call) Run(run func()) *MockConfigService_GetDataDir_Call { - _c.Call.Run(func(args mock.Arguments) { - run() - }) - return _c -} - -func (_c *MockConfigService_GetDataDir_Call) Return(s string) *MockConfigService_GetDataDir_Call { - _c.Call.Return(s) - return _c -} - -func (_c *MockConfigService_GetDataDir_Call) RunAndReturn(run func() string) *MockConfigService_GetDataDir_Call { - _c.Call.Return(run) - return _c -} - -// GetExternalRoutes provides a mock function for the type MockConfigService -func (_mock *MockConfigService) GetExternalRoutes() map[string]string { - ret := _mock.Called() - - if len(ret) == 0 { - panic("no return value specified for GetExternalRoutes") - } - - var r0 map[string]string - if returnFunc, ok := ret.Get(0).(func() map[string]string); ok { - r0 = returnFunc() - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).(map[string]string) - } - } - return r0 -} - -// MockConfigService_GetExternalRoutes_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetExternalRoutes' -type MockConfigService_GetExternalRoutes_Call struct { - *mock.Call -} - -// GetExternalRoutes is a helper method to define mock.On call -func (_e *MockConfigService_Expecter) GetExternalRoutes() *MockConfigService_GetExternalRoutes_Call { - return &MockConfigService_GetExternalRoutes_Call{Call: _e.mock.On("GetExternalRoutes")} -} - -func (_c *MockConfigService_GetExternalRoutes_Call) Run(run func()) *MockConfigService_GetExternalRoutes_Call { - _c.Call.Run(func(args mock.Arguments) { - run() - }) - return _c -} - -func (_c *MockConfigService_GetExternalRoutes_Call) Return(stringToString map[string]string) *MockConfigService_GetExternalRoutes_Call { - _c.Call.Return(stringToString) - return _c -} - -func (_c *MockConfigService_GetExternalRoutes_Call) RunAndReturn(run func() map[string]string) *MockConfigService_GetExternalRoutes_Call { - _c.Call.Return(run) - return _c -} - -// GetNetworkPrefix provides a mock function for the type MockConfigService -func (_mock *MockConfigService) GetNetworkPrefix() string { - ret := _mock.Called() - - if len(ret) == 0 { - panic("no return value specified for GetNetworkPrefix") - } - - var r0 string - if returnFunc, ok := ret.Get(0).(func() string); ok { - r0 = returnFunc() - } else { - r0 = ret.Get(0).(string) - } - return r0 -} - -// MockConfigService_GetNetworkPrefix_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetNetworkPrefix' -type MockConfigService_GetNetworkPrefix_Call struct { - *mock.Call -} - -// GetNetworkPrefix is a helper method to define mock.On call -func (_e *MockConfigService_Expecter) GetNetworkPrefix() *MockConfigService_GetNetworkPrefix_Call { - return &MockConfigService_GetNetworkPrefix_Call{Call: _e.mock.On("GetNetworkPrefix")} -} - -func (_c *MockConfigService_GetNetworkPrefix_Call) Run(run func()) *MockConfigService_GetNetworkPrefix_Call { - _c.Call.Run(func(args mock.Arguments) { - run() - }) - return _c -} - -func (_c *MockConfigService_GetNetworkPrefix_Call) Return(s string) *MockConfigService_GetNetworkPrefix_Call { - _c.Call.Return(s) - return _c -} - -func (_c *MockConfigService_GetNetworkPrefix_Call) RunAndReturn(run func() string) *MockConfigService_GetNetworkPrefix_Call { - _c.Call.Return(run) - return _c -} - -// GetPreviewConfig provides a mock function for the type MockConfigService -func (_mock *MockConfigService) GetPreviewConfig() domain.PreviewConfig { - ret := _mock.Called() - - if len(ret) == 0 { - panic("no return value specified for GetPreviewConfig") - } - - var r0 domain.PreviewConfig - if returnFunc, ok := ret.Get(0).(func() domain.PreviewConfig); ok { - r0 = returnFunc() - } else { - r0 = ret.Get(0).(domain.PreviewConfig) - } - return r0 -} - -// MockConfigService_GetPreviewConfig_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetPreviewConfig' -type MockConfigService_GetPreviewConfig_Call struct { - *mock.Call -} - -// GetPreviewConfig is a helper method to define mock.On call -func (_e *MockConfigService_Expecter) GetPreviewConfig() *MockConfigService_GetPreviewConfig_Call { - return &MockConfigService_GetPreviewConfig_Call{Call: _e.mock.On("GetPreviewConfig")} -} - -func (_c *MockConfigService_GetPreviewConfig_Call) Run(run func()) *MockConfigService_GetPreviewConfig_Call { - _c.Call.Run(func(args mock.Arguments) { - run() - }) - return _c -} - -func (_c *MockConfigService_GetPreviewConfig_Call) Return(previewConfig domain.PreviewConfig) *MockConfigService_GetPreviewConfig_Call { - _c.Call.Return(previewConfig) - return _c -} - -func (_c *MockConfigService_GetPreviewConfig_Call) RunAndReturn(run func() domain.PreviewConfig) *MockConfigService_GetPreviewConfig_Call { - _c.Call.Return(run) - return _c -} - -// GetPreviewTagPatterns provides a mock function for the type MockConfigService -func (_mock *MockConfigService) GetPreviewTagPatterns() []string { - ret := _mock.Called() - - if len(ret) == 0 { - panic("no return value specified for GetPreviewTagPatterns") - } - - var r0 []string - if returnFunc, ok := ret.Get(0).(func() []string); ok { - r0 = returnFunc() - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]string) - } - } - return r0 -} - -// MockConfigService_GetPreviewTagPatterns_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetPreviewTagPatterns' -type MockConfigService_GetPreviewTagPatterns_Call struct { - *mock.Call -} - -// GetPreviewTagPatterns is a helper method to define mock.On call -func (_e *MockConfigService_Expecter) GetPreviewTagPatterns() *MockConfigService_GetPreviewTagPatterns_Call { - return &MockConfigService_GetPreviewTagPatterns_Call{Call: _e.mock.On("GetPreviewTagPatterns")} -} - -func (_c *MockConfigService_GetPreviewTagPatterns_Call) Run(run func()) *MockConfigService_GetPreviewTagPatterns_Call { - _c.Call.Run(func(args mock.Arguments) { - run() - }) - return _c -} - -func (_c *MockConfigService_GetPreviewTagPatterns_Call) Return(strings []string) *MockConfigService_GetPreviewTagPatterns_Call { - _c.Call.Return(strings) - return _c -} - -func (_c *MockConfigService_GetPreviewTagPatterns_Call) RunAndReturn(run func() []string) *MockConfigService_GetPreviewTagPatterns_Call { - _c.Call.Return(run) - return _c -} - -// GetRegistryDomain provides a mock function for the type MockConfigService -func (_mock *MockConfigService) GetRegistryDomain() string { - ret := _mock.Called() - - if len(ret) == 0 { - panic("no return value specified for GetRegistryDomain") - } - - var r0 string - if returnFunc, ok := ret.Get(0).(func() string); ok { - r0 = returnFunc() - } else { - r0 = ret.Get(0).(string) - } - return r0 -} - -// MockConfigService_GetRegistryDomain_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetRegistryDomain' -type MockConfigService_GetRegistryDomain_Call struct { - *mock.Call -} - -// GetRegistryDomain is a helper method to define mock.On call -func (_e *MockConfigService_Expecter) GetRegistryDomain() *MockConfigService_GetRegistryDomain_Call { - return &MockConfigService_GetRegistryDomain_Call{Call: _e.mock.On("GetRegistryDomain")} -} - -func (_c *MockConfigService_GetRegistryDomain_Call) Run(run func()) *MockConfigService_GetRegistryDomain_Call { - _c.Call.Run(func(args mock.Arguments) { - run() - }) - return _c -} - -func (_c *MockConfigService_GetRegistryDomain_Call) Return(s string) *MockConfigService_GetRegistryDomain_Call { - _c.Call.Return(s) - return _c -} - -func (_c *MockConfigService_GetRegistryDomain_Call) RunAndReturn(run func() string) *MockConfigService_GetRegistryDomain_Call { - _c.Call.Return(run) - return _c -} - -// GetRegistryPort provides a mock function for the type MockConfigService -func (_mock *MockConfigService) GetRegistryPort() int { - ret := _mock.Called() - - if len(ret) == 0 { - panic("no return value specified for GetRegistryPort") - } - - var r0 int - if returnFunc, ok := ret.Get(0).(func() int); ok { - r0 = returnFunc() - } else { - r0 = ret.Get(0).(int) - } - return r0 -} - -// MockConfigService_GetRegistryPort_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetRegistryPort' -type MockConfigService_GetRegistryPort_Call struct { - *mock.Call -} - -// GetRegistryPort is a helper method to define mock.On call -func (_e *MockConfigService_Expecter) GetRegistryPort() *MockConfigService_GetRegistryPort_Call { - return &MockConfigService_GetRegistryPort_Call{Call: _e.mock.On("GetRegistryPort")} -} - -func (_c *MockConfigService_GetRegistryPort_Call) Run(run func()) *MockConfigService_GetRegistryPort_Call { - _c.Call.Run(func(args mock.Arguments) { - run() - }) - return _c -} - -func (_c *MockConfigService_GetRegistryPort_Call) Return(n int) *MockConfigService_GetRegistryPort_Call { - _c.Call.Return(n) - return _c -} - -func (_c *MockConfigService_GetRegistryPort_Call) RunAndReturn(run func() int) *MockConfigService_GetRegistryPort_Call { - _c.Call.Return(run) - return _c -} - -// GetRoute provides a mock function for the type MockConfigService -func (_mock *MockConfigService) GetRoute(ctx context.Context, domain1 string) (*domain.Route, error) { - ret := _mock.Called(ctx, domain1) - - if len(ret) == 0 { - panic("no return value specified for GetRoute") - } - - var r0 *domain.Route - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) (*domain.Route, error)); ok { - return returnFunc(ctx, domain1) - } - if returnFunc, ok := ret.Get(0).(func(context.Context, string) *domain.Route); ok { - r0 = returnFunc(ctx, domain1) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).(*domain.Route) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { - r1 = returnFunc(ctx, domain1) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockConfigService_GetRoute_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetRoute' -type MockConfigService_GetRoute_Call struct { - *mock.Call -} - -// GetRoute is a helper method to define mock.On call -// - ctx context.Context -// - domain1 string -func (_e *MockConfigService_Expecter) GetRoute(ctx any, domain1 any) *MockConfigService_GetRoute_Call { - return &MockConfigService_GetRoute_Call{Call: _e.mock.On("GetRoute", ctx, domain1)} -} - -func (_c *MockConfigService_GetRoute_Call) Run(run func(ctx context.Context, domain1 string)) *MockConfigService_GetRoute_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockConfigService_GetRoute_Call) Return(route *domain.Route, err error) *MockConfigService_GetRoute_Call { - _c.Call.Return(route, err) - return _c -} - -func (_c *MockConfigService_GetRoute_Call) RunAndReturn(run func(ctx context.Context, domain1 string) (*domain.Route, error)) *MockConfigService_GetRoute_Call { - _c.Call.Return(run) - return _c -} - -// GetRoutes provides a mock function for the type MockConfigService -func (_mock *MockConfigService) GetRoutes(ctx context.Context) []domain.Route { - ret := _mock.Called(ctx) - - if len(ret) == 0 { - panic("no return value specified for GetRoutes") - } - - var r0 []domain.Route - if returnFunc, ok := ret.Get(0).(func(context.Context) []domain.Route); ok { - r0 = returnFunc(ctx) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]domain.Route) - } - } - return r0 -} - -// MockConfigService_GetRoutes_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetRoutes' -type MockConfigService_GetRoutes_Call struct { - *mock.Call -} - -// GetRoutes is a helper method to define mock.On call -// - ctx context.Context -func (_e *MockConfigService_Expecter) GetRoutes(ctx any) *MockConfigService_GetRoutes_Call { - return &MockConfigService_GetRoutes_Call{Call: _e.mock.On("GetRoutes", ctx)} -} - -func (_c *MockConfigService_GetRoutes_Call) Run(run func(ctx context.Context)) *MockConfigService_GetRoutes_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - run( - arg0, - ) - }) - return _c -} - -func (_c *MockConfigService_GetRoutes_Call) Return(routes []domain.Route) *MockConfigService_GetRoutes_Call { - _c.Call.Return(routes) - return _c -} - -func (_c *MockConfigService_GetRoutes_Call) RunAndReturn(run func(ctx context.Context) []domain.Route) *MockConfigService_GetRoutes_Call { - _c.Call.Return(run) - return _c -} - -// GetServerPort provides a mock function for the type MockConfigService -func (_mock *MockConfigService) GetServerPort() int { - ret := _mock.Called() - - if len(ret) == 0 { - panic("no return value specified for GetServerPort") - } - - var r0 int - if returnFunc, ok := ret.Get(0).(func() int); ok { - r0 = returnFunc() - } else { - r0 = ret.Get(0).(int) - } - return r0 -} - -// MockConfigService_GetServerPort_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetServerPort' -type MockConfigService_GetServerPort_Call struct { - *mock.Call -} - -// GetServerPort is a helper method to define mock.On call -func (_e *MockConfigService_Expecter) GetServerPort() *MockConfigService_GetServerPort_Call { - return &MockConfigService_GetServerPort_Call{Call: _e.mock.On("GetServerPort")} -} - -func (_c *MockConfigService_GetServerPort_Call) Run(run func()) *MockConfigService_GetServerPort_Call { - _c.Call.Run(func(args mock.Arguments) { - run() - }) - return _c -} - -func (_c *MockConfigService_GetServerPort_Call) Return(n int) *MockConfigService_GetServerPort_Call { - _c.Call.Return(n) - return _c -} - -func (_c *MockConfigService_GetServerPort_Call) RunAndReturn(run func() int) *MockConfigService_GetServerPort_Call { - _c.Call.Return(run) - return _c -} - -// IsAutoEnabled provides a mock function for the type MockConfigService -func (_mock *MockConfigService) IsAutoEnabled() bool { - ret := _mock.Called() - - if len(ret) == 0 { - panic("no return value specified for IsAutoEnabled") - } - - var r0 bool - if returnFunc, ok := ret.Get(0).(func() bool); ok { - r0 = returnFunc() - } else { - r0 = ret.Get(0).(bool) - } - return r0 -} - -// MockConfigService_IsAutoEnabled_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'IsAutoEnabled' -type MockConfigService_IsAutoEnabled_Call struct { - *mock.Call -} - -// IsAutoEnabled is a helper method to define mock.On call -func (_e *MockConfigService_Expecter) IsAutoEnabled() *MockConfigService_IsAutoEnabled_Call { - return &MockConfigService_IsAutoEnabled_Call{Call: _e.mock.On("IsAutoEnabled")} -} - -func (_c *MockConfigService_IsAutoEnabled_Call) Run(run func()) *MockConfigService_IsAutoEnabled_Call { - _c.Call.Run(func(args mock.Arguments) { - run() - }) - return _c -} - -func (_c *MockConfigService_IsAutoEnabled_Call) Return(b bool) *MockConfigService_IsAutoEnabled_Call { - _c.Call.Return(b) - return _c -} - -func (_c *MockConfigService_IsAutoEnabled_Call) RunAndReturn(run func() bool) *MockConfigService_IsAutoEnabled_Call { - _c.Call.Return(run) - return _c -} - -// IsAutoRouteEnabled provides a mock function for the type MockConfigService -func (_mock *MockConfigService) IsAutoRouteEnabled() bool { - ret := _mock.Called() - - if len(ret) == 0 { - panic("no return value specified for IsAutoRouteEnabled") - } - - var r0 bool - if returnFunc, ok := ret.Get(0).(func() bool); ok { - r0 = returnFunc() - } else { - r0 = ret.Get(0).(bool) - } - return r0 -} - -// MockConfigService_IsAutoRouteEnabled_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'IsAutoRouteEnabled' -type MockConfigService_IsAutoRouteEnabled_Call struct { - *mock.Call -} - -// IsAutoRouteEnabled is a helper method to define mock.On call -func (_e *MockConfigService_Expecter) IsAutoRouteEnabled() *MockConfigService_IsAutoRouteEnabled_Call { - return &MockConfigService_IsAutoRouteEnabled_Call{Call: _e.mock.On("IsAutoRouteEnabled")} -} - -func (_c *MockConfigService_IsAutoRouteEnabled_Call) Run(run func()) *MockConfigService_IsAutoRouteEnabled_Call { - _c.Call.Run(func(args mock.Arguments) { - run() - }) - return _c -} - -func (_c *MockConfigService_IsAutoRouteEnabled_Call) Return(b bool) *MockConfigService_IsAutoRouteEnabled_Call { - _c.Call.Return(b) - return _c -} - -func (_c *MockConfigService_IsAutoRouteEnabled_Call) RunAndReturn(run func() bool) *MockConfigService_IsAutoRouteEnabled_Call { - _c.Call.Return(run) - return _c -} - -// IsNetworkIsolationEnabled provides a mock function for the type MockConfigService -func (_mock *MockConfigService) IsNetworkIsolationEnabled() bool { - ret := _mock.Called() - - if len(ret) == 0 { - panic("no return value specified for IsNetworkIsolationEnabled") - } - - var r0 bool - if returnFunc, ok := ret.Get(0).(func() bool); ok { - r0 = returnFunc() - } else { - r0 = ret.Get(0).(bool) - } - return r0 -} - -// MockConfigService_IsNetworkIsolationEnabled_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'IsNetworkIsolationEnabled' -type MockConfigService_IsNetworkIsolationEnabled_Call struct { - *mock.Call -} - -// IsNetworkIsolationEnabled is a helper method to define mock.On call -func (_e *MockConfigService_Expecter) IsNetworkIsolationEnabled() *MockConfigService_IsNetworkIsolationEnabled_Call { - return &MockConfigService_IsNetworkIsolationEnabled_Call{Call: _e.mock.On("IsNetworkIsolationEnabled")} -} - -func (_c *MockConfigService_IsNetworkIsolationEnabled_Call) Run(run func()) *MockConfigService_IsNetworkIsolationEnabled_Call { - _c.Call.Run(func(args mock.Arguments) { - run() - }) - return _c -} - -func (_c *MockConfigService_IsNetworkIsolationEnabled_Call) Return(b bool) *MockConfigService_IsNetworkIsolationEnabled_Call { - _c.Call.Return(b) - return _c -} - -func (_c *MockConfigService_IsNetworkIsolationEnabled_Call) RunAndReturn(run func() bool) *MockConfigService_IsNetworkIsolationEnabled_Call { - _c.Call.Return(run) - return _c -} - -// IsPreviewEnabled provides a mock function for the type MockConfigService -func (_mock *MockConfigService) IsPreviewEnabled() bool { - ret := _mock.Called() - - if len(ret) == 0 { - panic("no return value specified for IsPreviewEnabled") - } - - var r0 bool - if returnFunc, ok := ret.Get(0).(func() bool); ok { - r0 = returnFunc() - } else { - r0 = ret.Get(0).(bool) - } - return r0 -} - -// MockConfigService_IsPreviewEnabled_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'IsPreviewEnabled' -type MockConfigService_IsPreviewEnabled_Call struct { - *mock.Call -} - -// IsPreviewEnabled is a helper method to define mock.On call -func (_e *MockConfigService_Expecter) IsPreviewEnabled() *MockConfigService_IsPreviewEnabled_Call { - return &MockConfigService_IsPreviewEnabled_Call{Call: _e.mock.On("IsPreviewEnabled")} -} - -func (_c *MockConfigService_IsPreviewEnabled_Call) Run(run func()) *MockConfigService_IsPreviewEnabled_Call { - _c.Call.Run(func(args mock.Arguments) { - run() - }) - return _c -} - -func (_c *MockConfigService_IsPreviewEnabled_Call) Return(b bool) *MockConfigService_IsPreviewEnabled_Call { - _c.Call.Return(b) - return _c -} - -func (_c *MockConfigService_IsPreviewEnabled_Call) RunAndReturn(run func() bool) *MockConfigService_IsPreviewEnabled_Call { - _c.Call.Return(run) - return _c -} - -// Load provides a mock function for the type MockConfigService -func (_mock *MockConfigService) Load(ctx context.Context) error { - ret := _mock.Called(ctx) - - if len(ret) == 0 { - panic("no return value specified for Load") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context) error); ok { - r0 = returnFunc(ctx) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockConfigService_Load_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Load' -type MockConfigService_Load_Call struct { - *mock.Call -} - -// Load is a helper method to define mock.On call -// - ctx context.Context -func (_e *MockConfigService_Expecter) Load(ctx any) *MockConfigService_Load_Call { - return &MockConfigService_Load_Call{Call: _e.mock.On("Load", ctx)} -} - -func (_c *MockConfigService_Load_Call) Run(run func(ctx context.Context)) *MockConfigService_Load_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - run( - arg0, - ) - }) - return _c -} - -func (_c *MockConfigService_Load_Call) Return(err error) *MockConfigService_Load_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockConfigService_Load_Call) RunAndReturn(run func(ctx context.Context) error) *MockConfigService_Load_Call { - _c.Call.Return(run) - return _c -} - -// Reload provides a mock function for the type MockConfigService -func (_mock *MockConfigService) Reload(ctx context.Context) error { - ret := _mock.Called(ctx) - - if len(ret) == 0 { - panic("no return value specified for Reload") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context) error); ok { - r0 = returnFunc(ctx) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockConfigService_Reload_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Reload' -type MockConfigService_Reload_Call struct { - *mock.Call -} - -// Reload is a helper method to define mock.On call -// - ctx context.Context -func (_e *MockConfigService_Expecter) Reload(ctx any) *MockConfigService_Reload_Call { - return &MockConfigService_Reload_Call{Call: _e.mock.On("Reload", ctx)} -} - -func (_c *MockConfigService_Reload_Call) Run(run func(ctx context.Context)) *MockConfigService_Reload_Call { +func (_c *MockConfigService_Reload_Call) Run(run func(ctx context.Context)) *MockConfigService_Reload_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -1317,291 +458,6 @@ func (_c *MockConfigService_Reload_Call) RunAndReturn(run func(ctx context.Conte return _c } -// RemoveAttachment provides a mock function for the type MockConfigService -func (_mock *MockConfigService) RemoveAttachment(ctx context.Context, domainOrGroup string, image string) error { - ret := _mock.Called(ctx, domainOrGroup, image) - - if len(ret) == 0 { - panic("no return value specified for RemoveAttachment") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) error); ok { - r0 = returnFunc(ctx, domainOrGroup, image) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockConfigService_RemoveAttachment_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'RemoveAttachment' -type MockConfigService_RemoveAttachment_Call struct { - *mock.Call -} - -// RemoveAttachment is a helper method to define mock.On call -// - ctx context.Context -// - domainOrGroup string -// - image string -func (_e *MockConfigService_Expecter) RemoveAttachment(ctx any, domainOrGroup any, image any) *MockConfigService_RemoveAttachment_Call { - return &MockConfigService_RemoveAttachment_Call{Call: _e.mock.On("RemoveAttachment", ctx, domainOrGroup, image)} -} - -func (_c *MockConfigService_RemoveAttachment_Call) Run(run func(ctx context.Context, domainOrGroup string, image string)) *MockConfigService_RemoveAttachment_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - var arg2 string - if args[2] != nil { - arg2 = args[2].(string) - } - run( - arg0, - arg1, - arg2, - ) - }) - return _c -} - -func (_c *MockConfigService_RemoveAttachment_Call) Return(err error) *MockConfigService_RemoveAttachment_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockConfigService_RemoveAttachment_Call) RunAndReturn(run func(ctx context.Context, domainOrGroup string, image string) error) *MockConfigService_RemoveAttachment_Call { - _c.Call.Return(run) - return _c -} - -// RemoveAutoRouteAllowedDomain provides a mock function for the type MockConfigService -func (_mock *MockConfigService) RemoveAutoRouteAllowedDomain(ctx context.Context, pattern string) error { - ret := _mock.Called(ctx, pattern) - - if len(ret) == 0 { - panic("no return value specified for RemoveAutoRouteAllowedDomain") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) error); ok { - r0 = returnFunc(ctx, pattern) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockConfigService_RemoveAutoRouteAllowedDomain_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'RemoveAutoRouteAllowedDomain' -type MockConfigService_RemoveAutoRouteAllowedDomain_Call struct { - *mock.Call -} - -// RemoveAutoRouteAllowedDomain is a helper method to define mock.On call -// - ctx context.Context -// - pattern string -func (_e *MockConfigService_Expecter) RemoveAutoRouteAllowedDomain(ctx any, pattern any) *MockConfigService_RemoveAutoRouteAllowedDomain_Call { - return &MockConfigService_RemoveAutoRouteAllowedDomain_Call{Call: _e.mock.On("RemoveAutoRouteAllowedDomain", ctx, pattern)} -} - -func (_c *MockConfigService_RemoveAutoRouteAllowedDomain_Call) Run(run func(ctx context.Context, pattern string)) *MockConfigService_RemoveAutoRouteAllowedDomain_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockConfigService_RemoveAutoRouteAllowedDomain_Call) Return(err error) *MockConfigService_RemoveAutoRouteAllowedDomain_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockConfigService_RemoveAutoRouteAllowedDomain_Call) RunAndReturn(run func(ctx context.Context, pattern string) error) *MockConfigService_RemoveAutoRouteAllowedDomain_Call { - _c.Call.Return(run) - return _c -} - -// RemoveRoute provides a mock function for the type MockConfigService -func (_mock *MockConfigService) RemoveRoute(ctx context.Context, domain1 string) error { - ret := _mock.Called(ctx, domain1) - - if len(ret) == 0 { - panic("no return value specified for RemoveRoute") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) error); ok { - r0 = returnFunc(ctx, domain1) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockConfigService_RemoveRoute_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'RemoveRoute' -type MockConfigService_RemoveRoute_Call struct { - *mock.Call -} - -// RemoveRoute is a helper method to define mock.On call -// - ctx context.Context -// - domain1 string -func (_e *MockConfigService_Expecter) RemoveRoute(ctx any, domain1 any) *MockConfigService_RemoveRoute_Call { - return &MockConfigService_RemoveRoute_Call{Call: _e.mock.On("RemoveRoute", ctx, domain1)} -} - -func (_c *MockConfigService_RemoveRoute_Call) Run(run func(ctx context.Context, domain1 string)) *MockConfigService_RemoveRoute_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockConfigService_RemoveRoute_Call) Return(err error) *MockConfigService_RemoveRoute_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockConfigService_RemoveRoute_Call) RunAndReturn(run func(ctx context.Context, domain1 string) error) *MockConfigService_RemoveRoute_Call { - _c.Call.Return(run) - return _c -} - -// Save provides a mock function for the type MockConfigService -func (_mock *MockConfigService) Save(ctx context.Context) error { - ret := _mock.Called(ctx) - - if len(ret) == 0 { - panic("no return value specified for Save") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context) error); ok { - r0 = returnFunc(ctx) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockConfigService_Save_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Save' -type MockConfigService_Save_Call struct { - *mock.Call -} - -// Save is a helper method to define mock.On call -// - ctx context.Context -func (_e *MockConfigService_Expecter) Save(ctx any) *MockConfigService_Save_Call { - return &MockConfigService_Save_Call{Call: _e.mock.On("Save", ctx)} -} - -func (_c *MockConfigService_Save_Call) Run(run func(ctx context.Context)) *MockConfigService_Save_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - run( - arg0, - ) - }) - return _c -} - -func (_c *MockConfigService_Save_Call) Return(err error) *MockConfigService_Save_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockConfigService_Save_Call) RunAndReturn(run func(ctx context.Context) error) *MockConfigService_Save_Call { - _c.Call.Return(run) - return _c -} - -// UpdateRoute provides a mock function for the type MockConfigService -func (_mock *MockConfigService) UpdateRoute(ctx context.Context, route domain.Route) error { - ret := _mock.Called(ctx, route) - - if len(ret) == 0 { - panic("no return value specified for UpdateRoute") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, domain.Route) error); ok { - r0 = returnFunc(ctx, route) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockConfigService_UpdateRoute_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'UpdateRoute' -type MockConfigService_UpdateRoute_Call struct { - *mock.Call -} - -// UpdateRoute is a helper method to define mock.On call -// - ctx context.Context -// - route domain.Route -func (_e *MockConfigService_Expecter) UpdateRoute(ctx any, route any) *MockConfigService_UpdateRoute_Call { - return &MockConfigService_UpdateRoute_Call{Call: _e.mock.On("UpdateRoute", ctx, route)} -} - -func (_c *MockConfigService_UpdateRoute_Call) Run(run func(ctx context.Context, route domain.Route)) *MockConfigService_UpdateRoute_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 domain.Route - if args[1] != nil { - arg1 = args[1].(domain.Route) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockConfigService_UpdateRoute_Call) Return(err error) *MockConfigService_UpdateRoute_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockConfigService_UpdateRoute_Call) RunAndReturn(run func(ctx context.Context, route domain.Route) error) *MockConfigService_UpdateRoute_Call { - _c.Call.Return(run) - return _c -} - // Watch provides a mock function for the type MockConfigService func (_mock *MockConfigService) Watch(ctx context.Context, onChange func()) error { ret := _mock.Called(ctx, onChange) diff --git a/internal/boundaries/in/mocks/mock_container_service.go b/internal/boundaries/in/mocks/mock_container_service.go index 9afa84211..f5a6d9603 100644 --- a/internal/boundaries/in/mocks/mock_container_service.go +++ b/internal/boundaries/in/mocks/mock_container_service.go @@ -17,826 +17,94 @@ func NewMockContainerService(t interface { mock.TestingT Cleanup(func()) }) *MockContainerService { - mock := &MockContainerService{} - mock.Mock.Test(t) - - t.Cleanup(func() { mock.AssertExpectations(t) }) - - return mock -} - -// MockContainerService is an autogenerated mock type for the ContainerService type -type MockContainerService struct { - mock.Mock -} - -type MockContainerService_Expecter struct { - mock *mock.Mock -} - -func (_m *MockContainerService) EXPECT() *MockContainerService_Expecter { - return &MockContainerService_Expecter{mock: &_m.Mock} -} - -// AutoStart provides a mock function for the type MockContainerService -func (_mock *MockContainerService) AutoStart(ctx context.Context, routes []domain.Route) error { - ret := _mock.Called(ctx, routes) - - if len(ret) == 0 { - panic("no return value specified for AutoStart") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, []domain.Route) error); ok { - r0 = returnFunc(ctx, routes) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockContainerService_AutoStart_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'AutoStart' -type MockContainerService_AutoStart_Call struct { - *mock.Call -} - -// AutoStart is a helper method to define mock.On call -// - ctx context.Context -// - routes []domain.Route -func (_e *MockContainerService_Expecter) AutoStart(ctx any, routes any) *MockContainerService_AutoStart_Call { - return &MockContainerService_AutoStart_Call{Call: _e.mock.On("AutoStart", ctx, routes)} -} - -func (_c *MockContainerService_AutoStart_Call) Run(run func(ctx context.Context, routes []domain.Route)) *MockContainerService_AutoStart_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 []domain.Route - if args[1] != nil { - arg1 = args[1].([]domain.Route) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockContainerService_AutoStart_Call) Return(err error) *MockContainerService_AutoStart_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockContainerService_AutoStart_Call) RunAndReturn(run func(ctx context.Context, routes []domain.Route) error) *MockContainerService_AutoStart_Call { - _c.Call.Return(run) - return _c -} - -// CleanupOrphanedAttachments provides a mock function for the type MockContainerService -func (_mock *MockContainerService) CleanupOrphanedAttachments(ctx context.Context, owner string, stop bool) (*domain.CleanupReport, error) { - ret := _mock.Called(ctx, owner, stop) - - if len(ret) == 0 { - panic("no return value specified for CleanupOrphanedAttachments") - } - - var r0 *domain.CleanupReport - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, bool) (*domain.CleanupReport, error)); ok { - return returnFunc(ctx, owner, stop) - } - if returnFunc, ok := ret.Get(0).(func(context.Context, string, bool) *domain.CleanupReport); ok { - r0 = returnFunc(ctx, owner, stop) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).(*domain.CleanupReport) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context, string, bool) error); ok { - r1 = returnFunc(ctx, owner, stop) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockContainerService_CleanupOrphanedAttachments_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'CleanupOrphanedAttachments' -type MockContainerService_CleanupOrphanedAttachments_Call struct { - *mock.Call -} - -// CleanupOrphanedAttachments is a helper method to define mock.On call -// - ctx context.Context -// - owner string -// - stop bool -func (_e *MockContainerService_Expecter) CleanupOrphanedAttachments(ctx any, owner any, stop any) *MockContainerService_CleanupOrphanedAttachments_Call { - return &MockContainerService_CleanupOrphanedAttachments_Call{Call: _e.mock.On("CleanupOrphanedAttachments", ctx, owner, stop)} -} - -func (_c *MockContainerService_CleanupOrphanedAttachments_Call) Run(run func(ctx context.Context, owner string, stop bool)) *MockContainerService_CleanupOrphanedAttachments_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - var arg2 bool - if args[2] != nil { - arg2 = args[2].(bool) - } - run( - arg0, - arg1, - arg2, - ) - }) - return _c -} - -func (_c *MockContainerService_CleanupOrphanedAttachments_Call) Return(cleanupReport *domain.CleanupReport, err error) *MockContainerService_CleanupOrphanedAttachments_Call { - _c.Call.Return(cleanupReport, err) - return _c -} - -func (_c *MockContainerService_CleanupOrphanedAttachments_Call) RunAndReturn(run func(ctx context.Context, owner string, stop bool) (*domain.CleanupReport, error)) *MockContainerService_CleanupOrphanedAttachments_Call { - _c.Call.Return(run) - return _c -} - -// Deploy provides a mock function for the type MockContainerService -func (_mock *MockContainerService) Deploy(ctx context.Context, route domain.Route) (*domain.Container, error) { - ret := _mock.Called(ctx, route) - - if len(ret) == 0 { - panic("no return value specified for Deploy") - } - - var r0 *domain.Container - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, domain.Route) (*domain.Container, error)); ok { - return returnFunc(ctx, route) - } - if returnFunc, ok := ret.Get(0).(func(context.Context, domain.Route) *domain.Container); ok { - r0 = returnFunc(ctx, route) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).(*domain.Container) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context, domain.Route) error); ok { - r1 = returnFunc(ctx, route) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockContainerService_Deploy_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Deploy' -type MockContainerService_Deploy_Call struct { - *mock.Call -} - -// Deploy is a helper method to define mock.On call -// - ctx context.Context -// - route domain.Route -func (_e *MockContainerService_Expecter) Deploy(ctx any, route any) *MockContainerService_Deploy_Call { - return &MockContainerService_Deploy_Call{Call: _e.mock.On("Deploy", ctx, route)} -} - -func (_c *MockContainerService_Deploy_Call) Run(run func(ctx context.Context, route domain.Route)) *MockContainerService_Deploy_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 domain.Route - if args[1] != nil { - arg1 = args[1].(domain.Route) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockContainerService_Deploy_Call) Return(container *domain.Container, err error) *MockContainerService_Deploy_Call { - _c.Call.Return(container, err) - return _c -} - -func (_c *MockContainerService_Deploy_Call) RunAndReturn(run func(ctx context.Context, route domain.Route) (*domain.Container, error)) *MockContainerService_Deploy_Call { - _c.Call.Return(run) - return _c -} - -// Get provides a mock function for the type MockContainerService -func (_mock *MockContainerService) Get(ctx context.Context, domain1 string) (*domain.Container, bool) { - ret := _mock.Called(ctx, domain1) - - if len(ret) == 0 { - panic("no return value specified for Get") - } - - var r0 *domain.Container - var r1 bool - if returnFunc, ok := ret.Get(0).(func(context.Context, string) (*domain.Container, bool)); ok { - return returnFunc(ctx, domain1) - } - if returnFunc, ok := ret.Get(0).(func(context.Context, string) *domain.Container); ok { - r0 = returnFunc(ctx, domain1) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).(*domain.Container) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context, string) bool); ok { - r1 = returnFunc(ctx, domain1) - } else { - r1 = ret.Get(1).(bool) - } - return r0, r1 -} - -// MockContainerService_Get_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Get' -type MockContainerService_Get_Call struct { - *mock.Call -} - -// Get is a helper method to define mock.On call -// - ctx context.Context -// - domain1 string -func (_e *MockContainerService_Expecter) Get(ctx any, domain1 any) *MockContainerService_Get_Call { - return &MockContainerService_Get_Call{Call: _e.mock.On("Get", ctx, domain1)} -} - -func (_c *MockContainerService_Get_Call) Run(run func(ctx context.Context, domain1 string)) *MockContainerService_Get_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockContainerService_Get_Call) Return(container *domain.Container, b bool) *MockContainerService_Get_Call { - _c.Call.Return(container, b) - return _c -} - -func (_c *MockContainerService_Get_Call) RunAndReturn(run func(ctx context.Context, domain1 string) (*domain.Container, bool)) *MockContainerService_Get_Call { - _c.Call.Return(run) - return _c -} - -// HealthCheck provides a mock function for the type MockContainerService -func (_mock *MockContainerService) HealthCheck(ctx context.Context) map[string]bool { - ret := _mock.Called(ctx) - - if len(ret) == 0 { - panic("no return value specified for HealthCheck") - } - - var r0 map[string]bool - if returnFunc, ok := ret.Get(0).(func(context.Context) map[string]bool); ok { - r0 = returnFunc(ctx) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).(map[string]bool) - } - } - return r0 -} - -// MockContainerService_HealthCheck_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'HealthCheck' -type MockContainerService_HealthCheck_Call struct { - *mock.Call -} - -// HealthCheck is a helper method to define mock.On call -// - ctx context.Context -func (_e *MockContainerService_Expecter) HealthCheck(ctx any) *MockContainerService_HealthCheck_Call { - return &MockContainerService_HealthCheck_Call{Call: _e.mock.On("HealthCheck", ctx)} -} - -func (_c *MockContainerService_HealthCheck_Call) Run(run func(ctx context.Context)) *MockContainerService_HealthCheck_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - run( - arg0, - ) - }) - return _c -} - -func (_c *MockContainerService_HealthCheck_Call) Return(stringToBool map[string]bool) *MockContainerService_HealthCheck_Call { - _c.Call.Return(stringToBool) - return _c -} - -func (_c *MockContainerService_HealthCheck_Call) RunAndReturn(run func(ctx context.Context) map[string]bool) *MockContainerService_HealthCheck_Call { - _c.Call.Return(run) - return _c -} - -// List provides a mock function for the type MockContainerService -func (_mock *MockContainerService) List(ctx context.Context) map[string]*domain.Container { - ret := _mock.Called(ctx) - - if len(ret) == 0 { - panic("no return value specified for List") - } - - var r0 map[string]*domain.Container - if returnFunc, ok := ret.Get(0).(func(context.Context) map[string]*domain.Container); ok { - r0 = returnFunc(ctx) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).(map[string]*domain.Container) - } - } - return r0 -} - -// MockContainerService_List_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'List' -type MockContainerService_List_Call struct { - *mock.Call -} - -// List is a helper method to define mock.On call -// - ctx context.Context -func (_e *MockContainerService_Expecter) List(ctx any) *MockContainerService_List_Call { - return &MockContainerService_List_Call{Call: _e.mock.On("List", ctx)} -} - -func (_c *MockContainerService_List_Call) Run(run func(ctx context.Context)) *MockContainerService_List_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - run( - arg0, - ) - }) - return _c -} - -func (_c *MockContainerService_List_Call) Return(stringToContainer map[string]*domain.Container) *MockContainerService_List_Call { - _c.Call.Return(stringToContainer) - return _c -} - -func (_c *MockContainerService_List_Call) RunAndReturn(run func(ctx context.Context) map[string]*domain.Container) *MockContainerService_List_Call { - _c.Call.Return(run) - return _c -} - -// ListAttachments provides a mock function for the type MockContainerService -func (_mock *MockContainerService) ListAttachments(ctx context.Context, domain1 string) []domain.Attachment { - ret := _mock.Called(ctx, domain1) - - if len(ret) == 0 { - panic("no return value specified for ListAttachments") - } - - var r0 []domain.Attachment - if returnFunc, ok := ret.Get(0).(func(context.Context, string) []domain.Attachment); ok { - r0 = returnFunc(ctx, domain1) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]domain.Attachment) - } - } - return r0 -} - -// MockContainerService_ListAttachments_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListAttachments' -type MockContainerService_ListAttachments_Call struct { - *mock.Call -} - -// ListAttachments is a helper method to define mock.On call -// - ctx context.Context -// - domain1 string -func (_e *MockContainerService_Expecter) ListAttachments(ctx any, domain1 any) *MockContainerService_ListAttachments_Call { - return &MockContainerService_ListAttachments_Call{Call: _e.mock.On("ListAttachments", ctx, domain1)} -} - -func (_c *MockContainerService_ListAttachments_Call) Run(run func(ctx context.Context, domain1 string)) *MockContainerService_ListAttachments_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockContainerService_ListAttachments_Call) Return(attachments []domain.Attachment) *MockContainerService_ListAttachments_Call { - _c.Call.Return(attachments) - return _c -} - -func (_c *MockContainerService_ListAttachments_Call) RunAndReturn(run func(ctx context.Context, domain1 string) []domain.Attachment) *MockContainerService_ListAttachments_Call { - _c.Call.Return(run) - return _c -} - -// ListNetworks provides a mock function for the type MockContainerService -func (_mock *MockContainerService) ListNetworks(ctx context.Context) ([]*domain.NetworkInfo, error) { - ret := _mock.Called(ctx) - - if len(ret) == 0 { - panic("no return value specified for ListNetworks") - } - - var r0 []*domain.NetworkInfo - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context) ([]*domain.NetworkInfo, error)); ok { - return returnFunc(ctx) - } - if returnFunc, ok := ret.Get(0).(func(context.Context) []*domain.NetworkInfo); ok { - r0 = returnFunc(ctx) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]*domain.NetworkInfo) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { - r1 = returnFunc(ctx) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockContainerService_ListNetworks_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListNetworks' -type MockContainerService_ListNetworks_Call struct { - *mock.Call -} - -// ListNetworks is a helper method to define mock.On call -// - ctx context.Context -func (_e *MockContainerService_Expecter) ListNetworks(ctx any) *MockContainerService_ListNetworks_Call { - return &MockContainerService_ListNetworks_Call{Call: _e.mock.On("ListNetworks", ctx)} -} - -func (_c *MockContainerService_ListNetworks_Call) Run(run func(ctx context.Context)) *MockContainerService_ListNetworks_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - run( - arg0, - ) - }) - return _c -} - -func (_c *MockContainerService_ListNetworks_Call) Return(networkInfos []*domain.NetworkInfo, err error) *MockContainerService_ListNetworks_Call { - _c.Call.Return(networkInfos, err) - return _c -} - -func (_c *MockContainerService_ListNetworks_Call) RunAndReturn(run func(ctx context.Context) ([]*domain.NetworkInfo, error)) *MockContainerService_ListNetworks_Call { - _c.Call.Return(run) - return _c -} - -// ListOrphanedAttachments provides a mock function for the type MockContainerService -func (_mock *MockContainerService) ListOrphanedAttachments(ctx context.Context) ([]domain.CleanupAttachment, error) { - ret := _mock.Called(ctx) - - if len(ret) == 0 { - panic("no return value specified for ListOrphanedAttachments") - } - - var r0 []domain.CleanupAttachment - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context) ([]domain.CleanupAttachment, error)); ok { - return returnFunc(ctx) - } - if returnFunc, ok := ret.Get(0).(func(context.Context) []domain.CleanupAttachment); ok { - r0 = returnFunc(ctx) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]domain.CleanupAttachment) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { - r1 = returnFunc(ctx) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockContainerService_ListOrphanedAttachments_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListOrphanedAttachments' -type MockContainerService_ListOrphanedAttachments_Call struct { - *mock.Call -} - -// ListOrphanedAttachments is a helper method to define mock.On call -// - ctx context.Context -func (_e *MockContainerService_Expecter) ListOrphanedAttachments(ctx any) *MockContainerService_ListOrphanedAttachments_Call { - return &MockContainerService_ListOrphanedAttachments_Call{Call: _e.mock.On("ListOrphanedAttachments", ctx)} -} - -func (_c *MockContainerService_ListOrphanedAttachments_Call) Run(run func(ctx context.Context)) *MockContainerService_ListOrphanedAttachments_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - run( - arg0, - ) - }) - return _c -} - -func (_c *MockContainerService_ListOrphanedAttachments_Call) Return(cleanupAttachments []domain.CleanupAttachment, err error) *MockContainerService_ListOrphanedAttachments_Call { - _c.Call.Return(cleanupAttachments, err) - return _c -} - -func (_c *MockContainerService_ListOrphanedAttachments_Call) RunAndReturn(run func(ctx context.Context) ([]domain.CleanupAttachment, error)) *MockContainerService_ListOrphanedAttachments_Call { - _c.Call.Return(run) - return _c -} - -// ListRoutesWithDetails provides a mock function for the type MockContainerService -func (_mock *MockContainerService) ListRoutesWithDetails(ctx context.Context) []domain.RouteInfo { - ret := _mock.Called(ctx) - - if len(ret) == 0 { - panic("no return value specified for ListRoutesWithDetails") + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() } - var r0 []domain.RouteInfo - if returnFunc, ok := ret.Get(0).(func(context.Context) []domain.RouteInfo); ok { - r0 = returnFunc(ctx) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]domain.RouteInfo) - } - } - return r0 -} + mock := &MockContainerService{} + mock.Mock.Test(t) -// MockContainerService_ListRoutesWithDetails_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListRoutesWithDetails' -type MockContainerService_ListRoutesWithDetails_Call struct { - *mock.Call -} + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) -// ListRoutesWithDetails is a helper method to define mock.On call -// - ctx context.Context -func (_e *MockContainerService_Expecter) ListRoutesWithDetails(ctx any) *MockContainerService_ListRoutesWithDetails_Call { - return &MockContainerService_ListRoutesWithDetails_Call{Call: _e.mock.On("ListRoutesWithDetails", ctx)} + return mock } -func (_c *MockContainerService_ListRoutesWithDetails_Call) Run(run func(ctx context.Context)) *MockContainerService_ListRoutesWithDetails_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - run( - arg0, - ) - }) - return _c +// MockContainerService is an autogenerated mock type for the ContainerService type +type MockContainerService struct { + mock.Mock } -func (_c *MockContainerService_ListRoutesWithDetails_Call) Return(routeInfos []domain.RouteInfo) *MockContainerService_ListRoutesWithDetails_Call { - _c.Call.Return(routeInfos) - return _c +type MockContainerService_Expecter struct { + mock *mock.Mock } -func (_c *MockContainerService_ListRoutesWithDetails_Call) RunAndReturn(run func(ctx context.Context) []domain.RouteInfo) *MockContainerService_ListRoutesWithDetails_Call { - _c.Call.Return(run) - return _c +func (_m *MockContainerService) EXPECT() *MockContainerService_Expecter { + return &MockContainerService_Expecter{mock: &_m.Mock} } -// ReconcileRemovedRoute provides a mock function for the type MockContainerService -func (_mock *MockContainerService) ReconcileRemovedRoute(ctx context.Context, domain1 string) (*domain.CleanupReport, error) { - ret := _mock.Called(ctx, domain1) +// ListNetworks provides a mock function for the type MockContainerService +func (_mock *MockContainerService) ListNetworks(ctx context.Context) ([]*domain.NetworkInfo, error) { + ret := _mock.Called(ctx) if len(ret) == 0 { - panic("no return value specified for ReconcileRemovedRoute") + panic("no return value specified for ListNetworks") } - var r0 *domain.CleanupReport + var r0 []*domain.NetworkInfo var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) (*domain.CleanupReport, error)); ok { - return returnFunc(ctx, domain1) + if returnFunc, ok := ret.Get(0).(func(context.Context) ([]*domain.NetworkInfo, error)); ok { + return returnFunc(ctx) } - if returnFunc, ok := ret.Get(0).(func(context.Context, string) *domain.CleanupReport); ok { - r0 = returnFunc(ctx, domain1) + if returnFunc, ok := ret.Get(0).(func(context.Context) []*domain.NetworkInfo); ok { + r0 = returnFunc(ctx) } else { if ret.Get(0) != nil { - r0 = ret.Get(0).(*domain.CleanupReport) + r0 = ret.Get(0).([]*domain.NetworkInfo) } } - if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { - r1 = returnFunc(ctx, domain1) + if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { + r1 = returnFunc(ctx) } else { r1 = ret.Error(1) } return r0, r1 } -// MockContainerService_ReconcileRemovedRoute_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ReconcileRemovedRoute' -type MockContainerService_ReconcileRemovedRoute_Call struct { - *mock.Call -} - -// ReconcileRemovedRoute is a helper method to define mock.On call -// - ctx context.Context -// - domain1 string -func (_e *MockContainerService_Expecter) ReconcileRemovedRoute(ctx any, domain1 any) *MockContainerService_ReconcileRemovedRoute_Call { - return &MockContainerService_ReconcileRemovedRoute_Call{Call: _e.mock.On("ReconcileRemovedRoute", ctx, domain1)} -} - -func (_c *MockContainerService_ReconcileRemovedRoute_Call) Run(run func(ctx context.Context, domain1 string)) *MockContainerService_ReconcileRemovedRoute_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockContainerService_ReconcileRemovedRoute_Call) Return(cleanupReport *domain.CleanupReport, err error) *MockContainerService_ReconcileRemovedRoute_Call { - _c.Call.Return(cleanupReport, err) - return _c -} - -func (_c *MockContainerService_ReconcileRemovedRoute_Call) RunAndReturn(run func(ctx context.Context, domain1 string) (*domain.CleanupReport, error)) *MockContainerService_ReconcileRemovedRoute_Call { - _c.Call.Return(run) - return _c -} - -// Remove provides a mock function for the type MockContainerService -func (_mock *MockContainerService) Remove(ctx context.Context, containerID string, force bool) error { - ret := _mock.Called(ctx, containerID, force) - - if len(ret) == 0 { - panic("no return value specified for Remove") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, bool) error); ok { - r0 = returnFunc(ctx, containerID, force) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockContainerService_Remove_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Remove' -type MockContainerService_Remove_Call struct { - *mock.Call -} - -// Remove is a helper method to define mock.On call -// - ctx context.Context -// - containerID string -// - force bool -func (_e *MockContainerService_Expecter) Remove(ctx any, containerID any, force any) *MockContainerService_Remove_Call { - return &MockContainerService_Remove_Call{Call: _e.mock.On("Remove", ctx, containerID, force)} -} - -func (_c *MockContainerService_Remove_Call) Run(run func(ctx context.Context, containerID string, force bool)) *MockContainerService_Remove_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - var arg2 bool - if args[2] != nil { - arg2 = args[2].(bool) - } - run( - arg0, - arg1, - arg2, - ) - }) - return _c -} - -func (_c *MockContainerService_Remove_Call) Return(err error) *MockContainerService_Remove_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockContainerService_Remove_Call) RunAndReturn(run func(ctx context.Context, containerID string, force bool) error) *MockContainerService_Remove_Call { - _c.Call.Return(run) - return _c -} - -// Restart provides a mock function for the type MockContainerService -func (_mock *MockContainerService) Restart(ctx context.Context, domain1 string, withAttachments bool) error { - ret := _mock.Called(ctx, domain1, withAttachments) - - if len(ret) == 0 { - panic("no return value specified for Restart") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, bool) error); ok { - r0 = returnFunc(ctx, domain1, withAttachments) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockContainerService_Restart_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Restart' -type MockContainerService_Restart_Call struct { +// MockContainerService_ListNetworks_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListNetworks' +type MockContainerService_ListNetworks_Call struct { *mock.Call } -// Restart is a helper method to define mock.On call +// ListNetworks is a helper method to define mock.On call // - ctx context.Context -// - domain1 string -// - withAttachments bool -func (_e *MockContainerService_Expecter) Restart(ctx any, domain1 any, withAttachments any) *MockContainerService_Restart_Call { - return &MockContainerService_Restart_Call{Call: _e.mock.On("Restart", ctx, domain1, withAttachments)} +func (_e *MockContainerService_Expecter) ListNetworks(ctx any) *MockContainerService_ListNetworks_Call { + return &MockContainerService_ListNetworks_Call{Call: _e.mock.On("ListNetworks", ctx)} } -func (_c *MockContainerService_Restart_Call) Run(run func(ctx context.Context, domain1 string, withAttachments bool)) *MockContainerService_Restart_Call { +func (_c *MockContainerService_ListNetworks_Call) Run(run func(ctx context.Context)) *MockContainerService_ListNetworks_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { arg0 = args[0].(context.Context) } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - var arg2 bool - if args[2] != nil { - arg2 = args[2].(bool) - } run( arg0, - arg1, - arg2, ) }) return _c } -func (_c *MockContainerService_Restart_Call) Return(err error) *MockContainerService_Restart_Call { - _c.Call.Return(err) +func (_c *MockContainerService_ListNetworks_Call) Return(networkInfos []*domain.NetworkInfo, err error) *MockContainerService_ListNetworks_Call { + _c.Call.Return(networkInfos, err) return _c } -func (_c *MockContainerService_Restart_Call) RunAndReturn(run func(ctx context.Context, domain1 string, withAttachments bool) error) *MockContainerService_Restart_Call { +func (_c *MockContainerService_ListNetworks_Call) RunAndReturn(run func(ctx context.Context) ([]*domain.NetworkInfo, error)) *MockContainerService_ListNetworks_Call { _c.Call.Return(run) return _c } @@ -891,151 +159,3 @@ func (_c *MockContainerService_Shutdown_Call) RunAndReturn(run func(ctx context. _c.Call.Return(run) return _c } - -// Stop provides a mock function for the type MockContainerService -func (_mock *MockContainerService) Stop(ctx context.Context, containerID string) error { - ret := _mock.Called(ctx, containerID) - - if len(ret) == 0 { - panic("no return value specified for Stop") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) error); ok { - r0 = returnFunc(ctx, containerID) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockContainerService_Stop_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Stop' -type MockContainerService_Stop_Call struct { - *mock.Call -} - -// Stop is a helper method to define mock.On call -// - ctx context.Context -// - containerID string -func (_e *MockContainerService_Expecter) Stop(ctx any, containerID any) *MockContainerService_Stop_Call { - return &MockContainerService_Stop_Call{Call: _e.mock.On("Stop", ctx, containerID)} -} - -func (_c *MockContainerService_Stop_Call) Run(run func(ctx context.Context, containerID string)) *MockContainerService_Stop_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockContainerService_Stop_Call) Return(err error) *MockContainerService_Stop_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockContainerService_Stop_Call) RunAndReturn(run func(ctx context.Context, containerID string) error) *MockContainerService_Stop_Call { - _c.Call.Return(run) - return _c -} - -// SyncContainers provides a mock function for the type MockContainerService -func (_mock *MockContainerService) SyncContainers(ctx context.Context) error { - ret := _mock.Called(ctx) - - if len(ret) == 0 { - panic("no return value specified for SyncContainers") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context) error); ok { - r0 = returnFunc(ctx) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockContainerService_SyncContainers_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'SyncContainers' -type MockContainerService_SyncContainers_Call struct { - *mock.Call -} - -// SyncContainers is a helper method to define mock.On call -// - ctx context.Context -func (_e *MockContainerService_Expecter) SyncContainers(ctx any) *MockContainerService_SyncContainers_Call { - return &MockContainerService_SyncContainers_Call{Call: _e.mock.On("SyncContainers", ctx)} -} - -func (_c *MockContainerService_SyncContainers_Call) Run(run func(ctx context.Context)) *MockContainerService_SyncContainers_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - run( - arg0, - ) - }) - return _c -} - -func (_c *MockContainerService_SyncContainers_Call) Return(err error) *MockContainerService_SyncContainers_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockContainerService_SyncContainers_Call) RunAndReturn(run func(ctx context.Context) error) *MockContainerService_SyncContainers_Call { - _c.Call.Return(run) - return _c -} - -// UpdateAttachments provides a mock function for the type MockContainerService -func (_mock *MockContainerService) UpdateAttachments(attachments map[string][]string) { - _mock.Called(attachments) - return -} - -// MockContainerService_UpdateAttachments_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'UpdateAttachments' -type MockContainerService_UpdateAttachments_Call struct { - *mock.Call -} - -// UpdateAttachments is a helper method to define mock.On call -// - attachments map[string][]string -func (_e *MockContainerService_Expecter) UpdateAttachments(attachments any) *MockContainerService_UpdateAttachments_Call { - return &MockContainerService_UpdateAttachments_Call{Call: _e.mock.On("UpdateAttachments", attachments)} -} - -func (_c *MockContainerService_UpdateAttachments_Call) Run(run func(attachments map[string][]string)) *MockContainerService_UpdateAttachments_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 map[string][]string - if args[0] != nil { - arg0 = args[0].(map[string][]string) - } - run( - arg0, - ) - }) - return _c -} - -func (_c *MockContainerService_UpdateAttachments_Call) Return() *MockContainerService_UpdateAttachments_Call { - _c.Call.Return() - return _c -} - -func (_c *MockContainerService_UpdateAttachments_Call) RunAndReturn(run func(attachments map[string][]string)) *MockContainerService_UpdateAttachments_Call { - _c.Run(run) - return _c -} diff --git a/internal/boundaries/in/mocks/mock_health_service.go b/internal/boundaries/in/mocks/mock_health_service.go index 9fb907670..d30be7364 100644 --- a/internal/boundaries/in/mocks/mock_health_service.go +++ b/internal/boundaries/in/mocks/mock_health_service.go @@ -17,10 +17,19 @@ func NewMockHealthService(t interface { mock.TestingT Cleanup(func()) }) *MockHealthService { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockHealthService{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } @@ -90,62 +99,3 @@ func (_c *MockHealthService_CheckAllRoutes_Call) RunAndReturn(run func(ctx conte _c.Call.Return(run) return _c } - -// CheckRoute provides a mock function for the type MockHealthService -func (_mock *MockHealthService) CheckRoute(ctx context.Context, route domain.Route) *domain.RouteHealth { - ret := _mock.Called(ctx, route) - - if len(ret) == 0 { - panic("no return value specified for CheckRoute") - } - - var r0 *domain.RouteHealth - if returnFunc, ok := ret.Get(0).(func(context.Context, domain.Route) *domain.RouteHealth); ok { - r0 = returnFunc(ctx, route) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).(*domain.RouteHealth) - } - } - return r0 -} - -// MockHealthService_CheckRoute_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'CheckRoute' -type MockHealthService_CheckRoute_Call struct { - *mock.Call -} - -// CheckRoute is a helper method to define mock.On call -// - ctx context.Context -// - route domain.Route -func (_e *MockHealthService_Expecter) CheckRoute(ctx any, route any) *MockHealthService_CheckRoute_Call { - return &MockHealthService_CheckRoute_Call{Call: _e.mock.On("CheckRoute", ctx, route)} -} - -func (_c *MockHealthService_CheckRoute_Call) Run(run func(ctx context.Context, route domain.Route)) *MockHealthService_CheckRoute_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 domain.Route - if args[1] != nil { - arg1 = args[1].(domain.Route) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockHealthService_CheckRoute_Call) Return(routeHealth *domain.RouteHealth) *MockHealthService_CheckRoute_Call { - _c.Call.Return(routeHealth) - return _c -} - -func (_c *MockHealthService_CheckRoute_Call) RunAndReturn(run func(ctx context.Context, route domain.Route) *domain.RouteHealth) *MockHealthService_CheckRoute_Call { - _c.Call.Return(run) - return _c -} diff --git a/internal/boundaries/in/mocks/mock_http_prober.go b/internal/boundaries/in/mocks/mock_http_prober.go index 4d2523add..e671cd259 100644 --- a/internal/boundaries/in/mocks/mock_http_prober.go +++ b/internal/boundaries/in/mocks/mock_http_prober.go @@ -16,10 +16,19 @@ func NewMockHTTPProber(t interface { mock.TestingT Cleanup(func()) }) *MockHTTPProber { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockHTTPProber{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/in/mocks/mock_log_service.go b/internal/boundaries/in/mocks/mock_log_service.go index 84122c930..4c1c5408f 100644 --- a/internal/boundaries/in/mocks/mock_log_service.go +++ b/internal/boundaries/in/mocks/mock_log_service.go @@ -16,10 +16,19 @@ func NewMockLogService(t interface { mock.TestingT Cleanup(func()) }) *MockLogService { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockLogService{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/in/mocks/mock_proxy_service.go b/internal/boundaries/in/mocks/mock_proxy_service.go index e959c339a..10fa8d3b7 100644 --- a/internal/boundaries/in/mocks/mock_proxy_service.go +++ b/internal/boundaries/in/mocks/mock_proxy_service.go @@ -18,10 +18,19 @@ func NewMockProxyService(t interface { mock.TestingT Cleanup(func()) }) *MockProxyService { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockProxyService{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/in/mocks/mock_public_tls_service.go b/internal/boundaries/in/mocks/mock_public_tls_service.go index 35031d2d1..afec27a1a 100644 --- a/internal/boundaries/in/mocks/mock_public_tls_service.go +++ b/internal/boundaries/in/mocks/mock_public_tls_service.go @@ -18,10 +18,19 @@ func NewMockPublicTLSService(t interface { mock.TestingT Cleanup(func()) }) *MockPublicTLSService { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockPublicTLSService{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/in/mocks/mock_registry_service.go b/internal/boundaries/in/mocks/mock_registry_service.go index d43ef3013..d9e85eb28 100644 --- a/internal/boundaries/in/mocks/mock_registry_service.go +++ b/internal/boundaries/in/mocks/mock_registry_service.go @@ -18,10 +18,19 @@ func NewMockRegistryService(t interface { mock.TestingT Cleanup(func()) }) *MockRegistryService { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockRegistryService{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } @@ -187,16 +196,16 @@ func (_c *MockRegistryService_BlobExists_Call) RunAndReturn(run func(ctx context } // CancelUpload provides a mock function for the type MockRegistryService -func (_mock *MockRegistryService) CancelUpload(ctx context.Context, uuid string) error { - ret := _mock.Called(ctx, uuid) +func (_mock *MockRegistryService) CancelUpload(ctx context.Context, name string, uuid string) error { + ret := _mock.Called(ctx, name, uuid) if len(ret) == 0 { panic("no return value specified for CancelUpload") } var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) error); ok { - r0 = returnFunc(ctx, uuid) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) error); ok { + r0 = returnFunc(ctx, name, uuid) } else { r0 = ret.Error(0) } @@ -210,12 +219,13 @@ type MockRegistryService_CancelUpload_Call struct { // CancelUpload is a helper method to define mock.On call // - ctx context.Context +// - name string // - uuid string -func (_e *MockRegistryService_Expecter) CancelUpload(ctx any, uuid any) *MockRegistryService_CancelUpload_Call { - return &MockRegistryService_CancelUpload_Call{Call: _e.mock.On("CancelUpload", ctx, uuid)} +func (_e *MockRegistryService_Expecter) CancelUpload(ctx any, name any, uuid any) *MockRegistryService_CancelUpload_Call { + return &MockRegistryService_CancelUpload_Call{Call: _e.mock.On("CancelUpload", ctx, name, uuid)} } -func (_c *MockRegistryService_CancelUpload_Call) Run(run func(ctx context.Context, uuid string)) *MockRegistryService_CancelUpload_Call { +func (_c *MockRegistryService_CancelUpload_Call) Run(run func(ctx context.Context, name string, uuid string)) *MockRegistryService_CancelUpload_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -225,9 +235,14 @@ func (_c *MockRegistryService_CancelUpload_Call) Run(run func(ctx context.Contex if args[1] != nil { arg1 = args[1].(string) } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } run( arg0, arg1, + arg2, ) }) return _c @@ -238,7 +253,7 @@ func (_c *MockRegistryService_CancelUpload_Call) Return(err error) *MockRegistry return _c } -func (_c *MockRegistryService_CancelUpload_Call) RunAndReturn(run func(ctx context.Context, uuid string) error) *MockRegistryService_CancelUpload_Call { +func (_c *MockRegistryService_CancelUpload_Call) RunAndReturn(run func(ctx context.Context, name string, uuid string) error) *MockRegistryService_CancelUpload_Call { _c.Call.Return(run) return _c } @@ -307,16 +322,16 @@ func (_c *MockRegistryService_DeleteManifest_Call) RunAndReturn(run func(ctx con } // FinishUpload provides a mock function for the type MockRegistryService -func (_mock *MockRegistryService) FinishUpload(ctx context.Context, uuid string, digest string) error { - ret := _mock.Called(ctx, uuid, digest) +func (_mock *MockRegistryService) FinishUpload(ctx context.Context, name string, uuid string, digest string) error { + ret := _mock.Called(ctx, name, uuid, digest) if len(ret) == 0 { panic("no return value specified for FinishUpload") } var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) error); ok { - r0 = returnFunc(ctx, uuid, digest) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, string) error); ok { + r0 = returnFunc(ctx, name, uuid, digest) } else { r0 = ret.Error(0) } @@ -330,13 +345,14 @@ type MockRegistryService_FinishUpload_Call struct { // FinishUpload is a helper method to define mock.On call // - ctx context.Context +// - name string // - uuid string // - digest string -func (_e *MockRegistryService_Expecter) FinishUpload(ctx any, uuid any, digest any) *MockRegistryService_FinishUpload_Call { - return &MockRegistryService_FinishUpload_Call{Call: _e.mock.On("FinishUpload", ctx, uuid, digest)} +func (_e *MockRegistryService_Expecter) FinishUpload(ctx any, name any, uuid any, digest any) *MockRegistryService_FinishUpload_Call { + return &MockRegistryService_FinishUpload_Call{Call: _e.mock.On("FinishUpload", ctx, name, uuid, digest)} } -func (_c *MockRegistryService_FinishUpload_Call) Run(run func(ctx context.Context, uuid string, digest string)) *MockRegistryService_FinishUpload_Call { +func (_c *MockRegistryService_FinishUpload_Call) Run(run func(ctx context.Context, name string, uuid string, digest string)) *MockRegistryService_FinishUpload_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -350,10 +366,15 @@ func (_c *MockRegistryService_FinishUpload_Call) Run(run func(ctx context.Contex if args[2] != nil { arg2 = args[2].(string) } + var arg3 string + if args[3] != nil { + arg3 = args[3].(string) + } run( arg0, arg1, arg2, + arg3, ) }) return _c @@ -364,7 +385,7 @@ func (_c *MockRegistryService_FinishUpload_Call) Return(err error) *MockRegistry return _c } -func (_c *MockRegistryService_FinishUpload_Call) RunAndReturn(run func(ctx context.Context, uuid string, digest string) error) *MockRegistryService_FinishUpload_Call { +func (_c *MockRegistryService_FinishUpload_Call) RunAndReturn(run func(ctx context.Context, name string, uuid string, digest string) error) *MockRegistryService_FinishUpload_Call { _c.Call.Return(run) return _c } diff --git a/internal/boundaries/in/mocks/mock_secret_service.go b/internal/boundaries/in/mocks/mock_secret_service.go deleted file mode 100644 index 5cbe3c70c..000000000 --- a/internal/boundaries/in/mocks/mock_secret_service.go +++ /dev/null @@ -1,515 +0,0 @@ -// Code generated by mockery; DO NOT EDIT. -// github.com/vektra/mockery -// template: testify - -package mocks - -import ( - "context" - - "github.com/bnema/gordon/internal/boundaries/out" - mock "github.com/stretchr/testify/mock" -) - -// NewMockSecretService creates a new instance of MockSecretService. It also registers a testing interface on the mock and a cleanup function to assert the mocks expectations. -// The first argument is typically a *testing.T value. -func NewMockSecretService(t interface { - mock.TestingT - Cleanup(func()) -}) *MockSecretService { - mock := &MockSecretService{} - mock.Mock.Test(t) - - t.Cleanup(func() { mock.AssertExpectations(t) }) - - return mock -} - -// MockSecretService is an autogenerated mock type for the SecretService type -type MockSecretService struct { - mock.Mock -} - -type MockSecretService_Expecter struct { - mock *mock.Mock -} - -func (_m *MockSecretService) EXPECT() *MockSecretService_Expecter { - return &MockSecretService_Expecter{mock: &_m.Mock} -} - -// Delete provides a mock function for the type MockSecretService -func (_mock *MockSecretService) Delete(ctx context.Context, domain string, key string) error { - ret := _mock.Called(ctx, domain, key) - - if len(ret) == 0 { - panic("no return value specified for Delete") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) error); ok { - r0 = returnFunc(ctx, domain, key) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockSecretService_Delete_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Delete' -type MockSecretService_Delete_Call struct { - *mock.Call -} - -// Delete is a helper method to define mock.On call -// - ctx context.Context -// - domain string -// - key string -func (_e *MockSecretService_Expecter) Delete(ctx any, domain any, key any) *MockSecretService_Delete_Call { - return &MockSecretService_Delete_Call{Call: _e.mock.On("Delete", ctx, domain, key)} -} - -func (_c *MockSecretService_Delete_Call) Run(run func(ctx context.Context, domain string, key string)) *MockSecretService_Delete_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - var arg2 string - if args[2] != nil { - arg2 = args[2].(string) - } - run( - arg0, - arg1, - arg2, - ) - }) - return _c -} - -func (_c *MockSecretService_Delete_Call) Return(err error) *MockSecretService_Delete_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockSecretService_Delete_Call) RunAndReturn(run func(ctx context.Context, domain string, key string) error) *MockSecretService_Delete_Call { - _c.Call.Return(run) - return _c -} - -// DeleteAttachment provides a mock function for the type MockSecretService -func (_mock *MockSecretService) DeleteAttachment(ctx context.Context, domain string, service string, key string) error { - ret := _mock.Called(ctx, domain, service, key) - - if len(ret) == 0 { - panic("no return value specified for DeleteAttachment") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, string) error); ok { - r0 = returnFunc(ctx, domain, service, key) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockSecretService_DeleteAttachment_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'DeleteAttachment' -type MockSecretService_DeleteAttachment_Call struct { - *mock.Call -} - -// DeleteAttachment is a helper method to define mock.On call -// - ctx context.Context -// - domain string -// - service string -// - key string -func (_e *MockSecretService_Expecter) DeleteAttachment(ctx any, domain any, service any, key any) *MockSecretService_DeleteAttachment_Call { - return &MockSecretService_DeleteAttachment_Call{Call: _e.mock.On("DeleteAttachment", ctx, domain, service, key)} -} - -func (_c *MockSecretService_DeleteAttachment_Call) Run(run func(ctx context.Context, domain string, service string, key string)) *MockSecretService_DeleteAttachment_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - var arg2 string - if args[2] != nil { - arg2 = args[2].(string) - } - var arg3 string - if args[3] != nil { - arg3 = args[3].(string) - } - run( - arg0, - arg1, - arg2, - arg3, - ) - }) - return _c -} - -func (_c *MockSecretService_DeleteAttachment_Call) Return(err error) *MockSecretService_DeleteAttachment_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockSecretService_DeleteAttachment_Call) RunAndReturn(run func(ctx context.Context, domain string, service string, key string) error) *MockSecretService_DeleteAttachment_Call { - _c.Call.Return(run) - return _c -} - -// GetAll provides a mock function for the type MockSecretService -func (_mock *MockSecretService) GetAll(ctx context.Context, domain string) (map[string]string, error) { - ret := _mock.Called(ctx, domain) - - if len(ret) == 0 { - panic("no return value specified for GetAll") - } - - var r0 map[string]string - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) (map[string]string, error)); ok { - return returnFunc(ctx, domain) - } - if returnFunc, ok := ret.Get(0).(func(context.Context, string) map[string]string); ok { - r0 = returnFunc(ctx, domain) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).(map[string]string) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { - r1 = returnFunc(ctx, domain) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockSecretService_GetAll_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetAll' -type MockSecretService_GetAll_Call struct { - *mock.Call -} - -// GetAll is a helper method to define mock.On call -// - ctx context.Context -// - domain string -func (_e *MockSecretService_Expecter) GetAll(ctx any, domain any) *MockSecretService_GetAll_Call { - return &MockSecretService_GetAll_Call{Call: _e.mock.On("GetAll", ctx, domain)} -} - -func (_c *MockSecretService_GetAll_Call) Run(run func(ctx context.Context, domain string)) *MockSecretService_GetAll_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockSecretService_GetAll_Call) Return(stringToString map[string]string, err error) *MockSecretService_GetAll_Call { - _c.Call.Return(stringToString, err) - return _c -} - -func (_c *MockSecretService_GetAll_Call) RunAndReturn(run func(ctx context.Context, domain string) (map[string]string, error)) *MockSecretService_GetAll_Call { - _c.Call.Return(run) - return _c -} - -// ListKeys provides a mock function for the type MockSecretService -func (_mock *MockSecretService) ListKeys(ctx context.Context, domain string) ([]string, error) { - ret := _mock.Called(ctx, domain) - - if len(ret) == 0 { - panic("no return value specified for ListKeys") - } - - var r0 []string - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) ([]string, error)); ok { - return returnFunc(ctx, domain) - } - if returnFunc, ok := ret.Get(0).(func(context.Context, string) []string); ok { - r0 = returnFunc(ctx, domain) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]string) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { - r1 = returnFunc(ctx, domain) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockSecretService_ListKeys_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListKeys' -type MockSecretService_ListKeys_Call struct { - *mock.Call -} - -// ListKeys is a helper method to define mock.On call -// - ctx context.Context -// - domain string -func (_e *MockSecretService_Expecter) ListKeys(ctx any, domain any) *MockSecretService_ListKeys_Call { - return &MockSecretService_ListKeys_Call{Call: _e.mock.On("ListKeys", ctx, domain)} -} - -func (_c *MockSecretService_ListKeys_Call) Run(run func(ctx context.Context, domain string)) *MockSecretService_ListKeys_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockSecretService_ListKeys_Call) Return(strings []string, err error) *MockSecretService_ListKeys_Call { - _c.Call.Return(strings, err) - return _c -} - -func (_c *MockSecretService_ListKeys_Call) RunAndReturn(run func(ctx context.Context, domain string) ([]string, error)) *MockSecretService_ListKeys_Call { - _c.Call.Return(run) - return _c -} - -// ListKeysWithAttachments provides a mock function for the type MockSecretService -func (_mock *MockSecretService) ListKeysWithAttachments(ctx context.Context, domain string) ([]string, []out.AttachmentSecrets, error) { - ret := _mock.Called(ctx, domain) - - if len(ret) == 0 { - panic("no return value specified for ListKeysWithAttachments") - } - - var r0 []string - var r1 []out.AttachmentSecrets - var r2 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) ([]string, []out.AttachmentSecrets, error)); ok { - return returnFunc(ctx, domain) - } - if returnFunc, ok := ret.Get(0).(func(context.Context, string) []string); ok { - r0 = returnFunc(ctx, domain) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]string) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context, string) []out.AttachmentSecrets); ok { - r1 = returnFunc(ctx, domain) - } else { - if ret.Get(1) != nil { - r1 = ret.Get(1).([]out.AttachmentSecrets) - } - } - if returnFunc, ok := ret.Get(2).(func(context.Context, string) error); ok { - r2 = returnFunc(ctx, domain) - } else { - r2 = ret.Error(2) - } - return r0, r1, r2 -} - -// MockSecretService_ListKeysWithAttachments_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListKeysWithAttachments' -type MockSecretService_ListKeysWithAttachments_Call struct { - *mock.Call -} - -// ListKeysWithAttachments is a helper method to define mock.On call -// - ctx context.Context -// - domain string -func (_e *MockSecretService_Expecter) ListKeysWithAttachments(ctx any, domain any) *MockSecretService_ListKeysWithAttachments_Call { - return &MockSecretService_ListKeysWithAttachments_Call{Call: _e.mock.On("ListKeysWithAttachments", ctx, domain)} -} - -func (_c *MockSecretService_ListKeysWithAttachments_Call) Run(run func(ctx context.Context, domain string)) *MockSecretService_ListKeysWithAttachments_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockSecretService_ListKeysWithAttachments_Call) Return(strings []string, attachmentSecretss []out.AttachmentSecrets, err error) *MockSecretService_ListKeysWithAttachments_Call { - _c.Call.Return(strings, attachmentSecretss, err) - return _c -} - -func (_c *MockSecretService_ListKeysWithAttachments_Call) RunAndReturn(run func(ctx context.Context, domain string) ([]string, []out.AttachmentSecrets, error)) *MockSecretService_ListKeysWithAttachments_Call { - _c.Call.Return(run) - return _c -} - -// Set provides a mock function for the type MockSecretService -func (_mock *MockSecretService) Set(ctx context.Context, domain string, secrets map[string]string) error { - ret := _mock.Called(ctx, domain, secrets) - - if len(ret) == 0 { - panic("no return value specified for Set") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, map[string]string) error); ok { - r0 = returnFunc(ctx, domain, secrets) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockSecretService_Set_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Set' -type MockSecretService_Set_Call struct { - *mock.Call -} - -// Set is a helper method to define mock.On call -// - ctx context.Context -// - domain string -// - secrets map[string]string -func (_e *MockSecretService_Expecter) Set(ctx any, domain any, secrets any) *MockSecretService_Set_Call { - return &MockSecretService_Set_Call{Call: _e.mock.On("Set", ctx, domain, secrets)} -} - -func (_c *MockSecretService_Set_Call) Run(run func(ctx context.Context, domain string, secrets map[string]string)) *MockSecretService_Set_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - var arg2 map[string]string - if args[2] != nil { - arg2 = args[2].(map[string]string) - } - run( - arg0, - arg1, - arg2, - ) - }) - return _c -} - -func (_c *MockSecretService_Set_Call) Return(err error) *MockSecretService_Set_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockSecretService_Set_Call) RunAndReturn(run func(ctx context.Context, domain string, secrets map[string]string) error) *MockSecretService_Set_Call { - _c.Call.Return(run) - return _c -} - -// SetAttachment provides a mock function for the type MockSecretService -func (_mock *MockSecretService) SetAttachment(ctx context.Context, domain string, service string, secrets map[string]string) error { - ret := _mock.Called(ctx, domain, service, secrets) - - if len(ret) == 0 { - panic("no return value specified for SetAttachment") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, map[string]string) error); ok { - r0 = returnFunc(ctx, domain, service, secrets) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockSecretService_SetAttachment_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'SetAttachment' -type MockSecretService_SetAttachment_Call struct { - *mock.Call -} - -// SetAttachment is a helper method to define mock.On call -// - ctx context.Context -// - domain string -// - service string -// - secrets map[string]string -func (_e *MockSecretService_Expecter) SetAttachment(ctx any, domain any, service any, secrets any) *MockSecretService_SetAttachment_Call { - return &MockSecretService_SetAttachment_Call{Call: _e.mock.On("SetAttachment", ctx, domain, service, secrets)} -} - -func (_c *MockSecretService_SetAttachment_Call) Run(run func(ctx context.Context, domain string, service string, secrets map[string]string)) *MockSecretService_SetAttachment_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - var arg2 string - if args[2] != nil { - arg2 = args[2].(string) - } - var arg3 map[string]string - if args[3] != nil { - arg3 = args[3].(map[string]string) - } - run( - arg0, - arg1, - arg2, - arg3, - ) - }) - return _c -} - -func (_c *MockSecretService_SetAttachment_Call) Return(err error) *MockSecretService_SetAttachment_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockSecretService_SetAttachment_Call) RunAndReturn(run func(ctx context.Context, domain string, service string, secrets map[string]string) error) *MockSecretService_SetAttachment_Call { - _c.Call.Return(run) - return _c -} diff --git a/internal/boundaries/in/mocks/mock_standalone_service_service.go b/internal/boundaries/in/mocks/mock_standalone_service_service.go deleted file mode 100644 index e75acc3ed..000000000 --- a/internal/boundaries/in/mocks/mock_standalone_service_service.go +++ /dev/null @@ -1,158 +0,0 @@ -// Code generated by mockery; DO NOT EDIT. -// github.com/vektra/mockery -// template: testify - -package mocks - -import ( - "context" - - "github.com/bnema/gordon/internal/domain" - mock "github.com/stretchr/testify/mock" -) - -// NewMockStandaloneServiceService creates a new instance of MockStandaloneServiceService. It also registers a testing interface on the mock and a cleanup function to assert the mocks expectations. -// The first argument is typically a *testing.T value. -func NewMockStandaloneServiceService(t interface { - mock.TestingT - Cleanup(func()) -}) *MockStandaloneServiceService { - mock := &MockStandaloneServiceService{} - mock.Mock.Test(t) - - t.Cleanup(func() { mock.AssertExpectations(t) }) - - return mock -} - -// MockStandaloneServiceService is an autogenerated mock type for the StandaloneServiceService type -type MockStandaloneServiceService struct { - mock.Mock -} - -type MockStandaloneServiceService_Expecter struct { - mock *mock.Mock -} - -func (_m *MockStandaloneServiceService) EXPECT() *MockStandaloneServiceService_Expecter { - return &MockStandaloneServiceService_Expecter{mock: &_m.Mock} -} - -// Reconcile provides a mock function for the type MockStandaloneServiceService -func (_mock *MockStandaloneServiceService) Reconcile(ctx context.Context, services []domain.StandaloneService) error { - ret := _mock.Called(ctx, services) - - if len(ret) == 0 { - panic("no return value specified for Reconcile") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, []domain.StandaloneService) error); ok { - r0 = returnFunc(ctx, services) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockStandaloneServiceService_Reconcile_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Reconcile' -type MockStandaloneServiceService_Reconcile_Call struct { - *mock.Call -} - -// Reconcile is a helper method to define mock.On call -// - ctx context.Context -// - services []domain.StandaloneService -func (_e *MockStandaloneServiceService_Expecter) Reconcile(ctx any, services any) *MockStandaloneServiceService_Reconcile_Call { - return &MockStandaloneServiceService_Reconcile_Call{Call: _e.mock.On("Reconcile", ctx, services)} -} - -func (_c *MockStandaloneServiceService_Reconcile_Call) Run(run func(ctx context.Context, services []domain.StandaloneService)) *MockStandaloneServiceService_Reconcile_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 []domain.StandaloneService - if args[1] != nil { - arg1 = args[1].([]domain.StandaloneService) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockStandaloneServiceService_Reconcile_Call) Return(err error) *MockStandaloneServiceService_Reconcile_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockStandaloneServiceService_Reconcile_Call) RunAndReturn(run func(ctx context.Context, services []domain.StandaloneService) error) *MockStandaloneServiceService_Reconcile_Call { - _c.Call.Return(run) - return _c -} - -// Status provides a mock function for the type MockStandaloneServiceService -func (_mock *MockStandaloneServiceService) Status(ctx context.Context) ([]domain.StandaloneServiceStatus, error) { - ret := _mock.Called(ctx) - - if len(ret) == 0 { - panic("no return value specified for Status") - } - - var r0 []domain.StandaloneServiceStatus - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context) ([]domain.StandaloneServiceStatus, error)); ok { - return returnFunc(ctx) - } - if returnFunc, ok := ret.Get(0).(func(context.Context) []domain.StandaloneServiceStatus); ok { - r0 = returnFunc(ctx) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]domain.StandaloneServiceStatus) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { - r1 = returnFunc(ctx) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockStandaloneServiceService_Status_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Status' -type MockStandaloneServiceService_Status_Call struct { - *mock.Call -} - -// Status is a helper method to define mock.On call -// - ctx context.Context -func (_e *MockStandaloneServiceService_Expecter) Status(ctx any) *MockStandaloneServiceService_Status_Call { - return &MockStandaloneServiceService_Status_Call{Call: _e.mock.On("Status", ctx)} -} - -func (_c *MockStandaloneServiceService_Status_Call) Run(run func(ctx context.Context)) *MockStandaloneServiceService_Status_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - run( - arg0, - ) - }) - return _c -} - -func (_c *MockStandaloneServiceService_Status_Call) Return(standaloneServiceStatuss []domain.StandaloneServiceStatus, err error) *MockStandaloneServiceService_Status_Call { - _c.Call.Return(standaloneServiceStatuss, err) - return _c -} - -func (_c *MockStandaloneServiceService_Status_Call) RunAndReturn(run func(ctx context.Context) ([]domain.StandaloneServiceStatus, error)) *MockStandaloneServiceService_Status_Call { - _c.Call.Return(run) - return _c -} diff --git a/internal/boundaries/in/mocks/mock_traffic_status_service.go b/internal/boundaries/in/mocks/mock_traffic_status_service.go index 03666a804..26e3bf5d1 100644 --- a/internal/boundaries/in/mocks/mock_traffic_status_service.go +++ b/internal/boundaries/in/mocks/mock_traffic_status_service.go @@ -15,10 +15,19 @@ func NewMockTrafficStatusService(t interface { mock.TestingT Cleanup(func()) }) *MockTrafficStatusService { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockTrafficStatusService{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/in/mocks/mock_volume_backup_service.go b/internal/boundaries/in/mocks/mock_volume_backup_service.go index 6f0e8f0a6..49ac4be02 100644 --- a/internal/boundaries/in/mocks/mock_volume_backup_service.go +++ b/internal/boundaries/in/mocks/mock_volume_backup_service.go @@ -17,10 +17,19 @@ func NewMockVolumeBackupService(t interface { mock.TestingT Cleanup(func()) }) *MockVolumeBackupService { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockVolumeBackupService{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } @@ -39,8 +48,8 @@ func (_m *MockVolumeBackupService) EXPECT() *MockVolumeBackupService_Expecter { } // ListVolumeBackups provides a mock function for the type MockVolumeBackupService -func (_mock *MockVolumeBackupService) ListVolumeBackups(ctx context.Context, domainName string) ([]domain.VolumeBackupJob, error) { - ret := _mock.Called(ctx, domainName) +func (_mock *MockVolumeBackupService) ListVolumeBackups(ctx context.Context, app string) ([]domain.VolumeBackupJob, error) { + ret := _mock.Called(ctx, app) if len(ret) == 0 { panic("no return value specified for ListVolumeBackups") @@ -49,17 +58,17 @@ func (_mock *MockVolumeBackupService) ListVolumeBackups(ctx context.Context, dom var r0 []domain.VolumeBackupJob var r1 error if returnFunc, ok := ret.Get(0).(func(context.Context, string) ([]domain.VolumeBackupJob, error)); ok { - return returnFunc(ctx, domainName) + return returnFunc(ctx, app) } if returnFunc, ok := ret.Get(0).(func(context.Context, string) []domain.VolumeBackupJob); ok { - r0 = returnFunc(ctx, domainName) + r0 = returnFunc(ctx, app) } else { if ret.Get(0) != nil { r0 = ret.Get(0).([]domain.VolumeBackupJob) } } if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { - r1 = returnFunc(ctx, domainName) + r1 = returnFunc(ctx, app) } else { r1 = ret.Error(1) } @@ -73,12 +82,12 @@ type MockVolumeBackupService_ListVolumeBackups_Call struct { // ListVolumeBackups is a helper method to define mock.On call // - ctx context.Context -// - domainName string -func (_e *MockVolumeBackupService_Expecter) ListVolumeBackups(ctx any, domainName any) *MockVolumeBackupService_ListVolumeBackups_Call { - return &MockVolumeBackupService_ListVolumeBackups_Call{Call: _e.mock.On("ListVolumeBackups", ctx, domainName)} +// - app string +func (_e *MockVolumeBackupService_Expecter) ListVolumeBackups(ctx any, app any) *MockVolumeBackupService_ListVolumeBackups_Call { + return &MockVolumeBackupService_ListVolumeBackups_Call{Call: _e.mock.On("ListVolumeBackups", ctx, app)} } -func (_c *MockVolumeBackupService_ListVolumeBackups_Call) Run(run func(ctx context.Context, domainName string)) *MockVolumeBackupService_ListVolumeBackups_Call { +func (_c *MockVolumeBackupService_ListVolumeBackups_Call) Run(run func(ctx context.Context, app string)) *MockVolumeBackupService_ListVolumeBackups_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -101,14 +110,14 @@ func (_c *MockVolumeBackupService_ListVolumeBackups_Call) Return(volumeBackupJob return _c } -func (_c *MockVolumeBackupService_ListVolumeBackups_Call) RunAndReturn(run func(ctx context.Context, domainName string) ([]domain.VolumeBackupJob, error)) *MockVolumeBackupService_ListVolumeBackups_Call { +func (_c *MockVolumeBackupService_ListVolumeBackups_Call) RunAndReturn(run func(ctx context.Context, app string) ([]domain.VolumeBackupJob, error)) *MockVolumeBackupService_ListVolumeBackups_Call { _c.Call.Return(run) return _c } // RunVolumeBackups provides a mock function for the type MockVolumeBackupService -func (_mock *MockVolumeBackupService) RunVolumeBackups(ctx context.Context, domainName string, volumeName string) ([]domain.VolumeBackupJob, error) { - ret := _mock.Called(ctx, domainName, volumeName) +func (_mock *MockVolumeBackupService) RunVolumeBackups(ctx context.Context, app string, service string, volume string) ([]domain.VolumeBackupJob, error) { + ret := _mock.Called(ctx, app, service, volume) if len(ret) == 0 { panic("no return value specified for RunVolumeBackups") @@ -116,18 +125,18 @@ func (_mock *MockVolumeBackupService) RunVolumeBackups(ctx context.Context, doma var r0 []domain.VolumeBackupJob var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) ([]domain.VolumeBackupJob, error)); ok { - return returnFunc(ctx, domainName, volumeName) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, string) ([]domain.VolumeBackupJob, error)); ok { + return returnFunc(ctx, app, service, volume) } - if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) []domain.VolumeBackupJob); ok { - r0 = returnFunc(ctx, domainName, volumeName) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, string) []domain.VolumeBackupJob); ok { + r0 = returnFunc(ctx, app, service, volume) } else { if ret.Get(0) != nil { r0 = ret.Get(0).([]domain.VolumeBackupJob) } } - if returnFunc, ok := ret.Get(1).(func(context.Context, string, string) error); ok { - r1 = returnFunc(ctx, domainName, volumeName) + if returnFunc, ok := ret.Get(1).(func(context.Context, string, string, string) error); ok { + r1 = returnFunc(ctx, app, service, volume) } else { r1 = ret.Error(1) } @@ -141,13 +150,14 @@ type MockVolumeBackupService_RunVolumeBackups_Call struct { // RunVolumeBackups is a helper method to define mock.On call // - ctx context.Context -// - domainName string -// - volumeName string -func (_e *MockVolumeBackupService_Expecter) RunVolumeBackups(ctx any, domainName any, volumeName any) *MockVolumeBackupService_RunVolumeBackups_Call { - return &MockVolumeBackupService_RunVolumeBackups_Call{Call: _e.mock.On("RunVolumeBackups", ctx, domainName, volumeName)} +// - app string +// - service string +// - volume string +func (_e *MockVolumeBackupService_Expecter) RunVolumeBackups(ctx any, app any, service any, volume any) *MockVolumeBackupService_RunVolumeBackups_Call { + return &MockVolumeBackupService_RunVolumeBackups_Call{Call: _e.mock.On("RunVolumeBackups", ctx, app, service, volume)} } -func (_c *MockVolumeBackupService_RunVolumeBackups_Call) Run(run func(ctx context.Context, domainName string, volumeName string)) *MockVolumeBackupService_RunVolumeBackups_Call { +func (_c *MockVolumeBackupService_RunVolumeBackups_Call) Run(run func(ctx context.Context, app string, service string, volume string)) *MockVolumeBackupService_RunVolumeBackups_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -161,10 +171,15 @@ func (_c *MockVolumeBackupService_RunVolumeBackups_Call) Run(run func(ctx contex if args[2] != nil { arg2 = args[2].(string) } + var arg3 string + if args[3] != nil { + arg3 = args[3].(string) + } run( arg0, arg1, arg2, + arg3, ) }) return _c @@ -175,7 +190,7 @@ func (_c *MockVolumeBackupService_RunVolumeBackups_Call) Return(volumeBackupJobs return _c } -func (_c *MockVolumeBackupService_RunVolumeBackups_Call) RunAndReturn(run func(ctx context.Context, domainName string, volumeName string) ([]domain.VolumeBackupJob, error)) *MockVolumeBackupService_RunVolumeBackups_Call { +func (_c *MockVolumeBackupService_RunVolumeBackups_Call) RunAndReturn(run func(ctx context.Context, app string, service string, volume string) ([]domain.VolumeBackupJob, error)) *MockVolumeBackupService_RunVolumeBackups_Call { _c.Call.Return(run) return _c } diff --git a/internal/boundaries/in/mocks/mock_volume_service.go b/internal/boundaries/in/mocks/mock_volume_service.go index 4f8a65b93..11e719330 100644 --- a/internal/boundaries/in/mocks/mock_volume_service.go +++ b/internal/boundaries/in/mocks/mock_volume_service.go @@ -17,10 +17,19 @@ func NewMockVolumeService(t interface { mock.TestingT Cleanup(func()) }) *MockVolumeService { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockVolumeService{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/in/registry.go b/internal/boundaries/in/registry.go index d93af873b..57987ab5e 100644 --- a/internal/boundaries/in/registry.go +++ b/internal/boundaries/in/registry.go @@ -23,8 +23,8 @@ type RegistryService interface { // Upload operations StartUpload(ctx context.Context, name string) (string, error) AppendBlobChunk(ctx context.Context, name, uuid string, data io.Reader, contentLength, maxBlobSize int64) (int64, error) - FinishUpload(ctx context.Context, uuid, digest string) error - CancelUpload(ctx context.Context, uuid string) error + FinishUpload(ctx context.Context, name, uuid, digest string) error + CancelUpload(ctx context.Context, name, uuid string) error // Tag operations ListTags(ctx context.Context, name string) ([]string, error) diff --git a/internal/boundaries/in/secrets.go b/internal/boundaries/in/secrets.go deleted file mode 100644 index ae6726322..000000000 --- a/internal/boundaries/in/secrets.go +++ /dev/null @@ -1,39 +0,0 @@ -package in - -import ( - "context" - - "github.com/bnema/gordon/internal/boundaries/out" -) - -// SecretService defines the contract for managing domain-scoped secrets. -// These are environment variables stored per-domain for container injection. -type SecretService interface { - // ListKeys returns the list of secret keys for a domain (not values). - // Returns an error if the domain is invalid. - ListKeys(ctx context.Context, domain string) ([]string, error) - - // ListKeysWithAttachments returns the list of secret keys for a domain - // along with any attachment secrets for containers associated with the domain. - ListKeysWithAttachments(ctx context.Context, domain string) ([]string, []out.AttachmentSecrets, error) - - // GetAll returns all secrets for a domain as a key-value map. - // Returns an error if the domain is invalid. - GetAll(ctx context.Context, domain string) (map[string]string, error) - - // Set sets or updates multiple secrets for a domain, merging with existing. - // Returns an error if the domain is invalid. - Set(ctx context.Context, domain string, secrets map[string]string) error - - // Delete removes a specific secret key from a domain. - // Returns an error if the domain is invalid. - Delete(ctx context.Context, domain, key string) error - - // SetAttachment sets or updates multiple secrets for an attachment container. - // The container name is derived from the domain and service name. - SetAttachment(ctx context.Context, domain, service string, secrets map[string]string) error - - // DeleteAttachment removes a specific secret key from an attachment container. - // The container name is derived from the domain and service name. - DeleteAttachment(ctx context.Context, domain, service, key string) error -} diff --git a/internal/boundaries/in/services.go b/internal/boundaries/in/services.go deleted file mode 100644 index 977ddbe18..000000000 --- a/internal/boundaries/in/services.go +++ /dev/null @@ -1,16 +0,0 @@ -// Package in defines input ports (interfaces) for use cases. -// These interfaces define the contract between driving adapters (HTTP, CLI) -// and the business logic (use cases). -package in - -import ( - "context" - - "github.com/bnema/gordon/internal/domain" -) - -// StandaloneServiceService defines standalone service lifecycle operations. -type StandaloneServiceService interface { - Reconcile(ctx context.Context, services []domain.StandaloneService) error - Status(ctx context.Context) ([]domain.StandaloneServiceStatus, error) -} diff --git a/internal/boundaries/out/app_hosts.go b/internal/boundaries/out/app_hosts.go new file mode 100644 index 000000000..7dd0ed7c7 --- /dev/null +++ b/internal/boundaries/out/app_hosts.go @@ -0,0 +1,23 @@ +package out + +// AppHostSource is the shared ACTIVE-derived HTTP host projection for +// traffic, PKI and public-TLS consumers. Implemented by the apptraffic +// host index; read-only, never container IPs. +// +// Fail-closed contract: no apps, no ACTIVE record, stopped intent, or +// unresolved backends yield zero hosts. Installation external and +// management hosts are authorized independently, never through here. +// TLS-mode filtering (auto/always/never) is the consumer's decision: +// the projection preserves TLSMode per host. +type AppHostSource interface { + // AppHosts returns the sorted served hosts with resolved backends. + AppHosts() []AppHost +} + +// AppHost is one served HTTP host with its owning workload and TLS mode. +type AppHost struct { + Host string + App string + Service string + TLSMode string +} diff --git a/internal/boundaries/out/app_state.go b/internal/boundaries/out/app_state.go new file mode 100644 index 000000000..6798c7c61 --- /dev/null +++ b/internal/boundaries/out/app_state.go @@ -0,0 +1,179 @@ +package out + +import ( + "context" + + "github.com/bnema/gordon/internal/domain" +) + +// AppState persists desired/active app state, reservations, intents, +// operation journals, and ownership records. Implemented by the bbolt +// appstate adapter; consumed whole by the deployment engine. Other use +// cases depend on the narrower AppStateReader, AppCatalogReader, or +// AppApplyStore ports. +// All methods are safe for concurrent use within one process; +// cross-process coordination is owned by the implementation +// (store.lock flock). No secret values pass through this boundary. +type AppState interface { + // Close releases the durable store and its process lock. + Close() error + + // Recover completes committed-but-unmaterialized apply intents + // before any new mutation is accepted (recovery-before-mutation). + Recover(ctx context.Context) error + + // LoadRecoveryInhibitions returns every durable recovery-inhibition + // record of one app. Boot and periodic recovery must refuse to + // start or restart an inhibited container ID. + LoadRecoveryInhibitions(ctx context.Context, app string) ([]domain.AppRecoveryInhibition, error) + + // SaveRecoveryInhibition durably records one generation-scoped + // recovery inhibition. Repeated calls for the same + // (app, service, container ID) replace the existing record. + SaveRecoveryInhibition(ctx context.Context, inhibition domain.AppRecoveryInhibition) error + + // ClearRecoveryInhibition drops the inhibition of one exact + // generation. Clearing an absent record is not an error. + ClearRecoveryInhibition(ctx context.Context, app, service, containerID string) error + + // LoadCheckpoint returns the global reservation checkpoint. + LoadCheckpoint(ctx context.Context) (domain.AppStoreCheckpoint, error) + + // AppExists reports whether the app has a live identity: desired + // state, active state, or an ownership incarnation. It never creates + // state. A retired app whose only remaining record is an operation + // journal does not exist (its own key still replays through + // ClaimOperation). + AppExists(ctx context.Context, app string) (bool, error) + + // RegisterBackendBinds records Gordon-generated loopback backend + // binds (owner gordon-backend) in the global checkpoint. It fails + // closed on conflict with another app's desired/active/in-flight + // claim. Old binds stay claimed until ReleaseBackendBinds after + // verified withdrawal; same-app claims never self-conflict. + RegisterBackendBinds(ctx context.Context, binds []domain.AppListenerReservation) error + + // ReleaseBackendBinds drops Gordon-generated loopback claims for one + // retired container. Unknown claims are ignored. + ReleaseBackendBinds(ctx context.Context, app, containerID string) error + + // ListApps returns normalized names of apps with any state. + ListApps(ctx context.Context) ([]string, error) + + // LoadDesired returns the latest accepted desired revision. + // ok is false when the app has no desired state. + LoadDesired(ctx context.Context, app string) (domain.AppDesiredRevision, bool, error) + + // LoadRevision returns one immutable revision by id. + LoadRevision(ctx context.Context, app, revision string) (domain.AppDesiredRevision, error) + + // LoadActive returns the per-service effective state. + // ok is false when the app was never deployed. + LoadActive(ctx context.Context, app string) (domain.AppActive, bool, error) + + // LoadIntent returns the durable stopped/running intent. + LoadIntent(ctx context.Context, app string) (domain.AppStopIntent, error) + + // SaveIntent persists running/stopped intent. + SaveIntent(ctx context.Context, intent domain.AppStopIntent) error + + // LoadOwnership returns the ownership record (zero value when absent). + LoadOwnership(ctx context.Context, app string) (domain.AppOwnership, error) + + // SaveOwnership persists the ownership record. + SaveOwnership(ctx context.Context, ownership domain.AppOwnership) error + + // AcceptApply durably accepts one apply intent: it stages the full + // candidate, commits it (the single atomic commit point), and + // materializes the revision, desired pointer, and reservation + // checkpoint. Each step is its own transaction, so a crash after the + // commit is finished by Recover and a staged orphan is swept by GC. + // A garbage-collection failure after materialization is logged and + // does not fail the accepted apply. + AcceptApply(ctx context.Context, intent domain.AppApplyIntent) error + + // SaveOperation persists an operation journal record atomically. + SaveOperation(ctx context.Context, op domain.AppOperation) error + + // ClaimOperation is the single atomic check-and-write point for one + // mutation request key. Absent key: the candidate is persisted as an + // in-flight journal and claimed is true. Same key with the same + // request identity: the stored journal is returned and claimed is + // false, meaning the caller must not execute the request's effects + // again. Same key with a different request identity: + // ErrAppStateConflict. Absent key for an app with no desired, active, + // or ownership incarnation: ErrAppNotFound, writing nothing. A + // concurrent claim of the same key never creates a second journal. + ClaimOperation(ctx context.Context, candidate domain.AppOperation) (existing domain.AppOperation, claimed bool, err error) + + // LoadOperation returns one operation journal record. + LoadOperation(ctx context.Context, app, opID string) (domain.AppOperation, error) + + // LoadLatestOperation returns the most recently started operation. + LoadLatestOperation(ctx context.Context, app string) (domain.AppOperation, bool, error) + + // SaveActive persists the per-service effective state. + SaveActive(ctx context.Context, active domain.AppActive) error + + // RetireApp atomically ends an app's incarnation: it archives the + // live ownership record with every resource marked retained, drops + // the live ownership record, resets the app UUID so a reapply + // allocates a new incarnation, and clears desired, active, intent, + // staged intents, and revisions. Retained resources stay protected + // by the archived record and can never be adopted by a name reuse. + RetireApp(ctx context.Context, app string) error +} + +// AppStateReader is the ACTIVE/intent subset read paths need. +// Narrower than AppState so read paths never gain write access. +type AppStateReader interface { + ListApps(ctx context.Context) ([]string, error) + LoadActive(ctx context.Context, app string) (domain.AppActive, bool, error) + LoadIntent(ctx context.Context, app string) (domain.AppStopIntent, error) +} + +// AppCatalogReader is the read-only view the app administration read +// model needs: identity, desired/active state, and operation journals. +type AppCatalogReader interface { + AppStateReader + AppExists(ctx context.Context, app string) (bool, error) + LoadDesired(ctx context.Context, app string) (domain.AppDesiredRevision, bool, error) + LoadRevision(ctx context.Context, app, revision string) (domain.AppDesiredRevision, error) + LoadOwnership(ctx context.Context, app string) (domain.AppOwnership, error) + LoadOperation(ctx context.Context, app, opID string) (domain.AppOperation, error) + LoadLatestOperation(ctx context.Context, app string) (domain.AppOperation, bool, error) +} + +// AppApplyStore is what accepting desired configuration needs: recovery +// before mutation, conflict inputs, and one durable accept. +type AppApplyStore interface { + Recover(ctx context.Context) error + LoadCheckpoint(ctx context.Context) (domain.AppStoreCheckpoint, error) + LoadDesired(ctx context.Context, app string) (domain.AppDesiredRevision, bool, error) + LoadActive(ctx context.Context, app string) (domain.AppActive, bool, error) + AcceptApply(ctx context.Context, intent domain.AppApplyIntent) error +} + +// AppTrafficRefresher is the app-facing traffic contract: one serialized +// HTTP/L4 publish boundary that supports both fail-closed withdrawal of +// one service and full publication of current verified state. Both +// operations read the latest validated config at call time and are +// serialized against reload and other rebuilds so a stale config can +// never win. WithdrawService must leave no forwarding for the named +// service once it returns nil; failing loudly is preferred to leaving +// stale L4 forwarding behind. +type AppTrafficRefresher interface { + // RebuildTraffic re-projects verified ACTIVE state and applies the + // full HTTP/L4 graph. + RebuildTraffic(ctx context.Context) error + // WithdrawService publishes the current graph without the named + // app service, so a dead or unverified backend stops receiving + // traffic. A nil return means stale forwarding is proven disabled. + WithdrawService(ctx context.Context, app, service string) error + // WithdrawServiceState drops the named service's recorded backend + // binds in ACTIVE without applying the graph. It is the canonical + // state-only withdrawal: ready-path failures use it when the + // caller already owns publication (or none is due), so no second + // routing-state write races the serialized WithdrawService path. + WithdrawServiceState(ctx context.Context, app, service string) error +} diff --git a/internal/boundaries/out/attachment_config.go b/internal/boundaries/out/attachment_config.go deleted file mode 100644 index bd83925a2..000000000 --- a/internal/boundaries/out/attachment_config.go +++ /dev/null @@ -1,25 +0,0 @@ -package out - -// AttachmentConfigSnapshot holds a consistent point-in-time snapshot of -// attachment and network group configuration. -type AttachmentConfigSnapshot struct { - Attachments map[string][]string // domain/group -> []image - NetworkGroups map[string][]string // group -> []domain -} - -// AttachmentConfigProvider is a read-only view into the live config service -// for attachment and network group resolution. It decouples the container -// service from the full ConfigService interface while ensuring fresh reads -// at decision time (deploy, restart). -// -// Implementations must return deep copies of the underlying data so that -// callers cannot mutate provider state. Each call must produce an independent -// map with independently allocated slices. -type AttachmentConfigProvider interface { - // GetAttachmentConfig returns a consistent snapshot of attachments and - // network groups, read under a single lock to prevent cross-field races. - GetAttachmentConfig() AttachmentConfigSnapshot - - // GetNetworkGroups returns a deep copy of the current network group config: group -> []domain. - GetNetworkGroups() map[string][]string -} diff --git a/internal/boundaries/out/backupstore.go b/internal/boundaries/out/backupstore.go index 2e66be620..d776d20db 100644 --- a/internal/boundaries/out/backupstore.go +++ b/internal/boundaries/out/backupstore.go @@ -10,11 +10,11 @@ import ( // DatabaseBackupStorage defines persistence for database backup artifacts and metadata. type DatabaseBackupStorage interface { - Store(ctx context.Context, domainName, dbName string, schedule domain.BackupSchedule, timestamp time.Time, data io.Reader) (string, error) + Store(ctx context.Context, app, service, database string, schedule domain.BackupSchedule, timestamp time.Time, data io.Reader) (string, error) Get(ctx context.Context, path string) (io.ReadCloser, error) - List(ctx context.Context, domainName string, schedule *domain.BackupSchedule) ([]domain.DatabaseBackupJob, error) + List(ctx context.Context, app string, schedule *domain.BackupSchedule) ([]domain.DatabaseBackupJob, error) Delete(ctx context.Context, path string) error - ApplyRetention(ctx context.Context, domainName string, policy domain.DatabaseBackupRetentionPolicy) (int, error) + ApplyRetention(ctx context.Context, app string, policy domain.DatabaseBackupRetentionPolicy) (int, error) } // BackupStorage is kept as a compatibility alias for the existing database backup feature. diff --git a/internal/boundaries/out/domain_secrets.go b/internal/boundaries/out/domain_secrets.go deleted file mode 100644 index 495e63075..000000000 --- a/internal/boundaries/out/domain_secrets.go +++ /dev/null @@ -1,39 +0,0 @@ -package out - -// AttachmentSecrets represents secrets for an attachment container. -type AttachmentSecrets struct { - // Service is the attachment service name (e.g., "gitea-postgres") - Service string - // Keys is the list of secret keys for this attachment - Keys []string -} - -// DomainSecretStore defines the contract for managing domain-scoped secrets. -// These are environment variables stored per-domain for container injection. -type DomainSecretStore interface { - // ListKeys returns the list of secret keys for a domain (not values). - ListKeys(domain string) ([]string, error) - - // GetAll returns all secrets for a domain as a key-value map. - GetAll(domain string) (map[string]string, error) - - // Set sets or updates multiple secrets for a domain, merging with existing. - Set(domain string, secrets map[string]string) error - - // Delete removes a specific secret key from a domain. - Delete(domain, key string) error - - // SetAttachment sets or updates multiple secrets for an attachment container. - SetAttachment(containerName string, secrets map[string]string) error - - // GetAllAttachment returns all secrets for an attachment container as a key-value map. - GetAllAttachment(containerName string) (map[string]string, error) - - // DeleteAttachment removes a specific secret key from an attachment container. - DeleteAttachment(containerName, key string) error - - // ListAttachmentKeys finds and returns secret keys for attachment containers - // associated with the given domain. Returns a list of AttachmentSecrets, one - // for each attachment that has secrets configured. - ListAttachmentKeys(domain string) ([]AttachmentSecrets, error) -} diff --git a/internal/boundaries/out/envloader.go b/internal/boundaries/out/envloader.go deleted file mode 100644 index 297ae27b9..000000000 --- a/internal/boundaries/out/envloader.go +++ /dev/null @@ -1,16 +0,0 @@ -package out - -import "context" - -// EnvLoader defines the contract for loading environment variables for containers. -type EnvLoader interface { - // LoadEnv loads environment variables for a given domain. - // Returns a slice of "KEY=VALUE" strings. - LoadEnv(ctx context.Context, domain string) ([]string, error) - - // CreateEnvFile creates an empty environment file for a new domain. - CreateEnvFile(ctx context.Context, domain string) error - - // EnvFileExists checks if an environment file exists for a domain. - EnvFileExists(domain string) (bool, error) -} diff --git a/internal/boundaries/out/image_resolver.go b/internal/boundaries/out/image_resolver.go new file mode 100644 index 000000000..66d9825a4 --- /dev/null +++ b/internal/boundaries/out/image_resolver.go @@ -0,0 +1,15 @@ +package out + +import ( + "context" +) + +// ImageResolver resolves an image reference to a content digest without +// pulling or running anything. Installation-registry refs resolve locally; +// external refs honor allowlists and require-digest policy at the adapter. +// Consumed by deployment preflight; the adapter lands at cutover. +type ImageResolver interface { + // ResolveDigest returns the pinned digest for ref (e.g. sha256:…). + // Mutable tags are re-resolved on every call; callers pin the result. + ResolveDigest(ctx context.Context, ref string) (string, error) +} diff --git a/internal/boundaries/out/logs.go b/internal/boundaries/out/logs.go new file mode 100644 index 000000000..cce3326c5 --- /dev/null +++ b/internal/boundaries/out/logs.go @@ -0,0 +1,24 @@ +package out + +import ( + "context" + "time" + + "github.com/bnema/gordon/internal/domain" +) + +// LogExporter ships log records to an external log backend (OTLP). +// Export must not block on the network: implementations buffer and +// drop on overload rather than slow down callers. Safe for concurrent use. +type LogExporter interface { + Export(ctx context.Context, record domain.LogRecord) +} + +// ContainerLogStreamer follows the demultiplexed output of one container. +type ContainerLogStreamer interface { + // StreamContainerLogs calls emit for every line emitted after since, + // in order, and blocks until the stream ends (container stopped), + // ctx is canceled, or an error occurs. A missing container reports + // domain.ErrContainerNotFound. + StreamContainerLogs(ctx context.Context, containerID string, since time.Time, emit func(domain.ContainerLogLine)) error +} diff --git a/internal/boundaries/out/metrics.go b/internal/boundaries/out/metrics.go new file mode 100644 index 000000000..30608dcf7 --- /dev/null +++ b/internal/boundaries/out/metrics.go @@ -0,0 +1,18 @@ +package out + +import ( + "context" + + "github.com/bnema/gordon/internal/domain" +) + +// Metrics records Gordon operational metrics. +// Implementations must be safe for concurrent use. +type Metrics interface { + // RecordImagePush counts one stored manifest and its size in bytes. + RecordImagePush(ctx context.Context, name, reference string, sizeBytes int64) + // RecordEventProcessed counts one event handled successfully. + RecordEventProcessed(ctx context.Context, eventType domain.EventType) + // RecordEventDropped counts one event dropped because the bus was full. + RecordEventDropped(ctx context.Context, eventType domain.EventType) +} diff --git a/internal/boundaries/out/mocks/mock_app_state.go b/internal/boundaries/out/mocks/mock_app_state.go new file mode 100644 index 000000000..72a3b68b9 --- /dev/null +++ b/internal/boundaries/out/mocks/mock_app_state.go @@ -0,0 +1,1551 @@ +// Code generated by mockery; DO NOT EDIT. +// github.com/vektra/mockery +// template: testify + +package mocks + +import ( + "context" + + "github.com/bnema/gordon/internal/domain" + mock "github.com/stretchr/testify/mock" +) + +// NewMockAppState creates a new instance of MockAppState. It also registers a testing interface on the mock and a cleanup function to assert the mocks expectations. +// The first argument is typically a *testing.T value. +func NewMockAppState(t interface { + mock.TestingT + Cleanup(func()) +}) *MockAppState { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + + mock := &MockAppState{} + mock.Mock.Test(t) + + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) + + return mock +} + +// MockAppState is an autogenerated mock type for the AppState type +type MockAppState struct { + mock.Mock +} + +type MockAppState_Expecter struct { + mock *mock.Mock +} + +func (_m *MockAppState) EXPECT() *MockAppState_Expecter { + return &MockAppState_Expecter{mock: &_m.Mock} +} + +// AcceptApply provides a mock function for the type MockAppState +func (_mock *MockAppState) AcceptApply(ctx context.Context, intent domain.AppApplyIntent) error { + ret := _mock.Called(ctx, intent) + + if len(ret) == 0 { + panic("no return value specified for AcceptApply") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context, domain.AppApplyIntent) error); ok { + r0 = returnFunc(ctx, intent) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockAppState_AcceptApply_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'AcceptApply' +type MockAppState_AcceptApply_Call struct { + *mock.Call +} + +// AcceptApply is a helper method to define mock.On call +// - ctx context.Context +// - intent domain.AppApplyIntent +func (_e *MockAppState_Expecter) AcceptApply(ctx any, intent any) *MockAppState_AcceptApply_Call { + return &MockAppState_AcceptApply_Call{Call: _e.mock.On("AcceptApply", ctx, intent)} +} + +func (_c *MockAppState_AcceptApply_Call) Run(run func(ctx context.Context, intent domain.AppApplyIntent)) *MockAppState_AcceptApply_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 domain.AppApplyIntent + if args[1] != nil { + arg1 = args[1].(domain.AppApplyIntent) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockAppState_AcceptApply_Call) Return(err error) *MockAppState_AcceptApply_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockAppState_AcceptApply_Call) RunAndReturn(run func(ctx context.Context, intent domain.AppApplyIntent) error) *MockAppState_AcceptApply_Call { + _c.Call.Return(run) + return _c +} + +// AppExists provides a mock function for the type MockAppState +func (_mock *MockAppState) AppExists(ctx context.Context, app string) (bool, error) { + ret := _mock.Called(ctx, app) + + if len(ret) == 0 { + panic("no return value specified for AppExists") + } + + var r0 bool + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string) (bool, error)); ok { + return returnFunc(ctx, app) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string) bool); ok { + r0 = returnFunc(ctx, app) + } else { + r0 = ret.Get(0).(bool) + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { + r1 = returnFunc(ctx, app) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockAppState_AppExists_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'AppExists' +type MockAppState_AppExists_Call struct { + *mock.Call +} + +// AppExists is a helper method to define mock.On call +// - ctx context.Context +// - app string +func (_e *MockAppState_Expecter) AppExists(ctx any, app any) *MockAppState_AppExists_Call { + return &MockAppState_AppExists_Call{Call: _e.mock.On("AppExists", ctx, app)} +} + +func (_c *MockAppState_AppExists_Call) Run(run func(ctx context.Context, app string)) *MockAppState_AppExists_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockAppState_AppExists_Call) Return(b bool, err error) *MockAppState_AppExists_Call { + _c.Call.Return(b, err) + return _c +} + +func (_c *MockAppState_AppExists_Call) RunAndReturn(run func(ctx context.Context, app string) (bool, error)) *MockAppState_AppExists_Call { + _c.Call.Return(run) + return _c +} + +// ClaimOperation provides a mock function for the type MockAppState +func (_mock *MockAppState) ClaimOperation(ctx context.Context, candidate domain.AppOperation) (domain.AppOperation, bool, error) { + ret := _mock.Called(ctx, candidate) + + if len(ret) == 0 { + panic("no return value specified for ClaimOperation") + } + + var r0 domain.AppOperation + var r1 bool + var r2 error + if returnFunc, ok := ret.Get(0).(func(context.Context, domain.AppOperation) (domain.AppOperation, bool, error)); ok { + return returnFunc(ctx, candidate) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, domain.AppOperation) domain.AppOperation); ok { + r0 = returnFunc(ctx, candidate) + } else { + r0 = ret.Get(0).(domain.AppOperation) + } + if returnFunc, ok := ret.Get(1).(func(context.Context, domain.AppOperation) bool); ok { + r1 = returnFunc(ctx, candidate) + } else { + r1 = ret.Get(1).(bool) + } + if returnFunc, ok := ret.Get(2).(func(context.Context, domain.AppOperation) error); ok { + r2 = returnFunc(ctx, candidate) + } else { + r2 = ret.Error(2) + } + return r0, r1, r2 +} + +// MockAppState_ClaimOperation_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ClaimOperation' +type MockAppState_ClaimOperation_Call struct { + *mock.Call +} + +// ClaimOperation is a helper method to define mock.On call +// - ctx context.Context +// - candidate domain.AppOperation +func (_e *MockAppState_Expecter) ClaimOperation(ctx any, candidate any) *MockAppState_ClaimOperation_Call { + return &MockAppState_ClaimOperation_Call{Call: _e.mock.On("ClaimOperation", ctx, candidate)} +} + +func (_c *MockAppState_ClaimOperation_Call) Run(run func(ctx context.Context, candidate domain.AppOperation)) *MockAppState_ClaimOperation_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 domain.AppOperation + if args[1] != nil { + arg1 = args[1].(domain.AppOperation) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockAppState_ClaimOperation_Call) Return(existing domain.AppOperation, claimed bool, err error) *MockAppState_ClaimOperation_Call { + _c.Call.Return(existing, claimed, err) + return _c +} + +func (_c *MockAppState_ClaimOperation_Call) RunAndReturn(run func(ctx context.Context, candidate domain.AppOperation) (domain.AppOperation, bool, error)) *MockAppState_ClaimOperation_Call { + _c.Call.Return(run) + return _c +} + +// ClearRecoveryInhibition provides a mock function for the type MockAppState +func (_mock *MockAppState) ClearRecoveryInhibition(ctx context.Context, app string, service string, containerID string) error { + ret := _mock.Called(ctx, app, service, containerID) + + if len(ret) == 0 { + panic("no return value specified for ClearRecoveryInhibition") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, string) error); ok { + r0 = returnFunc(ctx, app, service, containerID) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockAppState_ClearRecoveryInhibition_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ClearRecoveryInhibition' +type MockAppState_ClearRecoveryInhibition_Call struct { + *mock.Call +} + +// ClearRecoveryInhibition is a helper method to define mock.On call +// - ctx context.Context +// - app string +// - service string +// - containerID string +func (_e *MockAppState_Expecter) ClearRecoveryInhibition(ctx any, app any, service any, containerID any) *MockAppState_ClearRecoveryInhibition_Call { + return &MockAppState_ClearRecoveryInhibition_Call{Call: _e.mock.On("ClearRecoveryInhibition", ctx, app, service, containerID)} +} + +func (_c *MockAppState_ClearRecoveryInhibition_Call) Run(run func(ctx context.Context, app string, service string, containerID string)) *MockAppState_ClearRecoveryInhibition_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } + var arg3 string + if args[3] != nil { + arg3 = args[3].(string) + } + run( + arg0, + arg1, + arg2, + arg3, + ) + }) + return _c +} + +func (_c *MockAppState_ClearRecoveryInhibition_Call) Return(err error) *MockAppState_ClearRecoveryInhibition_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockAppState_ClearRecoveryInhibition_Call) RunAndReturn(run func(ctx context.Context, app string, service string, containerID string) error) *MockAppState_ClearRecoveryInhibition_Call { + _c.Call.Return(run) + return _c +} + +// Close provides a mock function for the type MockAppState +func (_mock *MockAppState) Close() error { + ret := _mock.Called() + + if len(ret) == 0 { + panic("no return value specified for Close") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func() error); ok { + r0 = returnFunc() + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockAppState_Close_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Close' +type MockAppState_Close_Call struct { + *mock.Call +} + +// Close is a helper method to define mock.On call +func (_e *MockAppState_Expecter) Close() *MockAppState_Close_Call { + return &MockAppState_Close_Call{Call: _e.mock.On("Close")} +} + +func (_c *MockAppState_Close_Call) Run(run func()) *MockAppState_Close_Call { + _c.Call.Run(func(args mock.Arguments) { + run() + }) + return _c +} + +func (_c *MockAppState_Close_Call) Return(err error) *MockAppState_Close_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockAppState_Close_Call) RunAndReturn(run func() error) *MockAppState_Close_Call { + _c.Call.Return(run) + return _c +} + +// ListApps provides a mock function for the type MockAppState +func (_mock *MockAppState) ListApps(ctx context.Context) ([]string, error) { + ret := _mock.Called(ctx) + + if len(ret) == 0 { + panic("no return value specified for ListApps") + } + + var r0 []string + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context) ([]string, error)); ok { + return returnFunc(ctx) + } + if returnFunc, ok := ret.Get(0).(func(context.Context) []string); ok { + r0 = returnFunc(ctx) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).([]string) + } + } + if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { + r1 = returnFunc(ctx) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockAppState_ListApps_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListApps' +type MockAppState_ListApps_Call struct { + *mock.Call +} + +// ListApps is a helper method to define mock.On call +// - ctx context.Context +func (_e *MockAppState_Expecter) ListApps(ctx any) *MockAppState_ListApps_Call { + return &MockAppState_ListApps_Call{Call: _e.mock.On("ListApps", ctx)} +} + +func (_c *MockAppState_ListApps_Call) Run(run func(ctx context.Context)) *MockAppState_ListApps_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + run( + arg0, + ) + }) + return _c +} + +func (_c *MockAppState_ListApps_Call) Return(strings []string, err error) *MockAppState_ListApps_Call { + _c.Call.Return(strings, err) + return _c +} + +func (_c *MockAppState_ListApps_Call) RunAndReturn(run func(ctx context.Context) ([]string, error)) *MockAppState_ListApps_Call { + _c.Call.Return(run) + return _c +} + +// LoadActive provides a mock function for the type MockAppState +func (_mock *MockAppState) LoadActive(ctx context.Context, app string) (domain.AppActive, bool, error) { + ret := _mock.Called(ctx, app) + + if len(ret) == 0 { + panic("no return value specified for LoadActive") + } + + var r0 domain.AppActive + var r1 bool + var r2 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string) (domain.AppActive, bool, error)); ok { + return returnFunc(ctx, app) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string) domain.AppActive); ok { + r0 = returnFunc(ctx, app) + } else { + r0 = ret.Get(0).(domain.AppActive) + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string) bool); ok { + r1 = returnFunc(ctx, app) + } else { + r1 = ret.Get(1).(bool) + } + if returnFunc, ok := ret.Get(2).(func(context.Context, string) error); ok { + r2 = returnFunc(ctx, app) + } else { + r2 = ret.Error(2) + } + return r0, r1, r2 +} + +// MockAppState_LoadActive_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'LoadActive' +type MockAppState_LoadActive_Call struct { + *mock.Call +} + +// LoadActive is a helper method to define mock.On call +// - ctx context.Context +// - app string +func (_e *MockAppState_Expecter) LoadActive(ctx any, app any) *MockAppState_LoadActive_Call { + return &MockAppState_LoadActive_Call{Call: _e.mock.On("LoadActive", ctx, app)} +} + +func (_c *MockAppState_LoadActive_Call) Run(run func(ctx context.Context, app string)) *MockAppState_LoadActive_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockAppState_LoadActive_Call) Return(appActive domain.AppActive, b bool, err error) *MockAppState_LoadActive_Call { + _c.Call.Return(appActive, b, err) + return _c +} + +func (_c *MockAppState_LoadActive_Call) RunAndReturn(run func(ctx context.Context, app string) (domain.AppActive, bool, error)) *MockAppState_LoadActive_Call { + _c.Call.Return(run) + return _c +} + +// LoadCheckpoint provides a mock function for the type MockAppState +func (_mock *MockAppState) LoadCheckpoint(ctx context.Context) (domain.AppStoreCheckpoint, error) { + ret := _mock.Called(ctx) + + if len(ret) == 0 { + panic("no return value specified for LoadCheckpoint") + } + + var r0 domain.AppStoreCheckpoint + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context) (domain.AppStoreCheckpoint, error)); ok { + return returnFunc(ctx) + } + if returnFunc, ok := ret.Get(0).(func(context.Context) domain.AppStoreCheckpoint); ok { + r0 = returnFunc(ctx) + } else { + r0 = ret.Get(0).(domain.AppStoreCheckpoint) + } + if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { + r1 = returnFunc(ctx) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockAppState_LoadCheckpoint_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'LoadCheckpoint' +type MockAppState_LoadCheckpoint_Call struct { + *mock.Call +} + +// LoadCheckpoint is a helper method to define mock.On call +// - ctx context.Context +func (_e *MockAppState_Expecter) LoadCheckpoint(ctx any) *MockAppState_LoadCheckpoint_Call { + return &MockAppState_LoadCheckpoint_Call{Call: _e.mock.On("LoadCheckpoint", ctx)} +} + +func (_c *MockAppState_LoadCheckpoint_Call) Run(run func(ctx context.Context)) *MockAppState_LoadCheckpoint_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + run( + arg0, + ) + }) + return _c +} + +func (_c *MockAppState_LoadCheckpoint_Call) Return(appStoreCheckpoint domain.AppStoreCheckpoint, err error) *MockAppState_LoadCheckpoint_Call { + _c.Call.Return(appStoreCheckpoint, err) + return _c +} + +func (_c *MockAppState_LoadCheckpoint_Call) RunAndReturn(run func(ctx context.Context) (domain.AppStoreCheckpoint, error)) *MockAppState_LoadCheckpoint_Call { + _c.Call.Return(run) + return _c +} + +// LoadDesired provides a mock function for the type MockAppState +func (_mock *MockAppState) LoadDesired(ctx context.Context, app string) (domain.AppDesiredRevision, bool, error) { + ret := _mock.Called(ctx, app) + + if len(ret) == 0 { + panic("no return value specified for LoadDesired") + } + + var r0 domain.AppDesiredRevision + var r1 bool + var r2 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string) (domain.AppDesiredRevision, bool, error)); ok { + return returnFunc(ctx, app) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string) domain.AppDesiredRevision); ok { + r0 = returnFunc(ctx, app) + } else { + r0 = ret.Get(0).(domain.AppDesiredRevision) + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string) bool); ok { + r1 = returnFunc(ctx, app) + } else { + r1 = ret.Get(1).(bool) + } + if returnFunc, ok := ret.Get(2).(func(context.Context, string) error); ok { + r2 = returnFunc(ctx, app) + } else { + r2 = ret.Error(2) + } + return r0, r1, r2 +} + +// MockAppState_LoadDesired_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'LoadDesired' +type MockAppState_LoadDesired_Call struct { + *mock.Call +} + +// LoadDesired is a helper method to define mock.On call +// - ctx context.Context +// - app string +func (_e *MockAppState_Expecter) LoadDesired(ctx any, app any) *MockAppState_LoadDesired_Call { + return &MockAppState_LoadDesired_Call{Call: _e.mock.On("LoadDesired", ctx, app)} +} + +func (_c *MockAppState_LoadDesired_Call) Run(run func(ctx context.Context, app string)) *MockAppState_LoadDesired_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockAppState_LoadDesired_Call) Return(appDesiredRevision domain.AppDesiredRevision, b bool, err error) *MockAppState_LoadDesired_Call { + _c.Call.Return(appDesiredRevision, b, err) + return _c +} + +func (_c *MockAppState_LoadDesired_Call) RunAndReturn(run func(ctx context.Context, app string) (domain.AppDesiredRevision, bool, error)) *MockAppState_LoadDesired_Call { + _c.Call.Return(run) + return _c +} + +// LoadIntent provides a mock function for the type MockAppState +func (_mock *MockAppState) LoadIntent(ctx context.Context, app string) (domain.AppStopIntent, error) { + ret := _mock.Called(ctx, app) + + if len(ret) == 0 { + panic("no return value specified for LoadIntent") + } + + var r0 domain.AppStopIntent + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string) (domain.AppStopIntent, error)); ok { + return returnFunc(ctx, app) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string) domain.AppStopIntent); ok { + r0 = returnFunc(ctx, app) + } else { + r0 = ret.Get(0).(domain.AppStopIntent) + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { + r1 = returnFunc(ctx, app) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockAppState_LoadIntent_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'LoadIntent' +type MockAppState_LoadIntent_Call struct { + *mock.Call +} + +// LoadIntent is a helper method to define mock.On call +// - ctx context.Context +// - app string +func (_e *MockAppState_Expecter) LoadIntent(ctx any, app any) *MockAppState_LoadIntent_Call { + return &MockAppState_LoadIntent_Call{Call: _e.mock.On("LoadIntent", ctx, app)} +} + +func (_c *MockAppState_LoadIntent_Call) Run(run func(ctx context.Context, app string)) *MockAppState_LoadIntent_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockAppState_LoadIntent_Call) Return(appStopIntent domain.AppStopIntent, err error) *MockAppState_LoadIntent_Call { + _c.Call.Return(appStopIntent, err) + return _c +} + +func (_c *MockAppState_LoadIntent_Call) RunAndReturn(run func(ctx context.Context, app string) (domain.AppStopIntent, error)) *MockAppState_LoadIntent_Call { + _c.Call.Return(run) + return _c +} + +// LoadLatestOperation provides a mock function for the type MockAppState +func (_mock *MockAppState) LoadLatestOperation(ctx context.Context, app string) (domain.AppOperation, bool, error) { + ret := _mock.Called(ctx, app) + + if len(ret) == 0 { + panic("no return value specified for LoadLatestOperation") + } + + var r0 domain.AppOperation + var r1 bool + var r2 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string) (domain.AppOperation, bool, error)); ok { + return returnFunc(ctx, app) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string) domain.AppOperation); ok { + r0 = returnFunc(ctx, app) + } else { + r0 = ret.Get(0).(domain.AppOperation) + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string) bool); ok { + r1 = returnFunc(ctx, app) + } else { + r1 = ret.Get(1).(bool) + } + if returnFunc, ok := ret.Get(2).(func(context.Context, string) error); ok { + r2 = returnFunc(ctx, app) + } else { + r2 = ret.Error(2) + } + return r0, r1, r2 +} + +// MockAppState_LoadLatestOperation_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'LoadLatestOperation' +type MockAppState_LoadLatestOperation_Call struct { + *mock.Call +} + +// LoadLatestOperation is a helper method to define mock.On call +// - ctx context.Context +// - app string +func (_e *MockAppState_Expecter) LoadLatestOperation(ctx any, app any) *MockAppState_LoadLatestOperation_Call { + return &MockAppState_LoadLatestOperation_Call{Call: _e.mock.On("LoadLatestOperation", ctx, app)} +} + +func (_c *MockAppState_LoadLatestOperation_Call) Run(run func(ctx context.Context, app string)) *MockAppState_LoadLatestOperation_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockAppState_LoadLatestOperation_Call) Return(appOperation domain.AppOperation, b bool, err error) *MockAppState_LoadLatestOperation_Call { + _c.Call.Return(appOperation, b, err) + return _c +} + +func (_c *MockAppState_LoadLatestOperation_Call) RunAndReturn(run func(ctx context.Context, app string) (domain.AppOperation, bool, error)) *MockAppState_LoadLatestOperation_Call { + _c.Call.Return(run) + return _c +} + +// LoadOperation provides a mock function for the type MockAppState +func (_mock *MockAppState) LoadOperation(ctx context.Context, app string, opID string) (domain.AppOperation, error) { + ret := _mock.Called(ctx, app, opID) + + if len(ret) == 0 { + panic("no return value specified for LoadOperation") + } + + var r0 domain.AppOperation + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) (domain.AppOperation, error)); ok { + return returnFunc(ctx, app, opID) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) domain.AppOperation); ok { + r0 = returnFunc(ctx, app, opID) + } else { + r0 = ret.Get(0).(domain.AppOperation) + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string, string) error); ok { + r1 = returnFunc(ctx, app, opID) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockAppState_LoadOperation_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'LoadOperation' +type MockAppState_LoadOperation_Call struct { + *mock.Call +} + +// LoadOperation is a helper method to define mock.On call +// - ctx context.Context +// - app string +// - opID string +func (_e *MockAppState_Expecter) LoadOperation(ctx any, app any, opID any) *MockAppState_LoadOperation_Call { + return &MockAppState_LoadOperation_Call{Call: _e.mock.On("LoadOperation", ctx, app, opID)} +} + +func (_c *MockAppState_LoadOperation_Call) Run(run func(ctx context.Context, app string, opID string)) *MockAppState_LoadOperation_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } + run( + arg0, + arg1, + arg2, + ) + }) + return _c +} + +func (_c *MockAppState_LoadOperation_Call) Return(appOperation domain.AppOperation, err error) *MockAppState_LoadOperation_Call { + _c.Call.Return(appOperation, err) + return _c +} + +func (_c *MockAppState_LoadOperation_Call) RunAndReturn(run func(ctx context.Context, app string, opID string) (domain.AppOperation, error)) *MockAppState_LoadOperation_Call { + _c.Call.Return(run) + return _c +} + +// LoadOwnership provides a mock function for the type MockAppState +func (_mock *MockAppState) LoadOwnership(ctx context.Context, app string) (domain.AppOwnership, error) { + ret := _mock.Called(ctx, app) + + if len(ret) == 0 { + panic("no return value specified for LoadOwnership") + } + + var r0 domain.AppOwnership + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string) (domain.AppOwnership, error)); ok { + return returnFunc(ctx, app) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string) domain.AppOwnership); ok { + r0 = returnFunc(ctx, app) + } else { + r0 = ret.Get(0).(domain.AppOwnership) + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { + r1 = returnFunc(ctx, app) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockAppState_LoadOwnership_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'LoadOwnership' +type MockAppState_LoadOwnership_Call struct { + *mock.Call +} + +// LoadOwnership is a helper method to define mock.On call +// - ctx context.Context +// - app string +func (_e *MockAppState_Expecter) LoadOwnership(ctx any, app any) *MockAppState_LoadOwnership_Call { + return &MockAppState_LoadOwnership_Call{Call: _e.mock.On("LoadOwnership", ctx, app)} +} + +func (_c *MockAppState_LoadOwnership_Call) Run(run func(ctx context.Context, app string)) *MockAppState_LoadOwnership_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockAppState_LoadOwnership_Call) Return(appOwnership domain.AppOwnership, err error) *MockAppState_LoadOwnership_Call { + _c.Call.Return(appOwnership, err) + return _c +} + +func (_c *MockAppState_LoadOwnership_Call) RunAndReturn(run func(ctx context.Context, app string) (domain.AppOwnership, error)) *MockAppState_LoadOwnership_Call { + _c.Call.Return(run) + return _c +} + +// LoadRecoveryInhibitions provides a mock function for the type MockAppState +func (_mock *MockAppState) LoadRecoveryInhibitions(ctx context.Context, app string) ([]domain.AppRecoveryInhibition, error) { + ret := _mock.Called(ctx, app) + + if len(ret) == 0 { + panic("no return value specified for LoadRecoveryInhibitions") + } + + var r0 []domain.AppRecoveryInhibition + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string) ([]domain.AppRecoveryInhibition, error)); ok { + return returnFunc(ctx, app) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string) []domain.AppRecoveryInhibition); ok { + r0 = returnFunc(ctx, app) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).([]domain.AppRecoveryInhibition) + } + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { + r1 = returnFunc(ctx, app) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockAppState_LoadRecoveryInhibitions_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'LoadRecoveryInhibitions' +type MockAppState_LoadRecoveryInhibitions_Call struct { + *mock.Call +} + +// LoadRecoveryInhibitions is a helper method to define mock.On call +// - ctx context.Context +// - app string +func (_e *MockAppState_Expecter) LoadRecoveryInhibitions(ctx any, app any) *MockAppState_LoadRecoveryInhibitions_Call { + return &MockAppState_LoadRecoveryInhibitions_Call{Call: _e.mock.On("LoadRecoveryInhibitions", ctx, app)} +} + +func (_c *MockAppState_LoadRecoveryInhibitions_Call) Run(run func(ctx context.Context, app string)) *MockAppState_LoadRecoveryInhibitions_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockAppState_LoadRecoveryInhibitions_Call) Return(appRecoveryInhibitions []domain.AppRecoveryInhibition, err error) *MockAppState_LoadRecoveryInhibitions_Call { + _c.Call.Return(appRecoveryInhibitions, err) + return _c +} + +func (_c *MockAppState_LoadRecoveryInhibitions_Call) RunAndReturn(run func(ctx context.Context, app string) ([]domain.AppRecoveryInhibition, error)) *MockAppState_LoadRecoveryInhibitions_Call { + _c.Call.Return(run) + return _c +} + +// LoadRevision provides a mock function for the type MockAppState +func (_mock *MockAppState) LoadRevision(ctx context.Context, app string, revision string) (domain.AppDesiredRevision, error) { + ret := _mock.Called(ctx, app, revision) + + if len(ret) == 0 { + panic("no return value specified for LoadRevision") + } + + var r0 domain.AppDesiredRevision + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) (domain.AppDesiredRevision, error)); ok { + return returnFunc(ctx, app, revision) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) domain.AppDesiredRevision); ok { + r0 = returnFunc(ctx, app, revision) + } else { + r0 = ret.Get(0).(domain.AppDesiredRevision) + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string, string) error); ok { + r1 = returnFunc(ctx, app, revision) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockAppState_LoadRevision_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'LoadRevision' +type MockAppState_LoadRevision_Call struct { + *mock.Call +} + +// LoadRevision is a helper method to define mock.On call +// - ctx context.Context +// - app string +// - revision string +func (_e *MockAppState_Expecter) LoadRevision(ctx any, app any, revision any) *MockAppState_LoadRevision_Call { + return &MockAppState_LoadRevision_Call{Call: _e.mock.On("LoadRevision", ctx, app, revision)} +} + +func (_c *MockAppState_LoadRevision_Call) Run(run func(ctx context.Context, app string, revision string)) *MockAppState_LoadRevision_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } + run( + arg0, + arg1, + arg2, + ) + }) + return _c +} + +func (_c *MockAppState_LoadRevision_Call) Return(appDesiredRevision domain.AppDesiredRevision, err error) *MockAppState_LoadRevision_Call { + _c.Call.Return(appDesiredRevision, err) + return _c +} + +func (_c *MockAppState_LoadRevision_Call) RunAndReturn(run func(ctx context.Context, app string, revision string) (domain.AppDesiredRevision, error)) *MockAppState_LoadRevision_Call { + _c.Call.Return(run) + return _c +} + +// Recover provides a mock function for the type MockAppState +func (_mock *MockAppState) Recover(ctx context.Context) error { + ret := _mock.Called(ctx) + + if len(ret) == 0 { + panic("no return value specified for Recover") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context) error); ok { + r0 = returnFunc(ctx) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockAppState_Recover_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Recover' +type MockAppState_Recover_Call struct { + *mock.Call +} + +// Recover is a helper method to define mock.On call +// - ctx context.Context +func (_e *MockAppState_Expecter) Recover(ctx any) *MockAppState_Recover_Call { + return &MockAppState_Recover_Call{Call: _e.mock.On("Recover", ctx)} +} + +func (_c *MockAppState_Recover_Call) Run(run func(ctx context.Context)) *MockAppState_Recover_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + run( + arg0, + ) + }) + return _c +} + +func (_c *MockAppState_Recover_Call) Return(err error) *MockAppState_Recover_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockAppState_Recover_Call) RunAndReturn(run func(ctx context.Context) error) *MockAppState_Recover_Call { + _c.Call.Return(run) + return _c +} + +// RegisterBackendBinds provides a mock function for the type MockAppState +func (_mock *MockAppState) RegisterBackendBinds(ctx context.Context, binds []domain.AppListenerReservation) error { + ret := _mock.Called(ctx, binds) + + if len(ret) == 0 { + panic("no return value specified for RegisterBackendBinds") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context, []domain.AppListenerReservation) error); ok { + r0 = returnFunc(ctx, binds) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockAppState_RegisterBackendBinds_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'RegisterBackendBinds' +type MockAppState_RegisterBackendBinds_Call struct { + *mock.Call +} + +// RegisterBackendBinds is a helper method to define mock.On call +// - ctx context.Context +// - binds []domain.AppListenerReservation +func (_e *MockAppState_Expecter) RegisterBackendBinds(ctx any, binds any) *MockAppState_RegisterBackendBinds_Call { + return &MockAppState_RegisterBackendBinds_Call{Call: _e.mock.On("RegisterBackendBinds", ctx, binds)} +} + +func (_c *MockAppState_RegisterBackendBinds_Call) Run(run func(ctx context.Context, binds []domain.AppListenerReservation)) *MockAppState_RegisterBackendBinds_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 []domain.AppListenerReservation + if args[1] != nil { + arg1 = args[1].([]domain.AppListenerReservation) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockAppState_RegisterBackendBinds_Call) Return(err error) *MockAppState_RegisterBackendBinds_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockAppState_RegisterBackendBinds_Call) RunAndReturn(run func(ctx context.Context, binds []domain.AppListenerReservation) error) *MockAppState_RegisterBackendBinds_Call { + _c.Call.Return(run) + return _c +} + +// ReleaseBackendBinds provides a mock function for the type MockAppState +func (_mock *MockAppState) ReleaseBackendBinds(ctx context.Context, app string, containerID string) error { + ret := _mock.Called(ctx, app, containerID) + + if len(ret) == 0 { + panic("no return value specified for ReleaseBackendBinds") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) error); ok { + r0 = returnFunc(ctx, app, containerID) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockAppState_ReleaseBackendBinds_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ReleaseBackendBinds' +type MockAppState_ReleaseBackendBinds_Call struct { + *mock.Call +} + +// ReleaseBackendBinds is a helper method to define mock.On call +// - ctx context.Context +// - app string +// - containerID string +func (_e *MockAppState_Expecter) ReleaseBackendBinds(ctx any, app any, containerID any) *MockAppState_ReleaseBackendBinds_Call { + return &MockAppState_ReleaseBackendBinds_Call{Call: _e.mock.On("ReleaseBackendBinds", ctx, app, containerID)} +} + +func (_c *MockAppState_ReleaseBackendBinds_Call) Run(run func(ctx context.Context, app string, containerID string)) *MockAppState_ReleaseBackendBinds_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } + run( + arg0, + arg1, + arg2, + ) + }) + return _c +} + +func (_c *MockAppState_ReleaseBackendBinds_Call) Return(err error) *MockAppState_ReleaseBackendBinds_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockAppState_ReleaseBackendBinds_Call) RunAndReturn(run func(ctx context.Context, app string, containerID string) error) *MockAppState_ReleaseBackendBinds_Call { + _c.Call.Return(run) + return _c +} + +// RetireApp provides a mock function for the type MockAppState +func (_mock *MockAppState) RetireApp(ctx context.Context, app string) error { + ret := _mock.Called(ctx, app) + + if len(ret) == 0 { + panic("no return value specified for RetireApp") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string) error); ok { + r0 = returnFunc(ctx, app) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockAppState_RetireApp_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'RetireApp' +type MockAppState_RetireApp_Call struct { + *mock.Call +} + +// RetireApp is a helper method to define mock.On call +// - ctx context.Context +// - app string +func (_e *MockAppState_Expecter) RetireApp(ctx any, app any) *MockAppState_RetireApp_Call { + return &MockAppState_RetireApp_Call{Call: _e.mock.On("RetireApp", ctx, app)} +} + +func (_c *MockAppState_RetireApp_Call) Run(run func(ctx context.Context, app string)) *MockAppState_RetireApp_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockAppState_RetireApp_Call) Return(err error) *MockAppState_RetireApp_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockAppState_RetireApp_Call) RunAndReturn(run func(ctx context.Context, app string) error) *MockAppState_RetireApp_Call { + _c.Call.Return(run) + return _c +} + +// SaveActive provides a mock function for the type MockAppState +func (_mock *MockAppState) SaveActive(ctx context.Context, active domain.AppActive) error { + ret := _mock.Called(ctx, active) + + if len(ret) == 0 { + panic("no return value specified for SaveActive") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context, domain.AppActive) error); ok { + r0 = returnFunc(ctx, active) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockAppState_SaveActive_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'SaveActive' +type MockAppState_SaveActive_Call struct { + *mock.Call +} + +// SaveActive is a helper method to define mock.On call +// - ctx context.Context +// - active domain.AppActive +func (_e *MockAppState_Expecter) SaveActive(ctx any, active any) *MockAppState_SaveActive_Call { + return &MockAppState_SaveActive_Call{Call: _e.mock.On("SaveActive", ctx, active)} +} + +func (_c *MockAppState_SaveActive_Call) Run(run func(ctx context.Context, active domain.AppActive)) *MockAppState_SaveActive_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 domain.AppActive + if args[1] != nil { + arg1 = args[1].(domain.AppActive) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockAppState_SaveActive_Call) Return(err error) *MockAppState_SaveActive_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockAppState_SaveActive_Call) RunAndReturn(run func(ctx context.Context, active domain.AppActive) error) *MockAppState_SaveActive_Call { + _c.Call.Return(run) + return _c +} + +// SaveIntent provides a mock function for the type MockAppState +func (_mock *MockAppState) SaveIntent(ctx context.Context, intent domain.AppStopIntent) error { + ret := _mock.Called(ctx, intent) + + if len(ret) == 0 { + panic("no return value specified for SaveIntent") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context, domain.AppStopIntent) error); ok { + r0 = returnFunc(ctx, intent) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockAppState_SaveIntent_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'SaveIntent' +type MockAppState_SaveIntent_Call struct { + *mock.Call +} + +// SaveIntent is a helper method to define mock.On call +// - ctx context.Context +// - intent domain.AppStopIntent +func (_e *MockAppState_Expecter) SaveIntent(ctx any, intent any) *MockAppState_SaveIntent_Call { + return &MockAppState_SaveIntent_Call{Call: _e.mock.On("SaveIntent", ctx, intent)} +} + +func (_c *MockAppState_SaveIntent_Call) Run(run func(ctx context.Context, intent domain.AppStopIntent)) *MockAppState_SaveIntent_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 domain.AppStopIntent + if args[1] != nil { + arg1 = args[1].(domain.AppStopIntent) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockAppState_SaveIntent_Call) Return(err error) *MockAppState_SaveIntent_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockAppState_SaveIntent_Call) RunAndReturn(run func(ctx context.Context, intent domain.AppStopIntent) error) *MockAppState_SaveIntent_Call { + _c.Call.Return(run) + return _c +} + +// SaveOperation provides a mock function for the type MockAppState +func (_mock *MockAppState) SaveOperation(ctx context.Context, op domain.AppOperation) error { + ret := _mock.Called(ctx, op) + + if len(ret) == 0 { + panic("no return value specified for SaveOperation") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context, domain.AppOperation) error); ok { + r0 = returnFunc(ctx, op) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockAppState_SaveOperation_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'SaveOperation' +type MockAppState_SaveOperation_Call struct { + *mock.Call +} + +// SaveOperation is a helper method to define mock.On call +// - ctx context.Context +// - op domain.AppOperation +func (_e *MockAppState_Expecter) SaveOperation(ctx any, op any) *MockAppState_SaveOperation_Call { + return &MockAppState_SaveOperation_Call{Call: _e.mock.On("SaveOperation", ctx, op)} +} + +func (_c *MockAppState_SaveOperation_Call) Run(run func(ctx context.Context, op domain.AppOperation)) *MockAppState_SaveOperation_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 domain.AppOperation + if args[1] != nil { + arg1 = args[1].(domain.AppOperation) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockAppState_SaveOperation_Call) Return(err error) *MockAppState_SaveOperation_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockAppState_SaveOperation_Call) RunAndReturn(run func(ctx context.Context, op domain.AppOperation) error) *MockAppState_SaveOperation_Call { + _c.Call.Return(run) + return _c +} + +// SaveOwnership provides a mock function for the type MockAppState +func (_mock *MockAppState) SaveOwnership(ctx context.Context, ownership domain.AppOwnership) error { + ret := _mock.Called(ctx, ownership) + + if len(ret) == 0 { + panic("no return value specified for SaveOwnership") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context, domain.AppOwnership) error); ok { + r0 = returnFunc(ctx, ownership) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockAppState_SaveOwnership_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'SaveOwnership' +type MockAppState_SaveOwnership_Call struct { + *mock.Call +} + +// SaveOwnership is a helper method to define mock.On call +// - ctx context.Context +// - ownership domain.AppOwnership +func (_e *MockAppState_Expecter) SaveOwnership(ctx any, ownership any) *MockAppState_SaveOwnership_Call { + return &MockAppState_SaveOwnership_Call{Call: _e.mock.On("SaveOwnership", ctx, ownership)} +} + +func (_c *MockAppState_SaveOwnership_Call) Run(run func(ctx context.Context, ownership domain.AppOwnership)) *MockAppState_SaveOwnership_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 domain.AppOwnership + if args[1] != nil { + arg1 = args[1].(domain.AppOwnership) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockAppState_SaveOwnership_Call) Return(err error) *MockAppState_SaveOwnership_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockAppState_SaveOwnership_Call) RunAndReturn(run func(ctx context.Context, ownership domain.AppOwnership) error) *MockAppState_SaveOwnership_Call { + _c.Call.Return(run) + return _c +} + +// SaveRecoveryInhibition provides a mock function for the type MockAppState +func (_mock *MockAppState) SaveRecoveryInhibition(ctx context.Context, inhibition domain.AppRecoveryInhibition) error { + ret := _mock.Called(ctx, inhibition) + + if len(ret) == 0 { + panic("no return value specified for SaveRecoveryInhibition") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context, domain.AppRecoveryInhibition) error); ok { + r0 = returnFunc(ctx, inhibition) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockAppState_SaveRecoveryInhibition_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'SaveRecoveryInhibition' +type MockAppState_SaveRecoveryInhibition_Call struct { + *mock.Call +} + +// SaveRecoveryInhibition is a helper method to define mock.On call +// - ctx context.Context +// - inhibition domain.AppRecoveryInhibition +func (_e *MockAppState_Expecter) SaveRecoveryInhibition(ctx any, inhibition any) *MockAppState_SaveRecoveryInhibition_Call { + return &MockAppState_SaveRecoveryInhibition_Call{Call: _e.mock.On("SaveRecoveryInhibition", ctx, inhibition)} +} + +func (_c *MockAppState_SaveRecoveryInhibition_Call) Run(run func(ctx context.Context, inhibition domain.AppRecoveryInhibition)) *MockAppState_SaveRecoveryInhibition_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 domain.AppRecoveryInhibition + if args[1] != nil { + arg1 = args[1].(domain.AppRecoveryInhibition) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockAppState_SaveRecoveryInhibition_Call) Return(err error) *MockAppState_SaveRecoveryInhibition_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockAppState_SaveRecoveryInhibition_Call) RunAndReturn(run func(ctx context.Context, inhibition domain.AppRecoveryInhibition) error) *MockAppState_SaveRecoveryInhibition_Call { + _c.Call.Return(run) + return _c +} diff --git a/internal/boundaries/out/mocks/mock_app_state_reader.go b/internal/boundaries/out/mocks/mock_app_state_reader.go new file mode 100644 index 000000000..a3557321f --- /dev/null +++ b/internal/boundaries/out/mocks/mock_app_state_reader.go @@ -0,0 +1,248 @@ +// Code generated by mockery; DO NOT EDIT. +// github.com/vektra/mockery +// template: testify + +package mocks + +import ( + "context" + + "github.com/bnema/gordon/internal/domain" + mock "github.com/stretchr/testify/mock" +) + +// NewMockAppStateReader creates a new instance of MockAppStateReader. It also registers a testing interface on the mock and a cleanup function to assert the mocks expectations. +// The first argument is typically a *testing.T value. +func NewMockAppStateReader(t interface { + mock.TestingT + Cleanup(func()) +}) *MockAppStateReader { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + + mock := &MockAppStateReader{} + mock.Mock.Test(t) + + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) + + return mock +} + +// MockAppStateReader is an autogenerated mock type for the AppStateReader type +type MockAppStateReader struct { + mock.Mock +} + +type MockAppStateReader_Expecter struct { + mock *mock.Mock +} + +func (_m *MockAppStateReader) EXPECT() *MockAppStateReader_Expecter { + return &MockAppStateReader_Expecter{mock: &_m.Mock} +} + +// ListApps provides a mock function for the type MockAppStateReader +func (_mock *MockAppStateReader) ListApps(ctx context.Context) ([]string, error) { + ret := _mock.Called(ctx) + + if len(ret) == 0 { + panic("no return value specified for ListApps") + } + + var r0 []string + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context) ([]string, error)); ok { + return returnFunc(ctx) + } + if returnFunc, ok := ret.Get(0).(func(context.Context) []string); ok { + r0 = returnFunc(ctx) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).([]string) + } + } + if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { + r1 = returnFunc(ctx) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockAppStateReader_ListApps_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListApps' +type MockAppStateReader_ListApps_Call struct { + *mock.Call +} + +// ListApps is a helper method to define mock.On call +// - ctx context.Context +func (_e *MockAppStateReader_Expecter) ListApps(ctx any) *MockAppStateReader_ListApps_Call { + return &MockAppStateReader_ListApps_Call{Call: _e.mock.On("ListApps", ctx)} +} + +func (_c *MockAppStateReader_ListApps_Call) Run(run func(ctx context.Context)) *MockAppStateReader_ListApps_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + run( + arg0, + ) + }) + return _c +} + +func (_c *MockAppStateReader_ListApps_Call) Return(strings []string, err error) *MockAppStateReader_ListApps_Call { + _c.Call.Return(strings, err) + return _c +} + +func (_c *MockAppStateReader_ListApps_Call) RunAndReturn(run func(ctx context.Context) ([]string, error)) *MockAppStateReader_ListApps_Call { + _c.Call.Return(run) + return _c +} + +// LoadActive provides a mock function for the type MockAppStateReader +func (_mock *MockAppStateReader) LoadActive(ctx context.Context, app string) (domain.AppActive, bool, error) { + ret := _mock.Called(ctx, app) + + if len(ret) == 0 { + panic("no return value specified for LoadActive") + } + + var r0 domain.AppActive + var r1 bool + var r2 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string) (domain.AppActive, bool, error)); ok { + return returnFunc(ctx, app) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string) domain.AppActive); ok { + r0 = returnFunc(ctx, app) + } else { + r0 = ret.Get(0).(domain.AppActive) + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string) bool); ok { + r1 = returnFunc(ctx, app) + } else { + r1 = ret.Get(1).(bool) + } + if returnFunc, ok := ret.Get(2).(func(context.Context, string) error); ok { + r2 = returnFunc(ctx, app) + } else { + r2 = ret.Error(2) + } + return r0, r1, r2 +} + +// MockAppStateReader_LoadActive_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'LoadActive' +type MockAppStateReader_LoadActive_Call struct { + *mock.Call +} + +// LoadActive is a helper method to define mock.On call +// - ctx context.Context +// - app string +func (_e *MockAppStateReader_Expecter) LoadActive(ctx any, app any) *MockAppStateReader_LoadActive_Call { + return &MockAppStateReader_LoadActive_Call{Call: _e.mock.On("LoadActive", ctx, app)} +} + +func (_c *MockAppStateReader_LoadActive_Call) Run(run func(ctx context.Context, app string)) *MockAppStateReader_LoadActive_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockAppStateReader_LoadActive_Call) Return(appActive domain.AppActive, b bool, err error) *MockAppStateReader_LoadActive_Call { + _c.Call.Return(appActive, b, err) + return _c +} + +func (_c *MockAppStateReader_LoadActive_Call) RunAndReturn(run func(ctx context.Context, app string) (domain.AppActive, bool, error)) *MockAppStateReader_LoadActive_Call { + _c.Call.Return(run) + return _c +} + +// LoadIntent provides a mock function for the type MockAppStateReader +func (_mock *MockAppStateReader) LoadIntent(ctx context.Context, app string) (domain.AppStopIntent, error) { + ret := _mock.Called(ctx, app) + + if len(ret) == 0 { + panic("no return value specified for LoadIntent") + } + + var r0 domain.AppStopIntent + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string) (domain.AppStopIntent, error)); ok { + return returnFunc(ctx, app) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string) domain.AppStopIntent); ok { + r0 = returnFunc(ctx, app) + } else { + r0 = ret.Get(0).(domain.AppStopIntent) + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { + r1 = returnFunc(ctx, app) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockAppStateReader_LoadIntent_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'LoadIntent' +type MockAppStateReader_LoadIntent_Call struct { + *mock.Call +} + +// LoadIntent is a helper method to define mock.On call +// - ctx context.Context +// - app string +func (_e *MockAppStateReader_Expecter) LoadIntent(ctx any, app any) *MockAppStateReader_LoadIntent_Call { + return &MockAppStateReader_LoadIntent_Call{Call: _e.mock.On("LoadIntent", ctx, app)} +} + +func (_c *MockAppStateReader_LoadIntent_Call) Run(run func(ctx context.Context, app string)) *MockAppStateReader_LoadIntent_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockAppStateReader_LoadIntent_Call) Return(appStopIntent domain.AppStopIntent, err error) *MockAppStateReader_LoadIntent_Call { + _c.Call.Return(appStopIntent, err) + return _c +} + +func (_c *MockAppStateReader_LoadIntent_Call) RunAndReturn(run func(ctx context.Context, app string) (domain.AppStopIntent, error)) *MockAppStateReader_LoadIntent_Call { + _c.Call.Return(run) + return _c +} diff --git a/internal/boundaries/out/mocks/mock_app_traffic_refresher.go b/internal/boundaries/out/mocks/mock_app_traffic_refresher.go new file mode 100644 index 000000000..8ff2dbc42 --- /dev/null +++ b/internal/boundaries/out/mocks/mock_app_traffic_refresher.go @@ -0,0 +1,224 @@ +// Code generated by mockery; DO NOT EDIT. +// github.com/vektra/mockery +// template: testify + +package mocks + +import ( + "context" + + mock "github.com/stretchr/testify/mock" +) + +// NewMockAppTrafficRefresher creates a new instance of MockAppTrafficRefresher. It also registers a testing interface on the mock and a cleanup function to assert the mocks expectations. +// The first argument is typically a *testing.T value. +func NewMockAppTrafficRefresher(t interface { + mock.TestingT + Cleanup(func()) +}) *MockAppTrafficRefresher { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + + mock := &MockAppTrafficRefresher{} + mock.Mock.Test(t) + + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) + + return mock +} + +// MockAppTrafficRefresher is an autogenerated mock type for the AppTrafficRefresher type +type MockAppTrafficRefresher struct { + mock.Mock +} + +type MockAppTrafficRefresher_Expecter struct { + mock *mock.Mock +} + +func (_m *MockAppTrafficRefresher) EXPECT() *MockAppTrafficRefresher_Expecter { + return &MockAppTrafficRefresher_Expecter{mock: &_m.Mock} +} + +// RebuildTraffic provides a mock function for the type MockAppTrafficRefresher +func (_mock *MockAppTrafficRefresher) RebuildTraffic(ctx context.Context) error { + ret := _mock.Called(ctx) + + if len(ret) == 0 { + panic("no return value specified for RebuildTraffic") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context) error); ok { + r0 = returnFunc(ctx) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockAppTrafficRefresher_RebuildTraffic_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'RebuildTraffic' +type MockAppTrafficRefresher_RebuildTraffic_Call struct { + *mock.Call +} + +// RebuildTraffic is a helper method to define mock.On call +// - ctx context.Context +func (_e *MockAppTrafficRefresher_Expecter) RebuildTraffic(ctx any) *MockAppTrafficRefresher_RebuildTraffic_Call { + return &MockAppTrafficRefresher_RebuildTraffic_Call{Call: _e.mock.On("RebuildTraffic", ctx)} +} + +func (_c *MockAppTrafficRefresher_RebuildTraffic_Call) Run(run func(ctx context.Context)) *MockAppTrafficRefresher_RebuildTraffic_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + run( + arg0, + ) + }) + return _c +} + +func (_c *MockAppTrafficRefresher_RebuildTraffic_Call) Return(err error) *MockAppTrafficRefresher_RebuildTraffic_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockAppTrafficRefresher_RebuildTraffic_Call) RunAndReturn(run func(ctx context.Context) error) *MockAppTrafficRefresher_RebuildTraffic_Call { + _c.Call.Return(run) + return _c +} + +// WithdrawService provides a mock function for the type MockAppTrafficRefresher +func (_mock *MockAppTrafficRefresher) WithdrawService(ctx context.Context, app string, service string) error { + ret := _mock.Called(ctx, app, service) + + if len(ret) == 0 { + panic("no return value specified for WithdrawService") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) error); ok { + r0 = returnFunc(ctx, app, service) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockAppTrafficRefresher_WithdrawService_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'WithdrawService' +type MockAppTrafficRefresher_WithdrawService_Call struct { + *mock.Call +} + +// WithdrawService is a helper method to define mock.On call +// - ctx context.Context +// - app string +// - service string +func (_e *MockAppTrafficRefresher_Expecter) WithdrawService(ctx any, app any, service any) *MockAppTrafficRefresher_WithdrawService_Call { + return &MockAppTrafficRefresher_WithdrawService_Call{Call: _e.mock.On("WithdrawService", ctx, app, service)} +} + +func (_c *MockAppTrafficRefresher_WithdrawService_Call) Run(run func(ctx context.Context, app string, service string)) *MockAppTrafficRefresher_WithdrawService_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } + run( + arg0, + arg1, + arg2, + ) + }) + return _c +} + +func (_c *MockAppTrafficRefresher_WithdrawService_Call) Return(err error) *MockAppTrafficRefresher_WithdrawService_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockAppTrafficRefresher_WithdrawService_Call) RunAndReturn(run func(ctx context.Context, app string, service string) error) *MockAppTrafficRefresher_WithdrawService_Call { + _c.Call.Return(run) + return _c +} + +// WithdrawServiceState provides a mock function for the type MockAppTrafficRefresher +func (_mock *MockAppTrafficRefresher) WithdrawServiceState(ctx context.Context, app string, service string) error { + ret := _mock.Called(ctx, app, service) + + if len(ret) == 0 { + panic("no return value specified for WithdrawServiceState") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) error); ok { + r0 = returnFunc(ctx, app, service) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockAppTrafficRefresher_WithdrawServiceState_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'WithdrawServiceState' +type MockAppTrafficRefresher_WithdrawServiceState_Call struct { + *mock.Call +} + +// WithdrawServiceState is a helper method to define mock.On call +// - ctx context.Context +// - app string +// - service string +func (_e *MockAppTrafficRefresher_Expecter) WithdrawServiceState(ctx any, app any, service any) *MockAppTrafficRefresher_WithdrawServiceState_Call { + return &MockAppTrafficRefresher_WithdrawServiceState_Call{Call: _e.mock.On("WithdrawServiceState", ctx, app, service)} +} + +func (_c *MockAppTrafficRefresher_WithdrawServiceState_Call) Run(run func(ctx context.Context, app string, service string)) *MockAppTrafficRefresher_WithdrawServiceState_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } + run( + arg0, + arg1, + arg2, + ) + }) + return _c +} + +func (_c *MockAppTrafficRefresher_WithdrawServiceState_Call) Return(err error) *MockAppTrafficRefresher_WithdrawServiceState_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockAppTrafficRefresher_WithdrawServiceState_Call) RunAndReturn(run func(ctx context.Context, app string, service string) error) *MockAppTrafficRefresher_WithdrawServiceState_Call { + _c.Call.Return(run) + return _c +} diff --git a/internal/boundaries/out/mocks/mock_attachment_config_provider.go b/internal/boundaries/out/mocks/mock_attachment_config_provider.go deleted file mode 100644 index d37c2ad80..000000000 --- a/internal/boundaries/out/mocks/mock_attachment_config_provider.go +++ /dev/null @@ -1,127 +0,0 @@ -// Code generated by mockery; DO NOT EDIT. -// github.com/vektra/mockery -// template: testify - -package mocks - -import ( - "github.com/bnema/gordon/internal/boundaries/out" - mock "github.com/stretchr/testify/mock" -) - -// NewMockAttachmentConfigProvider creates a new instance of MockAttachmentConfigProvider. It also registers a testing interface on the mock and a cleanup function to assert the mocks expectations. -// The first argument is typically a *testing.T value. -func NewMockAttachmentConfigProvider(t interface { - mock.TestingT - Cleanup(func()) -}) *MockAttachmentConfigProvider { - mock := &MockAttachmentConfigProvider{} - mock.Mock.Test(t) - - t.Cleanup(func() { mock.AssertExpectations(t) }) - - return mock -} - -// MockAttachmentConfigProvider is an autogenerated mock type for the AttachmentConfigProvider type -type MockAttachmentConfigProvider struct { - mock.Mock -} - -type MockAttachmentConfigProvider_Expecter struct { - mock *mock.Mock -} - -func (_m *MockAttachmentConfigProvider) EXPECT() *MockAttachmentConfigProvider_Expecter { - return &MockAttachmentConfigProvider_Expecter{mock: &_m.Mock} -} - -// GetAttachmentConfig provides a mock function for the type MockAttachmentConfigProvider -func (_mock *MockAttachmentConfigProvider) GetAttachmentConfig() out.AttachmentConfigSnapshot { - ret := _mock.Called() - - if len(ret) == 0 { - panic("no return value specified for GetAttachmentConfig") - } - - var r0 out.AttachmentConfigSnapshot - if returnFunc, ok := ret.Get(0).(func() out.AttachmentConfigSnapshot); ok { - r0 = returnFunc() - } else { - r0 = ret.Get(0).(out.AttachmentConfigSnapshot) - } - return r0 -} - -// MockAttachmentConfigProvider_GetAttachmentConfig_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetAttachmentConfig' -type MockAttachmentConfigProvider_GetAttachmentConfig_Call struct { - *mock.Call -} - -// GetAttachmentConfig is a helper method to define mock.On call -func (_e *MockAttachmentConfigProvider_Expecter) GetAttachmentConfig() *MockAttachmentConfigProvider_GetAttachmentConfig_Call { - return &MockAttachmentConfigProvider_GetAttachmentConfig_Call{Call: _e.mock.On("GetAttachmentConfig")} -} - -func (_c *MockAttachmentConfigProvider_GetAttachmentConfig_Call) Run(run func()) *MockAttachmentConfigProvider_GetAttachmentConfig_Call { - _c.Call.Run(func(args mock.Arguments) { - run() - }) - return _c -} - -func (_c *MockAttachmentConfigProvider_GetAttachmentConfig_Call) Return(attachmentConfigSnapshot out.AttachmentConfigSnapshot) *MockAttachmentConfigProvider_GetAttachmentConfig_Call { - _c.Call.Return(attachmentConfigSnapshot) - return _c -} - -func (_c *MockAttachmentConfigProvider_GetAttachmentConfig_Call) RunAndReturn(run func() out.AttachmentConfigSnapshot) *MockAttachmentConfigProvider_GetAttachmentConfig_Call { - _c.Call.Return(run) - return _c -} - -// GetNetworkGroups provides a mock function for the type MockAttachmentConfigProvider -func (_mock *MockAttachmentConfigProvider) GetNetworkGroups() map[string][]string { - ret := _mock.Called() - - if len(ret) == 0 { - panic("no return value specified for GetNetworkGroups") - } - - var r0 map[string][]string - if returnFunc, ok := ret.Get(0).(func() map[string][]string); ok { - r0 = returnFunc() - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).(map[string][]string) - } - } - return r0 -} - -// MockAttachmentConfigProvider_GetNetworkGroups_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetNetworkGroups' -type MockAttachmentConfigProvider_GetNetworkGroups_Call struct { - *mock.Call -} - -// GetNetworkGroups is a helper method to define mock.On call -func (_e *MockAttachmentConfigProvider_Expecter) GetNetworkGroups() *MockAttachmentConfigProvider_GetNetworkGroups_Call { - return &MockAttachmentConfigProvider_GetNetworkGroups_Call{Call: _e.mock.On("GetNetworkGroups")} -} - -func (_c *MockAttachmentConfigProvider_GetNetworkGroups_Call) Run(run func()) *MockAttachmentConfigProvider_GetNetworkGroups_Call { - _c.Call.Run(func(args mock.Arguments) { - run() - }) - return _c -} - -func (_c *MockAttachmentConfigProvider_GetNetworkGroups_Call) Return(stringToStrings map[string][]string) *MockAttachmentConfigProvider_GetNetworkGroups_Call { - _c.Call.Return(stringToStrings) - return _c -} - -func (_c *MockAttachmentConfigProvider_GetNetworkGroups_Call) RunAndReturn(run func() map[string][]string) *MockAttachmentConfigProvider_GetNetworkGroups_Call { - _c.Call.Return(run) - return _c -} diff --git a/internal/boundaries/out/mocks/mock_backup_storage.go b/internal/boundaries/out/mocks/mock_backup_storage.go index 2da44f361..c0881eb03 100644 --- a/internal/boundaries/out/mocks/mock_backup_storage.go +++ b/internal/boundaries/out/mocks/mock_backup_storage.go @@ -19,10 +19,19 @@ func NewMockBackupStorage(t interface { mock.TestingT Cleanup(func()) }) *MockBackupStorage { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockBackupStorage{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } @@ -41,8 +50,8 @@ func (_m *MockBackupStorage) EXPECT() *MockBackupStorage_Expecter { } // ApplyRetention provides a mock function for the type MockBackupStorage -func (_mock *MockBackupStorage) ApplyRetention(ctx context.Context, domainName string, policy domain.DatabaseBackupRetentionPolicy) (int, error) { - ret := _mock.Called(ctx, domainName, policy) +func (_mock *MockBackupStorage) ApplyRetention(ctx context.Context, app string, policy domain.DatabaseBackupRetentionPolicy) (int, error) { + ret := _mock.Called(ctx, app, policy) if len(ret) == 0 { panic("no return value specified for ApplyRetention") @@ -51,15 +60,15 @@ func (_mock *MockBackupStorage) ApplyRetention(ctx context.Context, domainName s var r0 int var r1 error if returnFunc, ok := ret.Get(0).(func(context.Context, string, domain.DatabaseBackupRetentionPolicy) (int, error)); ok { - return returnFunc(ctx, domainName, policy) + return returnFunc(ctx, app, policy) } if returnFunc, ok := ret.Get(0).(func(context.Context, string, domain.DatabaseBackupRetentionPolicy) int); ok { - r0 = returnFunc(ctx, domainName, policy) + r0 = returnFunc(ctx, app, policy) } else { r0 = ret.Get(0).(int) } if returnFunc, ok := ret.Get(1).(func(context.Context, string, domain.DatabaseBackupRetentionPolicy) error); ok { - r1 = returnFunc(ctx, domainName, policy) + r1 = returnFunc(ctx, app, policy) } else { r1 = ret.Error(1) } @@ -73,13 +82,13 @@ type MockBackupStorage_ApplyRetention_Call struct { // ApplyRetention is a helper method to define mock.On call // - ctx context.Context -// - domainName string +// - app string // - policy domain.DatabaseBackupRetentionPolicy -func (_e *MockBackupStorage_Expecter) ApplyRetention(ctx any, domainName any, policy any) *MockBackupStorage_ApplyRetention_Call { - return &MockBackupStorage_ApplyRetention_Call{Call: _e.mock.On("ApplyRetention", ctx, domainName, policy)} +func (_e *MockBackupStorage_Expecter) ApplyRetention(ctx any, app any, policy any) *MockBackupStorage_ApplyRetention_Call { + return &MockBackupStorage_ApplyRetention_Call{Call: _e.mock.On("ApplyRetention", ctx, app, policy)} } -func (_c *MockBackupStorage_ApplyRetention_Call) Run(run func(ctx context.Context, domainName string, policy domain.DatabaseBackupRetentionPolicy)) *MockBackupStorage_ApplyRetention_Call { +func (_c *MockBackupStorage_ApplyRetention_Call) Run(run func(ctx context.Context, app string, policy domain.DatabaseBackupRetentionPolicy)) *MockBackupStorage_ApplyRetention_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -107,7 +116,7 @@ func (_c *MockBackupStorage_ApplyRetention_Call) Return(n int, err error) *MockB return _c } -func (_c *MockBackupStorage_ApplyRetention_Call) RunAndReturn(run func(ctx context.Context, domainName string, policy domain.DatabaseBackupRetentionPolicy) (int, error)) *MockBackupStorage_ApplyRetention_Call { +func (_c *MockBackupStorage_ApplyRetention_Call) RunAndReturn(run func(ctx context.Context, app string, policy domain.DatabaseBackupRetentionPolicy) (int, error)) *MockBackupStorage_ApplyRetention_Call { _c.Call.Return(run) return _c } @@ -238,8 +247,8 @@ func (_c *MockBackupStorage_Get_Call) RunAndReturn(run func(ctx context.Context, } // List provides a mock function for the type MockBackupStorage -func (_mock *MockBackupStorage) List(ctx context.Context, domainName string, schedule *domain.BackupSchedule) ([]domain.DatabaseBackupJob, error) { - ret := _mock.Called(ctx, domainName, schedule) +func (_mock *MockBackupStorage) List(ctx context.Context, app string, schedule *domain.BackupSchedule) ([]domain.DatabaseBackupJob, error) { + ret := _mock.Called(ctx, app, schedule) if len(ret) == 0 { panic("no return value specified for List") @@ -248,17 +257,17 @@ func (_mock *MockBackupStorage) List(ctx context.Context, domainName string, sch var r0 []domain.DatabaseBackupJob var r1 error if returnFunc, ok := ret.Get(0).(func(context.Context, string, *domain.BackupSchedule) ([]domain.DatabaseBackupJob, error)); ok { - return returnFunc(ctx, domainName, schedule) + return returnFunc(ctx, app, schedule) } if returnFunc, ok := ret.Get(0).(func(context.Context, string, *domain.BackupSchedule) []domain.DatabaseBackupJob); ok { - r0 = returnFunc(ctx, domainName, schedule) + r0 = returnFunc(ctx, app, schedule) } else { if ret.Get(0) != nil { r0 = ret.Get(0).([]domain.DatabaseBackupJob) } } if returnFunc, ok := ret.Get(1).(func(context.Context, string, *domain.BackupSchedule) error); ok { - r1 = returnFunc(ctx, domainName, schedule) + r1 = returnFunc(ctx, app, schedule) } else { r1 = ret.Error(1) } @@ -272,13 +281,13 @@ type MockBackupStorage_List_Call struct { // List is a helper method to define mock.On call // - ctx context.Context -// - domainName string +// - app string // - schedule *domain.BackupSchedule -func (_e *MockBackupStorage_Expecter) List(ctx any, domainName any, schedule any) *MockBackupStorage_List_Call { - return &MockBackupStorage_List_Call{Call: _e.mock.On("List", ctx, domainName, schedule)} +func (_e *MockBackupStorage_Expecter) List(ctx any, app any, schedule any) *MockBackupStorage_List_Call { + return &MockBackupStorage_List_Call{Call: _e.mock.On("List", ctx, app, schedule)} } -func (_c *MockBackupStorage_List_Call) Run(run func(ctx context.Context, domainName string, schedule *domain.BackupSchedule)) *MockBackupStorage_List_Call { +func (_c *MockBackupStorage_List_Call) Run(run func(ctx context.Context, app string, schedule *domain.BackupSchedule)) *MockBackupStorage_List_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -301,19 +310,19 @@ func (_c *MockBackupStorage_List_Call) Run(run func(ctx context.Context, domainN return _c } -func (_c *MockBackupStorage_List_Call) Return(vs []domain.DatabaseBackupJob, err error) *MockBackupStorage_List_Call { - _c.Call.Return(vs, err) +func (_c *MockBackupStorage_List_Call) Return(databaseBackupJobs []domain.DatabaseBackupJob, err error) *MockBackupStorage_List_Call { + _c.Call.Return(databaseBackupJobs, err) return _c } -func (_c *MockBackupStorage_List_Call) RunAndReturn(run func(ctx context.Context, domainName string, schedule *domain.BackupSchedule) ([]domain.DatabaseBackupJob, error)) *MockBackupStorage_List_Call { +func (_c *MockBackupStorage_List_Call) RunAndReturn(run func(ctx context.Context, app string, schedule *domain.BackupSchedule) ([]domain.DatabaseBackupJob, error)) *MockBackupStorage_List_Call { _c.Call.Return(run) return _c } // Store provides a mock function for the type MockBackupStorage -func (_mock *MockBackupStorage) Store(ctx context.Context, domainName string, dbName string, schedule domain.BackupSchedule, timestamp time.Time, data io.Reader) (string, error) { - ret := _mock.Called(ctx, domainName, dbName, schedule, timestamp, data) +func (_mock *MockBackupStorage) Store(ctx context.Context, app string, service string, database string, schedule domain.BackupSchedule, timestamp time.Time, data io.Reader) (string, error) { + ret := _mock.Called(ctx, app, service, database, schedule, timestamp, data) if len(ret) == 0 { panic("no return value specified for Store") @@ -321,16 +330,16 @@ func (_mock *MockBackupStorage) Store(ctx context.Context, domainName string, db var r0 string var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, domain.BackupSchedule, time.Time, io.Reader) (string, error)); ok { - return returnFunc(ctx, domainName, dbName, schedule, timestamp, data) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, string, domain.BackupSchedule, time.Time, io.Reader) (string, error)); ok { + return returnFunc(ctx, app, service, database, schedule, timestamp, data) } - if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, domain.BackupSchedule, time.Time, io.Reader) string); ok { - r0 = returnFunc(ctx, domainName, dbName, schedule, timestamp, data) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string, string, domain.BackupSchedule, time.Time, io.Reader) string); ok { + r0 = returnFunc(ctx, app, service, database, schedule, timestamp, data) } else { r0 = ret.Get(0).(string) } - if returnFunc, ok := ret.Get(1).(func(context.Context, string, string, domain.BackupSchedule, time.Time, io.Reader) error); ok { - r1 = returnFunc(ctx, domainName, dbName, schedule, timestamp, data) + if returnFunc, ok := ret.Get(1).(func(context.Context, string, string, string, domain.BackupSchedule, time.Time, io.Reader) error); ok { + r1 = returnFunc(ctx, app, service, database, schedule, timestamp, data) } else { r1 = ret.Error(1) } @@ -344,16 +353,17 @@ type MockBackupStorage_Store_Call struct { // Store is a helper method to define mock.On call // - ctx context.Context -// - domainName string -// - dbName string +// - app string +// - service string +// - database string // - schedule domain.BackupSchedule // - timestamp time.Time // - data io.Reader -func (_e *MockBackupStorage_Expecter) Store(ctx any, domainName any, dbName any, schedule any, timestamp any, data any) *MockBackupStorage_Store_Call { - return &MockBackupStorage_Store_Call{Call: _e.mock.On("Store", ctx, domainName, dbName, schedule, timestamp, data)} +func (_e *MockBackupStorage_Expecter) Store(ctx any, app any, service any, database any, schedule any, timestamp any, data any) *MockBackupStorage_Store_Call { + return &MockBackupStorage_Store_Call{Call: _e.mock.On("Store", ctx, app, service, database, schedule, timestamp, data)} } -func (_c *MockBackupStorage_Store_Call) Run(run func(ctx context.Context, domainName string, dbName string, schedule domain.BackupSchedule, timestamp time.Time, data io.Reader)) *MockBackupStorage_Store_Call { +func (_c *MockBackupStorage_Store_Call) Run(run func(ctx context.Context, app string, service string, database string, schedule domain.BackupSchedule, timestamp time.Time, data io.Reader)) *MockBackupStorage_Store_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -367,17 +377,21 @@ func (_c *MockBackupStorage_Store_Call) Run(run func(ctx context.Context, domain if args[2] != nil { arg2 = args[2].(string) } - var arg3 domain.BackupSchedule + var arg3 string if args[3] != nil { - arg3 = args[3].(domain.BackupSchedule) + arg3 = args[3].(string) } - var arg4 time.Time + var arg4 domain.BackupSchedule if args[4] != nil { - arg4 = args[4].(time.Time) + arg4 = args[4].(domain.BackupSchedule) } - var arg5 io.Reader + var arg5 time.Time if args[5] != nil { - arg5 = args[5].(io.Reader) + arg5 = args[5].(time.Time) + } + var arg6 io.Reader + if args[6] != nil { + arg6 = args[6].(io.Reader) } run( arg0, @@ -386,6 +400,7 @@ func (_c *MockBackupStorage_Store_Call) Run(run func(ctx context.Context, domain arg3, arg4, arg5, + arg6, ) }) return _c @@ -396,7 +411,7 @@ func (_c *MockBackupStorage_Store_Call) Return(s string, err error) *MockBackupS return _c } -func (_c *MockBackupStorage_Store_Call) RunAndReturn(run func(ctx context.Context, domainName string, dbName string, schedule domain.BackupSchedule, timestamp time.Time, data io.Reader) (string, error)) *MockBackupStorage_Store_Call { +func (_c *MockBackupStorage_Store_Call) RunAndReturn(run func(ctx context.Context, app string, service string, database string, schedule domain.BackupSchedule, timestamp time.Time, data io.Reader) (string, error)) *MockBackupStorage_Store_Call { _c.Call.Return(run) return _c } diff --git a/internal/boundaries/out/mocks/mock_blob_storage.go b/internal/boundaries/out/mocks/mock_blob_storage.go index 0a6058bd1..b7b8bca19 100644 --- a/internal/boundaries/out/mocks/mock_blob_storage.go +++ b/internal/boundaries/out/mocks/mock_blob_storage.go @@ -17,10 +17,19 @@ func NewMockBlobStorage(t interface { mock.TestingT Cleanup(func()) }) *MockBlobStorage { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockBlobStorage{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } @@ -173,17 +182,83 @@ func (_c *MockBlobStorage_BlobExists_Call) RunAndReturn(run func(digest string) return _c } +// BlobOwnedByRepository provides a mock function for the type MockBlobStorage +func (_mock *MockBlobStorage) BlobOwnedByRepository(name string, digest string) (bool, error) { + ret := _mock.Called(name, digest) + + if len(ret) == 0 { + panic("no return value specified for BlobOwnedByRepository") + } + + var r0 bool + var r1 error + if returnFunc, ok := ret.Get(0).(func(string, string) (bool, error)); ok { + return returnFunc(name, digest) + } + if returnFunc, ok := ret.Get(0).(func(string, string) bool); ok { + r0 = returnFunc(name, digest) + } else { + r0 = ret.Get(0).(bool) + } + if returnFunc, ok := ret.Get(1).(func(string, string) error); ok { + r1 = returnFunc(name, digest) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockBlobStorage_BlobOwnedByRepository_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'BlobOwnedByRepository' +type MockBlobStorage_BlobOwnedByRepository_Call struct { + *mock.Call +} + +// BlobOwnedByRepository is a helper method to define mock.On call +// - name string +// - digest string +func (_e *MockBlobStorage_Expecter) BlobOwnedByRepository(name any, digest any) *MockBlobStorage_BlobOwnedByRepository_Call { + return &MockBlobStorage_BlobOwnedByRepository_Call{Call: _e.mock.On("BlobOwnedByRepository", name, digest)} +} + +func (_c *MockBlobStorage_BlobOwnedByRepository_Call) Run(run func(name string, digest string)) *MockBlobStorage_BlobOwnedByRepository_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 string + if args[0] != nil { + arg0 = args[0].(string) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockBlobStorage_BlobOwnedByRepository_Call) Return(b bool, err error) *MockBlobStorage_BlobOwnedByRepository_Call { + _c.Call.Return(b, err) + return _c +} + +func (_c *MockBlobStorage_BlobOwnedByRepository_Call) RunAndReturn(run func(name string, digest string) (bool, error)) *MockBlobStorage_BlobOwnedByRepository_Call { + _c.Call.Return(run) + return _c +} + // CancelBlobUpload provides a mock function for the type MockBlobStorage -func (_mock *MockBlobStorage) CancelBlobUpload(uuid string) error { - ret := _mock.Called(uuid) +func (_mock *MockBlobStorage) CancelBlobUpload(name string, uuid string) error { + ret := _mock.Called(name, uuid) if len(ret) == 0 { panic("no return value specified for CancelBlobUpload") } var r0 error - if returnFunc, ok := ret.Get(0).(func(string) error); ok { - r0 = returnFunc(uuid) + if returnFunc, ok := ret.Get(0).(func(string, string) error); ok { + r0 = returnFunc(name, uuid) } else { r0 = ret.Error(0) } @@ -196,19 +271,25 @@ type MockBlobStorage_CancelBlobUpload_Call struct { } // CancelBlobUpload is a helper method to define mock.On call +// - name string // - uuid string -func (_e *MockBlobStorage_Expecter) CancelBlobUpload(uuid any) *MockBlobStorage_CancelBlobUpload_Call { - return &MockBlobStorage_CancelBlobUpload_Call{Call: _e.mock.On("CancelBlobUpload", uuid)} +func (_e *MockBlobStorage_Expecter) CancelBlobUpload(name any, uuid any) *MockBlobStorage_CancelBlobUpload_Call { + return &MockBlobStorage_CancelBlobUpload_Call{Call: _e.mock.On("CancelBlobUpload", name, uuid)} } -func (_c *MockBlobStorage_CancelBlobUpload_Call) Run(run func(uuid string)) *MockBlobStorage_CancelBlobUpload_Call { +func (_c *MockBlobStorage_CancelBlobUpload_Call) Run(run func(name string, uuid string)) *MockBlobStorage_CancelBlobUpload_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 string if args[0] != nil { arg0 = args[0].(string) } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } run( arg0, + arg1, ) }) return _c @@ -219,7 +300,7 @@ func (_c *MockBlobStorage_CancelBlobUpload_Call) Return(err error) *MockBlobStor return _c } -func (_c *MockBlobStorage_CancelBlobUpload_Call) RunAndReturn(run func(uuid string) error) *MockBlobStorage_CancelBlobUpload_Call { +func (_c *MockBlobStorage_CancelBlobUpload_Call) RunAndReturn(run func(name string, uuid string) error) *MockBlobStorage_CancelBlobUpload_Call { _c.Call.Return(run) return _c } @@ -351,16 +432,16 @@ func (_c *MockBlobStorage_DeleteBlob_Call) RunAndReturn(run func(digest string) } // FinishBlobUpload provides a mock function for the type MockBlobStorage -func (_mock *MockBlobStorage) FinishBlobUpload(uuid string, digest string) error { - ret := _mock.Called(uuid, digest) +func (_mock *MockBlobStorage) FinishBlobUpload(name string, uuid string, digest string) error { + ret := _mock.Called(name, uuid, digest) if len(ret) == 0 { panic("no return value specified for FinishBlobUpload") } var r0 error - if returnFunc, ok := ret.Get(0).(func(string, string) error); ok { - r0 = returnFunc(uuid, digest) + if returnFunc, ok := ret.Get(0).(func(string, string, string) error); ok { + r0 = returnFunc(name, uuid, digest) } else { r0 = ret.Error(0) } @@ -373,13 +454,14 @@ type MockBlobStorage_FinishBlobUpload_Call struct { } // FinishBlobUpload is a helper method to define mock.On call +// - name string // - uuid string // - digest string -func (_e *MockBlobStorage_Expecter) FinishBlobUpload(uuid any, digest any) *MockBlobStorage_FinishBlobUpload_Call { - return &MockBlobStorage_FinishBlobUpload_Call{Call: _e.mock.On("FinishBlobUpload", uuid, digest)} +func (_e *MockBlobStorage_Expecter) FinishBlobUpload(name any, uuid any, digest any) *MockBlobStorage_FinishBlobUpload_Call { + return &MockBlobStorage_FinishBlobUpload_Call{Call: _e.mock.On("FinishBlobUpload", name, uuid, digest)} } -func (_c *MockBlobStorage_FinishBlobUpload_Call) Run(run func(uuid string, digest string)) *MockBlobStorage_FinishBlobUpload_Call { +func (_c *MockBlobStorage_FinishBlobUpload_Call) Run(run func(name string, uuid string, digest string)) *MockBlobStorage_FinishBlobUpload_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 string if args[0] != nil { @@ -389,9 +471,14 @@ func (_c *MockBlobStorage_FinishBlobUpload_Call) Run(run func(uuid string, diges if args[1] != nil { arg1 = args[1].(string) } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } run( arg0, arg1, + arg2, ) }) return _c @@ -402,7 +489,7 @@ func (_c *MockBlobStorage_FinishBlobUpload_Call) Return(err error) *MockBlobStor return _c } -func (_c *MockBlobStorage_FinishBlobUpload_Call) RunAndReturn(run func(uuid string, digest string) error) *MockBlobStorage_FinishBlobUpload_Call { +func (_c *MockBlobStorage_FinishBlobUpload_Call) RunAndReturn(run func(name string, uuid string, digest string) error) *MockBlobStorage_FinishBlobUpload_Call { _c.Call.Return(run) return _c } @@ -469,6 +556,66 @@ func (_c *MockBlobStorage_GetBlob_Call) RunAndReturn(run func(digest string) (io return _c } +// GetBlobModTime provides a mock function for the type MockBlobStorage +func (_mock *MockBlobStorage) GetBlobModTime(digest string) (time.Time, error) { + ret := _mock.Called(digest) + + if len(ret) == 0 { + panic("no return value specified for GetBlobModTime") + } + + var r0 time.Time + var r1 error + if returnFunc, ok := ret.Get(0).(func(string) (time.Time, error)); ok { + return returnFunc(digest) + } + if returnFunc, ok := ret.Get(0).(func(string) time.Time); ok { + r0 = returnFunc(digest) + } else { + r0 = ret.Get(0).(time.Time) + } + if returnFunc, ok := ret.Get(1).(func(string) error); ok { + r1 = returnFunc(digest) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockBlobStorage_GetBlobModTime_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetBlobModTime' +type MockBlobStorage_GetBlobModTime_Call struct { + *mock.Call +} + +// GetBlobModTime is a helper method to define mock.On call +// - digest string +func (_e *MockBlobStorage_Expecter) GetBlobModTime(digest any) *MockBlobStorage_GetBlobModTime_Call { + return &MockBlobStorage_GetBlobModTime_Call{Call: _e.mock.On("GetBlobModTime", digest)} +} + +func (_c *MockBlobStorage_GetBlobModTime_Call) Run(run func(digest string)) *MockBlobStorage_GetBlobModTime_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 string + if args[0] != nil { + arg0 = args[0].(string) + } + run( + arg0, + ) + }) + return _c +} + +func (_c *MockBlobStorage_GetBlobModTime_Call) Return(time1 time.Time, err error) *MockBlobStorage_GetBlobModTime_Call { + _c.Call.Return(time1, err) + return _c +} + +func (_c *MockBlobStorage_GetBlobModTime_Call) RunAndReturn(run func(digest string) (time.Time, error)) *MockBlobStorage_GetBlobModTime_Call { + _c.Call.Return(run) + return _c +} + // GetBlobPath provides a mock function for the type MockBlobStorage func (_mock *MockBlobStorage) GetBlobPath(digest string) (string, error) { ret := _mock.Called(digest) @@ -646,63 +793,6 @@ func (_c *MockBlobStorage_ListBlobs_Call) RunAndReturn(run func() ([]string, err return _c } -// GetBlobModTime provides a mock function for the type MockBlobStorage. -func (_mock *MockBlobStorage) GetBlobModTime(digest string) (time.Time, error) { - ret := _mock.Called(digest) - - if len(ret) == 0 { - panic("no return value specified for GetBlobModTime") - } - - var r0 time.Time - var r1 error - if returnFunc, ok := ret.Get(0).(func(string) (time.Time, error)); ok { - return returnFunc(digest) - } - if returnFunc, ok := ret.Get(0).(func(string) time.Time); ok { - r0 = returnFunc(digest) - } else if ret.Get(0) != nil { - r0 = ret.Get(0).(time.Time) - } - if returnFunc, ok := ret.Get(1).(func(string) error); ok { - r1 = returnFunc(digest) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockBlobStorage_GetBlobModTime_Call is a typed expectation for GetBlobModTime. -type MockBlobStorage_GetBlobModTime_Call struct { - *mock.Call -} - -// GetBlobModTime is a helper method to define mock.On call. -func (_e *MockBlobStorage_Expecter) GetBlobModTime(digest any) *MockBlobStorage_GetBlobModTime_Call { - return &MockBlobStorage_GetBlobModTime_Call{Call: _e.mock.On("GetBlobModTime", digest)} -} - -func (_c *MockBlobStorage_GetBlobModTime_Call) Run(run func(digest string)) *MockBlobStorage_GetBlobModTime_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 string - if args[0] != nil { - arg0 = args[0].(string) - } - run(arg0) - }) - return _c -} - -func (_c *MockBlobStorage_GetBlobModTime_Call) Return(modTime time.Time, err error) *MockBlobStorage_GetBlobModTime_Call { - _c.Call.Return(modTime, err) - return _c -} - -func (_c *MockBlobStorage_GetBlobModTime_Call) RunAndReturn(run func(digest string) (time.Time, error)) *MockBlobStorage_GetBlobModTime_Call { - _c.Call.Return(run) - return _c -} - // PutBlob provides a mock function for the type MockBlobStorage func (_mock *MockBlobStorage) PutBlob(digest string, data io.Reader, size int64) error { ret := _mock.Called(digest, data, size) diff --git a/internal/boundaries/out/mocks/mock_certificate_authority.go b/internal/boundaries/out/mocks/mock_certificate_authority.go index 8bfe4ee00..b63035132 100644 --- a/internal/boundaries/out/mocks/mock_certificate_authority.go +++ b/internal/boundaries/out/mocks/mock_certificate_authority.go @@ -18,10 +18,19 @@ func NewMockCertificateAuthority(t interface { mock.TestingT Cleanup(func()) }) *MockCertificateAuthority { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockCertificateAuthority{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/out/mocks/mock_certificate_store.go b/internal/boundaries/out/mocks/mock_certificate_store.go index 871774c24..68022f713 100644 --- a/internal/boundaries/out/mocks/mock_certificate_store.go +++ b/internal/boundaries/out/mocks/mock_certificate_store.go @@ -17,10 +17,19 @@ func NewMockCertificateStore(t interface { mock.TestingT Cleanup(func()) }) *MockCertificateStore { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockCertificateStore{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/out/mocks/mock_cloudflare_zone_resolver.go b/internal/boundaries/out/mocks/mock_cloudflare_zone_resolver.go index 116f1e802..4da22ef01 100644 --- a/internal/boundaries/out/mocks/mock_cloudflare_zone_resolver.go +++ b/internal/boundaries/out/mocks/mock_cloudflare_zone_resolver.go @@ -17,10 +17,19 @@ func NewMockCloudflareZoneResolver(t interface { mock.TestingT Cleanup(func()) }) *MockCloudflareZoneResolver { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockCloudflareZoneResolver{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/out/mocks/mock_container_log_streamer.go b/internal/boundaries/out/mocks/mock_container_log_streamer.go new file mode 100644 index 000000000..63bc7c4e0 --- /dev/null +++ b/internal/boundaries/out/mocks/mock_container_log_streamer.go @@ -0,0 +1,118 @@ +// Code generated by mockery; DO NOT EDIT. +// github.com/vektra/mockery +// template: testify + +package mocks + +import ( + "context" + "time" + + "github.com/bnema/gordon/internal/domain" + mock "github.com/stretchr/testify/mock" +) + +// NewMockContainerLogStreamer creates a new instance of MockContainerLogStreamer. It also registers a testing interface on the mock and a cleanup function to assert the mocks expectations. +// The first argument is typically a *testing.T value. +func NewMockContainerLogStreamer(t interface { + mock.TestingT + Cleanup(func()) +}) *MockContainerLogStreamer { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + + mock := &MockContainerLogStreamer{} + mock.Mock.Test(t) + + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) + + return mock +} + +// MockContainerLogStreamer is an autogenerated mock type for the ContainerLogStreamer type +type MockContainerLogStreamer struct { + mock.Mock +} + +type MockContainerLogStreamer_Expecter struct { + mock *mock.Mock +} + +func (_m *MockContainerLogStreamer) EXPECT() *MockContainerLogStreamer_Expecter { + return &MockContainerLogStreamer_Expecter{mock: &_m.Mock} +} + +// StreamContainerLogs provides a mock function for the type MockContainerLogStreamer +func (_mock *MockContainerLogStreamer) StreamContainerLogs(ctx context.Context, containerID string, since time.Time, emit func(domain.ContainerLogLine)) error { + ret := _mock.Called(ctx, containerID, since, emit) + + if len(ret) == 0 { + panic("no return value specified for StreamContainerLogs") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, time.Time, func(domain.ContainerLogLine)) error); ok { + r0 = returnFunc(ctx, containerID, since, emit) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockContainerLogStreamer_StreamContainerLogs_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'StreamContainerLogs' +type MockContainerLogStreamer_StreamContainerLogs_Call struct { + *mock.Call +} + +// StreamContainerLogs is a helper method to define mock.On call +// - ctx context.Context +// - containerID string +// - since time.Time +// - emit func(domain.ContainerLogLine) +func (_e *MockContainerLogStreamer_Expecter) StreamContainerLogs(ctx any, containerID any, since any, emit any) *MockContainerLogStreamer_StreamContainerLogs_Call { + return &MockContainerLogStreamer_StreamContainerLogs_Call{Call: _e.mock.On("StreamContainerLogs", ctx, containerID, since, emit)} +} + +func (_c *MockContainerLogStreamer_StreamContainerLogs_Call) Run(run func(ctx context.Context, containerID string, since time.Time, emit func(domain.ContainerLogLine))) *MockContainerLogStreamer_StreamContainerLogs_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 time.Time + if args[2] != nil { + arg2 = args[2].(time.Time) + } + var arg3 func(domain.ContainerLogLine) + if args[3] != nil { + arg3 = args[3].(func(domain.ContainerLogLine)) + } + run( + arg0, + arg1, + arg2, + arg3, + ) + }) + return _c +} + +func (_c *MockContainerLogStreamer_StreamContainerLogs_Call) Return(err error) *MockContainerLogStreamer_StreamContainerLogs_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockContainerLogStreamer_StreamContainerLogs_Call) RunAndReturn(run func(ctx context.Context, containerID string, since time.Time, emit func(domain.ContainerLogLine)) error) *MockContainerLogStreamer_StreamContainerLogs_Call { + _c.Call.Return(run) + return _c +} diff --git a/internal/boundaries/out/mocks/mock_container_log_writer.go b/internal/boundaries/out/mocks/mock_container_log_writer.go index cddbbae06..ad7c88f5c 100644 --- a/internal/boundaries/out/mocks/mock_container_log_writer.go +++ b/internal/boundaries/out/mocks/mock_container_log_writer.go @@ -17,10 +17,19 @@ func NewMockContainerLogWriter(t interface { mock.TestingT Cleanup(func()) }) *MockContainerLogWriter { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockContainerLogWriter{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/out/mocks/mock_container_runtime.go b/internal/boundaries/out/mocks/mock_container_runtime.go index 837a35df9..37c3a17bd 100644 --- a/internal/boundaries/out/mocks/mock_container_runtime.go +++ b/internal/boundaries/out/mocks/mock_container_runtime.go @@ -7,6 +7,7 @@ package mocks import ( "context" "io" + "time" "github.com/bnema/gordon/internal/boundaries/out" "github.com/bnema/gordon/internal/domain" @@ -19,10 +20,19 @@ func NewMockContainerRuntime(t interface { mock.TestingT Cleanup(func()) }) *MockContainerRuntime { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockContainerRuntime{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } @@ -309,16 +319,16 @@ func (_c *MockContainerRuntime_CreateNetwork_Call) RunAndReturn(run func(ctx con } // CreateVolume provides a mock function for the type MockContainerRuntime -func (_mock *MockContainerRuntime) CreateVolume(ctx context.Context, volumeName string) error { - ret := _mock.Called(ctx, volumeName) +func (_mock *MockContainerRuntime) CreateVolume(ctx context.Context, volumeName string, labels map[string]string) error { + ret := _mock.Called(ctx, volumeName, labels) if len(ret) == 0 { panic("no return value specified for CreateVolume") } var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) error); ok { - r0 = returnFunc(ctx, volumeName) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, map[string]string) error); ok { + r0 = returnFunc(ctx, volumeName, labels) } else { r0 = ret.Error(0) } @@ -333,11 +343,12 @@ type MockContainerRuntime_CreateVolume_Call struct { // CreateVolume is a helper method to define mock.On call // - ctx context.Context // - volumeName string -func (_e *MockContainerRuntime_Expecter) CreateVolume(ctx any, volumeName any) *MockContainerRuntime_CreateVolume_Call { - return &MockContainerRuntime_CreateVolume_Call{Call: _e.mock.On("CreateVolume", ctx, volumeName)} +// - labels map[string]string +func (_e *MockContainerRuntime_Expecter) CreateVolume(ctx any, volumeName any, labels any) *MockContainerRuntime_CreateVolume_Call { + return &MockContainerRuntime_CreateVolume_Call{Call: _e.mock.On("CreateVolume", ctx, volumeName, labels)} } -func (_c *MockContainerRuntime_CreateVolume_Call) Run(run func(ctx context.Context, volumeName string)) *MockContainerRuntime_CreateVolume_Call { +func (_c *MockContainerRuntime_CreateVolume_Call) Run(run func(ctx context.Context, volumeName string, labels map[string]string)) *MockContainerRuntime_CreateVolume_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -347,9 +358,14 @@ func (_c *MockContainerRuntime_CreateVolume_Call) Run(run func(ctx context.Conte if args[1] != nil { arg1 = args[1].(string) } + var arg2 map[string]string + if args[2] != nil { + arg2 = args[2].(map[string]string) + } run( arg0, arg1, + arg2, ) }) return _c @@ -360,7 +376,7 @@ func (_c *MockContainerRuntime_CreateVolume_Call) Return(err error) *MockContain return _c } -func (_c *MockContainerRuntime_CreateVolume_Call) RunAndReturn(run func(ctx context.Context, volumeName string) error) *MockContainerRuntime_CreateVolume_Call { +func (_c *MockContainerRuntime_CreateVolume_Call) RunAndReturn(run func(ctx context.Context, volumeName string, labels map[string]string) error) *MockContainerRuntime_CreateVolume_Call { _c.Call.Return(run) return _c } @@ -502,6 +518,80 @@ func (_c *MockContainerRuntime_ExecInContainer_Call) RunAndReturn(run func(ctx c return _c } +// GetContainerBackendBinds provides a mock function for the type MockContainerRuntime +func (_mock *MockContainerRuntime) GetContainerBackendBinds(ctx context.Context, containerID string, ports []domain.ContainerBackendPort) ([]domain.ContainerBackendBind, error) { + ret := _mock.Called(ctx, containerID, ports) + + if len(ret) == 0 { + panic("no return value specified for GetContainerBackendBinds") + } + + var r0 []domain.ContainerBackendBind + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, []domain.ContainerBackendPort) ([]domain.ContainerBackendBind, error)); ok { + return returnFunc(ctx, containerID, ports) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string, []domain.ContainerBackendPort) []domain.ContainerBackendBind); ok { + r0 = returnFunc(ctx, containerID, ports) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).([]domain.ContainerBackendBind) + } + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string, []domain.ContainerBackendPort) error); ok { + r1 = returnFunc(ctx, containerID, ports) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockContainerRuntime_GetContainerBackendBinds_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetContainerBackendBinds' +type MockContainerRuntime_GetContainerBackendBinds_Call struct { + *mock.Call +} + +// GetContainerBackendBinds is a helper method to define mock.On call +// - ctx context.Context +// - containerID string +// - ports []domain.ContainerBackendPort +func (_e *MockContainerRuntime_Expecter) GetContainerBackendBinds(ctx any, containerID any, ports any) *MockContainerRuntime_GetContainerBackendBinds_Call { + return &MockContainerRuntime_GetContainerBackendBinds_Call{Call: _e.mock.On("GetContainerBackendBinds", ctx, containerID, ports)} +} + +func (_c *MockContainerRuntime_GetContainerBackendBinds_Call) Run(run func(ctx context.Context, containerID string, ports []domain.ContainerBackendPort)) *MockContainerRuntime_GetContainerBackendBinds_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 []domain.ContainerBackendPort + if args[2] != nil { + arg2 = args[2].([]domain.ContainerBackendPort) + } + run( + arg0, + arg1, + arg2, + ) + }) + return _c +} + +func (_c *MockContainerRuntime_GetContainerBackendBinds_Call) Return(containerBackendBinds []domain.ContainerBackendBind, err error) *MockContainerRuntime_GetContainerBackendBinds_Call { + _c.Call.Return(containerBackendBinds, err) + return _c +} + +func (_c *MockContainerRuntime_GetContainerBackendBinds_Call) RunAndReturn(run func(ctx context.Context, containerID string, ports []domain.ContainerBackendPort) ([]domain.ContainerBackendBind, error)) *MockContainerRuntime_GetContainerBackendBinds_Call { + _c.Call.Return(run) + return _c +} + // GetContainerExposedPorts provides a mock function for the type MockContainerRuntime func (_mock *MockContainerRuntime) GetContainerExposedPorts(ctx context.Context, containerID string) ([]int, error) { ret := _mock.Called(ctx, containerID) @@ -716,6 +806,86 @@ func (_c *MockContainerRuntime_GetContainerLogs_Call) RunAndReturn(run func(ctx return _c } +// GetContainerLogsSince provides a mock function for the type MockContainerRuntime +func (_mock *MockContainerRuntime) GetContainerLogsSince(ctx context.Context, containerID string, since time.Time, follow bool) (io.ReadCloser, error) { + ret := _mock.Called(ctx, containerID, since, follow) + + if len(ret) == 0 { + panic("no return value specified for GetContainerLogsSince") + } + + var r0 io.ReadCloser + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, time.Time, bool) (io.ReadCloser, error)); ok { + return returnFunc(ctx, containerID, since, follow) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string, time.Time, bool) io.ReadCloser); ok { + r0 = returnFunc(ctx, containerID, since, follow) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).(io.ReadCloser) + } + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string, time.Time, bool) error); ok { + r1 = returnFunc(ctx, containerID, since, follow) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockContainerRuntime_GetContainerLogsSince_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetContainerLogsSince' +type MockContainerRuntime_GetContainerLogsSince_Call struct { + *mock.Call +} + +// GetContainerLogsSince is a helper method to define mock.On call +// - ctx context.Context +// - containerID string +// - since time.Time +// - follow bool +func (_e *MockContainerRuntime_Expecter) GetContainerLogsSince(ctx any, containerID any, since any, follow any) *MockContainerRuntime_GetContainerLogsSince_Call { + return &MockContainerRuntime_GetContainerLogsSince_Call{Call: _e.mock.On("GetContainerLogsSince", ctx, containerID, since, follow)} +} + +func (_c *MockContainerRuntime_GetContainerLogsSince_Call) Run(run func(ctx context.Context, containerID string, since time.Time, follow bool)) *MockContainerRuntime_GetContainerLogsSince_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 time.Time + if args[2] != nil { + arg2 = args[2].(time.Time) + } + var arg3 bool + if args[3] != nil { + arg3 = args[3].(bool) + } + run( + arg0, + arg1, + arg2, + arg3, + ) + }) + return _c +} + +func (_c *MockContainerRuntime_GetContainerLogsSince_Call) Return(readCloser io.ReadCloser, err error) *MockContainerRuntime_GetContainerLogsSince_Call { + _c.Call.Return(readCloser, err) + return _c +} + +func (_c *MockContainerRuntime_GetContainerLogsSince_Call) RunAndReturn(run func(ctx context.Context, containerID string, since time.Time, follow bool) (io.ReadCloser, error)) *MockContainerRuntime_GetContainerLogsSince_Call { + _c.Call.Return(run) + return _c +} + // GetContainerNetwork provides a mock function for the type MockContainerRuntime func (_mock *MockContainerRuntime) GetContainerNetwork(ctx context.Context, containerID string) (string, error) { ret := _mock.Called(ctx, containerID) @@ -854,78 +1024,6 @@ func (_c *MockContainerRuntime_GetContainerNetworkInfo_Call) RunAndReturn(run fu return _c } -// GetContainerPort provides a mock function for the type MockContainerRuntime -func (_mock *MockContainerRuntime) GetContainerPort(ctx context.Context, containerID string, internalPort int) (int, error) { - ret := _mock.Called(ctx, containerID, internalPort) - - if len(ret) == 0 { - panic("no return value specified for GetContainerPort") - } - - var r0 int - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string, int) (int, error)); ok { - return returnFunc(ctx, containerID, internalPort) - } - if returnFunc, ok := ret.Get(0).(func(context.Context, string, int) int); ok { - r0 = returnFunc(ctx, containerID, internalPort) - } else { - r0 = ret.Get(0).(int) - } - if returnFunc, ok := ret.Get(1).(func(context.Context, string, int) error); ok { - r1 = returnFunc(ctx, containerID, internalPort) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockContainerRuntime_GetContainerPort_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetContainerPort' -type MockContainerRuntime_GetContainerPort_Call struct { - *mock.Call -} - -// GetContainerPort is a helper method to define mock.On call -// - ctx context.Context -// - containerID string -// - internalPort int -func (_e *MockContainerRuntime_Expecter) GetContainerPort(ctx any, containerID any, internalPort any) *MockContainerRuntime_GetContainerPort_Call { - return &MockContainerRuntime_GetContainerPort_Call{Call: _e.mock.On("GetContainerPort", ctx, containerID, internalPort)} -} - -func (_c *MockContainerRuntime_GetContainerPort_Call) Run(run func(ctx context.Context, containerID string, internalPort int)) *MockContainerRuntime_GetContainerPort_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - var arg2 int - if args[2] != nil { - arg2 = args[2].(int) - } - run( - arg0, - arg1, - arg2, - ) - }) - return _c -} - -func (_c *MockContainerRuntime_GetContainerPort_Call) Return(n int, err error) *MockContainerRuntime_GetContainerPort_Call { - _c.Call.Return(n, err) - return _c -} - -func (_c *MockContainerRuntime_GetContainerPort_Call) RunAndReturn(run func(ctx context.Context, containerID string, internalPort int) (int, error)) *MockContainerRuntime_GetContainerPort_Call { - _c.Call.Return(run) - return _c -} - // GetImageExposedPorts provides a mock function for the type MockContainerRuntime func (_mock *MockContainerRuntime) GetImageExposedPorts(ctx context.Context, imageRef string) ([]int, error) { ret := _mock.Called(ctx, imageRef) @@ -1769,6 +1867,72 @@ func (_c *MockContainerRuntime_Ping_Call) RunAndReturn(run func(ctx context.Cont return _c } +// ProbeContainerNetwork provides a mock function for the type MockContainerRuntime +func (_mock *MockContainerRuntime) ProbeContainerNetwork(ctx context.Context, request domain.ContainerNetworkProbeRequest) (domain.ContainerNetworkProbeResult, error) { + ret := _mock.Called(ctx, request) + + if len(ret) == 0 { + panic("no return value specified for ProbeContainerNetwork") + } + + var r0 domain.ContainerNetworkProbeResult + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, domain.ContainerNetworkProbeRequest) (domain.ContainerNetworkProbeResult, error)); ok { + return returnFunc(ctx, request) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, domain.ContainerNetworkProbeRequest) domain.ContainerNetworkProbeResult); ok { + r0 = returnFunc(ctx, request) + } else { + r0 = ret.Get(0).(domain.ContainerNetworkProbeResult) + } + if returnFunc, ok := ret.Get(1).(func(context.Context, domain.ContainerNetworkProbeRequest) error); ok { + r1 = returnFunc(ctx, request) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockContainerRuntime_ProbeContainerNetwork_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ProbeContainerNetwork' +type MockContainerRuntime_ProbeContainerNetwork_Call struct { + *mock.Call +} + +// ProbeContainerNetwork is a helper method to define mock.On call +// - ctx context.Context +// - request domain.ContainerNetworkProbeRequest +func (_e *MockContainerRuntime_Expecter) ProbeContainerNetwork(ctx any, request any) *MockContainerRuntime_ProbeContainerNetwork_Call { + return &MockContainerRuntime_ProbeContainerNetwork_Call{Call: _e.mock.On("ProbeContainerNetwork", ctx, request)} +} + +func (_c *MockContainerRuntime_ProbeContainerNetwork_Call) Run(run func(ctx context.Context, request domain.ContainerNetworkProbeRequest)) *MockContainerRuntime_ProbeContainerNetwork_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 domain.ContainerNetworkProbeRequest + if args[1] != nil { + arg1 = args[1].(domain.ContainerNetworkProbeRequest) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockContainerRuntime_ProbeContainerNetwork_Call) Return(containerNetworkProbeResult domain.ContainerNetworkProbeResult, err error) *MockContainerRuntime_ProbeContainerNetwork_Call { + _c.Call.Return(containerNetworkProbeResult, err) + return _c +} + +func (_c *MockContainerRuntime_ProbeContainerNetwork_Call) RunAndReturn(run func(ctx context.Context, request domain.ContainerNetworkProbeRequest) (domain.ContainerNetworkProbeResult, error)) *MockContainerRuntime_ProbeContainerNetwork_Call { + _c.Call.Return(run) + return _c +} + // PullImage provides a mock function for the type MockContainerRuntime func (_mock *MockContainerRuntime) PullImage(ctx context.Context, image string) error { ret := _mock.Called(ctx, image) @@ -1895,6 +2059,63 @@ func (_c *MockContainerRuntime_PullImageWithAuth_Call) RunAndReturn(run func(ctx return _c } +// PullImageWithOptions provides a mock function for the type MockContainerRuntime +func (_mock *MockContainerRuntime) PullImageWithOptions(ctx context.Context, request domain.ImagePullRequest) error { + ret := _mock.Called(ctx, request) + + if len(ret) == 0 { + panic("no return value specified for PullImageWithOptions") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context, domain.ImagePullRequest) error); ok { + r0 = returnFunc(ctx, request) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockContainerRuntime_PullImageWithOptions_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'PullImageWithOptions' +type MockContainerRuntime_PullImageWithOptions_Call struct { + *mock.Call +} + +// PullImageWithOptions is a helper method to define mock.On call +// - ctx context.Context +// - request domain.ImagePullRequest +func (_e *MockContainerRuntime_Expecter) PullImageWithOptions(ctx any, request any) *MockContainerRuntime_PullImageWithOptions_Call { + return &MockContainerRuntime_PullImageWithOptions_Call{Call: _e.mock.On("PullImageWithOptions", ctx, request)} +} + +func (_c *MockContainerRuntime_PullImageWithOptions_Call) Run(run func(ctx context.Context, request domain.ImagePullRequest)) *MockContainerRuntime_PullImageWithOptions_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 domain.ImagePullRequest + if args[1] != nil { + arg1 = args[1].(domain.ImagePullRequest) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockContainerRuntime_PullImageWithOptions_Call) Return(err error) *MockContainerRuntime_PullImageWithOptions_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockContainerRuntime_PullImageWithOptions_Call) RunAndReturn(run func(ctx context.Context, request domain.ImagePullRequest) error) *MockContainerRuntime_PullImageWithOptions_Call { + _c.Call.Return(run) + return _c +} + // RemoveContainer provides a mock function for the type MockContainerRuntime func (_mock *MockContainerRuntime) RemoveContainer(ctx context.Context, containerID string, force bool) error { ret := _mock.Called(ctx, containerID, force) @@ -2205,16 +2426,16 @@ func (_c *MockContainerRuntime_RenameContainer_Call) RunAndReturn(run func(ctx c } // RestartContainer provides a mock function for the type MockContainerRuntime -func (_mock *MockContainerRuntime) RestartContainer(ctx context.Context, containerID string) error { - ret := _mock.Called(ctx, containerID) +func (_mock *MockContainerRuntime) RestartContainer(ctx context.Context, containerID string, grace time.Duration) error { + ret := _mock.Called(ctx, containerID, grace) if len(ret) == 0 { panic("no return value specified for RestartContainer") } var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) error); ok { - r0 = returnFunc(ctx, containerID) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, time.Duration) error); ok { + r0 = returnFunc(ctx, containerID, grace) } else { r0 = ret.Error(0) } @@ -2229,11 +2450,12 @@ type MockContainerRuntime_RestartContainer_Call struct { // RestartContainer is a helper method to define mock.On call // - ctx context.Context // - containerID string -func (_e *MockContainerRuntime_Expecter) RestartContainer(ctx any, containerID any) *MockContainerRuntime_RestartContainer_Call { - return &MockContainerRuntime_RestartContainer_Call{Call: _e.mock.On("RestartContainer", ctx, containerID)} +// - grace time.Duration +func (_e *MockContainerRuntime_Expecter) RestartContainer(ctx any, containerID any, grace any) *MockContainerRuntime_RestartContainer_Call { + return &MockContainerRuntime_RestartContainer_Call{Call: _e.mock.On("RestartContainer", ctx, containerID, grace)} } -func (_c *MockContainerRuntime_RestartContainer_Call) Run(run func(ctx context.Context, containerID string)) *MockContainerRuntime_RestartContainer_Call { +func (_c *MockContainerRuntime_RestartContainer_Call) Run(run func(ctx context.Context, containerID string, grace time.Duration)) *MockContainerRuntime_RestartContainer_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -2243,9 +2465,14 @@ func (_c *MockContainerRuntime_RestartContainer_Call) Run(run func(ctx context.C if args[1] != nil { arg1 = args[1].(string) } + var arg2 time.Duration + if args[2] != nil { + arg2 = args[2].(time.Duration) + } run( arg0, arg1, + arg2, ) }) return _c @@ -2256,7 +2483,7 @@ func (_c *MockContainerRuntime_RestartContainer_Call) Return(err error) *MockCon return _c } -func (_c *MockContainerRuntime_RestartContainer_Call) RunAndReturn(run func(ctx context.Context, containerID string) error) *MockContainerRuntime_RestartContainer_Call { +func (_c *MockContainerRuntime_RestartContainer_Call) RunAndReturn(run func(ctx context.Context, containerID string, grace time.Duration) error) *MockContainerRuntime_RestartContainer_Call { _c.Call.Return(run) return _c } @@ -2319,16 +2546,16 @@ func (_c *MockContainerRuntime_StartContainer_Call) RunAndReturn(run func(ctx co } // StopContainer provides a mock function for the type MockContainerRuntime -func (_mock *MockContainerRuntime) StopContainer(ctx context.Context, containerID string) error { - ret := _mock.Called(ctx, containerID) +func (_mock *MockContainerRuntime) StopContainer(ctx context.Context, containerID string, grace time.Duration) error { + ret := _mock.Called(ctx, containerID, grace) if len(ret) == 0 { panic("no return value specified for StopContainer") } var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) error); ok { - r0 = returnFunc(ctx, containerID) + if returnFunc, ok := ret.Get(0).(func(context.Context, string, time.Duration) error); ok { + r0 = returnFunc(ctx, containerID, grace) } else { r0 = ret.Error(0) } @@ -2343,11 +2570,12 @@ type MockContainerRuntime_StopContainer_Call struct { // StopContainer is a helper method to define mock.On call // - ctx context.Context // - containerID string -func (_e *MockContainerRuntime_Expecter) StopContainer(ctx any, containerID any) *MockContainerRuntime_StopContainer_Call { - return &MockContainerRuntime_StopContainer_Call{Call: _e.mock.On("StopContainer", ctx, containerID)} +// - grace time.Duration +func (_e *MockContainerRuntime_Expecter) StopContainer(ctx any, containerID any, grace any) *MockContainerRuntime_StopContainer_Call { + return &MockContainerRuntime_StopContainer_Call{Call: _e.mock.On("StopContainer", ctx, containerID, grace)} } -func (_c *MockContainerRuntime_StopContainer_Call) Run(run func(ctx context.Context, containerID string)) *MockContainerRuntime_StopContainer_Call { +func (_c *MockContainerRuntime_StopContainer_Call) Run(run func(ctx context.Context, containerID string, grace time.Duration)) *MockContainerRuntime_StopContainer_Call { _c.Call.Run(func(args mock.Arguments) { var arg0 context.Context if args[0] != nil { @@ -2357,9 +2585,14 @@ func (_c *MockContainerRuntime_StopContainer_Call) Run(run func(ctx context.Cont if args[1] != nil { arg1 = args[1].(string) } + var arg2 time.Duration + if args[2] != nil { + arg2 = args[2].(time.Duration) + } run( arg0, arg1, + arg2, ) }) return _c @@ -2370,7 +2603,58 @@ func (_c *MockContainerRuntime_StopContainer_Call) Return(err error) *MockContai return _c } -func (_c *MockContainerRuntime_StopContainer_Call) RunAndReturn(run func(ctx context.Context, containerID string) error) *MockContainerRuntime_StopContainer_Call { +func (_c *MockContainerRuntime_StopContainer_Call) RunAndReturn(run func(ctx context.Context, containerID string, grace time.Duration) error) *MockContainerRuntime_StopContainer_Call { + _c.Call.Return(run) + return _c +} + +// SupportsCDIDevices provides a mock function for the type MockContainerRuntime +func (_mock *MockContainerRuntime) SupportsCDIDevices(ctx context.Context) error { + ret := _mock.Called(ctx) + + if len(ret) == 0 { + panic("no return value specified for SupportsCDIDevices") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context) error); ok { + r0 = returnFunc(ctx) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockContainerRuntime_SupportsCDIDevices_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'SupportsCDIDevices' +type MockContainerRuntime_SupportsCDIDevices_Call struct { + *mock.Call +} + +// SupportsCDIDevices is a helper method to define mock.On call +// - ctx context.Context +func (_e *MockContainerRuntime_Expecter) SupportsCDIDevices(ctx any) *MockContainerRuntime_SupportsCDIDevices_Call { + return &MockContainerRuntime_SupportsCDIDevices_Call{Call: _e.mock.On("SupportsCDIDevices", ctx)} +} + +func (_c *MockContainerRuntime_SupportsCDIDevices_Call) Run(run func(ctx context.Context)) *MockContainerRuntime_SupportsCDIDevices_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + run( + arg0, + ) + }) + return _c +} + +func (_c *MockContainerRuntime_SupportsCDIDevices_Call) Return(err error) *MockContainerRuntime_SupportsCDIDevices_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockContainerRuntime_SupportsCDIDevices_Call) RunAndReturn(run func(ctx context.Context) error) *MockContainerRuntime_SupportsCDIDevices_Call { _c.Call.Return(run) return _c } @@ -2495,6 +2779,69 @@ func (_c *MockContainerRuntime_UntagImage_Call) RunAndReturn(run func(ctx contex return _c } +// VerifyImageDigest provides a mock function for the type MockContainerRuntime +func (_mock *MockContainerRuntime) VerifyImageDigest(ctx context.Context, imageRef string, digest string) error { + ret := _mock.Called(ctx, imageRef, digest) + + if len(ret) == 0 { + panic("no return value specified for VerifyImageDigest") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) error); ok { + r0 = returnFunc(ctx, imageRef, digest) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockContainerRuntime_VerifyImageDigest_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'VerifyImageDigest' +type MockContainerRuntime_VerifyImageDigest_Call struct { + *mock.Call +} + +// VerifyImageDigest is a helper method to define mock.On call +// - ctx context.Context +// - imageRef string +// - digest string +func (_e *MockContainerRuntime_Expecter) VerifyImageDigest(ctx any, imageRef any, digest any) *MockContainerRuntime_VerifyImageDigest_Call { + return &MockContainerRuntime_VerifyImageDigest_Call{Call: _e.mock.On("VerifyImageDigest", ctx, imageRef, digest)} +} + +func (_c *MockContainerRuntime_VerifyImageDigest_Call) Run(run func(ctx context.Context, imageRef string, digest string)) *MockContainerRuntime_VerifyImageDigest_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } + run( + arg0, + arg1, + arg2, + ) + }) + return _c +} + +func (_c *MockContainerRuntime_VerifyImageDigest_Call) Return(err error) *MockContainerRuntime_VerifyImageDigest_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockContainerRuntime_VerifyImageDigest_Call) RunAndReturn(run func(ctx context.Context, imageRef string, digest string) error) *MockContainerRuntime_VerifyImageDigest_Call { + _c.Call.Return(run) + return _c +} + // Version provides a mock function for the type MockContainerRuntime func (_mock *MockContainerRuntime) Version(ctx context.Context) (string, error) { ret := _mock.Called(ctx) diff --git a/internal/boundaries/out/mocks/mock_domain_secret_store.go b/internal/boundaries/out/mocks/mock_domain_secret_store.go deleted file mode 100644 index 3ffe945b7..000000000 --- a/internal/boundaries/out/mocks/mock_domain_secret_store.go +++ /dev/null @@ -1,513 +0,0 @@ -// Code generated by mockery; DO NOT EDIT. -// github.com/vektra/mockery -// template: testify - -package mocks - -import ( - "github.com/bnema/gordon/internal/boundaries/out" - mock "github.com/stretchr/testify/mock" -) - -// NewMockDomainSecretStore creates a new instance of MockDomainSecretStore. It also registers a testing interface on the mock and a cleanup function to assert the mocks expectations. -// The first argument is typically a *testing.T value. -func NewMockDomainSecretStore(t interface { - mock.TestingT - Cleanup(func()) -}) *MockDomainSecretStore { - mock := &MockDomainSecretStore{} - mock.Mock.Test(t) - - t.Cleanup(func() { mock.AssertExpectations(t) }) - - return mock -} - -// MockDomainSecretStore is an autogenerated mock type for the DomainSecretStore type -type MockDomainSecretStore struct { - mock.Mock -} - -type MockDomainSecretStore_Expecter struct { - mock *mock.Mock -} - -func (_m *MockDomainSecretStore) EXPECT() *MockDomainSecretStore_Expecter { - return &MockDomainSecretStore_Expecter{mock: &_m.Mock} -} - -// Delete provides a mock function for the type MockDomainSecretStore -func (_mock *MockDomainSecretStore) Delete(domain string, key string) error { - ret := _mock.Called(domain, key) - - if len(ret) == 0 { - panic("no return value specified for Delete") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(string, string) error); ok { - r0 = returnFunc(domain, key) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockDomainSecretStore_Delete_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Delete' -type MockDomainSecretStore_Delete_Call struct { - *mock.Call -} - -// Delete is a helper method to define mock.On call -// - domain string -// - key string -func (_e *MockDomainSecretStore_Expecter) Delete(domain any, key any) *MockDomainSecretStore_Delete_Call { - return &MockDomainSecretStore_Delete_Call{Call: _e.mock.On("Delete", domain, key)} -} - -func (_c *MockDomainSecretStore_Delete_Call) Run(run func(domain string, key string)) *MockDomainSecretStore_Delete_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 string - if args[0] != nil { - arg0 = args[0].(string) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockDomainSecretStore_Delete_Call) Return(err error) *MockDomainSecretStore_Delete_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockDomainSecretStore_Delete_Call) RunAndReturn(run func(domain string, key string) error) *MockDomainSecretStore_Delete_Call { - _c.Call.Return(run) - return _c -} - -// DeleteAttachment provides a mock function for the type MockDomainSecretStore -func (_mock *MockDomainSecretStore) DeleteAttachment(containerName string, key string) error { - ret := _mock.Called(containerName, key) - - if len(ret) == 0 { - panic("no return value specified for DeleteAttachment") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(string, string) error); ok { - r0 = returnFunc(containerName, key) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockDomainSecretStore_DeleteAttachment_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'DeleteAttachment' -type MockDomainSecretStore_DeleteAttachment_Call struct { - *mock.Call -} - -// DeleteAttachment is a helper method to define mock.On call -// - containerName string -// - key string -func (_e *MockDomainSecretStore_Expecter) DeleteAttachment(containerName any, key any) *MockDomainSecretStore_DeleteAttachment_Call { - return &MockDomainSecretStore_DeleteAttachment_Call{Call: _e.mock.On("DeleteAttachment", containerName, key)} -} - -func (_c *MockDomainSecretStore_DeleteAttachment_Call) Run(run func(containerName string, key string)) *MockDomainSecretStore_DeleteAttachment_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 string - if args[0] != nil { - arg0 = args[0].(string) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockDomainSecretStore_DeleteAttachment_Call) Return(err error) *MockDomainSecretStore_DeleteAttachment_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockDomainSecretStore_DeleteAttachment_Call) RunAndReturn(run func(containerName string, key string) error) *MockDomainSecretStore_DeleteAttachment_Call { - _c.Call.Return(run) - return _c -} - -// GetAll provides a mock function for the type MockDomainSecretStore -func (_mock *MockDomainSecretStore) GetAll(domain string) (map[string]string, error) { - ret := _mock.Called(domain) - - if len(ret) == 0 { - panic("no return value specified for GetAll") - } - - var r0 map[string]string - var r1 error - if returnFunc, ok := ret.Get(0).(func(string) (map[string]string, error)); ok { - return returnFunc(domain) - } - if returnFunc, ok := ret.Get(0).(func(string) map[string]string); ok { - r0 = returnFunc(domain) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).(map[string]string) - } - } - if returnFunc, ok := ret.Get(1).(func(string) error); ok { - r1 = returnFunc(domain) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockDomainSecretStore_GetAll_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetAll' -type MockDomainSecretStore_GetAll_Call struct { - *mock.Call -} - -// GetAll is a helper method to define mock.On call -// - domain string -func (_e *MockDomainSecretStore_Expecter) GetAll(domain any) *MockDomainSecretStore_GetAll_Call { - return &MockDomainSecretStore_GetAll_Call{Call: _e.mock.On("GetAll", domain)} -} - -func (_c *MockDomainSecretStore_GetAll_Call) Run(run func(domain string)) *MockDomainSecretStore_GetAll_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 string - if args[0] != nil { - arg0 = args[0].(string) - } - run( - arg0, - ) - }) - return _c -} - -func (_c *MockDomainSecretStore_GetAll_Call) Return(stringToString map[string]string, err error) *MockDomainSecretStore_GetAll_Call { - _c.Call.Return(stringToString, err) - return _c -} - -func (_c *MockDomainSecretStore_GetAll_Call) RunAndReturn(run func(domain string) (map[string]string, error)) *MockDomainSecretStore_GetAll_Call { - _c.Call.Return(run) - return _c -} - -// GetAllAttachment provides a mock function for the type MockDomainSecretStore -func (_mock *MockDomainSecretStore) GetAllAttachment(containerName string) (map[string]string, error) { - ret := _mock.Called(containerName) - - if len(ret) == 0 { - panic("no return value specified for GetAllAttachment") - } - - var r0 map[string]string - var r1 error - if returnFunc, ok := ret.Get(0).(func(string) (map[string]string, error)); ok { - return returnFunc(containerName) - } - if returnFunc, ok := ret.Get(0).(func(string) map[string]string); ok { - r0 = returnFunc(containerName) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).(map[string]string) - } - } - if returnFunc, ok := ret.Get(1).(func(string) error); ok { - r1 = returnFunc(containerName) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockDomainSecretStore_GetAllAttachment_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetAllAttachment' -type MockDomainSecretStore_GetAllAttachment_Call struct { - *mock.Call -} - -// GetAllAttachment is a helper method to define mock.On call -// - containerName string -func (_e *MockDomainSecretStore_Expecter) GetAllAttachment(containerName any) *MockDomainSecretStore_GetAllAttachment_Call { - return &MockDomainSecretStore_GetAllAttachment_Call{Call: _e.mock.On("GetAllAttachment", containerName)} -} - -func (_c *MockDomainSecretStore_GetAllAttachment_Call) Run(run func(containerName string)) *MockDomainSecretStore_GetAllAttachment_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 string - if args[0] != nil { - arg0 = args[0].(string) - } - run( - arg0, - ) - }) - return _c -} - -func (_c *MockDomainSecretStore_GetAllAttachment_Call) Return(stringToString map[string]string, err error) *MockDomainSecretStore_GetAllAttachment_Call { - _c.Call.Return(stringToString, err) - return _c -} - -func (_c *MockDomainSecretStore_GetAllAttachment_Call) RunAndReturn(run func(containerName string) (map[string]string, error)) *MockDomainSecretStore_GetAllAttachment_Call { - _c.Call.Return(run) - return _c -} - -// ListAttachmentKeys provides a mock function for the type MockDomainSecretStore -func (_mock *MockDomainSecretStore) ListAttachmentKeys(domain string) ([]out.AttachmentSecrets, error) { - ret := _mock.Called(domain) - - if len(ret) == 0 { - panic("no return value specified for ListAttachmentKeys") - } - - var r0 []out.AttachmentSecrets - var r1 error - if returnFunc, ok := ret.Get(0).(func(string) ([]out.AttachmentSecrets, error)); ok { - return returnFunc(domain) - } - if returnFunc, ok := ret.Get(0).(func(string) []out.AttachmentSecrets); ok { - r0 = returnFunc(domain) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]out.AttachmentSecrets) - } - } - if returnFunc, ok := ret.Get(1).(func(string) error); ok { - r1 = returnFunc(domain) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockDomainSecretStore_ListAttachmentKeys_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListAttachmentKeys' -type MockDomainSecretStore_ListAttachmentKeys_Call struct { - *mock.Call -} - -// ListAttachmentKeys is a helper method to define mock.On call -// - domain string -func (_e *MockDomainSecretStore_Expecter) ListAttachmentKeys(domain any) *MockDomainSecretStore_ListAttachmentKeys_Call { - return &MockDomainSecretStore_ListAttachmentKeys_Call{Call: _e.mock.On("ListAttachmentKeys", domain)} -} - -func (_c *MockDomainSecretStore_ListAttachmentKeys_Call) Run(run func(domain string)) *MockDomainSecretStore_ListAttachmentKeys_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 string - if args[0] != nil { - arg0 = args[0].(string) - } - run( - arg0, - ) - }) - return _c -} - -func (_c *MockDomainSecretStore_ListAttachmentKeys_Call) Return(attachmentSecretss []out.AttachmentSecrets, err error) *MockDomainSecretStore_ListAttachmentKeys_Call { - _c.Call.Return(attachmentSecretss, err) - return _c -} - -func (_c *MockDomainSecretStore_ListAttachmentKeys_Call) RunAndReturn(run func(domain string) ([]out.AttachmentSecrets, error)) *MockDomainSecretStore_ListAttachmentKeys_Call { - _c.Call.Return(run) - return _c -} - -// ListKeys provides a mock function for the type MockDomainSecretStore -func (_mock *MockDomainSecretStore) ListKeys(domain string) ([]string, error) { - ret := _mock.Called(domain) - - if len(ret) == 0 { - panic("no return value specified for ListKeys") - } - - var r0 []string - var r1 error - if returnFunc, ok := ret.Get(0).(func(string) ([]string, error)); ok { - return returnFunc(domain) - } - if returnFunc, ok := ret.Get(0).(func(string) []string); ok { - r0 = returnFunc(domain) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]string) - } - } - if returnFunc, ok := ret.Get(1).(func(string) error); ok { - r1 = returnFunc(domain) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockDomainSecretStore_ListKeys_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ListKeys' -type MockDomainSecretStore_ListKeys_Call struct { - *mock.Call -} - -// ListKeys is a helper method to define mock.On call -// - domain string -func (_e *MockDomainSecretStore_Expecter) ListKeys(domain any) *MockDomainSecretStore_ListKeys_Call { - return &MockDomainSecretStore_ListKeys_Call{Call: _e.mock.On("ListKeys", domain)} -} - -func (_c *MockDomainSecretStore_ListKeys_Call) Run(run func(domain string)) *MockDomainSecretStore_ListKeys_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 string - if args[0] != nil { - arg0 = args[0].(string) - } - run( - arg0, - ) - }) - return _c -} - -func (_c *MockDomainSecretStore_ListKeys_Call) Return(strings []string, err error) *MockDomainSecretStore_ListKeys_Call { - _c.Call.Return(strings, err) - return _c -} - -func (_c *MockDomainSecretStore_ListKeys_Call) RunAndReturn(run func(domain string) ([]string, error)) *MockDomainSecretStore_ListKeys_Call { - _c.Call.Return(run) - return _c -} - -// Set provides a mock function for the type MockDomainSecretStore -func (_mock *MockDomainSecretStore) Set(domain string, secrets map[string]string) error { - ret := _mock.Called(domain, secrets) - - if len(ret) == 0 { - panic("no return value specified for Set") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(string, map[string]string) error); ok { - r0 = returnFunc(domain, secrets) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockDomainSecretStore_Set_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Set' -type MockDomainSecretStore_Set_Call struct { - *mock.Call -} - -// Set is a helper method to define mock.On call -// - domain string -// - secrets map[string]string -func (_e *MockDomainSecretStore_Expecter) Set(domain any, secrets any) *MockDomainSecretStore_Set_Call { - return &MockDomainSecretStore_Set_Call{Call: _e.mock.On("Set", domain, secrets)} -} - -func (_c *MockDomainSecretStore_Set_Call) Run(run func(domain string, secrets map[string]string)) *MockDomainSecretStore_Set_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 string - if args[0] != nil { - arg0 = args[0].(string) - } - var arg1 map[string]string - if args[1] != nil { - arg1 = args[1].(map[string]string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockDomainSecretStore_Set_Call) Return(err error) *MockDomainSecretStore_Set_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockDomainSecretStore_Set_Call) RunAndReturn(run func(domain string, secrets map[string]string) error) *MockDomainSecretStore_Set_Call { - _c.Call.Return(run) - return _c -} - -// SetAttachment provides a mock function for the type MockDomainSecretStore -func (_mock *MockDomainSecretStore) SetAttachment(containerName string, secrets map[string]string) error { - ret := _mock.Called(containerName, secrets) - - if len(ret) == 0 { - panic("no return value specified for SetAttachment") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(string, map[string]string) error); ok { - r0 = returnFunc(containerName, secrets) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockDomainSecretStore_SetAttachment_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'SetAttachment' -type MockDomainSecretStore_SetAttachment_Call struct { - *mock.Call -} - -// SetAttachment is a helper method to define mock.On call -// - containerName string -// - secrets map[string]string -func (_e *MockDomainSecretStore_Expecter) SetAttachment(containerName any, secrets any) *MockDomainSecretStore_SetAttachment_Call { - return &MockDomainSecretStore_SetAttachment_Call{Call: _e.mock.On("SetAttachment", containerName, secrets)} -} - -func (_c *MockDomainSecretStore_SetAttachment_Call) Run(run func(containerName string, secrets map[string]string)) *MockDomainSecretStore_SetAttachment_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 string - if args[0] != nil { - arg0 = args[0].(string) - } - var arg1 map[string]string - if args[1] != nil { - arg1 = args[1].(map[string]string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockDomainSecretStore_SetAttachment_Call) Return(err error) *MockDomainSecretStore_SetAttachment_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockDomainSecretStore_SetAttachment_Call) RunAndReturn(run func(containerName string, secrets map[string]string) error) *MockDomainSecretStore_SetAttachment_Call { - _c.Call.Return(run) - return _c -} diff --git a/internal/boundaries/out/mocks/mock_env_loader.go b/internal/boundaries/out/mocks/mock_env_loader.go deleted file mode 100644 index 105c06d1a..000000000 --- a/internal/boundaries/out/mocks/mock_env_loader.go +++ /dev/null @@ -1,223 +0,0 @@ -// Code generated by mockery; DO NOT EDIT. -// github.com/vektra/mockery -// template: testify - -package mocks - -import ( - "context" - - mock "github.com/stretchr/testify/mock" -) - -// NewMockEnvLoader creates a new instance of MockEnvLoader. It also registers a testing interface on the mock and a cleanup function to assert the mocks expectations. -// The first argument is typically a *testing.T value. -func NewMockEnvLoader(t interface { - mock.TestingT - Cleanup(func()) -}) *MockEnvLoader { - mock := &MockEnvLoader{} - mock.Mock.Test(t) - - t.Cleanup(func() { mock.AssertExpectations(t) }) - - return mock -} - -// MockEnvLoader is an autogenerated mock type for the EnvLoader type -type MockEnvLoader struct { - mock.Mock -} - -type MockEnvLoader_Expecter struct { - mock *mock.Mock -} - -func (_m *MockEnvLoader) EXPECT() *MockEnvLoader_Expecter { - return &MockEnvLoader_Expecter{mock: &_m.Mock} -} - -// CreateEnvFile provides a mock function for the type MockEnvLoader -func (_mock *MockEnvLoader) CreateEnvFile(ctx context.Context, domain string) error { - ret := _mock.Called(ctx, domain) - - if len(ret) == 0 { - panic("no return value specified for CreateEnvFile") - } - - var r0 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) error); ok { - r0 = returnFunc(ctx, domain) - } else { - r0 = ret.Error(0) - } - return r0 -} - -// MockEnvLoader_CreateEnvFile_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'CreateEnvFile' -type MockEnvLoader_CreateEnvFile_Call struct { - *mock.Call -} - -// CreateEnvFile is a helper method to define mock.On call -// - ctx context.Context -// - domain string -func (_e *MockEnvLoader_Expecter) CreateEnvFile(ctx any, domain any) *MockEnvLoader_CreateEnvFile_Call { - return &MockEnvLoader_CreateEnvFile_Call{Call: _e.mock.On("CreateEnvFile", ctx, domain)} -} - -func (_c *MockEnvLoader_CreateEnvFile_Call) Run(run func(ctx context.Context, domain string)) *MockEnvLoader_CreateEnvFile_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockEnvLoader_CreateEnvFile_Call) Return(err error) *MockEnvLoader_CreateEnvFile_Call { - _c.Call.Return(err) - return _c -} - -func (_c *MockEnvLoader_CreateEnvFile_Call) RunAndReturn(run func(ctx context.Context, domain string) error) *MockEnvLoader_CreateEnvFile_Call { - _c.Call.Return(run) - return _c -} - -// EnvFileExists provides a mock function for the type MockEnvLoader -func (_mock *MockEnvLoader) EnvFileExists(domain string) (bool, error) { - ret := _mock.Called(domain) - - if len(ret) == 0 { - panic("no return value specified for EnvFileExists") - } - - var r0 bool - var r1 error - if returnFunc, ok := ret.Get(0).(func(string) (bool, error)); ok { - return returnFunc(domain) - } - if returnFunc, ok := ret.Get(0).(func(string) bool); ok { - r0 = returnFunc(domain) - } else { - r0 = ret.Get(0).(bool) - } - if returnFunc, ok := ret.Get(1).(func(string) error); ok { - r1 = returnFunc(domain) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockEnvLoader_EnvFileExists_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'EnvFileExists' -type MockEnvLoader_EnvFileExists_Call struct { - *mock.Call -} - -// EnvFileExists is a helper method to define mock.On call -// - domain string -func (_e *MockEnvLoader_Expecter) EnvFileExists(domain any) *MockEnvLoader_EnvFileExists_Call { - return &MockEnvLoader_EnvFileExists_Call{Call: _e.mock.On("EnvFileExists", domain)} -} - -func (_c *MockEnvLoader_EnvFileExists_Call) Run(run func(domain string)) *MockEnvLoader_EnvFileExists_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 string - if args[0] != nil { - arg0 = args[0].(string) - } - run( - arg0, - ) - }) - return _c -} - -func (_c *MockEnvLoader_EnvFileExists_Call) Return(b bool, err error) *MockEnvLoader_EnvFileExists_Call { - _c.Call.Return(b, err) - return _c -} - -func (_c *MockEnvLoader_EnvFileExists_Call) RunAndReturn(run func(domain string) (bool, error)) *MockEnvLoader_EnvFileExists_Call { - _c.Call.Return(run) - return _c -} - -// LoadEnv provides a mock function for the type MockEnvLoader -func (_mock *MockEnvLoader) LoadEnv(ctx context.Context, domain string) ([]string, error) { - ret := _mock.Called(ctx, domain) - - if len(ret) == 0 { - panic("no return value specified for LoadEnv") - } - - var r0 []string - var r1 error - if returnFunc, ok := ret.Get(0).(func(context.Context, string) ([]string, error)); ok { - return returnFunc(ctx, domain) - } - if returnFunc, ok := ret.Get(0).(func(context.Context, string) []string); ok { - r0 = returnFunc(ctx, domain) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]string) - } - } - if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { - r1 = returnFunc(ctx, domain) - } else { - r1 = ret.Error(1) - } - return r0, r1 -} - -// MockEnvLoader_LoadEnv_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'LoadEnv' -type MockEnvLoader_LoadEnv_Call struct { - *mock.Call -} - -// LoadEnv is a helper method to define mock.On call -// - ctx context.Context -// - domain string -func (_e *MockEnvLoader_Expecter) LoadEnv(ctx any, domain any) *MockEnvLoader_LoadEnv_Call { - return &MockEnvLoader_LoadEnv_Call{Call: _e.mock.On("LoadEnv", ctx, domain)} -} - -func (_c *MockEnvLoader_LoadEnv_Call) Run(run func(ctx context.Context, domain string)) *MockEnvLoader_LoadEnv_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run( - arg0, - arg1, - ) - }) - return _c -} - -func (_c *MockEnvLoader_LoadEnv_Call) Return(strings []string, err error) *MockEnvLoader_LoadEnv_Call { - _c.Call.Return(strings, err) - return _c -} - -func (_c *MockEnvLoader_LoadEnv_Call) RunAndReturn(run func(ctx context.Context, domain string) ([]string, error)) *MockEnvLoader_LoadEnv_Call { - _c.Call.Return(run) - return _c -} diff --git a/internal/boundaries/out/mocks/mock_event_bus.go b/internal/boundaries/out/mocks/mock_event_bus.go index b1c2b12cd..f206262b9 100644 --- a/internal/boundaries/out/mocks/mock_event_bus.go +++ b/internal/boundaries/out/mocks/mock_event_bus.go @@ -16,10 +16,19 @@ func NewMockEventBus(t interface { mock.TestingT Cleanup(func()) }) *MockEventBus { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockEventBus{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/out/mocks/mock_event_handler.go b/internal/boundaries/out/mocks/mock_event_handler.go index becf1b820..0712ff687 100644 --- a/internal/boundaries/out/mocks/mock_event_handler.go +++ b/internal/boundaries/out/mocks/mock_event_handler.go @@ -17,10 +17,19 @@ func NewMockEventHandler(t interface { mock.TestingT Cleanup(func()) }) *MockEventHandler { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockEventHandler{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/out/mocks/mock_event_publisher.go b/internal/boundaries/out/mocks/mock_event_publisher.go index b5f45115d..5a4cc7a11 100644 --- a/internal/boundaries/out/mocks/mock_event_publisher.go +++ b/internal/boundaries/out/mocks/mock_event_publisher.go @@ -15,10 +15,19 @@ func NewMockEventPublisher(t interface { mock.TestingT Cleanup(func()) }) *MockEventPublisher { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockEventPublisher{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/out/mocks/mock_event_subscriber.go b/internal/boundaries/out/mocks/mock_event_subscriber.go index ee9576e66..3c7759d09 100644 --- a/internal/boundaries/out/mocks/mock_event_subscriber.go +++ b/internal/boundaries/out/mocks/mock_event_subscriber.go @@ -15,10 +15,19 @@ func NewMockEventSubscriber(t interface { mock.TestingT Cleanup(func()) }) *MockEventSubscriber { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockEventSubscriber{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/out/mocks/mock_gc_barrier.go b/internal/boundaries/out/mocks/mock_gc_barrier.go new file mode 100644 index 000000000..75b2542f0 --- /dev/null +++ b/internal/boundaries/out/mocks/mock_gc_barrier.go @@ -0,0 +1,172 @@ +// Code generated by mockery; DO NOT EDIT. +// github.com/vektra/mockery +// template: testify + +package mocks + +import ( + "context" + + "github.com/bnema/gordon/internal/boundaries/out" + mock "github.com/stretchr/testify/mock" +) + +// NewMockGCBarrier creates a new instance of MockGCBarrier. It also registers a testing interface on the mock and a cleanup function to assert the mocks expectations. +// The first argument is typically a *testing.T value. +func NewMockGCBarrier(t interface { + mock.TestingT + Cleanup(func()) +}) *MockGCBarrier { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + + mock := &MockGCBarrier{} + mock.Mock.Test(t) + + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) + + return mock +} + +// MockGCBarrier is an autogenerated mock type for the GCBarrier type +type MockGCBarrier struct { + mock.Mock +} + +type MockGCBarrier_Expecter struct { + mock *mock.Mock +} + +func (_m *MockGCBarrier) EXPECT() *MockGCBarrier_Expecter { + return &MockGCBarrier_Expecter{mock: &_m.Mock} +} + +// AcquireExclusive provides a mock function for the type MockGCBarrier +func (_mock *MockGCBarrier) AcquireExclusive(ctx context.Context) (out.GCLease, error) { + ret := _mock.Called(ctx) + + if len(ret) == 0 { + panic("no return value specified for AcquireExclusive") + } + + var r0 out.GCLease + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context) (out.GCLease, error)); ok { + return returnFunc(ctx) + } + if returnFunc, ok := ret.Get(0).(func(context.Context) out.GCLease); ok { + r0 = returnFunc(ctx) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).(out.GCLease) + } + } + if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { + r1 = returnFunc(ctx) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockGCBarrier_AcquireExclusive_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'AcquireExclusive' +type MockGCBarrier_AcquireExclusive_Call struct { + *mock.Call +} + +// AcquireExclusive is a helper method to define mock.On call +// - ctx context.Context +func (_e *MockGCBarrier_Expecter) AcquireExclusive(ctx any) *MockGCBarrier_AcquireExclusive_Call { + return &MockGCBarrier_AcquireExclusive_Call{Call: _e.mock.On("AcquireExclusive", ctx)} +} + +func (_c *MockGCBarrier_AcquireExclusive_Call) Run(run func(ctx context.Context)) *MockGCBarrier_AcquireExclusive_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + run( + arg0, + ) + }) + return _c +} + +func (_c *MockGCBarrier_AcquireExclusive_Call) Return(gCLease out.GCLease, err error) *MockGCBarrier_AcquireExclusive_Call { + _c.Call.Return(gCLease, err) + return _c +} + +func (_c *MockGCBarrier_AcquireExclusive_Call) RunAndReturn(run func(ctx context.Context) (out.GCLease, error)) *MockGCBarrier_AcquireExclusive_Call { + _c.Call.Return(run) + return _c +} + +// AcquireShared provides a mock function for the type MockGCBarrier +func (_mock *MockGCBarrier) AcquireShared(ctx context.Context) (out.GCLease, error) { + ret := _mock.Called(ctx) + + if len(ret) == 0 { + panic("no return value specified for AcquireShared") + } + + var r0 out.GCLease + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context) (out.GCLease, error)); ok { + return returnFunc(ctx) + } + if returnFunc, ok := ret.Get(0).(func(context.Context) out.GCLease); ok { + r0 = returnFunc(ctx) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).(out.GCLease) + } + } + if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { + r1 = returnFunc(ctx) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockGCBarrier_AcquireShared_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'AcquireShared' +type MockGCBarrier_AcquireShared_Call struct { + *mock.Call +} + +// AcquireShared is a helper method to define mock.On call +// - ctx context.Context +func (_e *MockGCBarrier_Expecter) AcquireShared(ctx any) *MockGCBarrier_AcquireShared_Call { + return &MockGCBarrier_AcquireShared_Call{Call: _e.mock.On("AcquireShared", ctx)} +} + +func (_c *MockGCBarrier_AcquireShared_Call) Run(run func(ctx context.Context)) *MockGCBarrier_AcquireShared_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + run( + arg0, + ) + }) + return _c +} + +func (_c *MockGCBarrier_AcquireShared_Call) Return(gCLease out.GCLease, err error) *MockGCBarrier_AcquireShared_Call { + _c.Call.Return(gCLease, err) + return _c +} + +func (_c *MockGCBarrier_AcquireShared_Call) RunAndReturn(run func(ctx context.Context) (out.GCLease, error)) *MockGCBarrier_AcquireShared_Call { + _c.Call.Return(run) + return _c +} diff --git a/internal/boundaries/out/mocks/mock_gc_lease.go b/internal/boundaries/out/mocks/mock_gc_lease.go new file mode 100644 index 000000000..109bf2cfc --- /dev/null +++ b/internal/boundaries/out/mocks/mock_gc_lease.go @@ -0,0 +1,78 @@ +// Code generated by mockery; DO NOT EDIT. +// github.com/vektra/mockery +// template: testify + +package mocks + +import ( + mock "github.com/stretchr/testify/mock" +) + +// NewMockGCLease creates a new instance of MockGCLease. It also registers a testing interface on the mock and a cleanup function to assert the mocks expectations. +// The first argument is typically a *testing.T value. +func NewMockGCLease(t interface { + mock.TestingT + Cleanup(func()) +}) *MockGCLease { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + + mock := &MockGCLease{} + mock.Mock.Test(t) + + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) + + return mock +} + +// MockGCLease is an autogenerated mock type for the GCLease type +type MockGCLease struct { + mock.Mock +} + +type MockGCLease_Expecter struct { + mock *mock.Mock +} + +func (_m *MockGCLease) EXPECT() *MockGCLease_Expecter { + return &MockGCLease_Expecter{mock: &_m.Mock} +} + +// Release provides a mock function for the type MockGCLease +func (_mock *MockGCLease) Release() { + _mock.Called() + return +} + +// MockGCLease_Release_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Release' +type MockGCLease_Release_Call struct { + *mock.Call +} + +// Release is a helper method to define mock.On call +func (_e *MockGCLease_Expecter) Release() *MockGCLease_Release_Call { + return &MockGCLease_Release_Call{Call: _e.mock.On("Release")} +} + +func (_c *MockGCLease_Release_Call) Run(run func()) *MockGCLease_Release_Call { + _c.Call.Run(func(args mock.Arguments) { + run() + }) + return _c +} + +func (_c *MockGCLease_Release_Call) Return() *MockGCLease_Release_Call { + _c.Call.Return() + return _c +} + +func (_c *MockGCLease_Release_Call) RunAndReturn(run func()) *MockGCLease_Release_Call { + _c.Run(run) + return _c +} diff --git a/internal/boundaries/out/mocks/mock_http_challenge_sink.go b/internal/boundaries/out/mocks/mock_http_challenge_sink.go index c1ef22bbe..eacd4f20c 100644 --- a/internal/boundaries/out/mocks/mock_http_challenge_sink.go +++ b/internal/boundaries/out/mocks/mock_http_challenge_sink.go @@ -14,10 +14,19 @@ func NewMockHTTPChallengeSink(t interface { mock.TestingT Cleanup(func()) }) *MockHTTPChallengeSink { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockHTTPChallengeSink{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/out/mocks/mock_image_resolver.go b/internal/boundaries/out/mocks/mock_image_resolver.go new file mode 100644 index 000000000..31dcb6e0c --- /dev/null +++ b/internal/boundaries/out/mocks/mock_image_resolver.go @@ -0,0 +1,113 @@ +// Code generated by mockery; DO NOT EDIT. +// github.com/vektra/mockery +// template: testify + +package mocks + +import ( + "context" + + mock "github.com/stretchr/testify/mock" +) + +// NewMockImageResolver creates a new instance of MockImageResolver. It also registers a testing interface on the mock and a cleanup function to assert the mocks expectations. +// The first argument is typically a *testing.T value. +func NewMockImageResolver(t interface { + mock.TestingT + Cleanup(func()) +}) *MockImageResolver { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + + mock := &MockImageResolver{} + mock.Mock.Test(t) + + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) + + return mock +} + +// MockImageResolver is an autogenerated mock type for the ImageResolver type +type MockImageResolver struct { + mock.Mock +} + +type MockImageResolver_Expecter struct { + mock *mock.Mock +} + +func (_m *MockImageResolver) EXPECT() *MockImageResolver_Expecter { + return &MockImageResolver_Expecter{mock: &_m.Mock} +} + +// ResolveDigest provides a mock function for the type MockImageResolver +func (_mock *MockImageResolver) ResolveDigest(ctx context.Context, ref string) (string, error) { + ret := _mock.Called(ctx, ref) + + if len(ret) == 0 { + panic("no return value specified for ResolveDigest") + } + + var r0 string + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string) (string, error)); ok { + return returnFunc(ctx, ref) + } + if returnFunc, ok := ret.Get(0).(func(context.Context, string) string); ok { + r0 = returnFunc(ctx, ref) + } else { + r0 = ret.Get(0).(string) + } + if returnFunc, ok := ret.Get(1).(func(context.Context, string) error); ok { + r1 = returnFunc(ctx, ref) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockImageResolver_ResolveDigest_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ResolveDigest' +type MockImageResolver_ResolveDigest_Call struct { + *mock.Call +} + +// ResolveDigest is a helper method to define mock.On call +// - ctx context.Context +// - ref string +func (_e *MockImageResolver_Expecter) ResolveDigest(ctx any, ref any) *MockImageResolver_ResolveDigest_Call { + return &MockImageResolver_ResolveDigest_Call{Call: _e.mock.On("ResolveDigest", ctx, ref)} +} + +func (_c *MockImageResolver_ResolveDigest_Call) Run(run func(ctx context.Context, ref string)) *MockImageResolver_ResolveDigest_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockImageResolver_ResolveDigest_Call) Return(s string, err error) *MockImageResolver_ResolveDigest_Call { + _c.Call.Return(s, err) + return _c +} + +func (_c *MockImageResolver_ResolveDigest_Call) RunAndReturn(run func(ctx context.Context, ref string) (string, error)) *MockImageResolver_ResolveDigest_Call { + _c.Call.Return(run) + return _c +} diff --git a/internal/boundaries/out/mocks/mock_log_exporter.go b/internal/boundaries/out/mocks/mock_log_exporter.go new file mode 100644 index 000000000..110d5e1bc --- /dev/null +++ b/internal/boundaries/out/mocks/mock_log_exporter.go @@ -0,0 +1,94 @@ +// Code generated by mockery; DO NOT EDIT. +// github.com/vektra/mockery +// template: testify + +package mocks + +import ( + "context" + + "github.com/bnema/gordon/internal/domain" + mock "github.com/stretchr/testify/mock" +) + +// NewMockLogExporter creates a new instance of MockLogExporter. It also registers a testing interface on the mock and a cleanup function to assert the mocks expectations. +// The first argument is typically a *testing.T value. +func NewMockLogExporter(t interface { + mock.TestingT + Cleanup(func()) +}) *MockLogExporter { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + + mock := &MockLogExporter{} + mock.Mock.Test(t) + + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) + + return mock +} + +// MockLogExporter is an autogenerated mock type for the LogExporter type +type MockLogExporter struct { + mock.Mock +} + +type MockLogExporter_Expecter struct { + mock *mock.Mock +} + +func (_m *MockLogExporter) EXPECT() *MockLogExporter_Expecter { + return &MockLogExporter_Expecter{mock: &_m.Mock} +} + +// Export provides a mock function for the type MockLogExporter +func (_mock *MockLogExporter) Export(ctx context.Context, record domain.LogRecord) { + _mock.Called(ctx, record) + return +} + +// MockLogExporter_Export_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'Export' +type MockLogExporter_Export_Call struct { + *mock.Call +} + +// Export is a helper method to define mock.On call +// - ctx context.Context +// - record domain.LogRecord +func (_e *MockLogExporter_Expecter) Export(ctx any, record any) *MockLogExporter_Export_Call { + return &MockLogExporter_Export_Call{Call: _e.mock.On("Export", ctx, record)} +} + +func (_c *MockLogExporter_Export_Call) Run(run func(ctx context.Context, record domain.LogRecord)) *MockLogExporter_Export_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 domain.LogRecord + if args[1] != nil { + arg1 = args[1].(domain.LogRecord) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockLogExporter_Export_Call) Return() *MockLogExporter_Export_Call { + _c.Call.Return() + return _c +} + +func (_c *MockLogExporter_Export_Call) RunAndReturn(run func(ctx context.Context, record domain.LogRecord)) *MockLogExporter_Export_Call { + _c.Run(run) + return _c +} diff --git a/internal/boundaries/out/mocks/mock_manifest_storage.go b/internal/boundaries/out/mocks/mock_manifest_storage.go index f6aee0aad..8857ea0b3 100644 --- a/internal/boundaries/out/mocks/mock_manifest_storage.go +++ b/internal/boundaries/out/mocks/mock_manifest_storage.go @@ -16,10 +16,19 @@ func NewMockManifestStorage(t interface { mock.TestingT Cleanup(func()) }) *MockManifestStorage { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockManifestStorage{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/out/mocks/mock_metrics.go b/internal/boundaries/out/mocks/mock_metrics.go new file mode 100644 index 000000000..b22adb88a --- /dev/null +++ b/internal/boundaries/out/mocks/mock_metrics.go @@ -0,0 +1,198 @@ +// Code generated by mockery; DO NOT EDIT. +// github.com/vektra/mockery +// template: testify + +package mocks + +import ( + "context" + + "github.com/bnema/gordon/internal/domain" + mock "github.com/stretchr/testify/mock" +) + +// NewMockMetrics creates a new instance of MockMetrics. It also registers a testing interface on the mock and a cleanup function to assert the mocks expectations. +// The first argument is typically a *testing.T value. +func NewMockMetrics(t interface { + mock.TestingT + Cleanup(func()) +}) *MockMetrics { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + + mock := &MockMetrics{} + mock.Mock.Test(t) + + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) + + return mock +} + +// MockMetrics is an autogenerated mock type for the Metrics type +type MockMetrics struct { + mock.Mock +} + +type MockMetrics_Expecter struct { + mock *mock.Mock +} + +func (_m *MockMetrics) EXPECT() *MockMetrics_Expecter { + return &MockMetrics_Expecter{mock: &_m.Mock} +} + +// RecordEventDropped provides a mock function for the type MockMetrics +func (_mock *MockMetrics) RecordEventDropped(ctx context.Context, eventType domain.EventType) { + _mock.Called(ctx, eventType) + return +} + +// MockMetrics_RecordEventDropped_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'RecordEventDropped' +type MockMetrics_RecordEventDropped_Call struct { + *mock.Call +} + +// RecordEventDropped is a helper method to define mock.On call +// - ctx context.Context +// - eventType domain.EventType +func (_e *MockMetrics_Expecter) RecordEventDropped(ctx any, eventType any) *MockMetrics_RecordEventDropped_Call { + return &MockMetrics_RecordEventDropped_Call{Call: _e.mock.On("RecordEventDropped", ctx, eventType)} +} + +func (_c *MockMetrics_RecordEventDropped_Call) Run(run func(ctx context.Context, eventType domain.EventType)) *MockMetrics_RecordEventDropped_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 domain.EventType + if args[1] != nil { + arg1 = args[1].(domain.EventType) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockMetrics_RecordEventDropped_Call) Return() *MockMetrics_RecordEventDropped_Call { + _c.Call.Return() + return _c +} + +func (_c *MockMetrics_RecordEventDropped_Call) RunAndReturn(run func(ctx context.Context, eventType domain.EventType)) *MockMetrics_RecordEventDropped_Call { + _c.Run(run) + return _c +} + +// RecordEventProcessed provides a mock function for the type MockMetrics +func (_mock *MockMetrics) RecordEventProcessed(ctx context.Context, eventType domain.EventType) { + _mock.Called(ctx, eventType) + return +} + +// MockMetrics_RecordEventProcessed_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'RecordEventProcessed' +type MockMetrics_RecordEventProcessed_Call struct { + *mock.Call +} + +// RecordEventProcessed is a helper method to define mock.On call +// - ctx context.Context +// - eventType domain.EventType +func (_e *MockMetrics_Expecter) RecordEventProcessed(ctx any, eventType any) *MockMetrics_RecordEventProcessed_Call { + return &MockMetrics_RecordEventProcessed_Call{Call: _e.mock.On("RecordEventProcessed", ctx, eventType)} +} + +func (_c *MockMetrics_RecordEventProcessed_Call) Run(run func(ctx context.Context, eventType domain.EventType)) *MockMetrics_RecordEventProcessed_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 domain.EventType + if args[1] != nil { + arg1 = args[1].(domain.EventType) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockMetrics_RecordEventProcessed_Call) Return() *MockMetrics_RecordEventProcessed_Call { + _c.Call.Return() + return _c +} + +func (_c *MockMetrics_RecordEventProcessed_Call) RunAndReturn(run func(ctx context.Context, eventType domain.EventType)) *MockMetrics_RecordEventProcessed_Call { + _c.Run(run) + return _c +} + +// RecordImagePush provides a mock function for the type MockMetrics +func (_mock *MockMetrics) RecordImagePush(ctx context.Context, name string, reference string, sizeBytes int64) { + _mock.Called(ctx, name, reference, sizeBytes) + return +} + +// MockMetrics_RecordImagePush_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'RecordImagePush' +type MockMetrics_RecordImagePush_Call struct { + *mock.Call +} + +// RecordImagePush is a helper method to define mock.On call +// - ctx context.Context +// - name string +// - reference string +// - sizeBytes int64 +func (_e *MockMetrics_Expecter) RecordImagePush(ctx any, name any, reference any, sizeBytes any) *MockMetrics_RecordImagePush_Call { + return &MockMetrics_RecordImagePush_Call{Call: _e.mock.On("RecordImagePush", ctx, name, reference, sizeBytes)} +} + +func (_c *MockMetrics_RecordImagePush_Call) Run(run func(ctx context.Context, name string, reference string, sizeBytes int64)) *MockMetrics_RecordImagePush_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } + var arg3 int64 + if args[3] != nil { + arg3 = args[3].(int64) + } + run( + arg0, + arg1, + arg2, + arg3, + ) + }) + return _c +} + +func (_c *MockMetrics_RecordImagePush_Call) Return() *MockMetrics_RecordImagePush_Call { + _c.Call.Return() + return _c +} + +func (_c *MockMetrics_RecordImagePush_Call) RunAndReturn(run func(ctx context.Context, name string, reference string, sizeBytes int64)) *MockMetrics_RecordImagePush_Call { + _c.Run(run) + return _c +} diff --git a/internal/boundaries/out/mocks/mock_proxy_cache_invalidator.go b/internal/boundaries/out/mocks/mock_proxy_cache_invalidator.go deleted file mode 100644 index 7c3db5b4d..000000000 --- a/internal/boundaries/out/mocks/mock_proxy_cache_invalidator.go +++ /dev/null @@ -1,80 +0,0 @@ -// Code generated by mockery; DO NOT EDIT. -// github.com/vektra/mockery -// template: testify - -package mocks - -import ( - "context" - - mock "github.com/stretchr/testify/mock" -) - -// NewMockProxyCacheInvalidator creates a new instance of MockProxyCacheInvalidator. It also registers a testing interface on the mock and a cleanup function to assert the mocks expectations. -// The first argument is typically a *testing.T value. -func NewMockProxyCacheInvalidator(t interface { - mock.TestingT - Cleanup(func()) -}) *MockProxyCacheInvalidator { - mock := &MockProxyCacheInvalidator{} - mock.Mock.Test(t) - - t.Cleanup(func() { mock.AssertExpectations(t) }) - - return mock -} - -// MockProxyCacheInvalidator is an autogenerated mock type for the ProxyCacheInvalidator type -type MockProxyCacheInvalidator struct { - mock.Mock -} - -type MockProxyCacheInvalidator_Expecter struct { - mock *mock.Mock -} - -func (_m *MockProxyCacheInvalidator) EXPECT() *MockProxyCacheInvalidator_Expecter { - return &MockProxyCacheInvalidator_Expecter{mock: &_m.Mock} -} - -// InvalidateTarget provides a mock function for the type MockProxyCacheInvalidator -func (_mock *MockProxyCacheInvalidator) InvalidateTarget(ctx context.Context, domainName string) { - _mock.Called(ctx, domainName) -} - -// MockProxyCacheInvalidator_InvalidateTarget_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'InvalidateTarget' -type MockProxyCacheInvalidator_InvalidateTarget_Call struct { - *mock.Call -} - -// InvalidateTarget is a helper method to define mock.On call -// - ctx context.Context -// - domainName string -func (_e *MockProxyCacheInvalidator_Expecter) InvalidateTarget(ctx any, domainName any) *MockProxyCacheInvalidator_InvalidateTarget_Call { - return &MockProxyCacheInvalidator_InvalidateTarget_Call{Call: _e.mock.On("InvalidateTarget", ctx, domainName)} -} - -func (_c *MockProxyCacheInvalidator_InvalidateTarget_Call) Run(run func(ctx context.Context, domainName string)) *MockProxyCacheInvalidator_InvalidateTarget_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - var arg1 string - if args[1] != nil { - arg1 = args[1].(string) - } - run(arg0, arg1) - }) - return _c -} - -func (_c *MockProxyCacheInvalidator_InvalidateTarget_Call) Return() *MockProxyCacheInvalidator_InvalidateTarget_Call { - _c.Call.Return() - return _c -} - -func (_c *MockProxyCacheInvalidator_InvalidateTarget_Call) RunAndReturn(run func(ctx context.Context, domainName string)) *MockProxyCacheInvalidator_InvalidateTarget_Call { - _c.Run(run) - return _c -} diff --git a/internal/boundaries/out/mocks/mock_prune_protection_store.go b/internal/boundaries/out/mocks/mock_prune_protection_store.go new file mode 100644 index 000000000..c335cdb2e --- /dev/null +++ b/internal/boundaries/out/mocks/mock_prune_protection_store.go @@ -0,0 +1,110 @@ +// Code generated by mockery; DO NOT EDIT. +// github.com/vektra/mockery +// template: testify + +package mocks + +import ( + "context" + + "github.com/bnema/gordon/internal/domain" + mock "github.com/stretchr/testify/mock" +) + +// NewMockPruneProtectionStore creates a new instance of MockPruneProtectionStore. It also registers a testing interface on the mock and a cleanup function to assert the mocks expectations. +// The first argument is typically a *testing.T value. +func NewMockPruneProtectionStore(t interface { + mock.TestingT + Cleanup(func()) +}) *MockPruneProtectionStore { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + + mock := &MockPruneProtectionStore{} + mock.Mock.Test(t) + + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) + + return mock +} + +// MockPruneProtectionStore is an autogenerated mock type for the PruneProtectionStore type +type MockPruneProtectionStore struct { + mock.Mock +} + +type MockPruneProtectionStore_Expecter struct { + mock *mock.Mock +} + +func (_m *MockPruneProtectionStore) EXPECT() *MockPruneProtectionStore_Expecter { + return &MockPruneProtectionStore_Expecter{mock: &_m.Mock} +} + +// ProtectionSnapshot provides a mock function for the type MockPruneProtectionStore +func (_mock *MockPruneProtectionStore) ProtectionSnapshot(ctx context.Context) (*domain.PruneProtectionSnapshot, error) { + ret := _mock.Called(ctx) + + if len(ret) == 0 { + panic("no return value specified for ProtectionSnapshot") + } + + var r0 *domain.PruneProtectionSnapshot + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context) (*domain.PruneProtectionSnapshot, error)); ok { + return returnFunc(ctx) + } + if returnFunc, ok := ret.Get(0).(func(context.Context) *domain.PruneProtectionSnapshot); ok { + r0 = returnFunc(ctx) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).(*domain.PruneProtectionSnapshot) + } + } + if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { + r1 = returnFunc(ctx) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockPruneProtectionStore_ProtectionSnapshot_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'ProtectionSnapshot' +type MockPruneProtectionStore_ProtectionSnapshot_Call struct { + *mock.Call +} + +// ProtectionSnapshot is a helper method to define mock.On call +// - ctx context.Context +func (_e *MockPruneProtectionStore_Expecter) ProtectionSnapshot(ctx any) *MockPruneProtectionStore_ProtectionSnapshot_Call { + return &MockPruneProtectionStore_ProtectionSnapshot_Call{Call: _e.mock.On("ProtectionSnapshot", ctx)} +} + +func (_c *MockPruneProtectionStore_ProtectionSnapshot_Call) Run(run func(ctx context.Context)) *MockPruneProtectionStore_ProtectionSnapshot_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + run( + arg0, + ) + }) + return _c +} + +func (_c *MockPruneProtectionStore_ProtectionSnapshot_Call) Return(pruneProtectionSnapshot *domain.PruneProtectionSnapshot, err error) *MockPruneProtectionStore_ProtectionSnapshot_Call { + _c.Call.Return(pruneProtectionSnapshot, err) + return _c +} + +func (_c *MockPruneProtectionStore_ProtectionSnapshot_Call) RunAndReturn(run func(ctx context.Context) (*domain.PruneProtectionSnapshot, error)) *MockPruneProtectionStore_ProtectionSnapshot_Call { + _c.Call.Return(run) + return _c +} diff --git a/internal/boundaries/out/mocks/mock_prune_runtime.go b/internal/boundaries/out/mocks/mock_prune_runtime.go new file mode 100644 index 000000000..75dbd97ca --- /dev/null +++ b/internal/boundaries/out/mocks/mock_prune_runtime.go @@ -0,0 +1,224 @@ +// Code generated by mockery; DO NOT EDIT. +// github.com/vektra/mockery +// template: testify + +package mocks + +import ( + "context" + + "github.com/bnema/gordon/internal/domain" + mock "github.com/stretchr/testify/mock" +) + +// NewMockPruneRuntime creates a new instance of MockPruneRuntime. It also registers a testing interface on the mock and a cleanup function to assert the mocks expectations. +// The first argument is typically a *testing.T value. +func NewMockPruneRuntime(t interface { + mock.TestingT + Cleanup(func()) +}) *MockPruneRuntime { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + + mock := &MockPruneRuntime{} + mock.Mock.Test(t) + + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) + + return mock +} + +// MockPruneRuntime is an autogenerated mock type for the PruneRuntime type +type MockPruneRuntime struct { + mock.Mock +} + +type MockPruneRuntime_Expecter struct { + mock *mock.Mock +} + +func (_m *MockPruneRuntime) EXPECT() *MockPruneRuntime_Expecter { + return &MockPruneRuntime_Expecter{mock: &_m.Mock} +} + +// InventoryRuntime provides a mock function for the type MockPruneRuntime +func (_mock *MockPruneRuntime) InventoryRuntime(ctx context.Context) (*domain.RuntimeInventory, error) { + ret := _mock.Called(ctx) + + if len(ret) == 0 { + panic("no return value specified for InventoryRuntime") + } + + var r0 *domain.RuntimeInventory + var r1 error + if returnFunc, ok := ret.Get(0).(func(context.Context) (*domain.RuntimeInventory, error)); ok { + return returnFunc(ctx) + } + if returnFunc, ok := ret.Get(0).(func(context.Context) *domain.RuntimeInventory); ok { + r0 = returnFunc(ctx) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).(*domain.RuntimeInventory) + } + } + if returnFunc, ok := ret.Get(1).(func(context.Context) error); ok { + r1 = returnFunc(ctx) + } else { + r1 = ret.Error(1) + } + return r0, r1 +} + +// MockPruneRuntime_InventoryRuntime_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'InventoryRuntime' +type MockPruneRuntime_InventoryRuntime_Call struct { + *mock.Call +} + +// InventoryRuntime is a helper method to define mock.On call +// - ctx context.Context +func (_e *MockPruneRuntime_Expecter) InventoryRuntime(ctx any) *MockPruneRuntime_InventoryRuntime_Call { + return &MockPruneRuntime_InventoryRuntime_Call{Call: _e.mock.On("InventoryRuntime", ctx)} +} + +func (_c *MockPruneRuntime_InventoryRuntime_Call) Run(run func(ctx context.Context)) *MockPruneRuntime_InventoryRuntime_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + run( + arg0, + ) + }) + return _c +} + +func (_c *MockPruneRuntime_InventoryRuntime_Call) Return(runtimeInventory *domain.RuntimeInventory, err error) *MockPruneRuntime_InventoryRuntime_Call { + _c.Call.Return(runtimeInventory, err) + return _c +} + +func (_c *MockPruneRuntime_InventoryRuntime_Call) RunAndReturn(run func(ctx context.Context) (*domain.RuntimeInventory, error)) *MockPruneRuntime_InventoryRuntime_Call { + _c.Call.Return(run) + return _c +} + +// RemoveImageExact provides a mock function for the type MockPruneRuntime +func (_mock *MockPruneRuntime) RemoveImageExact(ctx context.Context, ref domain.RuntimeImageRef) error { + ret := _mock.Called(ctx, ref) + + if len(ret) == 0 { + panic("no return value specified for RemoveImageExact") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context, domain.RuntimeImageRef) error); ok { + r0 = returnFunc(ctx, ref) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockPruneRuntime_RemoveImageExact_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'RemoveImageExact' +type MockPruneRuntime_RemoveImageExact_Call struct { + *mock.Call +} + +// RemoveImageExact is a helper method to define mock.On call +// - ctx context.Context +// - ref domain.RuntimeImageRef +func (_e *MockPruneRuntime_Expecter) RemoveImageExact(ctx any, ref any) *MockPruneRuntime_RemoveImageExact_Call { + return &MockPruneRuntime_RemoveImageExact_Call{Call: _e.mock.On("RemoveImageExact", ctx, ref)} +} + +func (_c *MockPruneRuntime_RemoveImageExact_Call) Run(run func(ctx context.Context, ref domain.RuntimeImageRef)) *MockPruneRuntime_RemoveImageExact_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 domain.RuntimeImageRef + if args[1] != nil { + arg1 = args[1].(domain.RuntimeImageRef) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockPruneRuntime_RemoveImageExact_Call) Return(err error) *MockPruneRuntime_RemoveImageExact_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockPruneRuntime_RemoveImageExact_Call) RunAndReturn(run func(ctx context.Context, ref domain.RuntimeImageRef) error) *MockPruneRuntime_RemoveImageExact_Call { + _c.Call.Return(run) + return _c +} + +// RemoveVolumeExact provides a mock function for the type MockPruneRuntime +func (_mock *MockPruneRuntime) RemoveVolumeExact(ctx context.Context, ref domain.RuntimeVolumeRef) error { + ret := _mock.Called(ctx, ref) + + if len(ret) == 0 { + panic("no return value specified for RemoveVolumeExact") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context, domain.RuntimeVolumeRef) error); ok { + r0 = returnFunc(ctx, ref) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockPruneRuntime_RemoveVolumeExact_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'RemoveVolumeExact' +type MockPruneRuntime_RemoveVolumeExact_Call struct { + *mock.Call +} + +// RemoveVolumeExact is a helper method to define mock.On call +// - ctx context.Context +// - ref domain.RuntimeVolumeRef +func (_e *MockPruneRuntime_Expecter) RemoveVolumeExact(ctx any, ref any) *MockPruneRuntime_RemoveVolumeExact_Call { + return &MockPruneRuntime_RemoveVolumeExact_Call{Call: _e.mock.On("RemoveVolumeExact", ctx, ref)} +} + +func (_c *MockPruneRuntime_RemoveVolumeExact_Call) Run(run func(ctx context.Context, ref domain.RuntimeVolumeRef)) *MockPruneRuntime_RemoveVolumeExact_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 domain.RuntimeVolumeRef + if args[1] != nil { + arg1 = args[1].(domain.RuntimeVolumeRef) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockPruneRuntime_RemoveVolumeExact_Call) Return(err error) *MockPruneRuntime_RemoveVolumeExact_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockPruneRuntime_RemoveVolumeExact_Call) RunAndReturn(run func(ctx context.Context, ref domain.RuntimeVolumeRef) error) *MockPruneRuntime_RemoveVolumeExact_Call { + _c.Call.Return(run) + return _c +} diff --git a/internal/boundaries/out/mocks/mock_public_certificate_issuer.go b/internal/boundaries/out/mocks/mock_public_certificate_issuer.go index 6b737ae55..021dee899 100644 --- a/internal/boundaries/out/mocks/mock_public_certificate_issuer.go +++ b/internal/boundaries/out/mocks/mock_public_certificate_issuer.go @@ -17,10 +17,19 @@ func NewMockPublicCertificateIssuer(t interface { mock.TestingT Cleanup(func()) }) *MockPublicCertificateIssuer { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockPublicCertificateIssuer{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/out/mocks/mock_rate_limiter.go b/internal/boundaries/out/mocks/mock_rate_limiter.go index a6d349773..38a3b4817 100644 --- a/internal/boundaries/out/mocks/mock_rate_limiter.go +++ b/internal/boundaries/out/mocks/mock_rate_limiter.go @@ -16,10 +16,19 @@ func NewMockRateLimiter(t interface { mock.TestingT Cleanup(func()) }) *MockRateLimiter { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockRateLimiter{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/out/mocks/mock_route_checker.go b/internal/boundaries/out/mocks/mock_route_checker.go deleted file mode 100644 index a3c7b4fb8..000000000 --- a/internal/boundaries/out/mocks/mock_route_checker.go +++ /dev/null @@ -1,138 +0,0 @@ -// Code generated by mockery; DO NOT EDIT. -// github.com/vektra/mockery -// template: testify - -package mocks - -import ( - "context" - - "github.com/bnema/gordon/internal/domain" - mock "github.com/stretchr/testify/mock" -) - -// NewMockRouteChecker creates a new instance of MockRouteChecker. It also registers a testing interface on the mock and a cleanup function to assert the mocks expectations. -// The first argument is typically a *testing.T value. -func NewMockRouteChecker(t interface { - mock.TestingT - Cleanup(func()) -}) *MockRouteChecker { - mock := &MockRouteChecker{} - mock.Mock.Test(t) - - t.Cleanup(func() { mock.AssertExpectations(t) }) - - return mock -} - -// MockRouteChecker is an autogenerated mock type for the RouteChecker type -type MockRouteChecker struct { - mock.Mock -} - -type MockRouteChecker_Expecter struct { - mock *mock.Mock -} - -func (_m *MockRouteChecker) EXPECT() *MockRouteChecker_Expecter { - return &MockRouteChecker_Expecter{mock: &_m.Mock} -} - -// GetExternalRoutes provides a mock function for the type MockRouteChecker -func (_mock *MockRouteChecker) GetExternalRoutes() map[string]string { - ret := _mock.Called() - - if len(ret) == 0 { - panic("no return value specified for GetExternalRoutes") - } - - var r0 map[string]string - if returnFunc, ok := ret.Get(0).(func() map[string]string); ok { - r0 = returnFunc() - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).(map[string]string) - } - } - return r0 -} - -// MockRouteChecker_GetExternalRoutes_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetExternalRoutes' -type MockRouteChecker_GetExternalRoutes_Call struct { - *mock.Call -} - -// GetExternalRoutes is a helper method to define mock.On call -func (_e *MockRouteChecker_Expecter) GetExternalRoutes() *MockRouteChecker_GetExternalRoutes_Call { - return &MockRouteChecker_GetExternalRoutes_Call{Call: _e.mock.On("GetExternalRoutes")} -} - -func (_c *MockRouteChecker_GetExternalRoutes_Call) Run(run func()) *MockRouteChecker_GetExternalRoutes_Call { - _c.Call.Run(func(args mock.Arguments) { - run() - }) - return _c -} - -func (_c *MockRouteChecker_GetExternalRoutes_Call) Return(stringToString map[string]string) *MockRouteChecker_GetExternalRoutes_Call { - _c.Call.Return(stringToString) - return _c -} - -func (_c *MockRouteChecker_GetExternalRoutes_Call) RunAndReturn(run func() map[string]string) *MockRouteChecker_GetExternalRoutes_Call { - _c.Call.Return(run) - return _c -} - -// GetRoutes provides a mock function for the type MockRouteChecker -func (_mock *MockRouteChecker) GetRoutes(ctx context.Context) []domain.Route { - ret := _mock.Called(ctx) - - if len(ret) == 0 { - panic("no return value specified for GetRoutes") - } - - var r0 []domain.Route - if returnFunc, ok := ret.Get(0).(func(context.Context) []domain.Route); ok { - r0 = returnFunc(ctx) - } else { - if ret.Get(0) != nil { - r0 = ret.Get(0).([]domain.Route) - } - } - return r0 -} - -// MockRouteChecker_GetRoutes_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'GetRoutes' -type MockRouteChecker_GetRoutes_Call struct { - *mock.Call -} - -// GetRoutes is a helper method to define mock.On call -// - ctx context.Context -func (_e *MockRouteChecker_Expecter) GetRoutes(ctx any) *MockRouteChecker_GetRoutes_Call { - return &MockRouteChecker_GetRoutes_Call{Call: _e.mock.On("GetRoutes", ctx)} -} - -func (_c *MockRouteChecker_GetRoutes_Call) Run(run func(ctx context.Context)) *MockRouteChecker_GetRoutes_Call { - _c.Call.Run(func(args mock.Arguments) { - var arg0 context.Context - if args[0] != nil { - arg0 = args[0].(context.Context) - } - run( - arg0, - ) - }) - return _c -} - -func (_c *MockRouteChecker_GetRoutes_Call) Return(routes []domain.Route) *MockRouteChecker_GetRoutes_Call { - _c.Call.Return(routes) - return _c -} - -func (_c *MockRouteChecker_GetRoutes_Call) RunAndReturn(run func(ctx context.Context) []domain.Route) *MockRouteChecker_GetRoutes_Call { - _c.Call.Return(run) - return _c -} diff --git a/internal/boundaries/out/mocks/mock_secret_provider.go b/internal/boundaries/out/mocks/mock_secret_provider.go index fa8368873..c9b2f2e50 100644 --- a/internal/boundaries/out/mocks/mock_secret_provider.go +++ b/internal/boundaries/out/mocks/mock_secret_provider.go @@ -16,10 +16,19 @@ func NewMockSecretProvider(t interface { mock.TestingT Cleanup(func()) }) *MockSecretProvider { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockSecretProvider{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/out/mocks/mock_secret_resolver.go b/internal/boundaries/out/mocks/mock_secret_resolver.go index bffc80c76..bfb5c35f1 100644 --- a/internal/boundaries/out/mocks/mock_secret_resolver.go +++ b/internal/boundaries/out/mocks/mock_secret_resolver.go @@ -17,10 +17,19 @@ func NewMockSecretResolver(t interface { mock.TestingT Cleanup(func()) }) *MockSecretResolver { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockSecretResolver{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/out/mocks/mock_secret_writer.go b/internal/boundaries/out/mocks/mock_secret_writer.go new file mode 100644 index 000000000..ad66f7edd --- /dev/null +++ b/internal/boundaries/out/mocks/mock_secret_writer.go @@ -0,0 +1,167 @@ +// Code generated by mockery; DO NOT EDIT. +// github.com/vektra/mockery +// template: testify + +package mocks + +import ( + "context" + + mock "github.com/stretchr/testify/mock" +) + +// NewMockSecretWriter creates a new instance of MockSecretWriter. It also registers a testing interface on the mock and a cleanup function to assert the mocks expectations. +// The first argument is typically a *testing.T value. +func NewMockSecretWriter(t interface { + mock.TestingT + Cleanup(func()) +}) *MockSecretWriter { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + + mock := &MockSecretWriter{} + mock.Mock.Test(t) + + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) + + return mock +} + +// MockSecretWriter is an autogenerated mock type for the SecretWriter type +type MockSecretWriter struct { + mock.Mock +} + +type MockSecretWriter_Expecter struct { + mock *mock.Mock +} + +func (_m *MockSecretWriter) EXPECT() *MockSecretWriter_Expecter { + return &MockSecretWriter_Expecter{mock: &_m.Mock} +} + +// DeleteSecret provides a mock function for the type MockSecretWriter +func (_mock *MockSecretWriter) DeleteSecret(ctx context.Context, path string) error { + ret := _mock.Called(ctx, path) + + if len(ret) == 0 { + panic("no return value specified for DeleteSecret") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string) error); ok { + r0 = returnFunc(ctx, path) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockSecretWriter_DeleteSecret_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'DeleteSecret' +type MockSecretWriter_DeleteSecret_Call struct { + *mock.Call +} + +// DeleteSecret is a helper method to define mock.On call +// - ctx context.Context +// - path string +func (_e *MockSecretWriter_Expecter) DeleteSecret(ctx any, path any) *MockSecretWriter_DeleteSecret_Call { + return &MockSecretWriter_DeleteSecret_Call{Call: _e.mock.On("DeleteSecret", ctx, path)} +} + +func (_c *MockSecretWriter_DeleteSecret_Call) Run(run func(ctx context.Context, path string)) *MockSecretWriter_DeleteSecret_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + run( + arg0, + arg1, + ) + }) + return _c +} + +func (_c *MockSecretWriter_DeleteSecret_Call) Return(err error) *MockSecretWriter_DeleteSecret_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockSecretWriter_DeleteSecret_Call) RunAndReturn(run func(ctx context.Context, path string) error) *MockSecretWriter_DeleteSecret_Call { + _c.Call.Return(run) + return _c +} + +// SetSecret provides a mock function for the type MockSecretWriter +func (_mock *MockSecretWriter) SetSecret(ctx context.Context, path string, value string) error { + ret := _mock.Called(ctx, path, value) + + if len(ret) == 0 { + panic("no return value specified for SetSecret") + } + + var r0 error + if returnFunc, ok := ret.Get(0).(func(context.Context, string, string) error); ok { + r0 = returnFunc(ctx, path, value) + } else { + r0 = ret.Error(0) + } + return r0 +} + +// MockSecretWriter_SetSecret_Call is a *mock.Call that shadows Run/Return methods with type explicit version for method 'SetSecret' +type MockSecretWriter_SetSecret_Call struct { + *mock.Call +} + +// SetSecret is a helper method to define mock.On call +// - ctx context.Context +// - path string +// - value string +func (_e *MockSecretWriter_Expecter) SetSecret(ctx any, path any, value any) *MockSecretWriter_SetSecret_Call { + return &MockSecretWriter_SetSecret_Call{Call: _e.mock.On("SetSecret", ctx, path, value)} +} + +func (_c *MockSecretWriter_SetSecret_Call) Run(run func(ctx context.Context, path string, value string)) *MockSecretWriter_SetSecret_Call { + _c.Call.Run(func(args mock.Arguments) { + var arg0 context.Context + if args[0] != nil { + arg0 = args[0].(context.Context) + } + var arg1 string + if args[1] != nil { + arg1 = args[1].(string) + } + var arg2 string + if args[2] != nil { + arg2 = args[2].(string) + } + run( + arg0, + arg1, + arg2, + ) + }) + return _c +} + +func (_c *MockSecretWriter_SetSecret_Call) Return(err error) *MockSecretWriter_SetSecret_Call { + _c.Call.Return(err) + return _c +} + +func (_c *MockSecretWriter_SetSecret_Call) RunAndReturn(run func(ctx context.Context, path string, value string) error) *MockSecretWriter_SetSecret_Call { + _c.Call.Return(run) + return _c +} diff --git a/internal/boundaries/out/mocks/mock_token_store.go b/internal/boundaries/out/mocks/mock_token_store.go index 91534f54d..98771f7f0 100644 --- a/internal/boundaries/out/mocks/mock_token_store.go +++ b/internal/boundaries/out/mocks/mock_token_store.go @@ -17,10 +17,19 @@ func NewMockTokenStore(t interface { mock.TestingT Cleanup(func()) }) *MockTokenStore { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock := &MockTokenStore{} mock.Mock.Test(t) - t.Cleanup(func() { mock.AssertExpectations(t) }) + t.Cleanup(func() { + if helper, ok := t.(interface{ Helper() }); ok { + helper.Helper() + } + mock.AssertExpectations(t) + }) return mock } diff --git a/internal/boundaries/out/pki.go b/internal/boundaries/out/pki.go index 89550936f..74f440f0e 100644 --- a/internal/boundaries/out/pki.go +++ b/internal/boundaries/out/pki.go @@ -1,17 +1,18 @@ package out import ( - "context" "crypto/tls" "crypto/x509" "time" - - "github.com/bnema/gordon/internal/domain" ) -// RouteChecker provides route lookup for domain validation. -type RouteChecker interface { - GetRoutes(ctx context.Context) []domain.Route +// AppRoutes provides ACTIVE-derived host lookup for domain validation. +// Implemented by the apptraffic host index via AppHostSource plus the +// installation external routes. Replaces the retired RouteChecker +// (config-file routes are gone with the declarative-apps cutover). +type AppRoutes interface { + AppHostSource + // GetExternalRoutes returns installation external routes. GetExternalRoutes() map[string]string } diff --git a/internal/boundaries/out/ports_test.go b/internal/boundaries/out/ports_test.go new file mode 100644 index 000000000..5c0d98da6 --- /dev/null +++ b/internal/boundaries/out/ports_test.go @@ -0,0 +1,166 @@ +package out_test + +import ( + "context" + "errors" + "sync" + "testing" + "time" + + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/boundaries/out/mocks" +) + +// The generated mocks must always satisfy their boundary ports. +var ( + _ out.PruneProtectionStore = (*mocks.MockPruneProtectionStore)(nil) + _ out.PruneRuntime = (*mocks.MockPruneRuntime)(nil) + _ out.GCBarrier = (*mocks.MockGCBarrier)(nil) +) + +// referenceBarrier is the minimal shape any GCBarrier implementation +// must have: shared leases exclude exclusive leases, exclusive leases +// exclude everything, and a blocked acquisition honors context +// cancellation instead of wedging shutdown. +type referenceBarrier struct { + mu sync.Mutex + shared int + exclusive bool + // released is closed whenever a lease is released so waiters retry. + released chan struct{} +} + +func newReferenceBarrier() *referenceBarrier { + return &referenceBarrier{released: make(chan struct{})} +} + +type referenceLease struct { + barrier *referenceBarrier + exclusive bool + once sync.Once +} + +func (l *referenceLease) Release() { + l.once.Do(func() { + l.barrier.mu.Lock() + if l.exclusive { + l.barrier.exclusive = false + } else { + l.barrier.shared-- + } + close(l.barrier.released) + l.barrier.released = make(chan struct{}) + l.barrier.mu.Unlock() + }) +} + +func (b *referenceBarrier) acquire(ctx context.Context, exclusive bool) (out.GCLease, error) { + for { + b.mu.Lock() + if b.canAcquire(exclusive) { + if exclusive { + b.exclusive = true + } else { + b.shared++ + } + b.mu.Unlock() + return &referenceLease{barrier: b, exclusive: exclusive}, nil + } + wait := b.released + b.mu.Unlock() + + select { + case <-ctx.Done(): + return nil, ctx.Err() + case <-wait: + } + } +} + +func (b *referenceBarrier) canAcquire(exclusive bool) bool { + if exclusive { + return !b.exclusive && b.shared == 0 + } + return !b.exclusive +} + +func (b *referenceBarrier) AcquireShared(ctx context.Context) (out.GCLease, error) { + return b.acquire(ctx, false) +} + +func (b *referenceBarrier) AcquireExclusive(ctx context.Context) (out.GCLease, error) { + return b.acquire(ctx, true) +} + +// releaseAll exercises the contract through the interface, so a change +// to the port breaks this test rather than silently changing semantics. +func acquireShared(t *testing.T, barrier out.GCBarrier, ctx context.Context) out.GCLease { + t.Helper() + lease, err := barrier.AcquireShared(ctx) + if err != nil { + t.Fatalf("AcquireShared: %v", err) + } + return lease +} + +func acquireExclusive(t *testing.T, barrier out.GCBarrier, ctx context.Context) out.GCLease { + t.Helper() + lease, err := barrier.AcquireExclusive(ctx) + if err != nil { + t.Fatalf("AcquireExclusive: %v", err) + } + return lease +} + +func TestGCBarrierContractExcludesExclusiveDuringShared(t *testing.T) { + barrier := newReferenceBarrier() + ctx := context.Background() + + shared := acquireShared(t, barrier, ctx) + defer shared.Release() + + waitCtx, cancel := context.WithTimeout(ctx, 50*time.Millisecond) + defer cancel() + if _, err := barrier.AcquireExclusive(waitCtx); !errors.Is(err, context.DeadlineExceeded) { + t.Fatalf("exclusive acquisition during a shared lease returned %v, want deadline exceeded", err) + } + + shared.Release() + exclusive := acquireExclusive(t, barrier, ctx) + exclusive.Release() + + // Release is idempotent: a double release must not corrupt counts. + shared.Release() + again := acquireShared(t, barrier, ctx) + again.Release() +} + +func TestGCBarrierContractExcludesSharedDuringExclusive(t *testing.T) { + barrier := newReferenceBarrier() + ctx := context.Background() + + exclusive := acquireExclusive(t, barrier, ctx) + defer exclusive.Release() + + waitCtx, cancel := context.WithTimeout(ctx, 50*time.Millisecond) + defer cancel() + if _, err := barrier.AcquireShared(waitCtx); !errors.Is(err, context.DeadlineExceeded) { + t.Fatalf("shared acquisition during an exclusive lease returned %v, want deadline exceeded", err) + } + + exclusive.Release() + shared := acquireShared(t, barrier, ctx) + shared.Release() +} + +func TestGCBarrierContractPreCanceledContext(t *testing.T) { + barrier := newReferenceBarrier() + held := acquireExclusive(t, barrier, context.Background()) + defer held.Release() + + ctx, cancel := context.WithCancel(context.Background()) + cancel() + if _, err := barrier.AcquireShared(ctx); !errors.Is(err, context.Canceled) { + t.Fatalf("canceled acquisition returned %v, want context.Canceled", err) + } +} diff --git a/internal/boundaries/out/previewstore.go b/internal/boundaries/out/previewstore.go deleted file mode 100644 index 11593b25c..000000000 --- a/internal/boundaries/out/previewstore.go +++ /dev/null @@ -1,13 +0,0 @@ -package out - -import ( - "context" - - "github.com/bnema/gordon/internal/domain" -) - -// PreviewStore persists active preview routes. -type PreviewStore interface { - Load(ctx context.Context) ([]domain.PreviewRoute, error) - Save(ctx context.Context, previews []domain.PreviewRoute) error -} diff --git a/internal/boundaries/out/proxy_cache.go b/internal/boundaries/out/proxy_cache.go deleted file mode 100644 index 91c47a1e7..000000000 --- a/internal/boundaries/out/proxy_cache.go +++ /dev/null @@ -1,19 +0,0 @@ -package out - -import ( - "context" - "time" -) - -// ProxyCacheInvalidator defines the contract for synchronously invalidating -// proxy target cache entries. This is used during zero-downtime deployments -// to ensure the proxy stops routing to an old container before it is stopped. -type ProxyCacheInvalidator interface { - InvalidateTarget(ctx context.Context, domainName string) -} - -// ProxyDrainWaiter defines the contract for waiting until no in-flight -// requests remain for a container. -type ProxyDrainWaiter interface { - WaitForNoInFlight(ctx context.Context, containerID string, timeout time.Duration) bool -} diff --git a/internal/boundaries/out/prune.go b/internal/boundaries/out/prune.go new file mode 100644 index 000000000..5496d4c0c --- /dev/null +++ b/internal/boundaries/out/prune.go @@ -0,0 +1,69 @@ +package out + +import ( + "context" + + "github.com/bnema/gordon/internal/domain" +) + +// PruneProtectionStore builds the coherent protection snapshot prune +// plans against. The implementation must read every durable fact in one +// transaction so no candidate is judged against facts from a different +// moment, and must report unreadable state as an inventory gap instead +// of returning an empty (apparently unprotected) snapshot. +type PruneProtectionStore interface { + // ProtectionSnapshot returns every durable fact that protects prune + // candidates, including desired and active app state, recovery + // inhibitions, staged/committed apply intents, unfinished + // operations, and ownership records. + ProtectionSnapshot(ctx context.Context) (*domain.PruneProtectionSnapshot, error) +} + +// PruneRuntime inventories runtime resources and deletes exactly the +// planned targets. It is deliberately not part of ContainerRuntime: +// prune needs enriched inventory and exact, non-force deletion, and no +// other use case may reach an implicit runtime-wide prune. +type PruneRuntime interface { + // InventoryRuntime returns one coherent read of runtime images, + // container image usage (running and stopped), and volumes with + // labels and usage. A failure to inspect part of the runtime is + // reported as an inventory gap, never as absence. + InventoryRuntime(ctx context.Context) (*domain.RuntimeInventory, error) + + // RemoveImageExact removes exactly the named runtime image ID with + // force=false. It never removes images selected by a filter. + RemoveImageExact(ctx context.Context, ref domain.RuntimeImageRef) error + + // RemoveVolumeExact removes exactly the named runtime volume with + // force=false. It never removes volumes selected by a filter. + RemoveVolumeExact(ctx context.Context, ref domain.RuntimeVolumeRef) error +} + +// GCBarrier serializes resource acquisition and its durable protection +// publication against destructive garbage collection. +// +// A shared lease must be held from the moment a deploy, start, restart, +// remove, recovery, or restore path selects or acquires a runtime +// resource until that resource's protection is durably published +// (journal step, ACTIVE state, ownership record). An exclusive lease +// covers prune's snapshot read, planning, and deletion. +// +// Lock order is fixed and documented so the two never deadlock: +// GC barrier → per-app coordinator → registry mutation lock → short +// bbolt transaction. +type GCBarrier interface { + // AcquireShared blocks until no exclusive lease is held, then + // returns a lease the caller must release. It returns ctx.Err() + // when ctx is done or when shutdown happens first. + AcquireShared(ctx context.Context) (GCLease, error) + // AcquireExclusive blocks until no lease of either kind is held, + // then returns a lease the caller must release. It returns ctx.Err() + // when ctx is done or when shutdown happens first. + AcquireExclusive(ctx context.Context) (GCLease, error) +} + +// GCLease is one held GC barrier lease. Release is idempotent; callers +// should still defer exactly one Release per acquired lease. +type GCLease interface { + Release() +} diff --git a/internal/boundaries/out/runtime.go b/internal/boundaries/out/runtime.go index e424ce5d7..bde656ce4 100644 --- a/internal/boundaries/out/runtime.go +++ b/internal/boundaries/out/runtime.go @@ -6,6 +6,7 @@ package out import ( "context" "io" + "time" "github.com/bnema/gordon/internal/domain" ) @@ -16,8 +17,14 @@ type ContainerRuntime interface { // Container lifecycle CreateContainer(ctx context.Context, config *domain.ContainerConfig) (*domain.Container, error) StartContainer(ctx context.Context, containerID string) error - StopContainer(ctx context.Context, containerID string) error - RestartContainer(ctx context.Context, containerID string) error + // StopContainer stops one exact container, giving it grace to exit + // before the runtime kills it. A non-positive grace keeps the + // runtime's own default. + StopContainer(ctx context.Context, containerID string, grace time.Duration) error + // RestartContainer restarts one exact container, giving it grace to + // exit before the runtime kills it. A non-positive grace keeps the + // runtime's own default. + RestartContainer(ctx context.Context, containerID string, grace time.Duration) error RemoveContainer(ctx context.Context, containerID string, force bool) error RenameContainer(ctx context.Context, containerID, newName string) error @@ -25,9 +32,15 @@ type ContainerRuntime interface { ListContainers(ctx context.Context, all bool) ([]*domain.Container, error) InspectContainer(ctx context.Context, containerID string) (*domain.Container, error) GetContainerLogs(ctx context.Context, containerID string, follow bool) (io.ReadCloser, error) + // GetContainerLogsSince returns logs emitted at or after since, so + // readiness can scope a probe to the current execution of one + // exact container ID. A missing container reports the + // domain.ErrContainerNotFound sentinel. + GetContainerLogsSince(ctx context.Context, containerID string, since time.Time, follow bool) (io.ReadCloser, error) // Image operations PullImage(ctx context.Context, image string) error + PullImageWithOptions(ctx context.Context, request domain.ImagePullRequest) error PullImageWithAuth(ctx context.Context, image, username, password string) error TagImage(ctx context.Context, sourceRef, targetRef string) error UntagImage(ctx context.Context, imageRef string) error @@ -37,11 +50,20 @@ type ContainerRuntime interface { // Runtime information Ping(ctx context.Context) error Version(ctx context.Context) (string, error) + // SupportsCDIDevices reports whether the connected engine can serve + // native CDI device requests. It fails closed with + // domain.ErrRuntimeUnsupported on engines below the support matrix + // (Podman 5.4+, Docker 28.3+). Callers invoke it before any workload + // mutation so an unsupported engine fails in preflight, never after + // withdrawing the serving generation. + SupportsCDIDevices(ctx context.Context) error // Health and status IsContainerRunning(ctx context.Context, containerID string) (bool, error) GetContainerHealthStatus(ctx context.Context, containerID string) (status string, hasHealthcheck bool, err error) - GetContainerPort(ctx context.Context, containerID string, internalPort int) (int, error) + // GetContainerBackendBinds resolves protocol-specific container ports to + // their host binds in one container inspection. + GetContainerBackendBinds(ctx context.Context, containerID string, ports []domain.ContainerBackendPort) ([]domain.ContainerBackendBind, error) // Image and port inspection GetImageExposedPorts(ctx context.Context, imageRef string) ([]int, error) @@ -52,7 +74,10 @@ type ContainerRuntime interface { // Volume management InspectImageVolumes(ctx context.Context, imageRef string) ([]string, error) VolumeExists(ctx context.Context, volumeName string) (bool, error) - CreateVolume(ctx context.Context, volumeName string) error + // CreateVolume creates one named volume with the given labels. + // Labels carry ownership provenance; a caller that omits them + // creates an unmanaged volume that prune never adopts. + CreateVolume(ctx context.Context, volumeName string, labels map[string]string) error RemoveVolume(ctx context.Context, volumeName string, force bool) error ListVolumes(ctx context.Context) ([]*domain.VolumeInfo, error) @@ -64,6 +89,7 @@ type ContainerRuntime interface { // Image identity GetImageID(ctx context.Context, imageRef string) (string, error) + VerifyImageDigest(ctx context.Context, imageRef, digest string) error // In-container operations ExecInContainer(ctx context.Context, containerID string, cmd []string) (*ExecResult, error) @@ -76,6 +102,17 @@ type ContainerRuntime interface { NetworkExists(ctx context.Context, name string) (bool, error) ConnectContainerToNetwork(ctx context.Context, containerName, networkName string) error DisconnectContainerFromNetwork(ctx context.Context, containerName, networkName string) error + + // Bounded network readiness + // + // ProbeContainerNetwork runs one bounded readiness session against one + // declared internal container port, which has no host publication. The + // adapter owns the helper lifecycle: it resolves the target endpoint on + // the exact network in the request, runs one hardened helper attached + // only to that network, and always removes the helper under an + // independent bounded cleanup context. It never publishes a host port + // and never targets a service alias. + ProbeContainerNetwork(ctx context.Context, request domain.ContainerNetworkProbeRequest) (domain.ContainerNetworkProbeResult, error) } // ContainerLister is the subset of ContainerRuntime needed by the orphan GC. diff --git a/internal/boundaries/out/secrets.go b/internal/boundaries/out/secrets.go index 9be3ded00..610ff66f1 100644 --- a/internal/boundaries/out/secrets.go +++ b/internal/boundaries/out/secrets.go @@ -13,3 +13,14 @@ type SecretProvider interface { // IsAvailable checks if this provider is available in the current environment. IsAvailable() bool } + +// SecretWriter defines the contract for writing app secret values by path. +// v3 app secrets live in pass at gordon/apps///; +// values cross this boundary only on the explicit SetSecrets path. +type SecretWriter interface { + // SetSecret writes one secret value by path. + SetSecret(ctx context.Context, path, value string) error + + // DeleteSecret removes one secret value by path. + DeleteSecret(ctx context.Context, path string) error +} diff --git a/internal/boundaries/out/storage.go b/internal/boundaries/out/storage.go index df91885d8..d51298401 100644 --- a/internal/boundaries/out/storage.go +++ b/internal/boundaries/out/storage.go @@ -38,11 +38,19 @@ type BlobStorage interface { // GetBlobUpload returns a writer for the upload. GetBlobUpload(uuid string) (io.WriteCloser, error) - // FinishBlobUpload completes an upload and moves it to blob storage. - FinishBlobUpload(uuid, digest string) error - - // CancelBlobUpload cancels an in-progress upload. - CancelBlobUpload(uuid string) error + // FinishBlobUpload completes an upload for the named repository and + // records the repository/blob association. The upload must belong to + // that repository; a mismatched repository reports ErrUploadNotFound. + FinishBlobUpload(name, uuid, digest string) error + + // CancelBlobUpload cancels an in-progress upload owned by the named + // repository. + CancelBlobUpload(name, uuid string) error + + // BlobOwnedByRepository reports whether the repository completed an + // upload of the digest. Ownership is never inferred from manifest + // references, which a repository writer controls. + BlobOwnedByRepository(name, digest string) (bool, error) // CleanupStaleUploads removes upload files older than maxAge. // Returns the number of uploads removed and total bytes reclaimed. diff --git a/internal/domain/app_bind_policy.go b/internal/domain/app_bind_policy.go new file mode 100644 index 000000000..49c727a80 --- /dev/null +++ b/internal/domain/app_bind_policy.go @@ -0,0 +1,189 @@ +package domain + +import ( + "fmt" + "os" + "path/filepath" + "strings" +) + +// AppBindPolicy is one named administrative bind policy. The installation +// declares policies; app manifests reference them by name. Source is the +// only host location a bind governed by this policy may come from. Root, +// when set, is the administrative boundary the resolved source must remain +// under. AllowedApps and AllowedServices are exact, non-empty allowlists. +type AppBindPolicy struct { + Name string + Source string + ReadOnly bool + AllowedApps []string + AllowedServices []string + Root string +} + +// ResolvedAppBind is a bind whose source has been resolved and verified for +// the runtime to mount. It carries no logging, DTO, or transport concerns. +type ResolvedAppBind struct { + Name string + Source string + Destination string + ReadOnly bool +} + +// ValidateBindPolicyName checks a stable policy name (service charset plus -- ban). +func ValidateBindPolicyName(name string) error { + if !serviceNamePattern.MatchString(name) { + return fmt.Errorf("%w: policy name %q must match [a-z0-9_.-], max 63", ErrBindPolicy, name) + } + if strings.Contains(name, "--") { + return fmt.Errorf("%w: policy name %q must not contain -- (reserved separator)", ErrBindPolicy, name) + } + return nil +} + +// Validate checks the policy shape without touching the filesystem. +func (p AppBindPolicy) Validate() error { + if err := ValidateBindPolicyName(p.Name); err != nil { + return err + } + if err := validatePolicyPath("source", p.Source); err != nil { + return fmt.Errorf("%w: policy %q: %v", ErrBindPolicy, p.Name, err) + } + if p.Root != "" { + if err := validatePolicyPath("root", p.Root); err != nil { + return fmt.Errorf("%w: policy %q: %v", ErrBindPolicy, p.Name, err) + } + } + if err := validateAllowlist("app", p.AllowedApps, ValidateAppName); err != nil { + return fmt.Errorf("%w: policy %q: %v", ErrBindPolicy, p.Name, err) + } + if err := validateAllowlist("service", p.AllowedServices, ValidateServiceName); err != nil { + return fmt.Errorf("%w: policy %q: %v", ErrBindPolicy, p.Name, err) + } + return nil +} + +// validatePolicyPath requires a non-empty, absolute, normalized host path. +func validatePolicyPath(kind, value string) error { + if value == "" { + return fmt.Errorf("%s must not be empty", kind) + } + if !filepath.IsAbs(value) { + return fmt.Errorf("%s %q must be absolute", kind, value) + } + if filepath.Clean(value) != value { + return fmt.Errorf("%s %q must be a normalized clean path", kind, value) + } + if kind == "source" && value == string(filepath.Separator) { + return fmt.Errorf("source must not be the filesystem root") + } + return nil +} + +// validateAllowlist requires a non-empty exact list of valid, unique names. +func validateAllowlist(kind string, names []string, validate func(string) error) error { + if len(names) == 0 { + return fmt.Errorf("allowed %ss must not be empty", kind) + } + seen := map[string]struct{}{} + for _, name := range names { + if err := validate(name); err != nil { + return fmt.Errorf("allowed %s %q: %w", kind, name, err) + } + if _, ok := seen[name]; ok { + return fmt.Errorf("allowed %s %q is duplicated", kind, name) + } + seen[name] = struct{}{} + } + return nil +} + +// ResolveAppBind resolves one AppBind under this policy for the given app and +// service. It validates the policy and bind shapes, enforces the exact +// app/service allowlists, resolves the source through symlinks, verifies it is +// a regular file or directory that remains under the configured root, and +// never weakens read-only intent (policy read-only and bind read-only both +// force the resolved bind read-only). The result is runtime-ready: callers +// mount ResolvedAppBind.Source at Destination. +func (p AppBindPolicy) ResolveAppBind(app, service string, bind AppBind) (ResolvedAppBind, error) { + if err := p.Validate(); err != nil { + return ResolvedAppBind{}, err + } + if !allowlistContains(p.AllowedApps, app) { + return ResolvedAppBind{}, fmt.Errorf("%w: policy %q does not allow app %q", ErrBindPolicy, p.Name, app) + } + if !allowlistContains(p.AllowedServices, service) { + return ResolvedAppBind{}, fmt.Errorf("%w: policy %q does not allow service %q", ErrBindPolicy, p.Name, service) + } + if err := ValidateBindName(bind.Name); err != nil { + return ResolvedAppBind{}, err + } + if err := ValidateBindDestination(bind.Path); err != nil { + return ResolvedAppBind{}, err + } + source, err := p.resolveSource() + if err != nil { + return ResolvedAppBind{}, err + } + return ResolvedAppBind{ + Name: bind.Name, + Source: source, + Destination: bind.Path, + ReadOnly: p.ReadOnly || bind.ReadOnly, + }, nil +} + +// allowlistContains reports exact membership. +func allowlistContains(names []string, candidate string) bool { + for _, name := range names { + if name == candidate { + return true + } + } + return false +} + +// resolveSource resolves and verifies the policy source against the filesystem. +func (p AppBindPolicy) resolveSource() (string, error) { + resolvedSource, err := filepath.EvalSymlinks(p.Source) + if err != nil { + return "", fmt.Errorf("%w: policy %q source %q: %v", ErrBindPolicy, p.Name, p.Source, err) + } + info, err := os.Lstat(resolvedSource) + if err != nil { + return "", fmt.Errorf("%w: policy %q source %q: %v", ErrBindPolicy, p.Name, p.Source, err) + } + if !bindSourceTypeAllowed(info.Mode()) { + return "", fmt.Errorf("%w: policy %q source %q must be a regular file or directory", ErrBindPolicy, p.Name, p.Source) + } + if p.Root == "" { + return resolvedSource, nil + } + resolvedRoot, err := filepath.EvalSymlinks(p.Root) + if err != nil { + return "", fmt.Errorf("%w: policy %q root %q: %v", ErrBindPolicy, p.Name, p.Root, err) + } + if !resolvedUnderRoot(resolvedRoot, resolvedSource) { + return "", fmt.Errorf("%w: policy %q source %q resolves outside root %q", ErrBindPolicy, p.Name, p.Source, p.Root) + } + return resolvedSource, nil +} + +// bindSourceTypeAllowed accepts only regular files and directories. Devices, +// sockets, FIFOs, symlinks, and other special files are refused. +func bindSourceTypeAllowed(mode os.FileMode) bool { + return mode.IsRegular() || mode.IsDir() +} + +// resolvedUnderRoot reports whether target is root itself or a descendant of +// root. Both paths must already be resolved. +func resolvedUnderRoot(root, target string) bool { + rel, err := filepath.Rel(root, target) + if err != nil { + return false + } + if rel == "." { + return true + } + return rel != ".." && !strings.HasPrefix(rel, ".."+string(filepath.Separator)) +} diff --git a/internal/domain/app_bind_policy_test.go b/internal/domain/app_bind_policy_test.go new file mode 100644 index 000000000..aec5f4ed7 --- /dev/null +++ b/internal/domain/app_bind_policy_test.go @@ -0,0 +1,258 @@ +package domain + +import ( + "net" + "os" + "path/filepath" + "syscall" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func validPolicy(root, source string) AppBindPolicy { + return AppBindPolicy{ + Name: "config", + Source: source, + AllowedApps: []string{"blog"}, + AllowedServices: []string{"web"}, + Root: root, + } +} + +func TestAppBindPolicy_Validate(t *testing.T) { + root := t.TempDir() + source := filepath.Join(root, "binds") + require.NoError(t, os.MkdirAll(source, 0o755)) + + require.NoError(t, validPolicy(root, source).Validate()) + + cases := []struct { + name string + mutate func(*AppBindPolicy) + wantErr string + }{ + {"empty name", func(p *AppBindPolicy) { p.Name = "" }, "policy name"}, + {"bad name", func(p *AppBindPolicy) { p.Name = "Bad" }, "policy name"}, + {"reserved name", func(p *AppBindPolicy) { p.Name = "a--b" }, "reserved separator"}, + {"relative source", func(p *AppBindPolicy) { p.Source = "binds" }, "must be absolute"}, + {"unclean source", func(p *AppBindPolicy) { p.Source = root + "/binds/../etc" }, "normalized clean path"}, + {"empty source", func(p *AppBindPolicy) { p.Source = "" }, "source must not be empty"}, + {"relative root", func(p *AppBindPolicy) { p.Root = "root" }, "must be absolute"}, + {"unclean root", func(p *AppBindPolicy) { p.Root = root + "/x/.." }, "normalized clean path"}, + {"empty allowed apps", func(p *AppBindPolicy) { p.AllowedApps = nil }, "allowed apps must not be empty"}, + {"empty allowed services", func(p *AppBindPolicy) { p.AllowedServices = nil }, "allowed services must not be empty"}, + {"bad allowed app", func(p *AppBindPolicy) { p.AllowedApps = []string{"Bad!"} }, "allowed app"}, + {"bad allowed service", func(p *AppBindPolicy) { p.AllowedServices = []string{"Bad!"} }, "allowed service"}, + {"duplicate allowed app", func(p *AppBindPolicy) { p.AllowedApps = []string{"blog", "blog"} }, "duplicated"}, + {"duplicate allowed service", func(p *AppBindPolicy) { p.AllowedServices = []string{"web", "web"} }, "duplicated"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + policy := validPolicy(root, source) + tc.mutate(&policy) + err := policy.Validate() + require.Error(t, err) + assert.ErrorIs(t, err, ErrBindPolicy) + assert.Contains(t, err.Error(), tc.wantErr) + }) + } +} + +func TestAppBindPolicy_ResolveAppBind(t *testing.T) { + root := t.TempDir() + source := filepath.Join(root, "binds", "config") + require.NoError(t, os.MkdirAll(source, 0o755)) + + policy := validPolicy(root, source) + resolved, err := policy.ResolveAppBind("blog", "web", AppBind{Name: "config", Path: "/etc/app.conf"}) + require.NoError(t, err) + assert.Equal(t, "config", resolved.Name) + assert.Equal(t, "/etc/app.conf", resolved.Destination) + assert.False(t, resolved.ReadOnly) + realSource, err := filepath.EvalSymlinks(source) + require.NoError(t, err) + assert.Equal(t, realSource, resolved.Source) +} + +func TestAppBindPolicy_ResolveAppBindReadOnlyNonWeakening(t *testing.T) { + root := t.TempDir() + source := filepath.Join(root, "file.txt") + require.NoError(t, os.WriteFile(source, []byte("data"), 0o644)) + + cases := []struct { + name string + policyRO bool + bindRO bool + wantReadOnly bool + }{ + {"policy enforces read-only", true, false, true}, + {"bind requests read-only", false, true, true}, + {"both read-only", true, true, true}, + {"writable by default", false, false, false}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + policy := validPolicy(root, source) + policy.ReadOnly = tc.policyRO + resolved, err := policy.ResolveAppBind("blog", "web", AppBind{Name: "config", Path: "/etc/app.conf", ReadOnly: tc.bindRO}) + require.NoError(t, err) + assert.Equal(t, tc.wantReadOnly, resolved.ReadOnly) + }) + } +} + +func TestAppBindPolicy_ResolveAppBindAllowlists(t *testing.T) { + root := t.TempDir() + source := filepath.Join(root, "binds") + require.NoError(t, os.MkdirAll(source, 0o755)) + policy := validPolicy(root, source) + + _, err := policy.ResolveAppBind("other", "web", AppBind{Name: "config", Path: "/etc/app.conf"}) + require.ErrorIs(t, err, ErrBindPolicy) + assert.Contains(t, err.Error(), "does not allow app") + + _, err = policy.ResolveAppBind("blog", "other", AppBind{Name: "config", Path: "/etc/app.conf"}) + require.ErrorIs(t, err, ErrBindPolicy) + assert.Contains(t, err.Error(), "does not allow service") + + // Exact, not prefix, matching. + _, err = policy.ResolveAppBind("blog-staging", "web", AppBind{Name: "config", Path: "/etc/app.conf"}) + require.ErrorIs(t, err, ErrBindPolicy) +} + +func TestAppBindPolicy_ResolveAppBindBindShape(t *testing.T) { + root := t.TempDir() + source := filepath.Join(root, "binds") + require.NoError(t, os.MkdirAll(source, 0o755)) + policy := validPolicy(root, source) + + _, err := policy.ResolveAppBind("blog", "web", AppBind{Name: "Bad", Path: "/etc/app.conf"}) + require.ErrorIs(t, err, ErrInvalidAppSpec) + + _, err = policy.ResolveAppBind("blog", "web", AppBind{Name: "config", Path: "relative"}) + require.ErrorIs(t, err, ErrInvalidAppSpec) +} + +func TestAppBindPolicy_ResolveAppBindRootBoundary(t *testing.T) { + root := t.TempDir() + outside := t.TempDir() + require.NoError(t, os.MkdirAll(filepath.Join(root, "binds"), 0o755)) + require.NoError(t, os.MkdirAll(filepath.Join(outside, "binds"), 0o755)) + + t.Run("source outside root", func(t *testing.T) { + policy := validPolicy(root, filepath.Join(outside, "binds")) + _, err := policy.ResolveAppBind("blog", "web", AppBind{Name: "config", Path: "/etc/app.conf"}) + require.ErrorIs(t, err, ErrBindPolicy) + assert.Contains(t, err.Error(), "outside root") + }) + + t.Run("sibling prefix is not inside", func(t *testing.T) { + // root/../root-sibling must not be treated as a descendant of root. + sibling := root + "-sibling" + require.NoError(t, os.MkdirAll(sibling, 0o755)) + t.Cleanup(func() { _ = os.RemoveAll(sibling) }) + policy := validPolicy(root, sibling) + _, err := policy.ResolveAppBind("blog", "web", AppBind{Name: "config", Path: "/etc/app.conf"}) + require.ErrorIs(t, err, ErrBindPolicy) + assert.Contains(t, err.Error(), "outside root") + }) + + t.Run("symlink escaping root is refused", func(t *testing.T) { + link := filepath.Join(root, "escape") + require.NoError(t, os.Symlink(filepath.Join(outside, "binds"), link)) + policy := validPolicy(root, link) + _, err := policy.ResolveAppBind("blog", "web", AppBind{Name: "config", Path: "/etc/app.conf"}) + require.ErrorIs(t, err, ErrBindPolicy) + assert.Contains(t, err.Error(), "outside root") + }) + + t.Run("symlink inside root resolves", func(t *testing.T) { + real := filepath.Join(root, "binds") + link := filepath.Join(root, "link") + require.NoError(t, os.Symlink(real, link)) + policy := validPolicy(root, link) + resolved, err := policy.ResolveAppBind("blog", "web", AppBind{Name: "config", Path: "/etc/app.conf"}) + require.NoError(t, err) + realSource, err := filepath.EvalSymlinks(real) + require.NoError(t, err) + assert.Equal(t, realSource, resolved.Source) + }) + + t.Run("root itself is inside", func(t *testing.T) { + policy := validPolicy(root, root) + _, err := policy.ResolveAppBind("blog", "web", AppBind{Name: "config", Path: "/etc/app.conf"}) + require.NoError(t, err) + }) + + t.Run("optional root is not required", func(t *testing.T) { + policy := validPolicy("", filepath.Join(outside, "binds")) + _, err := policy.ResolveAppBind("blog", "web", AppBind{Name: "config", Path: "/etc/app.conf"}) + require.NoError(t, err) + }) +} + +func TestAppBindPolicy_ResolveAppBindSourceTypes(t *testing.T) { + root := t.TempDir() + require.NoError(t, os.MkdirAll(filepath.Join(root, "binds"), 0o755)) + + t.Run("missing source", func(t *testing.T) { + policy := validPolicy(root, filepath.Join(root, "binds", "nope")) + _, err := policy.ResolveAppBind("blog", "web", AppBind{Name: "config", Path: "/etc/app.conf"}) + require.ErrorIs(t, err, ErrBindPolicy) + }) + + t.Run("fifo refused", func(t *testing.T) { + fifo := filepath.Join(root, "binds", "fifo") + require.NoError(t, syscall.Mkfifo(fifo, 0o600)) + policy := validPolicy(root, fifo) + _, err := policy.ResolveAppBind("blog", "web", AppBind{Name: "config", Path: "/etc/app.conf"}) + require.ErrorIs(t, err, ErrBindPolicy) + assert.Contains(t, err.Error(), "regular file or directory") + }) + + t.Run("socket refused", func(t *testing.T) { + socket := filepath.Join(root, "binds", "sock") + listener, err := net.Listen("unix", socket) + require.NoError(t, err) + defer func() { _ = listener.Close() }() + policy := validPolicy(root, socket) + _, err = policy.ResolveAppBind("blog", "web", AppBind{Name: "config", Path: "/etc/app.conf"}) + require.ErrorIs(t, err, ErrBindPolicy) + assert.Contains(t, err.Error(), "regular file or directory") + }) + + t.Run("regular file accepted", func(t *testing.T) { + file := filepath.Join(root, "binds", "app.conf") + require.NoError(t, os.WriteFile(file, []byte("x"), 0o644)) + policy := validPolicy(root, file) + resolved, err := policy.ResolveAppBind("blog", "web", AppBind{Name: "config", Path: "/etc/app.conf"}) + require.NoError(t, err) + assert.Equal(t, file, resolved.Source) + }) +} + +func TestBindSourceTypeAllowed(t *testing.T) { + assert.True(t, bindSourceTypeAllowed(0)) + assert.True(t, bindSourceTypeAllowed(os.ModeDir)) + refused := []os.FileMode{ + os.ModeSymlink, + os.ModeDevice, + os.ModeCharDevice, + os.ModeNamedPipe, + os.ModeSocket, + os.ModeIrregular, + } + for _, mode := range refused { + assert.False(t, bindSourceTypeAllowed(mode), mode.String()) + } +} + +func TestResolvedUnderRoot(t *testing.T) { + assert.True(t, resolvedUnderRoot("/srv/binds", "/srv/binds")) + assert.True(t, resolvedUnderRoot("/srv/binds", "/srv/binds/conf")) + assert.False(t, resolvedUnderRoot("/srv/binds", "/srv/binds-sibling")) + assert.False(t, resolvedUnderRoot("/srv/binds", "/srv/other")) + assert.False(t, resolvedUnderRoot("/srv/binds", "/etc")) +} diff --git a/internal/domain/app_bind_test.go b/internal/domain/app_bind_test.go new file mode 100644 index 000000000..95932ab1d --- /dev/null +++ b/internal/domain/app_bind_test.go @@ -0,0 +1,111 @@ +package domain_test + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +func TestAppSpec_ValidateBindsOK(t *testing.T) { + spec := validSpec() + spec.Services[0].Binds = []domain.AppBind{ + {Name: "config", Path: "/etc/app.conf", ReadOnly: true}, + {Name: "data", Path: "/var/lib/app"}, + } + require.NoError(t, spec.Validate()) +} + +func TestAppSpec_ValidateBindErrors(t *testing.T) { + cases := []struct { + name string + mutate func(*domain.AppSpec) + wantErr string + }{ + {"bad bind name", func(s *domain.AppSpec) { + s.Services[0].Binds = []domain.AppBind{{Name: "Bad_Name", Path: "/data"}} + }, "bind name"}, + {"reserved separator name", func(s *domain.AppSpec) { + s.Services[0].Binds = []domain.AppBind{{Name: "a--b", Path: "/data"}} + }, "reserved separator"}, + {"relative destination", func(s *domain.AppSpec) { + s.Services[0].Binds = []domain.AppBind{{Name: "data", Path: "relative/path"}} + }, "must be absolute"}, + {"unnormalized destination", func(s *domain.AppSpec) { + s.Services[0].Binds = []domain.AppBind{{Name: "data", Path: "/var/../data"}} + }, "normalized clean path"}, + {"trailing slash destination", func(s *domain.AppSpec) { + s.Services[0].Binds = []domain.AppBind{{Name: "data", Path: "/data/"}} + }, "normalized clean path"}, + {"sensitive destination", func(s *domain.AppSpec) { + s.Services[0].Binds = []domain.AppBind{{Name: "dev", Path: "/dev"}} + }, "sensitive container path"}, + {"duplicate bind name", func(s *domain.AppSpec) { + s.Services[0].Binds = []domain.AppBind{ + {Name: "data", Path: "/data"}, + {Name: "data", Path: "/other"}, + } + }, "duplicate bind name"}, + {"duplicate bind destination", func(s *domain.AppSpec) { + s.Services[0].Binds = []domain.AppBind{ + {Name: "one", Path: "/data"}, + {Name: "two", Path: "/data"}, + } + }, "duplicate bind destination"}, + {"collision with volume", func(s *domain.AppSpec) { + s.Services[0].Volumes = []domain.AppVolume{{Name: "data", Path: "/data"}} + s.Services[0].Binds = []domain.AppBind{{Name: "data-bind", Path: "/data"}} + }, "collides with a declared volume"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + spec := validSpec() + tc.mutate(&spec) + err := spec.Validate() + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrInvalidAppSpec) + assert.Contains(t, err.Error(), tc.wantErr) + }) + } +} + +func TestValidateBindName(t *testing.T) { + assert.NoError(t, domain.ValidateBindName("data")) + assert.NoError(t, domain.ValidateBindName("app.config")) + assert.ErrorIs(t, domain.ValidateBindName(""), domain.ErrInvalidAppSpec) + assert.ErrorIs(t, domain.ValidateBindName("Bad"), domain.ErrInvalidAppSpec) + assert.ErrorIs(t, domain.ValidateBindName("a--b"), domain.ErrInvalidAppSpec) +} + +func TestValidateBindDestination(t *testing.T) { + valid := []string{"/data", "/etc/app.conf", "/var/lib/app"} + for _, dest := range valid { + assert.NoError(t, domain.ValidateBindDestination(dest), dest) + } + invalid := []string{"", "data", "/data/", "/data/../etc", "//data", "/./data", "/", "/proc", "/sys", "/dev", "/boot"} + for _, dest := range invalid { + err := domain.ValidateBindDestination(dest) + require.Error(t, err, dest) + assert.ErrorIs(t, err, domain.ErrInvalidAppSpec, dest) + } +} + +func TestIsSensitiveBindDestination(t *testing.T) { + for _, dest := range []string{"/", "/proc", "/sys", "/dev", "/boot"} { + assert.True(t, domain.IsSensitiveBindDestination(dest), dest) + } + assert.False(t, domain.IsSensitiveBindDestination("/etc/app.conf")) + assert.True(t, domain.IsSensitiveBindDestination("/proc/self")) + assert.True(t, domain.IsSensitiveBindDestination("/dev/shm")) +} + +func TestDiffAppSpec_Binds(t *testing.T) { + desired := validSpec() + desired.Services[0].Binds = []domain.AppBind{{Name: "data", Path: "/data", ReadOnly: true}} + effective := validSpec() + effective.Services[0].Binds = []domain.AppBind{{Name: "data", Path: "/data"}} + diff := domain.DiffAppSpec(desired, effective) + assert.Contains(t, diff.Changed, "service/web/binds") +} diff --git a/internal/domain/app_device_policy.go b/internal/domain/app_device_policy.go new file mode 100644 index 000000000..be0b5d4da --- /dev/null +++ b/internal/domain/app_device_policy.go @@ -0,0 +1,165 @@ +package domain + +import ( + "fmt" + "regexp" + "slices" + "strings" +) + +// cdiDeviceNamePattern matches the CDI device-name grammar: an alphanumeric +// start followed by alphanumerics, '_', '.', or '-'. Host paths and raw +// device nodes never match. +var cdiDeviceNamePattern = regexp.MustCompile(`^[A-Za-z0-9][A-Za-z0-9_.\-]*$`) + +// AppDevicePolicy is one named administrative device grant declared under +// [app_devices.]. The installation maps a logical device name to +// explicit CDI device IDs and exact app/service allowlists. App manifests +// request devices by logical name; Gordon resolves the CDI IDs at +// activation time. Host CDI resolution is never persisted as desired +// intent: revisions keep the logical names. +type AppDevicePolicy struct { + Name string + // CDI holds the explicit CDI device IDs (for example + // "nvidia.com/gpu=GPU-..."). Aggregate selectors such as "=all" + // are rejected: every granted device must be named explicitly. + CDI []string + // AllowedApps and AllowedServices are exact, non-empty allowlists. + AllowedApps []string + AllowedServices []string +} + +// ValidateDeviceName checks a logical device name (service charset plus -- ban). +func ValidateDeviceName(name string) error { + if !serviceNamePattern.MatchString(name) { + return fmt.Errorf("%w: device name %q must match [a-z0-9_.-], max 63", ErrInvalidAppSpec, name) + } + if strings.Contains(name, "--") { + return fmt.Errorf("%w: device name %q must not contain -- (reserved separator)", ErrInvalidAppSpec, name) + } + return nil +} + +// Validate checks the policy shape without touching the runtime. +func (p AppDevicePolicy) Validate() error { + if err := ValidateDevicePolicyName(p.Name); err != nil { + return err + } + if err := validateDeviceCDIIDs(p.Name, p.CDI); err != nil { + return err + } + if err := validateAllowlist("app", p.AllowedApps, ValidateAppName); err != nil { + return fmt.Errorf("%w: device policy %q: %w", ErrDevicePolicy, p.Name, err) + } + if err := validateAllowlist("service", p.AllowedServices, ValidateServiceName); err != nil { + return fmt.Errorf("%w: device policy %q: %w", ErrDevicePolicy, p.Name, err) + } + return nil +} + +// ValidateDevicePolicyName checks a stable policy name (service charset plus -- ban). +func ValidateDevicePolicyName(name string) error { + if !serviceNamePattern.MatchString(name) { + return fmt.Errorf("%w: device policy name %q must match [a-z0-9_.-], max 63", ErrDevicePolicy, name) + } + if strings.Contains(name, "--") { + return fmt.Errorf("%w: device policy name %q must not contain -- (reserved separator)", ErrDevicePolicy, name) + } + return nil +} + +// validateDeviceCDIIDs requires a non-empty list of unique, well-formed CDI +// device IDs. Raw /dev paths, empty entries, and aggregate "=all" +// selectors are rejected. +func validateDeviceCDIIDs(policy string, ids []string) error { + if len(ids) == 0 { + return fmt.Errorf("%w: device policy %q: cdi must not be empty", ErrDevicePolicy, policy) + } + seen := map[string]struct{}{} + for _, id := range ids { + if err := validateDeviceCDIID(id); err != nil { + return fmt.Errorf("%w: device policy %q: %v", ErrDevicePolicy, policy, err) + } + if _, ok := seen[id]; ok { + return fmt.Errorf("%w: device policy %q: duplicate cdi device %q", ErrDevicePolicy, policy, id) + } + seen[id] = struct{}{} + } + return nil +} + +// validateDeviceCDIID checks one qualified CDI device ID of the form +// "/=" (for example "nvidia.com/gpu=GPU-..."). +// The device name must not be empty, "all", or a host path. +func validateDeviceCDIID(id string) error { + if id == "" { + return fmt.Errorf("cdi device must not be empty") + } + if strings.HasPrefix(id, "/") { + return fmt.Errorf("cdi device %q must be a qualified CDI ID, not a host path", id) + } + kind, name, ok := strings.Cut(id, "=") + if !ok || kind == "" || name == "" { + return fmt.Errorf("cdi device %q must be a qualified CDI ID of the form \"/=\"", id) + } + if !strings.Contains(kind, "/") { + return fmt.Errorf("cdi device %q must be a qualified CDI ID of the form \"/=\"", id) + } + if name == "all" { + return fmt.Errorf("cdi device %q: aggregate \"=all\" selectors are not allowed, name each device explicitly", id) + } + if !cdiDeviceNamePattern.MatchString(name) { + return fmt.Errorf("cdi device %q: name %q must match [A-Za-z0-9][A-Za-z0-9_.-]*, not a host path or raw device node", id, name) + } + return nil +} + +// ResolveAppDevice resolves the logical device grant under this policy +// for the given app and service. It validates the policy shape, enforces +// the exact app/service allowlists, and returns the explicit CDI IDs. +// The result is runtime-ready: callers encode the IDs as one native CDI +// DeviceRequest. +func (p AppDevicePolicy) ResolveAppDevice(app, service string) ([]string, error) { + if err := p.Validate(); err != nil { + return nil, err + } + if !allowlistContains(p.AllowedApps, app) { + return nil, fmt.Errorf("%w: device policy %q does not allow app %q", ErrDevicePolicy, p.Name, app) + } + if !allowlistContains(p.AllowedServices, service) { + return nil, fmt.Errorf("%w: device policy %q does not allow service %q", ErrDevicePolicy, p.Name, service) + } + return slices.Clone(p.CDI), nil +} + +// ResolveAppDevices resolves every logical device in names against the +// given policies for one app service. It returns the deterministic union +// of CDI IDs (sorted) and rejects duplicate resolved IDs across logical +// resources. Errors name only the app, service, and logical resource: +// host inventory never appears. +func ResolveAppDevices(app, service string, names []string, policies map[string]AppDevicePolicy) ([]string, error) { + if len(names) == 0 { + return nil, nil + } + var resolved []string + seen := map[string]struct{}{} + for _, name := range names { + policy, ok := policies[name] + if !ok { + return nil, fmt.Errorf("%w: app %q service %q device %q: no administrative device policy configured", ErrDevicePolicy, app, service, name) + } + ids, err := policy.ResolveAppDevice(app, service) + if err != nil { + return nil, fmt.Errorf("%w: app %q service %q device %q: refused by administrative device policy", ErrDevicePolicy, app, service, name) + } + for _, id := range ids { + if _, dup := seen[id]; dup { + return nil, fmt.Errorf("%w: app %q service %q device %q: cdi device already granted by another device", ErrDevicePolicy, app, service, name) + } + seen[id] = struct{}{} + resolved = append(resolved, id) + } + } + slices.Sort(resolved) + return resolved, nil +} diff --git a/internal/domain/app_device_policy_test.go b/internal/domain/app_device_policy_test.go new file mode 100644 index 000000000..751f2eb6e --- /dev/null +++ b/internal/domain/app_device_policy_test.go @@ -0,0 +1,188 @@ +package domain + +import ( + "errors" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func validDevicePolicy() AppDevicePolicy { + return AppDevicePolicy{ + Name: "test_gpu", + CDI: []string{"example.com/gpu=GPU-test-uuid"}, + AllowedApps: []string{"demo"}, + AllowedServices: []string{"worker"}, + } +} + +func TestDevicePolicyValid(t *testing.T) { + p := validDevicePolicy() + require.NoError(t, p.Validate()) +} + +func TestDevicePolicyNameRejected(t *testing.T) { + for _, name := range []string{"", "Bad", "a--b", "has space"} { + p := validDevicePolicy() + p.Name = name + err := p.Validate() + require.Error(t, err, "name %q", name) + assert.True(t, errors.Is(err, ErrDevicePolicy)) + } +} + +func TestDevicePolicyCDIRejected(t *testing.T) { + cases := map[string][]string{ + "empty": {}, + "empty entry": {""}, + "raw path": {"/dev/card0"}, + "unqualified": {"gpu0"}, + "missing name": {"example.com/gpu="}, + "missing kind": {"=gpu0"}, + "aggregate": {"example.com/gpu=all"}, + "duplicate": {"example.com/gpu=0", "example.com/gpu=0"}, + "qualified host path": {"example.com/gpu=/dev/card0"}, + } + for name, cdi := range cases { + p := validDevicePolicy() + p.CDI = cdi + err := p.Validate() + require.Error(t, err, "case %s", name) + assert.True(t, errors.Is(err, ErrDevicePolicy), "case %s", name) + } +} + +func TestDevicePolicyAllowlistRejected(t *testing.T) { + p := validDevicePolicy() + p.AllowedApps = nil + err := p.Validate() + require.Error(t, err) + assert.True(t, errors.Is(err, ErrDevicePolicy)) + + p = validDevicePolicy() + p.AllowedApps = []string{"demo", "demo"} + require.Error(t, p.Validate()) + + p = validDevicePolicy() + p.AllowedServices = []string{"worker", "worker"} + require.Error(t, p.Validate()) + + // An invalid allowlist entry is both a policy violation and an invalid + // app spec: callers must be able to detect either sentinel. + p = validDevicePolicy() + p.AllowedServices = []string{"Bad Name"} + err = p.Validate() + require.Error(t, err) + assert.True(t, errors.Is(err, ErrDevicePolicy)) + assert.True(t, errors.Is(err, ErrInvalidAppSpec), "the invalid-spec chain must survive allowlist validation") +} + +func TestResolveAppDeviceDenyByDefault(t *testing.T) { + p := validDevicePolicy() + _, err := p.ResolveAppDevice("other", "worker") + require.Error(t, err) + assert.True(t, errors.Is(err, ErrDevicePolicy)) + + _, err = p.ResolveAppDevice("demo", "helper") + require.Error(t, err) + assert.True(t, errors.Is(err, ErrDevicePolicy)) +} + +func TestResolveAppDeviceExactGrant(t *testing.T) { + p := validDevicePolicy() + ids, err := p.ResolveAppDevice("demo", "worker") + require.NoError(t, err) + assert.Equal(t, []string{"example.com/gpu=GPU-test-uuid"}, ids) + // The returned slice must be a copy: mutating it must not affect + // later resolutions. + ids[0] = "mutated" + again, err := p.ResolveAppDevice("demo", "worker") + require.NoError(t, err) + assert.Equal(t, []string{"example.com/gpu=GPU-test-uuid"}, again) +} + +func TestResolveAppDevicesUnknownDenied(t *testing.T) { + policies := map[string]AppDevicePolicy{"test_gpu": validDevicePolicy()} + _, err := ResolveAppDevices("demo", "worker", []string{"unknown"}, policies) + require.Error(t, err) + assert.True(t, errors.Is(err, ErrDevicePolicy)) +} + +func TestResolveAppDevicesDeterministicUnion(t *testing.T) { + policies := map[string]AppDevicePolicy{ + "b": {Name: "b", CDI: []string{"example.com/gpu=GPU-b"}, AllowedApps: []string{"demo"}, AllowedServices: []string{"worker"}}, + "a": {Name: "a", CDI: []string{"example.com/gpu=GPU-a"}, AllowedApps: []string{"demo"}, AllowedServices: []string{"worker"}}, + } + ids, err := ResolveAppDevices("demo", "worker", []string{"b", "a"}, policies) + require.NoError(t, err) + assert.Equal(t, []string{"example.com/gpu=GPU-a", "example.com/gpu=GPU-b"}, ids) +} + +func TestResolveAppDevicesDuplicateIDRejected(t *testing.T) { + shared := "example.com/gpu=GPU-shared" + policies := map[string]AppDevicePolicy{ + "a": {Name: "a", CDI: []string{shared}, AllowedApps: []string{"demo"}, AllowedServices: []string{"worker"}}, + "b": {Name: "b", CDI: []string{shared}, AllowedApps: []string{"demo"}, AllowedServices: []string{"worker"}}, + } + _, err := ResolveAppDevices("demo", "worker", []string{"a", "b"}, policies) + require.Error(t, err) + assert.True(t, errors.Is(err, ErrDevicePolicy)) + assert.NotContains(t, err.Error(), shared, "duplicate-grant errors must not echo resolved CDI IDs") +} + +func TestResolveAppDevicesEmptyIsNil(t *testing.T) { + ids, err := ResolveAppDevices("demo", "worker", nil, nil) + require.NoError(t, err) + assert.Nil(t, ids) +} + +func TestValidateDeviceName(t *testing.T) { + require.NoError(t, ValidateDeviceName("test_gpu")) + for _, name := range []string{"", "Bad", "a--b", "/dev/card0"} { + require.Error(t, ValidateDeviceName(name), "name %q", name) + } +} + +func TestAppServiceDevicesValidation(t *testing.T) { + base := AppSpec{Name: "demo", Services: []AppService{{Name: "worker", Image: "registry.example.com/demo/worker:1", StopGrace: 10 * time.Second, Readiness: AppReadiness{Type: "http", Path: "/healthz", Timeout: 30 * time.Second}}}} + valid := base + valid.Services[0].Devices = []string{"test_gpu"} + require.NoError(t, valid.Validate()) + + dup := base + dup.Services[0].Devices = []string{"test_gpu", "test_gpu"} + err := dup.Validate() + require.Error(t, err) + assert.True(t, errors.Is(err, ErrInvalidAppSpec)) + assert.Contains(t, err.Error(), "duplicate device") + + bad := base + bad.Services[0].Devices = []string{"Bad!"} + err = bad.Validate() + require.Error(t, err) + assert.True(t, errors.Is(err, ErrInvalidAppSpec)) +} + +func TestDiffAppSpecDevices(t *testing.T) { + svc := func(devices ...string) AppService { + return AppService{Name: "worker", Image: "registry.example.com/demo/worker:1", StopGrace: 10 * time.Second, Readiness: AppReadiness{Type: "http", Path: "/healthz", Timeout: 30 * time.Second}, Devices: devices} + } + mk := func(services ...AppService) AppSpec { return AppSpec{Name: "demo", Services: services} } + + // Add/remove report service/worker/devices. + oldSpec, nextSpec := mk(svc()), mk(svc("test_gpu")) + diff := DiffAppSpec(oldSpec, nextSpec) + assert.Contains(t, diff.Changed, "service/worker/devices") + + // Reorder-only is a no-op. + oldSpec, nextSpec = mk(svc("a", "b")), mk(svc("b", "a")) + diff = DiffAppSpec(oldSpec, nextSpec) + assert.NotContains(t, diff.Changed, "service/worker/devices") + + // Nil vs empty stays a no-op. + oldSpec, nextSpec = mk(svc()), mk(svc([]string{}...)) + diff = DiffAppSpec(oldSpec, nextSpec) + assert.NotContains(t, diff.Changed, "service/worker/devices") +} diff --git a/internal/domain/app_l4.go b/internal/domain/app_l4.go new file mode 100644 index 000000000..195750470 --- /dev/null +++ b/internal/domain/app_l4.go @@ -0,0 +1,96 @@ +package domain + +import ( + "fmt" + "net" + "strconv" + "strings" +) + +// EntryPointListener is the canonical public bind of one installation +// entrypoint. An app L4 publish declaration must be exactly compatible +// with the entrypoint that will actually bind it: the runtime binds the +// entrypoint address, so a narrower publish (for example loopback) would +// silently be exposed on the entrypoint listener instead. +type EntryPointListener struct { + // Address is the configured listen address, "host:port" or ":port". + Address string + // Protocol is the entrypoint transport. + Protocol EntryPointProtocol +} + +// CanonicalBindHost normalizes a bind host: an empty host means "all +// interfaces" and is reported as 0.0.0.0 so an unset host and an explicit +// wildcard compare equal. +func CanonicalBindHost(host string) string { + host = strings.TrimSpace(host) + if host == "" { + return "0.0.0.0" + } + return host +} + +// EntryPointCarriesTransport reports whether an entrypoint protocol can +// carry the given app interface transport. +func EntryPointCarriesTransport(protocol EntryPointProtocol, transport NetworkProtocol) bool { + switch transport { + case NetworkProtocolTCP: + switch protocol { + case EntryPointProtocolTCP, EntryPointProtocolTLSMux, EntryPointProtocolSmartTCP: + return true + } + return false + case NetworkProtocolUDP: + return protocol == EntryPointProtocolUDP + default: + return false + } +} + +// ValidatePublishForListener requires a declared publish bind to equal the +// entrypoint listener address and the declared transport to be supported +// by the entrypoint protocol. Any mismatch is rejected before persistence +// or container effects; a caller must never substitute the entrypoint +// address for a narrower declared bind. +func ValidatePublishForListener(publish string, transport NetworkProtocol, listener EntryPointListener, entrypointName string) error { + host, port, err := ParsePublish(publish) + if err != nil { + return err + } + listenerHost, listenerPort, err := SplitListenerAddress(listener.Address) + if err != nil { + return fmt.Errorf("%w: entrypoint %q listener address %q is invalid: %s", ErrInvalidAppSpec, entrypointName, listener.Address, err) + } + if port != listenerPort || CanonicalBindHost(host) != CanonicalBindHost(listenerHost) { + return fmt.Errorf( + "%w: publish %q for entrypoint %q must match the listener bind %q exactly", + ErrInvalidAppSpec, publish, entrypointName, listener.Address, + ) + } + if !EntryPointCarriesTransport(listener.Protocol, transport) { + return fmt.Errorf( + "%w: entrypoint %q protocol %q cannot carry %s traffic", + ErrInvalidAppSpec, entrypointName, listener.Protocol, transport, + ) + } + return nil +} + +// SplitListenerAddress parses a configured listen address into its +// canonical host and port. An address without a port is invalid: the +// runtime always binds a concrete port. +func SplitListenerAddress(address string) (string, int, error) { + address = strings.TrimSpace(address) + if address == "" { + return "", 0, fmt.Errorf("empty listen address") + } + host, portText, err := net.SplitHostPort(address) + if err != nil { + return "", 0, err + } + port, err := strconv.Atoi(portText) + if err != nil || port < 1 || port > 65535 { + return "", 0, fmt.Errorf("port %q must be 1-65535", portText) + } + return host, port, nil +} diff --git a/internal/domain/app_l4_test.go b/internal/domain/app_l4_test.go new file mode 100644 index 000000000..173b8e58f --- /dev/null +++ b/internal/domain/app_l4_test.go @@ -0,0 +1,77 @@ +package domain_test + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +func TestValidatePublishForListener_ExactTupleAccepted(t *testing.T) { + cases := []struct { + name string + publish string + transport domain.NetworkProtocol + listener domain.EntryPointListener + entrypoint string + }{ + { + name: "tcp wildcard explicit", publish: "0.0.0.0:25432", transport: domain.NetworkProtocolTCP, + listener: domain.EntryPointListener{Address: "0.0.0.0:25432", Protocol: domain.EntryPointProtocolTCP}, entrypoint: "tcp", + }, + { + name: "tcp empty host equals wildcard listener", publish: "25432", transport: domain.NetworkProtocolTCP, + listener: domain.EntryPointListener{Address: ":25432", Protocol: domain.EntryPointProtocolTCP}, entrypoint: "tcp", + }, + { + name: "tcp smart entrypoint", publish: "0.0.0.0:25432", transport: domain.NetworkProtocolTCP, + listener: domain.EntryPointListener{Address: "0.0.0.0:25432", Protocol: domain.EntryPointProtocolSmartTCP}, entrypoint: "edge", + }, + { + name: "udp", publish: "0.0.0.0:28015", transport: domain.NetworkProtocolUDP, + listener: domain.EntryPointListener{Address: "0.0.0.0:28015", Protocol: domain.EntryPointProtocolUDP}, entrypoint: "udp", + }, + { + name: "loopback listener with loopback publish", publish: "127.0.0.1:1543", transport: domain.NetworkProtocolTCP, + listener: domain.EntryPointListener{Address: "127.0.0.1:1543", Protocol: domain.EntryPointProtocolTCP}, entrypoint: "local", + }, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + require.NoError(t, domain.ValidatePublishForListener(tc.publish, tc.transport, tc.listener, tc.entrypoint)) + }) + } +} + +func TestValidatePublishForListener_RejectsMismatch(t *testing.T) { + wildcard := domain.EntryPointListener{Address: "0.0.0.0:25432", Protocol: domain.EntryPointProtocolTCP} + cases := []struct { + name string + publish string + transport domain.NetworkProtocol + listener domain.EntryPointListener + }{ + {name: "loopback publish on wildcard listener", publish: "127.0.0.1:25432", transport: domain.NetworkProtocolTCP, listener: wildcard}, + {name: "port mismatch", publish: "0.0.0.0:15432", transport: domain.NetworkProtocolTCP, listener: wildcard}, + {name: "loopback vs wildcard host", publish: "127.0.0.1:25432", transport: domain.NetworkProtocolTCP, listener: wildcard}, + {name: "udp on tcp entrypoint", publish: "0.0.0.0:25432", transport: domain.NetworkProtocolUDP, listener: wildcard}, + {name: "tcp on udp entrypoint", publish: "0.0.0.0:25432", transport: domain.NetworkProtocolTCP, listener: domain.EntryPointListener{Address: "0.0.0.0:25432", Protocol: domain.EntryPointProtocolUDP}}, + {name: "hostname publish", publish: "db.internal:25432", transport: domain.NetworkProtocolTCP, listener: wildcard}, + {name: "missing port", publish: "0.0.0.0", transport: domain.NetworkProtocolTCP, listener: wildcard}, + {name: "invalid listener address", publish: "0.0.0.0:25432", transport: domain.NetworkProtocolTCP, listener: domain.EntryPointListener{Address: "25432", Protocol: domain.EntryPointProtocolTCP}}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + assert.ErrorIs(t, domain.ValidatePublishForListener(tc.publish, tc.transport, tc.listener, "tcp"), domain.ErrInvalidAppSpec) + }) + } +} + +func TestCanonicalBindHost(t *testing.T) { + assert.Equal(t, "0.0.0.0", domain.CanonicalBindHost("")) + assert.Equal(t, "0.0.0.0", domain.CanonicalBindHost(" ")) + assert.Equal(t, "127.0.0.1", domain.CanonicalBindHost("127.0.0.1")) + assert.Equal(t, "::1", domain.CanonicalBindHost("::1")) +} diff --git a/internal/domain/app_network.go b/internal/domain/app_network.go new file mode 100644 index 000000000..cd6f1bb78 --- /dev/null +++ b/internal/domain/app_network.go @@ -0,0 +1,120 @@ +package domain + +import ( + "crypto/sha256" + "fmt" + "strings" +) + +// DefaultNetworkPrefix applies when no installation prefix is configured. +const DefaultNetworkPrefix = "gordon" + +// Network roles identifying what a Gordon-managed network is for. +const ( + // AppNetworkRolePrivate marks the incarnation-owned network of one app. + // Containers of different apps never share it. + AppNetworkRolePrivate = "private" + // AppNetworkRoleShared marks a declared cross-app shared network. + AppNetworkRoleShared = "shared" +) + +// Network ownership label keys. Management follows the same fail-closed +// rule as volumes: an existing network is reused only when its labels +// prove Gordon ownership; anything else is refused, never adopted. +const ( + // LabelAppNetworkRole carries AppNetworkRolePrivate/AppNetworkRoleShared. + LabelAppNetworkRole = "gordon.app.network.role" + // LabelAppNetworkName carries the declared shared-network name, so a + // derived runtime name can be tied back to exactly one declaration. + LabelAppNetworkName = "gordon.app.network.name" +) + +// NormalizeNetworkPrefix returns the configured prefix or the default. +func NormalizeNetworkPrefix(prefix string) string { + if strings.TrimSpace(prefix) == "" { + return DefaultNetworkPrefix + } + return prefix +} + +// AppPrivateNetworkName derives the incarnation-owned network name from +// the app's stable internal UUID. The name is deterministic so recovery +// reattaches the same network, and it never derives from the public name: +// removing and re-adding an app under the same name yields a different +// incarnation UUID and therefore a different network. +func AppPrivateNetworkName(prefix, appID string) string { + sum := sha256.Sum256([]byte(appID)) + return fmt.Sprintf("%s-app-%x", NormalizeNetworkPrefix(prefix), sum[:6]) +} + +// AppSharedNetworkName derives the Gordon-managed name of a declared +// shared network. Deriving from the declaration means two apps that +// declare the same name join the same network, while a foreign network +// that happens to use the name is never mistaken for it. +func AppSharedNetworkName(prefix, declared string) string { + sum := sha256.Sum256([]byte(declared)) + return fmt.Sprintf("%s-shared-%x", NormalizeNetworkPrefix(prefix), sum[:6]) +} + +// AppPrivateNetworkLabels is the ownership label set stamped on an +// incarnation-owned network. +func AppPrivateNetworkLabels(app, appID string) map[string]string { + labels := map[string]string{ + LabelManaged: "true", + LabelApp: app, + LabelAppNetworkRole: AppNetworkRolePrivate, + } + if appID != "" { + labels[LabelAppID] = appID + } + return labels +} + +// AppSharedNetworkLabels is the ownership label set stamped on a declared +// shared network. A shared network belongs to the installation, not to a +// single app, so it carries no app identity. +func AppSharedNetworkLabels(declared string) map[string]string { + return map[string]string{ + LabelManaged: "true", + LabelAppNetworkRole: AppNetworkRoleShared, + LabelAppNetworkName: declared, + } +} + +// NetworkOwnedBy reports whether an observed network's labels prove the +// exact expected Gordon ownership. Extra labels are tolerated; every +// expected key must be present and equal, so a network created with +// different ownership (or none) is never adopted. +func NetworkOwnedBy(observed, expected map[string]string) bool { + if len(expected) == 0 { + return false + } + for key, want := range expected { + if observed[key] != want { + return false + } + } + return true +} + +// AppServiceSharedNetworks returns the declared shared-network memberships +// that the named service joins, preserving declaration order and +// de-duplicating by network name. +func AppServiceSharedNetworks(spec AppSpec, service string) []AppSharedNetwork { + var memberships []AppSharedNetwork + seen := map[string]struct{}{} + for _, network := range spec.Networks { + if _, ok := seen[network.Network]; ok { + continue + } + for _, member := range network.Services { + if member != service { + continue + } + seen[network.Network] = struct{}{} + memberships = append(memberships, network) + break + } + } + return memberships +} diff --git a/internal/domain/app_network_test.go b/internal/domain/app_network_test.go new file mode 100644 index 000000000..ee8f4e56e --- /dev/null +++ b/internal/domain/app_network_test.go @@ -0,0 +1,75 @@ +package domain_test + +import ( + "testing" + + "github.com/stretchr/testify/assert" + + "github.com/bnema/gordon/internal/domain" +) + +func TestAppPrivateNetworkName_DerivesFromUUIDNotName(t *testing.T) { + first := domain.AppPrivateNetworkName("gordon", "app-1") + second := domain.AppPrivateNetworkName("gordon", "app-2") + assert.NotEqual(t, first, second, "distinct incarnations must not share a network") + assert.Equal(t, first, domain.AppPrivateNetworkName("gordon", "app-1"), "derivation is deterministic for recovery") + assert.NotEqual(t, "default", first) + assert.NotEqual(t, "bridge", first) + + // Same public name, new incarnation UUID: a different network. + reused := domain.AppPrivateNetworkName("gordon", "app-9") + assert.NotEqual(t, first, reused, "name reuse must not reattach the old network") +} + +func TestAppPrivateNetworkName_AppliesDefaultPrefix(t *testing.T) { + assert.Equal(t, + domain.AppPrivateNetworkName("gordon", "app-1"), + domain.AppPrivateNetworkName("", "app-1"), + ) +} + +func TestAppSharedNetworkName_StablePerDeclaration(t *testing.T) { + assert.Equal(t, domain.AppSharedNetworkName("gordon", "db"), domain.AppSharedNetworkName("gordon", "db")) + assert.NotEqual(t, domain.AppSharedNetworkName("gordon", "db"), domain.AppSharedNetworkName("gordon", "cache")) +} + +func TestNetworkOwnedBy_RequiresEveryExpectedLabel(t *testing.T) { + expected := domain.AppPrivateNetworkLabels("blog", "app-1") + assert.True(t, domain.NetworkOwnedBy(expected, expected)) + assert.True(t, domain.NetworkOwnedBy(map[string]string{ + domain.LabelManaged: "true", + domain.LabelApp: "blog", + domain.LabelAppNetworkRole: domain.AppNetworkRolePrivate, + domain.LabelAppID: "app-1", + "extra": "tolerated", + }, expected)) + assert.False(t, domain.NetworkOwnedBy(map[string]string{domain.LabelManaged: "true"}, expected)) + assert.False(t, domain.NetworkOwnedBy(nil, expected)) + assert.False(t, domain.NetworkOwnedBy(expected, nil), "empty expectation is never proof of ownership") + assert.False(t, domain.NetworkOwnedBy(map[string]string{ + domain.LabelManaged: "true", + domain.LabelApp: "blog", + domain.LabelAppNetworkRole: domain.AppNetworkRolePrivate, + domain.LabelAppID: "app-2", + }, expected), "a different incarnation is foreign") +} + +func TestAppServiceSharedNetworks_FiltersAndDeduplicates(t *testing.T) { + spec := domain.AppSpec{ + Name: "blog", + Networks: []domain.AppSharedNetwork{ + {Network: "db", Services: []string{"web", "worker"}}, + {Network: "cache", Services: []string{"worker"}}, + {Network: "db", Services: []string{"web"}}, + }, + } + web := domain.AppServiceSharedNetworks(spec, "web") + assert.Equal(t, []domain.AppSharedNetwork{{Network: "db", Services: []string{"web", "worker"}}}, web) + + worker := domain.AppServiceSharedNetworks(spec, "worker") + assert.Len(t, worker, 2) + assert.Equal(t, "db", worker[0].Network) + assert.Equal(t, "cache", worker[1].Network) + + assert.Empty(t, domain.AppServiceSharedNetworks(spec, "absent")) +} diff --git a/internal/domain/app_readiness_test.go b/internal/domain/app_readiness_test.go new file mode 100644 index 000000000..4dd66b564 --- /dev/null +++ b/internal/domain/app_readiness_test.go @@ -0,0 +1,50 @@ +package domain_test + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +func TestValidateReadinessPath(t *testing.T) { + valid := []string{"/", "/healthz", "/healthz?probe=1", "/a/b-c_d.e"} + for _, path := range valid { + require.NoError(t, domain.ValidateReadinessPath(path), "path %q should be accepted", path) + } + + invalid := []string{ + "", + " ", + "healthz", + "@127.0.0.1:9/internal", + "//evil.example.com/path", + "http://evil.example.com/path", + "https://evil.example.com/path", + "/healthz\r\nHost: evil", + "/healthz\x00", + "/healthz\x7f", + "/he\talthz", + " /healthz", + "/healthz ", + "\t/healthz", + "/healthz\n", + } + for _, path := range invalid { + assert.Error(t, domain.ValidateReadinessPath(path), "path %q should be rejected", path) + } +} + +func TestAppSpecValidate_RejectsAuthorityInReadinessPath(t *testing.T) { + spec := domain.AppSpec{ + Name: "blog", + Services: []domain.AppService{{ + Name: "web", + Image: "img:1", + Readiness: domain.AppReadiness{Type: domain.AppReadinessHTTP, Path: "@127.0.0.1:9/internal", Timeout: 30}, + }}, + } + assert.ErrorIs(t, spec.Validate(), domain.ErrInvalidAppSpec) +} diff --git a/internal/domain/app_reservations.go b/internal/domain/app_reservations.go new file mode 100644 index 000000000..b13b550d1 --- /dev/null +++ b/internal/domain/app_reservations.go @@ -0,0 +1,142 @@ +package domain + +import ( + "fmt" + "sort" +) + +// ReservationsFor computes the global listener claims for a normalized +// spec. It is the single owner of reservation derivation; the apps and +// deployment use cases delegate to it. Only effective-public HTTP +// interfaces claim a host: internal interfaces are never published, so +// they hold no global listener reservation. +func ReservationsFor(spec AppSpec) []AppListenerReservation { + var reservations []AppListenerReservation + for _, svc := range spec.Services { + for _, h := range svc.HTTP { + if !h.IsPublic() { + continue + } + reservations = append(reservations, AppListenerReservation{ + Proto: "http", + Host: h.Host, + Service: svc.Name, + App: spec.Name, + }) + } + for _, t := range svc.TCP { + host, port, err := ParsePublish(t.Publish) + if err != nil { + continue + } + reservations = append(reservations, AppListenerReservation{ + Proto: "tcp", + IP: reservationIP(host), + Port: port, + Service: svc.Name, + App: spec.Name, + }) + } + for _, u := range svc.UDP { + host, port, err := ParsePublish(u.Publish) + if err != nil { + continue + } + reservations = append(reservations, AppListenerReservation{ + Proto: "udp", + IP: reservationIP(host), + Port: port, + Service: svc.Name, + App: spec.Name, + }) + } + } + sort.Slice(reservations, func(i, j int) bool { + if reservations[i].Proto != reservations[j].Proto { + return reservations[i].Proto < reservations[j].Proto + } + if reservations[i].Host != reservations[j].Host { + return reservations[i].Host < reservations[j].Host + } + if reservations[i].Port != reservations[j].Port { + return reservations[i].Port < reservations[j].Port + } + return reservations[i].IP < reservations[j].IP + }) + return reservations +} + +// reservationIP maps an empty publish host to the wildcard entry. +func reservationIP(host string) string { + if host == "" { + return "dual" + } + return host +} + +// CheckReservations validates a candidate reservation set against the +// global checkpoint plus wildcard-aware overlay rules. Same-app entries +// in existing are retained until withdrawal, so the candidate's own app +// entries never self-conflict. Intra-candidate claims overlap like +// cross-app claims: a wildcard L4 bind collides with a specific bind on +// the same proto and port. +func CheckReservations(existing []AppListenerReservation, candidate []AppListenerReservation, app string) error { + // Duplicate and overlapping claims within the candidate itself. + for i, res := range candidate { + for _, other := range candidate[:i] { + if ReservationsOverlap(res, other) { + return fmt.Errorf( + "%w: %s claims %s twice (service %s vs %s)", + ErrAppReservationConflict, + app, describeReservation(res), other.Service, res.Service, + ) + } + } + } + for _, res := range candidate { + for _, other := range existing { + if other.App == app { + continue + } + if ReservationsOverlap(res, other) { + return fmt.Errorf( + "%w: %s/%s conflicts with %s/%s on %s", + ErrAppReservationConflict, + app, res.Service, other.App, other.Service, describeReservation(res), + ) + } + } + } + return nil +} + +// ReservationsOverlap reports wildcard/specific conflicts within one +// proto namespace. +func ReservationsOverlap(a, b AppListenerReservation) bool { + if a.Proto != b.Proto { + return false + } + if a.Proto == "http" { + return a.Host == b.Host + } + if a.Port != b.Port { + return false + } + if IsReservationWildcard(a.IP) || IsReservationWildcard(b.IP) { + return true + } + return a.IP == b.IP +} + +// IsReservationWildcard matches wildcard and dual-family entries. +func IsReservationWildcard(ip string) bool { + return ip == "" || ip == "0.0.0.0" || ip == "::" || ip == "dual" +} + +// describeReservation renders a human-readable claim. +func describeReservation(res AppListenerReservation) string { + if res.Proto == "http" { + return "http://" + res.Host + } + return fmt.Sprintf("%s://%s:%d", res.Proto, res.IP, res.Port) +} diff --git a/internal/domain/app_reservations_test.go b/internal/domain/app_reservations_test.go new file mode 100644 index 000000000..9f7b2bc42 --- /dev/null +++ b/internal/domain/app_reservations_test.go @@ -0,0 +1,83 @@ +package domain_test + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +func reservationSpec() domain.AppSpec { + return domain.AppSpec{ + Name: "blog", + Services: []domain.AppService{ + { + Name: "web", Image: "img:1", + HTTP: []domain.AppHTTPInterface{{Host: "b.example.com", Port: 8080, TLS: "auto"}}, + TCP: []domain.AppTCPInterface{{Entrypoint: "tcp", Port: 9000, Publish: "9000"}}, + }, + }, + } +} + +func TestReservationsFor_SortedAndNamespaced(t *testing.T) { + reservations := domain.ReservationsFor(reservationSpec()) + require.Len(t, reservations, 2) + assert.Equal(t, "http", reservations[0].Proto) + assert.Equal(t, "tcp", reservations[1].Proto) + assert.Equal(t, "dual", reservations[1].IP) + assert.Equal(t, "web", reservations[0].Service) + assert.Equal(t, "blog", reservations[0].App) +} + +func TestCheckReservations_RejectsDuplicateInsideCandidate(t *testing.T) { + spec := reservationSpec() + dup := spec.Services[0] + dup.Name = "web2" + spec.Services = append(spec.Services, dup) + + err := domain.CheckReservations(nil, domain.ReservationsFor(spec), "blog") + require.ErrorIs(t, err, domain.ErrAppReservationConflict) + assert.Contains(t, err.Error(), "claims") +} + +func TestCheckReservations_RejectsWildcardOverlapInsideCandidate(t *testing.T) { + candidate := []domain.AppListenerReservation{ + {Proto: "tcp", IP: "0.0.0.0", Port: 9000, Service: "one", App: "tcp-overlap"}, + {Proto: "tcp", IP: "127.0.0.1", Port: 9000, Service: "two", App: "tcp-overlap"}, + } + err := domain.CheckReservations(nil, candidate, "tcp-overlap") + require.ErrorIs(t, err, domain.ErrAppReservationConflict) +} + +func TestCheckReservations_RejectsCrossAppWildcardConflict(t *testing.T) { + existing := []domain.AppListenerReservation{ + {Proto: "tcp", IP: "dual", Port: 9000, Service: "s", App: "other"}, + } + candidate := []domain.AppListenerReservation{ + {Proto: "tcp", IP: "127.0.0.1", Port: 9000, Service: "s", App: "shop"}, + } + err := domain.CheckReservations(existing, candidate, "shop") + require.ErrorIs(t, err, domain.ErrAppReservationConflict) + assert.Contains(t, err.Error(), "conflicts with") +} + +func TestCheckReservations_SameAppNeverSelfConflicts(t *testing.T) { + existing := []domain.AppListenerReservation{ + {Proto: "http", Host: "blog.example.com", Service: "web", App: "blog"}, + } + candidate := []domain.AppListenerReservation{ + {Proto: "http", Host: "blog.example.com", Service: "web", App: "blog"}, + } + require.NoError(t, domain.CheckReservations(existing, candidate, "blog")) +} + +func TestCheckReservations_SamePortDifferentProtoCoexists(t *testing.T) { + candidate := []domain.AppListenerReservation{ + {Proto: "tcp", IP: "0.0.0.0", Port: 9000, Service: "s", App: "a"}, + {Proto: "udp", IP: "0.0.0.0", Port: 9000, Service: "s", App: "a"}, + } + require.NoError(t, domain.CheckReservations(nil, candidate, "a")) +} diff --git a/internal/domain/app_reservations_visibility_test.go b/internal/domain/app_reservations_visibility_test.go new file mode 100644 index 000000000..f9460ee57 --- /dev/null +++ b/internal/domain/app_reservations_visibility_test.go @@ -0,0 +1,37 @@ +package domain_test + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +// TestReservationsFor_InternalHTTPClaimsNothing proves internal HTTP holds +// no global listener reservation: there is no host to reserve, so a second +// app may use the same internal port without conflict. +func TestReservationsFor_InternalHTTPClaimsNothing(t *testing.T) { + spec := domain.AppSpec{ + Name: "blog", + Services: []domain.AppService{{ + Name: "api", + Image: "registry.example.com/blog/api:1.0.0", + HTTP: []domain.AppHTTPInterface{{Port: 8080, Visibility: domain.AppVisibilityInternal}}, + }}, + } + assert.Empty(t, domain.ReservationsFor(spec)) + + public := domain.AppSpec{ + Name: "blog", + Services: []domain.AppService{{ + Name: "web", + Image: "registry.example.com/blog/web:1.0.0", + HTTP: []domain.AppHTTPInterface{{Host: "blog.example.com", Port: 8080, TLS: "auto"}}, + }}, + } + reservations := domain.ReservationsFor(public) + require.Len(t, reservations, 1) + assert.Equal(t, "blog.example.com", reservations[0].Host) +} diff --git a/internal/domain/app_spec.go b/internal/domain/app_spec.go new file mode 100644 index 000000000..af7358a00 --- /dev/null +++ b/internal/domain/app_spec.go @@ -0,0 +1,1204 @@ +package domain + +import ( + "fmt" + "net" + "path" + "reflect" + "regexp" + "slices" + "sort" + "strconv" + "strings" + "time" +) + +// App manifest validation is owned by the domain. +// The parser adapter (internal/adapters/in/appmanifest) decodes TOML +// into these types; all semantic rules live here, not in the parser. + +var ( + appNamePattern = regexp.MustCompile(`^[a-z0-9]([a-z0-9-]{0,61}[a-z0-9])?$`) + serviceNamePattern = regexp.MustCompile(`^[a-z0-9]([a-z0-9_.\-]{0,61}[a-z0-9])?$`) + dnsLabelPattern = regexp.MustCompile(`^[a-z0-9]([a-z0-9-]{0,61}[a-z0-9])?$`) +) + +// Reserved app names that collide with installation identity. +var reservedAppNames = map[string]struct{}{ + "gordon": {}, + "registry": {}, + "admin": {}, + "localhost": {}, +} + +// App readiness types. HTTP is new in v3. +const ( + AppReadinessNone = "none" + AppReadinessTCP = "tcp" + AppReadinessHTTP = "http" + AppReadinessLog = "log" +) + +// App TLS modes for HTTP interfaces. +const ( + AppTLSAuto = "auto" + AppTLSAlways = "always" + AppTLSNever = "never" +) + +// App visibility modes for HTTP interfaces. public keeps the historical +// host/TLS plane; internal is reachable only from the app private network. +const ( + AppVisibilityPublic = "public" + AppVisibilityInternal = "internal" +) + +// App database engines. Only postgres in v3. +const ( + AppDBPostgres = "postgres" +) + +// Backup schedules reuse the BackupSchedule vocabulary. +var appBackupSchedules = map[string]struct{}{ + "hourly": {}, + "daily": {}, + "weekly": {}, + "monthly": {}, +} + +// App size/time bounds frozen by the contract. +const ( + // MaxAppEnvValueLen caps a single [env] value. + MaxAppEnvValueLen = 64 * 1024 + // AppDefaultReadinessTimeout applies when timeout is unset. + AppDefaultReadinessTimeout = 30 * time.Second + // AppMinReadinessTimeout bounds readiness timeout below. + AppMinReadinessTimeout = time.Second + // AppMaxReadinessTimeout bounds readiness timeout above. + AppMaxReadinessTimeout = 10 * time.Minute + // AppDefaultStopGrace applies when stop_grace is unset. + AppDefaultStopGrace = 30 * time.Second + // AppMaxStopGrace caps stop_grace. + AppMaxStopGrace = 5 * time.Minute +) + +// AppSpec is the normalized form of one app manifest file. +type AppSpec struct { + Name string + Env map[string]string + Services []AppService + Networks []AppSharedNetwork +} + +// AppService is one explicitly named image-backed service. +// v3 runs exactly one container per service: no replicas field. +// RCON is ordinary TCP: use TCP interfaces, no special RCON kind. +type AppService struct { + Name string + Image string + Command []string + StopGrace time.Duration + Readiness AppReadiness + HTTP []AppHTTPInterface + TCP []AppTCPInterface + UDP []AppUDPInterface + Secrets map[string]string + Volumes []AppVolume + Binds []AppBind + // Devices lists logical device names granted by administrative + // [app_devices] policy. Names resolve to CDI device IDs at activation + // time; the logical names (not host resolution) persist in revisions. + Devices []string + Databases []AppDatabase + Backup AppBackup + // LogExportDisabled opts the service out of OTLP log export. The + // manifest resolves it from [services..telemetry] logs, then + // the app-level [telemetry] logs, then the default (export on). The + // zero value keeps export enabled for states persisted before the + // field existed. + LogExportDisabled bool +} + +// AppReadiness is the explicit readiness check for a service. +type AppReadiness struct { + Type string + Path string + Contains string + Port int + Timeout time.Duration +} + +// AppHTTPInterface is one HTTP service interface. Visibility is explicit +// interface state: public HTTP keeps the historical host/TLS behavior, +// internal HTTP declares neither host nor tls and is reachable only from +// the app private network. +type AppHTTPInterface struct { + Host string + Port int + TLS string + // Visibility is public or internal. The zero value normalizes to + // public when read or projected, so manifests and stored state + // written before visibility existed keep their meaning. + Visibility string +} + +// EffectiveVisibility normalizes absent visibility to public. +func (h AppHTTPInterface) EffectiveVisibility() string { + if h.Visibility == "" { + return AppVisibilityPublic + } + return h.Visibility +} + +// IsPublic reports whether the interface is served on the public plane. +func (h AppHTTPInterface) IsPublic() bool { + return h.EffectiveVisibility() == AppVisibilityPublic +} + +// IsInternal reports whether the interface is reachable only from the +// app private network. +func (h AppHTTPInterface) IsInternal() bool { + return h.EffectiveVisibility() == AppVisibilityInternal +} + +// IsPublicHTTP reports whether a service declares at least one +// effective-public HTTP interface. +func (s AppService) IsPublicHTTP() bool { + for _, h := range s.HTTP { + if h.IsPublic() { + return true + } + } + return false +} + +// InternallyOnlyPort reports whether port is declared by internal HTTP +// interfaces and by no externally backed HTTP/TCP interface. Such a port +// is reached over the private network only and must never gain a host +// publication, including through readiness metadata. +func (s AppService) InternallyOnlyPort(port int) bool { + internal := false + for _, h := range s.HTTP { + if h.Port != port { + continue + } + if !h.IsInternal() { + return false + } + internal = true + } + for _, t := range s.TCP { + if t.Port == port { + return false + } + } + return internal +} + +// AppTCPInterface is one TCP service interface. +type AppTCPInterface struct { + Entrypoint string + Port int + Publish string +} + +// AppUDPInterface is one UDP service interface. +type AppUDPInterface struct { + Entrypoint string + Port int + Publish string +} + +// AppVolume is one named volume mount. +type AppVolume struct { + Name string + Path string + ReadOnly bool +} + +// AppBind is one named read-only host bind mount. +// Name is the stable identity reused across deploys; Path is the +// container destination. +type AppBind struct { + Name string + Path string + ReadOnly bool +} + +// AppDatabase is one explicit database declaration. +type AppDatabase struct { + Name string + Type string + Schedule string +} + +// AppBackup declares backup targets by reference. +type AppBackup struct { + Postgres []string + Volume []string +} + +// AppSharedNetwork is one shared-network membership declaration. +type AppSharedNetwork struct { + Network string + Services []string + Aliases []string +} + +// ValidateAppName checks the app identity rules. +func ValidateAppName(name string) error { + if !appNamePattern.MatchString(name) { + return fmt.Errorf("%w: app name %q must be a DNS label (lowercase alphanumerics and hyphens, max 63)", ErrInvalidAppSpec, name) + } + if strings.Contains(name, "--") { + return fmt.Errorf("%w: app name %q must not contain -- (reserved separator)", ErrInvalidAppSpec, name) + } + if _, reserved := reservedAppNames[strings.ToLower(name)]; reserved { + return fmt.Errorf("%w: app name %q is reserved", ErrInvalidAppSpec, name) + } + return nil +} + +// ValidateServiceName checks the service identity rules. +func ValidateServiceName(name string) error { + if !serviceNamePattern.MatchString(name) { + return fmt.Errorf("%w: service name %q must match [a-z0-9_.-], max 63", ErrInvalidAppSpec, name) + } + return nil +} + +// ValidateVolumeName checks volume names (service charset plus -- ban). +func ValidateVolumeName(name string) error { + if err := ValidateServiceName(name); err != nil { + return err + } + if strings.Contains(name, "--") { + return fmt.Errorf("%w: volume name %q must not contain -- (reserved separator)", ErrInvalidAppSpec, name) + } + return nil +} + +// ValidateSecretName checks service-local secret names. +func ValidateSecretName(name string) error { + return ValidateServiceName(name) +} + +// ValidateBindName checks the stable bind identity (service charset plus -- ban). +func ValidateBindName(name string) error { + if !serviceNamePattern.MatchString(name) { + return fmt.Errorf("%w: bind name %q must match [a-z0-9_.-], max 63", ErrInvalidAppSpec, name) + } + if strings.Contains(name, "--") { + return fmt.Errorf("%w: bind name %q must not contain -- (reserved separator)", ErrInvalidAppSpec, name) + } + return nil +} + +// sensitiveBindDestinations lists container paths a manifest bind must never +// shadow. Server policy may refuse more, but these are always sensitive. +var sensitiveBindDestinations = map[string]struct{}{ + "/": {}, + "/proc": {}, + "/sys": {}, + "/dev": {}, + "/boot": {}, +} + +// IsSensitiveBindDestination reports whether dest is a reserved container path +// or lies below one. Mounting a child such as /proc/self is as dangerous as +// shadowing the reserved root itself. +func IsSensitiveBindDestination(dest string) bool { + if dest == "/" { + return true + } + for reserved := range sensitiveBindDestinations { + if reserved != "/" && (dest == reserved || strings.HasPrefix(dest, reserved+"/")) { + return true + } + } + return false +} + +// ValidateBindDestination checks one bind destination: absolute, clean, +// normalized, and never a sensitive container path. It is pure and reusable +// by manifest validation and by server-side source policy. +func ValidateBindDestination(dest string) error { + if dest == "" { + return fmt.Errorf("%w: bind destination must not be empty", ErrInvalidAppSpec) + } + if !path.IsAbs(dest) { + return fmt.Errorf("%w: bind destination %q must be absolute", ErrInvalidAppSpec, dest) + } + if path.Clean(dest) != dest { + return fmt.Errorf("%w: bind destination %q must be a normalized clean path", ErrInvalidAppSpec, dest) + } + if IsSensitiveBindDestination(dest) { + return fmt.Errorf("%w: bind destination %q is a sensitive container path", ErrInvalidAppSpec, dest) + } + return nil +} + +// NormalizeServiceName applies the runtime-identifier normalization. +func NormalizeServiceName(name string) string { + return strings.NewReplacer(".", "-", "_", "-", "/", "-").Replace(name) +} + +// LogicalServiceIdentity returns the stable gordon--- identity. +func LogicalServiceIdentity(app, service string) string { + return "gordon-" + app + "--" + NormalizeServiceName(service) +} + +// RuntimeVolumeName returns the generated gordon-----vol--. +func RuntimeVolumeName(app, service, volume string) string { + return LogicalServiceIdentity(app, service) + "--vol--" + volume +} + +// AppSecretPath returns the pass path gordon/apps///. +// Prefer AppSecretPathForID when the stable internal UUID is known: +// the UUID-keyed path survives name reuse without implicit adoption. +func AppSecretPath(app, service, name string) string { + return AppSecretPathForID(app, app, service, name) +} + +// AppSecretPathForID returns the pass path for one secret. id is the +// stable internal UUID; app is the public name used as a fallback when +// id is empty (legacy records written before UUID assignment). New +// writes must always pass the record ID so a removed app's secrets are +// never implicitly adopted by a new app reusing the name. +func AppSecretPathForID(id, app, service, name string) string { + key := id + if key == "" { + key = app + } + return "gordon/apps/" + key + "/" + NormalizeServiceName(service) + "/" + name +} + +// CanonicalHTTPHost lowercases and strips a trailing dot. +func CanonicalHTTPHost(host string) string { + host = strings.ToLower(strings.TrimSpace(host)) + return strings.TrimSuffix(host, ".") +} + +// ParsePublish splits a publish bind into host and port. +// Acceptable forms: "IP:port" or "port". Hostnames are rejected. +func ParsePublish(publish string) (string, int, error) { + publish = strings.TrimSpace(publish) + if publish == "" { + return "", 0, fmt.Errorf("%w: publish must not be empty", ErrInvalidAppSpec) + } + host := "" + portText := publish + if h, p, err := net.SplitHostPort(publish); err == nil { + host = h + portText = p + } + if host != "" && net.ParseIP(host) == nil { + return "", 0, fmt.Errorf("%w: publish host %q must be a literal IP, not a hostname", ErrInvalidAppSpec, host) + } + port, err := strconv.Atoi(portText) + if err != nil || port < 1 || port > 65535 { + return "", 0, fmt.Errorf("%w: publish port %q must be 1-65535", ErrInvalidAppSpec, portText) + } + return host, port, nil +} + +// Validate checks the full normalized spec. +func (s AppSpec) Validate() error { + return s.validateFull() +} + +// CheckEnvSecretCollisions re-validates [env]/secret key disjointness on a +// stored spec. Deployment preflight calls it defensively before any +// workload mutation; overlapping keys are invalid with no precedence. +func (s AppSpec) CheckEnvSecretCollisions() error { + return checkEnvSecretKeyDisjoint(s) +} + +// validateFull runs the full normalized validation. +func (s AppSpec) validateFull() error { + if err := ValidateAppName(s.Name); err != nil { + return err + } + if len(s.Services) == 0 { + return fmt.Errorf("%w: app %q must declare at least one service", ErrInvalidAppSpec, s.Name) + } + if err := s.validateEnv(); err != nil { + return err + } + if err := s.validateServices(); err != nil { + return err + } + if err := checkEnvSecretKeyDisjoint(s); err != nil { + return err + } + if err := checkVolumeOwnership(s); err != nil { + return err + } + for i := range s.Networks { + if err := s.Networks[i].validate(s); err != nil { + return err + } + } + return nil +} + +// validateEnv checks app-wide public environment values. +func (s AppSpec) validateEnv() error { + for key, value := range s.Env { + if err := ValidateEnvKey(key); err != nil { + return fmt.Errorf("%w: env key %q: %v", ErrInvalidAppSpec, key, err) + } + if strings.Contains(value, "\n") { + return fmt.Errorf("%w: env value for %q must not contain newlines", ErrInvalidAppSpec, key) + } + if len(value) == 0 || len(value) > MaxAppEnvValueLen { + return fmt.Errorf("%w: env value for %q must be 1-%d bytes", ErrInvalidAppSpec, key, MaxAppEnvValueLen) + } + if ContainsSecretReference(value) { + return fmt.Errorf("%w: env value for %q must not contain secret references", ErrInvalidAppSpec, key) + } + } + return nil +} + +// validateServices checks service identity uniqueness and delegates per-service rules. +func (s AppSpec) validateServices() error { + seenServices := map[string]struct{}{} + seenNormalized := map[string]string{} + for i := range s.Services { + svc := &s.Services[i] + if err := ValidateServiceName(svc.Name); err != nil { + return err + } + if _, ok := seenServices[svc.Name]; ok { + return fmt.Errorf("%w: duplicate service name %q", ErrInvalidAppSpec, svc.Name) + } + seenServices[svc.Name] = struct{}{} + normalized := NormalizeServiceName(svc.Name) + if prev, ok := seenNormalized[normalized]; ok { + return fmt.Errorf("%w: service name %q normalizes to the same runtime identifier as %q", ErrInvalidAppSpec, svc.Name, prev) + } + seenNormalized[normalized] = svc.Name + if err := svc.validate(); err != nil { + return err + } + } + return nil +} + +// checkEnvSecretKeyDisjoint enforces [env] keys disjoint from secret map keys. +func checkEnvSecretKeyDisjoint(s AppSpec) error { + for key := range s.Env { + for _, svc := range s.Services { + if _, ok := svc.Secrets[key]; ok { + return fmt.Errorf("%w: env key %q collides with a secret key in service %q", ErrInvalidAppSpec, key, svc.Name) + } + } + } + return nil +} + +// checkVolumeOwnership enforces one claimant per volume name. +func checkVolumeOwnership(s AppSpec) error { + claimants := map[string]string{} + for _, svc := range s.Services { + for _, vol := range svc.Volumes { + if prev, ok := claimants[vol.Name]; ok { + return fmt.Errorf("%w: volume %q claimed by both %q and %q", ErrInvalidAppSpec, vol.Name, prev, svc.Name) + } + claimants[vol.Name] = svc.Name + } + } + return nil +} + +// validate checks one service. +func (s *AppService) validate() error { + if strings.TrimSpace(s.Image) == "" { + return fmt.Errorf("%w: service %q requires an image", ErrInvalidAppSpec, s.Name) + } + if s.StopGrace <= 0 || s.StopGrace > AppMaxStopGrace { + return fmt.Errorf("%w: service %q stop_grace must be within (0, 5m]", ErrInvalidAppSpec, s.Name) + } + if err := s.Readiness.validate(s); err != nil { + return err + } + if err := s.validateInterfaces(); err != nil { + return err + } + if err := s.validateSecrets(); err != nil { + return err + } + volumes, err := s.validateVolumes() + if err != nil { + return err + } + if err := s.validateBinds(); err != nil { + return err + } + if err := s.validateDevices(); err != nil { + return err + } + if err := s.validateDatabases(volumes); err != nil { + return err + } + return nil +} + +// validateInterfaces checks HTTP/TCP/UDP entries and their port ownership. +func (s *AppService) validateInterfaces() error { + for i := range s.HTTP { + if err := s.HTTP[i].validate(s.Name); err != nil { + return err + } + } + for i := range s.TCP { + if err := s.TCP[i].validate(s.Name); err != nil { + return err + } + } + for i := range s.UDP { + if err := s.UDP[i].validate(s.Name); err != nil { + return err + } + } + return s.validateInterfacePorts() +} + +// validateInterfacePorts rejects one container port claimed by both an +// internal HTTP interface and an externally backed HTTP/TCP interface: +// publication is socket-level, so publishing the port would expose the +// internal listener beyond the private network. Internal ports must also +// be declared at most once, since they carry no host to tell them apart. +func (s *AppService) validateInterfacePorts() error { + external := map[int]string{} + for _, h := range s.HTTP { + if h.IsPublic() { + external[h.Port] = "http" + } + } + for _, t := range s.TCP { + external[t.Port] = "tcp" + } + seenInternal := map[int]struct{}{} + for _, h := range s.HTTP { + if !h.IsInternal() { + continue + } + if kind, ok := external[h.Port]; ok { + return fmt.Errorf("%w: service %q internal http port %d is also declared by an externally backed %s interface", ErrInvalidAppSpec, s.Name, h.Port, kind) + } + if _, ok := seenInternal[h.Port]; ok { + return fmt.Errorf("%w: service %q internal http port %d is declared more than once", ErrInvalidAppSpec, s.Name, h.Port) + } + seenInternal[h.Port] = struct{}{} + } + return nil +} + +// validateSecrets checks the ENV-key to secret-name map. +func (s *AppService) validateSecrets() error { + for envKey, secretName := range s.Secrets { + if err := ValidateEnvKey(envKey); err != nil { + return fmt.Errorf("%w: service %q secret env key %q: %v", ErrInvalidAppSpec, s.Name, envKey, err) + } + if err := ValidateSecretName(secretName); err != nil { + return fmt.Errorf("%w: service %q secret name %q: %v", ErrInvalidAppSpec, s.Name, secretName, err) + } + } + return nil +} + +// validateVolumes checks volume declarations and returns known names. +func (s *AppService) validateVolumes() (map[string]struct{}, error) { + seenVolumes := map[string]struct{}{} + for i := range s.Volumes { + vol := &s.Volumes[i] + if err := ValidateVolumeName(vol.Name); err != nil { + return nil, fmt.Errorf("%w: service %q: %v", ErrInvalidAppSpec, s.Name, err) + } + if _, ok := seenVolumes[vol.Name]; ok { + return nil, fmt.Errorf("%w: service %q duplicate volume %q", ErrInvalidAppSpec, s.Name, vol.Name) + } + seenVolumes[vol.Name] = struct{}{} + if !path.IsAbs(vol.Path) || path.Clean(vol.Path) != vol.Path { + return nil, fmt.Errorf("%w: service %q volume %q path must be absolute and normalized", ErrInvalidAppSpec, s.Name, vol.Name) + } + } + return seenVolumes, nil +} + +// validateBinds checks bind names, destinations, duplicates, and volume collisions. +func (s *AppService) validateBinds() error { + volumePaths := make(map[string]struct{}, len(s.Volumes)) + for _, vol := range s.Volumes { + volumePaths[vol.Path] = struct{}{} + } + seenNames := map[string]struct{}{} + seenDests := map[string]struct{}{} + for i := range s.Binds { + bind := &s.Binds[i] + if err := ValidateBindName(bind.Name); err != nil { + return fmt.Errorf("%w: service %q: %v", ErrInvalidAppSpec, s.Name, err) + } + if _, ok := seenNames[bind.Name]; ok { + return fmt.Errorf("%w: service %q duplicate bind name %q", ErrInvalidAppSpec, s.Name, bind.Name) + } + seenNames[bind.Name] = struct{}{} + if err := ValidateBindDestination(bind.Path); err != nil { + return fmt.Errorf("%w: service %q bind %q: %v", ErrInvalidAppSpec, s.Name, bind.Name, err) + } + if _, ok := seenDests[bind.Path]; ok { + return fmt.Errorf("%w: service %q duplicate bind destination %q", ErrInvalidAppSpec, s.Name, bind.Path) + } + seenDests[bind.Path] = struct{}{} + if _, ok := volumePaths[bind.Path]; ok { + return fmt.Errorf("%w: service %q bind destination %q collides with a declared volume", ErrInvalidAppSpec, s.Name, bind.Path) + } + } + return nil +} + +// validateDevices checks logical device names and duplicates. Authorization +// against administrative policy happens at apply/deploy time, not here: +// the manifest shape stays valid while policy may refuse to serve it. +func (s *AppService) validateDevices() error { + seen := map[string]struct{}{} + for _, name := range s.Devices { + if err := ValidateDeviceName(name); err != nil { + return fmt.Errorf("%w: service %q: %v", ErrInvalidAppSpec, s.Name, err) + } + if _, ok := seen[name]; ok { + return fmt.Errorf("%w: service %q duplicate device %q", ErrInvalidAppSpec, s.Name, name) + } + seen[name] = struct{}{} + } + return nil +} + +// validateDatabases checks database declarations and backup references. +func (s *AppService) validateDatabases(seenVolumes map[string]struct{}) error { + seenDBs := map[string]struct{}{} + for i := range s.Databases { + db := &s.Databases[i] + if strings.TrimSpace(db.Name) == "" { + return fmt.Errorf("%w: service %q database name must not be empty", ErrInvalidAppSpec, s.Name) + } + if _, ok := seenDBs[db.Name]; ok { + return fmt.Errorf("%w: service %q duplicate database %q", ErrInvalidAppSpec, s.Name, db.Name) + } + seenDBs[db.Name] = struct{}{} + if db.Type != AppDBPostgres { + return fmt.Errorf("%w: service %q database %q type must be postgres", ErrInvalidAppSpec, s.Name, db.Name) + } + if _, ok := appBackupSchedules[db.Schedule]; !ok { + return fmt.Errorf("%w: service %q database %q schedule must be hourly|daily|weekly|monthly", ErrInvalidAppSpec, s.Name, db.Name) + } + } + for _, ref := range s.Backup.Postgres { + if _, ok := seenDBs[ref]; !ok { + return fmt.Errorf("%w: service %q backup references unknown database %q", ErrInvalidAppSpec, s.Name, ref) + } + } + for _, ref := range s.Backup.Volume { + if _, ok := seenVolumes[ref]; !ok { + return fmt.Errorf("%w: service %q backup references unknown volume %q", ErrInvalidAppSpec, s.Name, ref) + } + } + return nil +} + +// validate checks readiness rules including UDP restrictions and port selection. +func (r AppReadiness) validate(svc *AppService) error { + if err := r.validateType(svc); err != nil { + return err + } + if r.Timeout < AppMinReadinessTimeout || r.Timeout > AppMaxReadinessTimeout { + return fmt.Errorf("%w: service %q readiness timeout must be within [1s, 10m]", ErrInvalidAppSpec, svc.Name) + } + return nil +} + +// validateType checks the type-specific readiness constraints. +func (r AppReadiness) validateType(svc *AppService) error { + switch r.Type { + case "", AppReadinessNone: + if r.Path != "" || r.Contains != "" || r.Port != 0 { + return fmt.Errorf("%w: service %q readiness none takes no path/contains/port", ErrInvalidAppSpec, svc.Name) + } + return nil + case AppReadinessTCP, AppReadinessHTTP: + return r.validateL4(svc) + case AppReadinessLog: + if strings.TrimSpace(r.Path) == "" || strings.TrimSpace(r.Contains) == "" { + return fmt.Errorf("%w: service %q log readiness requires path and contains", ErrInvalidAppSpec, svc.Name) + } + return nil + default: + return fmt.Errorf("%w: service %q readiness type must be none|tcp|http|log", ErrInvalidAppSpec, svc.Name) + } +} + +// validateL4 checks TCP/HTTP readiness including UDP exclusion and port selection. +func (r AppReadiness) validateL4(svc *AppService) error { + if hasUDPOnly(svc) { + return fmt.Errorf("%w: service %q with UDP-only interfaces must use none or log readiness", ErrInvalidAppSpec, svc.Name) + } + if r.Port == 0 && readinessPortRequired(svc) { + return fmt.Errorf("%w: service %q interfaces do not select one readiness port, readiness.port is required", ErrInvalidAppSpec, svc.Name) + } + if r.Port != 0 && !hasTCPContainerPort(svc, r.Port) { + return fmt.Errorf("%w: service %q readiness.port %d matches no declared container port", ErrInvalidAppSpec, svc.Name, r.Port) + } + if r.Type == AppReadinessHTTP { + if err := ValidateReadinessPath(r.Path); err != nil { + return fmt.Errorf("%w: service %q: %s", ErrInvalidAppSpec, svc.Name, err) + } + } + return nil +} + +// ValidateReadinessPath requires an origin-form request path: a single +// leading slash, no scheme or authority, and no control characters. The +// path is appended to an immutable loopback URL, so rejecting everything +// that could be read as a different authority keeps the probe on the +// declared backend. Surrounding whitespace is rejected rather than +// trimmed: the stored path is used verbatim, and a silently trimmed probe +// would target a different path than the manifest declares. +func ValidateReadinessPath(path string) error { + if strings.TrimSpace(path) == "" { + return fmt.Errorf("http readiness requires path") + } + if path != strings.TrimSpace(path) { + return fmt.Errorf("http readiness path must not have leading or trailing whitespace") + } + if !strings.HasPrefix(path, "/") || strings.HasPrefix(path, "//") { + return fmt.Errorf("http readiness path must be an origin-form path beginning with one slash") + } + for _, r := range path { + if r < 0x20 || r == 0x7f { + return fmt.Errorf("http readiness path must not contain control characters") + } + } + return nil +} + +// hasUDPOnly reports services with UDP interfaces and no TCP/HTTP interface. +func hasUDPOnly(svc *AppService) bool { + if len(svc.UDP) == 0 { + return false + } + return len(svc.HTTP) == 0 && len(svc.TCP) == 0 +} + +// readinessPortRequired reports whether an omitted readiness.port leaves +// the probe target ambiguous. One effective-public HTTP port stays +// selectable even when internal HTTP interfaces are also declared, and a +// lone TCP-capable interface selects itself. Several public HTTP ports, any +// mix of TCP and HTTP, or several TCP interfaces must name it explicitly. +func readinessPortRequired(svc *AppService) bool { + if _, ok := singleEffectivePublicHTTPPort(svc); ok { + return false + } + return countTCPInterfaces(svc) > 1 +} + +// singleEffectivePublicHTTPPort returns the only distinct effective-public +// HTTP container port when the service declares at least one and no TCP +// interface. Internal HTTP interfaces never contribute: they are not +// reachable through a host publication, so they cannot make the public +// backend ambiguous. +func singleEffectivePublicHTTPPort(svc *AppService) (int, bool) { + if len(svc.TCP) > 0 { + return 0, false + } + port := 0 + for _, h := range svc.HTTP { + if !h.IsPublic() { + continue + } + if port != 0 && port != h.Port { + return 0, false + } + port = h.Port + } + if port == 0 { + return 0, false + } + return port, true +} + +// countTCPInterfaces counts http/tcp container ports. +func countTCPInterfaces(svc *AppService) int { + return len(svc.HTTP) + len(svc.TCP) +} + +// hasTCPContainerPort checks readiness.port against declared ports. +func hasTCPContainerPort(svc *AppService, port int) bool { + for _, h := range svc.HTTP { + if h.Port == port { + return true + } + } + for _, t := range svc.TCP { + if t.Port == port { + return true + } + } + return false +} + +// validate checks one HTTP interface against its visibility. Public +// interfaces keep the historical host/TLS rules; internal interfaces are +// private-network only and must not claim a host or a TLS mode. +func (h AppHTTPInterface) validate(service string) error { + switch h.EffectiveVisibility() { + case AppVisibilityPublic: + if h.Host == "" { + return fmt.Errorf("%w: service %q http host is required", ErrInvalidAppSpec, service) + } + if _, ok := CanonicalRouteDomain(h.Host); !ok { + return fmt.Errorf("%w: service %q http host %q is not a valid public hostname", ErrInvalidAppSpec, service, h.Host) + } + if h.Port < 1 || h.Port > 65535 { + return fmt.Errorf("%w: service %q http port must be 1-65535", ErrInvalidAppSpec, service) + } + switch h.TLS { + case AppTLSAuto, AppTLSAlways, AppTLSNever: + default: + return fmt.Errorf("%w: service %q http tls must be auto|always|never", ErrInvalidAppSpec, service) + } + case AppVisibilityInternal: + if h.Host != "" { + return fmt.Errorf("%w: service %q internal http interface must not declare host", ErrInvalidAppSpec, service) + } + if h.TLS != "" { + return fmt.Errorf("%w: service %q internal http interface must not declare tls", ErrInvalidAppSpec, service) + } + if h.Port < 1 || h.Port > 65535 { + return fmt.Errorf("%w: service %q http port must be 1-65535", ErrInvalidAppSpec, service) + } + default: + return fmt.Errorf("%w: service %q http visibility must be public|internal", ErrInvalidAppSpec, service) + } + return nil +} + +// validate checks one TCP interface. +func (t AppTCPInterface) validate(service string) error { + if strings.TrimSpace(t.Entrypoint) == "" { + return fmt.Errorf("%w: service %q tcp entrypoint is required", ErrInvalidAppSpec, service) + } + if t.Port < 1 || t.Port > 65535 { + return fmt.Errorf("%w: service %q tcp port must be 1-65535", ErrInvalidAppSpec, service) + } + if _, _, err := ParsePublish(t.Publish); err != nil { + return fmt.Errorf("%w: service %q tcp: %v", ErrInvalidAppSpec, service, err) + } + return nil +} + +// validate checks one UDP interface. +func (u AppUDPInterface) validate(service string) error { + if strings.TrimSpace(u.Entrypoint) == "" { + return fmt.Errorf("%w: service %q udp entrypoint is required", ErrInvalidAppSpec, service) + } + if u.Port < 1 || u.Port > 65535 { + return fmt.Errorf("%w: service %q udp port must be 1-65535", ErrInvalidAppSpec, service) + } + if _, _, err := ParsePublish(u.Publish); err != nil { + return fmt.Errorf("%w: service %q udp: %v", ErrInvalidAppSpec, service, err) + } + return nil +} + +// validate checks one shared-network declaration. +func (n AppSharedNetwork) validate(s AppSpec) error { + if strings.TrimSpace(n.Network) == "" { + return fmt.Errorf("%w: shared network name must not be empty", ErrInvalidAppSpec) + } + if len(n.Services) == 0 { + return fmt.Errorf("%w: shared network %q must list at least one service", ErrInvalidAppSpec, n.Network) + } + seen := map[string]struct{}{} + for _, name := range n.Services { + if _, ok := seen[name]; ok { + return fmt.Errorf("%w: shared network %q duplicate service %q", ErrInvalidAppSpec, n.Network, name) + } + seen[name] = struct{}{} + found := false + for _, svc := range s.Services { + if svc.Name == name { + found = true + break + } + } + if !found { + return fmt.Errorf("%w: shared network %q references unknown service %q", ErrInvalidAppSpec, n.Network, name) + } + } + for _, alias := range n.Aliases { + if !dnsLabelPattern.MatchString(alias) { + return fmt.Errorf("%w: shared network %q alias %q must be a DNS label", ErrInvalidAppSpec, n.Network, alias) + } + } + return nil +} + +// AppDiff describes normalized differences between desired and effective specs. +type AppDiff struct { + Added []string + Removed []string + Changed []string +} + +// DiffAppSpec returns a deterministic, stably sorted diff. +// It carries no secret values: only secret PATHS appear. +func DiffAppSpec(desired, effective AppSpec) AppDiff { + diff := AppDiff{} + desiredServices := map[string]AppService{} + for _, svc := range desired.Services { + desiredServices[svc.Name] = svc + } + effectiveServices := map[string]AppService{} + for _, svc := range effective.Services { + effectiveServices[svc.Name] = svc + } + for name := range desiredServices { + if _, ok := effectiveServices[name]; !ok { + diff.Added = append(diff.Added, "service/"+name) + } + } + for name := range effectiveServices { + if _, ok := desiredServices[name]; !ok { + diff.Removed = append(diff.Removed, "service/"+name) + } + } + for name, desiredSvc := range desiredServices { + effectiveSvc, ok := effectiveServices[name] + if !ok { + continue + } + diff.Changed = append(diff.Changed, diffService(name, desiredSvc, effectiveSvc)...) + } + if !equalStringMaps(desired.Env, effective.Env) { + diff.Changed = append(diff.Changed, "env") + } + diff.Changed = append(diff.Changed, diffNetworks(desired.Networks, effective.Networks)...) + sort.Strings(diff.Added) + sort.Strings(diff.Removed) + sort.Strings(diff.Changed) + return diff +} + +func diffNetworks(desired, effective []AppSharedNetwork) []string { + desiredByName := make(map[string]AppSharedNetwork, len(desired)) + for _, network := range desired { + desiredByName[network.Network] = network + } + effectiveByName := make(map[string]AppSharedNetwork, len(effective)) + for _, network := range effective { + effectiveByName[network.Network] = network + } + + var changed []string + for name, network := range desiredByName { + previous, ok := effectiveByName[name] + switch { + case !ok: + changed = append(changed, "network/"+name+"/added") + case !equalStringSets(network.Services, previous.Services): + changed = append(changed, "network/"+name+"/services") + case !equalStringSets(network.Aliases, previous.Aliases): + changed = append(changed, "network/"+name+"/aliases") + } + } + for name := range effectiveByName { + if _, ok := desiredByName[name]; !ok { + changed = append(changed, "network/"+name+"/removed") + } + } + return changed +} + +func equalStringSets(a, b []string) bool { + if len(a) != len(b) { + return false + } + counts := make(map[string]int, len(a)) + for _, value := range a { + counts[value]++ + } + for _, value := range b { + counts[value]-- + if counts[value] < 0 { + return false + } + } + return true +} + +// SameAppService reports whether two service specs are equal under the +// same normalized comparison DiffAppSpec uses. +func SameAppService(a, b AppService) bool { + return len(diffService(a.Name, a, b)) == 0 +} + +// diffService compares two services field by field. +func diffService(name string, desired, effective AppService) []string { + var changed []string + if desired.Image != effective.Image { + changed = append(changed, "service/"+name+"/image") + } + if strings.Join(desired.Command, "\x00") != strings.Join(effective.Command, "\x00") { + changed = append(changed, "service/"+name+"/command") + } + if desired.StopGrace != effective.StopGrace { + changed = append(changed, "service/"+name+"/stop_grace") + } + if desired.Readiness != effective.Readiness { + changed = append(changed, "service/"+name+"/readiness") + } + if len(desired.HTTP) != len(effective.HTTP) || len(desired.TCP) != len(effective.TCP) || + len(desired.UDP) != len(effective.UDP) { + changed = append(changed, "service/"+name+"/interfaces") + } else if interfacesChanged(desired, effective) { + changed = append(changed, "service/"+name+"/interfaces") + } + if !equalStringMaps(desired.Secrets, effective.Secrets) { + for path := range secretPathChanges(desired, effective) { + changed = append(changed, path) + } + } + if !equalVolumes(desired.Volumes, effective.Volumes) { + changed = append(changed, "service/"+name+"/volumes") + } + if !equalBinds(desired.Binds, effective.Binds) { + changed = append(changed, "service/"+name+"/binds") + } + changed = appendDeviceChanges(changed, name, desired.Devices, effective.Devices) + return appendOperationChanges(changed, name, desired, effective) +} + +// appendOperationChanges compares the operational (non-runtime) service +// settings: databases, backups, and telemetry. +func appendOperationChanges(changed []string, name string, desired, effective AppService) []string { + if !reflect.DeepEqual(desired.Databases, effective.Databases) { + changed = append(changed, "service/"+name+"/databases") + } + if !reflect.DeepEqual(desired.Backup, effective.Backup) { + changed = append(changed, "service/"+name+"/backup") + } + if desired.LogExportDisabled != effective.LogExportDisabled { + changed = append(changed, "service/"+name+"/telemetry") + } + return changed +} + +// secretPathChanges lists secret PATH changes (never values). +func secretPathChanges(desired, effective AppService) map[string]struct{} { + paths := map[string]struct{}{} + for envKey, secretName := range desired.Secrets { + if effective.Secrets[envKey] != secretName { + paths["service/"+desired.Name+"/secret/"+secretName] = struct{}{} + } + } + for envKey, secretName := range effective.Secrets { + if desired.Secrets[envKey] != secretName { + paths["service/"+effective.Name+"/secret/"+secretName] = struct{}{} + } + } + return paths +} + +// interfacesChanged compares interface slices. +func interfacesChanged(desired, effective AppService) bool { + for i := range desired.HTTP { + if !httpInterfacesEqual(desired.HTTP[i], effective.HTTP[i]) { + return true + } + } + for i := range desired.TCP { + if desired.TCP[i] != effective.TCP[i] { + return true + } + } + for i := range desired.UDP { + if desired.UDP[i] != effective.UDP[i] { + return true + } + } + return false +} + +// httpInterfacesEqual compares two HTTP interfaces with visibility +// normalized. State written before the field existed reads as the zero +// value and must equal an explicit public, while a real public/internal +// flip is still a change. +func httpInterfacesEqual(a, b AppHTTPInterface) bool { + if a.EffectiveVisibility() != b.EffectiveVisibility() { + return false + } + a.Visibility, b.Visibility = "", "" + return a == b +} + +// equalStringMaps compares string maps. +func equalStringMaps(a, b map[string]string) bool { + if len(a) != len(b) { + return false + } + for k, v := range a { + if b[k] != v { + return false + } + } + return true +} + +// equalVolumes compares volume slices by value. +func equalVolumes(a, b []AppVolume) bool { + if len(a) != len(b) { + return false + } + for i := range a { + if a[i] != b[i] { + return false + } + } + return true +} + +// equalBinds compares bind slices by value. +func equalBinds(a, b []AppBind) bool { + if len(a) != len(b) { + return false + } + for i := range a { + if a[i] != b[i] { + return false + } + } + return true +} + +// equalDevices compares device slices as sets: reorder-only input is a +// no-op, while add/remove reshuffles the sorted comparison. +func equalDevices(a, b []string) bool { + if len(a) != len(b) { + return false + } + sortedA := slices.Sorted(slices.Values(a)) + sortedB := slices.Sorted(slices.Values(b)) + return slices.Equal(sortedA, sortedB) +} + +// appendDeviceChanges appends the service devices diff entry when the +// device sets differ. Split from diffService to keep its complexity +// within budget. +func appendDeviceChanges(changed []string, name string, desired, effective []string) []string { + if !equalDevices(desired, effective) { + changed = append(changed, "service/"+name+"/devices") + } + return changed +} diff --git a/internal/domain/app_spec_test.go b/internal/domain/app_spec_test.go new file mode 100644 index 000000000..4a6623359 --- /dev/null +++ b/internal/domain/app_spec_test.go @@ -0,0 +1,179 @@ +package domain_test + +import ( + "errors" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +func validSpec() domain.AppSpec { + return domain.AppSpec{ + Name: "blog", + Env: map[string]string{"APP_ENV": "production"}, + Services: []domain.AppService{ + { + Name: "web", + Image: "registry.example.com/blog/web:1.4.2", + StopGrace: 10 * time.Second, + Readiness: domain.AppReadiness{Type: "http", Path: "/healthz", Timeout: 30 * time.Second}, + HTTP: []domain.AppHTTPInterface{{Host: "blog.example.com", Port: 8080, TLS: "auto"}}, + Secrets: map[string]string{"DATABASE_URL": "database-url"}, + }, + }, + } +} + +func TestAppSpec_ValidateOK(t *testing.T) { + require.NoError(t, validSpec().Validate()) +} + +func TestAppSpec_ValidateErrors(t *testing.T) { + cases := []struct { + name string + mutate func(*domain.AppSpec) + wantErr string + }{ + {"bad app", func(s *domain.AppSpec) { s.Name = "Bad!" }, "app name"}, + {"no services", func(s *domain.AppSpec) { s.Services = nil }, "at least one service"}, + {"no image", func(s *domain.AppSpec) { s.Services[0].Image = "" }, "requires an image"}, + {"env newline", func(s *domain.AppSpec) { s.Env["X"] = "a\nb" }, "newlines"}, + {"env secret ref", func(s *domain.AppSpec) { s.Env["X"] = "${sops:y}" }, "secret references"}, + {"http bad tls", func(s *domain.AppSpec) { s.Services[0].HTTP[0].TLS = "sometimes" }, "auto|always|never"}, + {"http readiness authority", func(s *domain.AppSpec) { s.Services[0].Readiness.Path = "@127.0.0.1:9000/private" }, "origin-form path"}, + {"http readiness absolute URL", func(s *domain.AppSpec) { s.Services[0].Readiness.Path = "http://127.0.0.1/private" }, "origin-form path"}, + {"http readiness scheme relative", func(s *domain.AppSpec) { s.Services[0].Readiness.Path = "//127.0.0.1/private" }, "origin-form path"}, + {"network unknown service", func(s *domain.AppSpec) { + s.Networks = []domain.AppSharedNetwork{{Network: "n", Services: []string{"ghost"}}} + }, "unknown service"}, + {"network bad alias", func(s *domain.AppSpec) { + s.Networks = []domain.AppSharedNetwork{{Network: "n", Services: []string{"web"}, Aliases: []string{"Bad_Alias!"}}} + }, "DNS label"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + spec := validSpec() + tc.mutate(&spec) + err := spec.Validate() + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrInvalidAppSpec) + assert.Contains(t, err.Error(), tc.wantErr) + }) + } +} + +func TestAppIdentityHelpers(t *testing.T) { + assert.Equal(t, "gordon-blog--web", domain.LogicalServiceIdentity("blog", "web")) + assert.Equal(t, "gordon-blog--a-b", domain.LogicalServiceIdentity("blog", "a.b")) + assert.Equal(t, + "gordon-blog--db--vol--pgdata", + domain.RuntimeVolumeName("blog", "db", "pgdata")) + assert.Equal(t, + "gordon/apps/blog/web/database-url", + domain.AppSecretPath("blog", "web", "database-url")) + assert.Equal(t, + "gordon/apps/app-uuid-1/web/database-url", + domain.AppSecretPathForID("app-uuid-1", "blog", "web", "database-url")) + assert.Equal(t, + "gordon/apps/blog/web/database-url", + domain.AppSecretPathForID("", "blog", "web", "database-url")) + assert.Equal(t, "blog.example.com", domain.CanonicalHTTPHost("Blog.Example.COM.")) +} + +func TestParsePublish(t *testing.T) { + host, port, err := domain.ParsePublish("127.0.0.1:8080") + require.NoError(t, err) + assert.Equal(t, "127.0.0.1", host) + assert.Equal(t, 8080, port) + + host, port, err = domain.ParsePublish("8080") + require.NoError(t, err) + assert.Empty(t, host) + assert.Equal(t, 8080, port) + + _, _, err = domain.ParsePublish("example.com:8080") + require.ErrorIs(t, err, domain.ErrInvalidAppSpec) + + _, _, err = domain.ParsePublish("8080:9090:extra") + require.Error(t, err) + + _, _, err = domain.ParsePublish("") + require.ErrorIs(t, err, domain.ErrInvalidAppSpec) + + _, _, err = domain.ParsePublish("0") + require.ErrorIs(t, err, domain.ErrInvalidAppSpec) +} + +func TestDiffAppSpec_DeterministicAndRedacted(t *testing.T) { + before := validSpec() + after := validSpec() + after.Services = append(after.Services, domain.AppService{ + Name: "worker", Image: "img:1", + StopGrace: 10 * time.Second, + Readiness: domain.AppReadiness{Type: "none", Timeout: 30 * time.Second}, + }) + after.Services[0].Image = "registry.example.com/blog/web:1.5.0" + after.Services[0].StopGrace = 20 * time.Second + after.Services[0].Secrets = map[string]string{"DATABASE_URL": "database-url-v2"} + after.Services[0].Databases = []domain.AppDatabase{{Name: "main", Type: "postgres", Schedule: "daily"}} + after.Services[0].Backup = domain.AppBackup{Postgres: []string{"main"}} + after.Networks = []domain.AppSharedNetwork{{Network: "shared", Services: []string{"web"}}} + after.Env["NEW_KEY"] = "v" + + diff := domain.DiffAppSpec(after, before) + assert.Equal(t, []string{"service/worker"}, diff.Added) + assert.Empty(t, diff.Removed) + assert.Contains(t, diff.Changed, "service/web/image") + assert.Contains(t, diff.Changed, "service/web/stop_grace") + assert.Contains(t, diff.Changed, "service/web/databases") + assert.Contains(t, diff.Changed, "service/web/backup") + assert.Contains(t, diff.Changed, "network/shared/added") + assert.Contains(t, diff.Changed, "service/web/secret/database-url-v2") + assert.Contains(t, diff.Changed, "service/web/secret/database-url") + assert.Contains(t, diff.Changed, "env") + // Sorted output. + assert.IsIncreasing(t, diff.Changed) + // No secret values leak: only names/paths. + for _, entry := range append(append(diff.Added, diff.Removed...), diff.Changed...) { + assert.NotContains(t, entry, "s3cr3t") + } + + // Identical specs produce an empty diff. + empty := domain.DiffAppSpec(before, validSpec()) + assert.Empty(t, empty.Added) + assert.Empty(t, empty.Removed) + assert.Empty(t, empty.Changed) + + // Removed service. + removed := domain.DiffAppSpec(domain.AppSpec{Name: "x"}, before) + assert.Equal(t, []string{"service/web"}, removed.Removed) +} + +func TestDiffAppSpec_NetworkDetailsAndOrderNormalization(t *testing.T) { + base := validSpec() + base.Networks = []domain.AppSharedNetwork{{Network: "database", Services: []string{"web"}, Aliases: []string{"db", "primary"}}} + + reordered := validSpec() + reordered.Networks = []domain.AppSharedNetwork{{Network: "database", Services: []string{"web"}, Aliases: []string{"primary", "db"}}} + assert.Empty(t, domain.DiffAppSpec(reordered, base).Changed) + + servicesChanged := validSpec() + servicesChanged.Services = append(servicesChanged.Services, domain.AppService{Name: "worker", Image: "img:1", Readiness: domain.AppReadiness{Type: "none", Timeout: 30 * time.Second}}) + servicesChanged.Networks = []domain.AppSharedNetwork{{Network: "database", Services: []string{"web", "worker"}, Aliases: []string{"db", "primary"}}} + assert.Contains(t, domain.DiffAppSpec(servicesChanged, base).Changed, "network/database/services") + + aliasesChanged := base + aliasesChanged.Networks = []domain.AppSharedNetwork{{Network: "database", Services: []string{"web"}, Aliases: []string{"db"}}} + assert.Equal(t, []string{"network/database/aliases"}, domain.DiffAppSpec(aliasesChanged, base).Changed) + + removed := validSpec() + assert.Equal(t, []string{"network/database/removed"}, domain.DiffAppSpec(removed, base).Changed) +} + +func TestDiffAppSpec_ErrorsWrapped(t *testing.T) { + assert.True(t, errors.Is(domain.AppSpec{}.Validate(), domain.ErrInvalidAppSpec)) +} diff --git a/internal/domain/app_state.go b/internal/domain/app_state.go new file mode 100644 index 000000000..55d405e41 --- /dev/null +++ b/internal/domain/app_state.go @@ -0,0 +1,346 @@ +package domain + +import "time" + +// App state records are persistence shapes: the store adapter serializes them as +// JSON, the apps use case orchestrates transitions. No secret VALUES +// ever appear in these records — only secret paths. + +// AppStoreVersion is the single supported app-state format version. +const AppStoreVersion = 1 + +// AppDefaultRevisionRetention is the default count of unreferenced +// revisions kept per app. Revisions referenced by desired state, +// active services, or unfinished operations are always protected, +// even beyond this limit. Configurable in main installation config. +const AppDefaultRevisionRetention = 8 + +// App intent states for the durable apply protocol. +const ( + AppIntentStaged = "staged" + AppIntentCommitted = "committed" + AppIntentApplied = "applied" +) + +// App operation step states. +const ( + AppStepPending = "pending" + AppStepSucceeded = "succeeded" + AppStepFailed = "failed" + AppStepNotRun = "not-run" +) + +// AppServiceUnchanged is the result of a deploy step that kept the running +// container because its image, spec, and environment were already current. +// A succeeded step whose Detail starts with it reports this result. +const AppServiceUnchanged = "unchanged" + +// App operation outcomes over terminal per-service results. +const ( + AppOutcomeSuccess = "success" + AppOutcomePartial = "partial" + AppOutcomeFailed = "failed" +) + +// App owned-resource states. +const ( + AppResourceAttached = "attached" + AppResourceRetained = "retained" + // AppResourceReleased marks a volume whose owning app explicitly + // released it. Only this state makes a volume prune-eligible, and + // only when the runtime labels agree with the durable record. + AppResourceReleased = "released" +) + +// OwnerGordonBackend marks Gordon-generated loopback backend binds in +// the global reservation checkpoint (plan D3: loopback-only, +// generated, reserved, distinguishable from public routes). +const OwnerGordonBackend = "gordon-backend" + +// AppListenerReservation is one global listener claim. +type AppListenerReservation struct { + Proto string `json:"proto"` + IP string `json:"ip,omitempty"` + Port int `json:"port,omitempty"` + Host string `json:"host,omitempty"` + Service string `json:"service"` + App string `json:"app"` + Owner string `json:"owner,omitempty"` + Dual *bool `json:"dual,omitempty"` + // ContainerID ties a Gordon-generated backend claim to the exact + // container holding the bind, so old and replacement binds coexist + // and release targets exactly one retired container. + ContainerID string `json:"container_id,omitempty"` +} + +// AppDesiredRevision is one accepted desired revision. +type AppDesiredRevision struct { + Revision string `json:"revision"` + App string `json:"app"` + Supersedes string `json:"supersedes,omitempty"` + AcceptedAt time.Time `json:"accepted_at"` + SourceSHA256 string `json:"source_sha256"` + Spec AppSpec `json:"spec"` + Reservations []AppListenerReservation `json:"reservations"` + SecretsRequired []string `json:"secrets_required"` + SecretsEnv map[string]string `json:"secrets_env"` + Status string `json:"status"` +} + +// AppBackend is one resolved service backend endpoint: the loopback +// address plus the Gordon-published host port for one container port, +// tied to the exact container holding the bind. The proxy and readiness +// dial this endpoint rootless-first; a zero Backend means unbound +// (fail closed, never fall back to container IPs). +type AppBackend struct { + // Host is the loopback address (127.0.0.1) when bound, empty when not. + Host string `json:"host,omitempty"` + // Port is the Gordon-published host port. + Port int `json:"port,omitempty"` + // ContainerPort is the container port this endpoint serves. + ContainerPort int `json:"container_port,omitempty"` + // ContainerID is the exact container holding the bind. + ContainerID string `json:"container_id,omitempty"` +} + +// Resolved reports whether the backend endpoint is dialable. +func (b AppBackend) Resolved() bool { + return b.Host != "" && b.Port > 0 && b.ContainerID != "" +} + +// BackendFor resolves one container port to its loopback endpoint from +// the recorded binds. ContainerPort is always set (declared manifest +// port); Host/Port/ContainerID resolve only when the bind is recorded +// (zero Backend otherwise: fail closed). +func (s AppEffectiveService) BackendFor(containerPort int) AppBackend { + return s.BackendForProtocol(containerPort, NetworkProtocolTCP) +} + +// BackendForProtocol resolves a protocol-specific container port bind. +func (s AppEffectiveService) BackendForProtocol(containerPort int, protocol NetworkProtocol) AppBackend { + backend := AppBackend{ContainerPort: containerPort} + binds := s.BackendBinds + if protocol == NetworkProtocolUDP { + binds = s.UDPBackendBinds + } + hostPort, ok := binds[containerPort] + if !ok || hostPort <= 0 || s.Container == "" { + return backend + } + backend.Host = "127.0.0.1" + backend.Port = hostPort + backend.ContainerID = s.Container + return backend +} + +type AppEffectiveService struct { + EffectiveRevision string `json:"effective_revision"` + ActivatedBy string `json:"activated_by"` + ActivatedAt time.Time `json:"activated_at"` + Image string `json:"image"` + Digest string `json:"digest,omitempty"` + Container string `json:"container,omitempty"` + Spec AppService `json:"spec"` + // BackendBinds maps container port -> 127.0.0.1 host port for the + // Gordon-published loopback backends. Readiness and the proxy dial + // these binds: they work rootless-first (no container-IP route + // required) on Docker and Podman alike, and never expose backends + // beyond loopback (04-network.md §4). + BackendBinds map[int]int `json:"backend_binds,omitempty"` + UDPBackendBinds map[int]int `json:"udp_backend_binds,omitempty"` +} + +// AppActive is the per-service effective state of one app. +type AppActive struct { + App string `json:"app"` + ConvergedRevision string `json:"converged_revision"` + Converged bool `json:"converged"` + Services map[string]AppEffectiveService `json:"services"` + Networks []AppSharedNetwork `json:"networks,omitempty"` + StopIntent bool `json:"stop_intent"` +} + +// AppStopIntent is the durable running/stopped intent. +type AppStopIntent struct { + App string `json:"app"` + Stopped bool `json:"stopped"` + UpdatedBy string `json:"updated_by"` + UpdatedAt time.Time `json:"updated_at"` +} + +// AppOperationStep is one durable journal step. Digest and Image are +// structured fields: a step that acquires or replaces an image records +// the pinned content digest and the runtime image reference here, so +// prune protection never has to parse human-readable Detail. +// AppOperationStep is one journaled step. Error carries a stable, +// log-free message; Diagnostics carries bounded application log output +// and is only returned to callers holding the logs read scope. +type AppOperationStep struct { + ID string `json:"id"` + State string `json:"state"` + Detail string `json:"detail,omitempty"` + Error string `json:"error,omitempty"` + Before string `json:"before,omitempty"` + After string `json:"after,omitempty"` + // Digest pins the registry content digest this step resolved. + Digest string `json:"digest,omitempty"` + // Image is the runtime image reference this step acquired. + Image string `json:"image,omitempty"` + // Service names the app service this step applies to. + Service string `json:"service,omitempty"` + // Diagnostics is bounded, redacted failure output kept separate from + // the stable Error. Retrieval requires the logs read scope. + Diagnostics []string `json:"diagnostics,omitempty"` +} + +// AppOperationRequest is the immutable identity of the mutation request +// one journal record answers: kind, app, requested revision, requested +// service. It is persisted with the claim so reusing one idempotency key +// for a different request is rejected instead of silently replaying +// another request's result. It never carries a secret value. +type AppOperationRequest struct { + Kind string `json:"kind"` + App string `json:"app"` + Revision string `json:"revision,omitempty"` + Service string `json:"service,omitempty"` +} + +// AppOperationRequestFor builds the request identity of one app +// mutation from the caller's inputs: the resolved revision is never +// part of it, so a repeat of the same request always matches. +func AppOperationRequestFor(kind, app, revision, service string) AppOperationRequest { + return AppOperationRequest{Kind: kind, App: app, Revision: revision, Service: service} +} + +// AppOperation is one durable operation journal record. +type AppOperation struct { + Op string `json:"op"` + Kind string `json:"kind"` + App string `json:"app"` + InputRevision string `json:"input_revision"` + StartedAt time.Time `json:"started_at"` + Steps []AppOperationStep `json:"steps"` + Outcome string `json:"outcome,omitempty"` + // Warnings are bounded, operator-actionable leftovers of this + // operation: a container that could not be removed, or a backend + // claim that could not be released. They never carry log content. + Warnings []AppOperationWarning `json:"warnings,omitempty"` + // Request is the immutable request identity this journal answers. + // Records written by older binaries decode it as the zero value; + // such a record is only reachable through its own key. + Request AppOperationRequest `json:"request,omitempty"` +} + +// AppOperationWarning records one bounded leftover of an operation. +type AppOperationWarning struct { + Service string `json:"service,omitempty"` + Leftover string `json:"leftover,omitempty"` + Detail string `json:"detail"` +} + +// Terminal reports whether the operation reached a terminal outcome. A +// non-terminal operation is either in flight or was interrupted; its +// effects must never be re-executed under the same key. +func (o AppOperation) Terminal() bool { + return o.Outcome != "" +} + +// AppApplyIntent is one durable apply intent. +type AppApplyIntent struct { + Intent string `json:"intent"` + App string `json:"app"` + State string `json:"state"` + Revision string `json:"revision"` + Supersedes string `json:"supersedes,omitempty"` + SourceSHA256 string `json:"source_sha256"` + Spec AppSpec `json:"spec"` + Reservations []AppListenerReservation `json:"reservations"` + CreatedAt time.Time `json:"created_at"` +} + +// AppOwnedVolume tracks one app-owned volume. +type AppOwnedVolume struct { + Name string `json:"name"` + Service string `json:"service"` + RuntimeName string `json:"runtime_name"` + State string `json:"state"` +} + +// AppOwnedSecret tracks one app-referenced secret (path only, never value). +type AppOwnedSecret struct { + Service string `json:"service"` + Env string `json:"env"` + Name string `json:"name"` + Path string `json:"path"` + State string `json:"state"` +} + +// AppOwnedNetwork tracks one app network. +type AppOwnedNetwork struct { + Name string `json:"name"` + Role string `json:"role"` +} + +// AppServiceRecovery tracks per-service restart-safety flags. +type AppServiceRecovery struct { + RestartUnsafe bool `json:"restart_unsafe"` +} + +// AppRecoveryInhibition is a durable, generation-scoped recovery +// inhibition record. It names one exact container ID that must never be +// started or restarted by boot or periodic recovery, because a +// replacement may already have written to a volume that this generation +// still owns. It carries no secret values. RestartUnsafe is a different +// fact (a successful volume deployment also sets it) and never implies +// inhibition. +type AppRecoveryInhibition struct { + App string `json:"app"` + AppID string `json:"app_id,omitempty"` + Service string `json:"service"` + ContainerID string `json:"container_id"` + Reason string `json:"reason"` + Operation string `json:"operation,omitempty"` + CreatedAt time.Time `json:"created_at"` +} + +// AppRecoveryInhibition reasons. +const ( + // AppInhibitReplacementPending marks the generation being replaced + // while a volume-owning candidate may already be writing. + AppInhibitReplacementPending = "replacement-pending" + // AppInhibitRetirementPending marks a published predecessor that must + // remain stopped and be retired by a later reconciliation pass. + AppInhibitRetirementPending = "retirement-pending" +) + +// AppOwnership is the ownership record for one app. +// ID is the stable internal UUID, distinct from the public name. +// Remove frees the name but retains volumes/secrets under the old ID +// with their exact ownership records; a new app reusing the name +// never implicitly adopts them. +// AppOwnedImage tracks one app-pinned image (reference and digest only, +// never credentials). +// +//nolint:revive // Stutter matches the sibling AppOwned* record names. +type AppOwnedImage struct { + Service string `json:"service"` + Reference string `json:"reference"` + Digest string `json:"digest,omitempty"` + State string `json:"state"` +} + +type AppOwnership struct { + App string `json:"app"` + ID string `json:"id,omitempty"` + Volumes []AppOwnedVolume `json:"volumes"` + Services map[string]AppServiceRecovery `json:"services"` + Secrets []AppOwnedSecret `json:"secrets"` + Networks []AppOwnedNetwork `json:"networks"` + Images []AppOwnedImage `json:"images"` +} + +// AppStoreCheckpoint is the global reservation checkpoint. +type AppStoreCheckpoint struct { + Version int `json:"version"` + Reservations []AppListenerReservation `json:"reservations"` +} diff --git a/internal/domain/app_visibility_diff_test.go b/internal/domain/app_visibility_diff_test.go new file mode 100644 index 000000000..43da3a299 --- /dev/null +++ b/internal/domain/app_visibility_diff_test.go @@ -0,0 +1,39 @@ +package domain_test + +import ( + "testing" + + "github.com/stretchr/testify/assert" + + "github.com/bnema/gordon/internal/domain" +) + +// TestDiffAppSpec_VisibilityChangeIsDetected proves flipping an interface +// between public and internal is a meaningful change: it must force a +// redeploy instead of being silently ignored. +func TestDiffAppSpec_VisibilityChangeIsDetected(t *testing.T) { + desired := visibilitySpec(domain.AppHTTPInterface{Port: 8080, Visibility: domain.AppVisibilityInternal}) + effective := visibilitySpec(domain.AppHTTPInterface{ + Host: "blog.example.com", Port: 8080, TLS: "auto", Visibility: domain.AppVisibilityPublic, + }) + + diff := domain.DiffAppSpec(desired, effective) + assert.Contains(t, diff.Changed, "service/web/interfaces") +} + +// TestDiffAppSpec_ZeroVisibilityEqualsPublic proves state written before +// the visibility field existed (zero value) is not a phantom change +// against an explicit public interface, so applying an unchanged manifest +// stays a no-op after upgrade. +func TestDiffAppSpec_ZeroVisibilityEqualsPublic(t *testing.T) { + // Legacy stored state: same interface, visibility never serialized. + legacy := visibilitySpec(domain.AppHTTPInterface{ + Host: "blog.example.com", Port: 8080, TLS: "auto", + }) + desired := visibilitySpec(domain.AppHTTPInterface{ + Host: "blog.example.com", Port: 8080, TLS: "auto", Visibility: domain.AppVisibilityPublic, + }) + + diff := domain.DiffAppSpec(desired, legacy) + assert.NotContains(t, diff.Changed, "service/web/interfaces") +} diff --git a/internal/domain/app_visibility_test.go b/internal/domain/app_visibility_test.go new file mode 100644 index 000000000..02cb1a332 --- /dev/null +++ b/internal/domain/app_visibility_test.go @@ -0,0 +1,191 @@ +package domain_test + +import ( + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +func visibilitySpec(interfaces ...domain.AppHTTPInterface) domain.AppSpec { + return domain.AppSpec{ + Name: "blog", + Services: []domain.AppService{{ + Name: "web", + Image: "registry.example.com/blog/web:1.4.2", + StopGrace: 10 * time.Second, + Readiness: domain.AppReadiness{Type: domain.AppReadinessHTTP, Path: "/healthz", Timeout: 30 * time.Second}, + HTTP: interfaces, + }}, + } +} + +// TestAppHTTPInterface_EffectiveVisibility proves the zero value normalizes +// to public so manifests and stored state written before visibility existed +// keep their meaning. +func TestAppHTTPInterface_EffectiveVisibility(t *testing.T) { + absent := domain.AppHTTPInterface{Host: "blog.example.com", Port: 8080, TLS: "auto"} + assert.Equal(t, domain.AppVisibilityPublic, absent.EffectiveVisibility()) + assert.True(t, absent.IsPublic()) + assert.False(t, absent.IsInternal()) + + explicit := absent + explicit.Visibility = domain.AppVisibilityInternal + assert.Equal(t, domain.AppVisibilityInternal, explicit.EffectiveVisibility()) + assert.True(t, explicit.IsInternal()) + assert.False(t, explicit.IsPublic()) +} + +// TestAppSpec_ValidateVisibility proves public and absent visibility stay +// valid, internal visibility accepts a port without host or tls, and every +// malformed combination is rejected. +func TestAppSpec_ValidateVisibility(t *testing.T) { + t.Run("absent visibility remains a valid public manifest", func(t *testing.T) { + require.NoError(t, visibilitySpec(domain.AppHTTPInterface{ + Host: "blog.example.com", Port: 8080, TLS: "auto", + }).Validate()) + }) + + t.Run("explicit public remains valid", func(t *testing.T) { + require.NoError(t, visibilitySpec(domain.AppHTTPInterface{ + Host: "blog.example.com", Port: 8080, TLS: "auto", Visibility: domain.AppVisibilityPublic, + }).Validate()) + }) + + t.Run("internal accepts a port without host or tls", func(t *testing.T) { + require.NoError(t, visibilitySpec(domain.AppHTTPInterface{ + Port: 8080, Visibility: domain.AppVisibilityInternal, + }).Validate()) + }) + + reject := map[string]domain.AppHTTPInterface{ + "internal forbids host": {Host: "blog.example.com", Port: 8080, Visibility: domain.AppVisibilityInternal}, + "internal forbids tls": {Port: 8080, TLS: "auto", Visibility: domain.AppVisibilityInternal}, + "unknown visibility": {Host: "blog.example.com", Port: 8080, TLS: "auto", Visibility: "vpn"}, + "internal without port": {Visibility: domain.AppVisibilityInternal}, + "public without host": {Port: 8080, TLS: "auto"}, + "public without valid host": {Host: "not a host", Port: 8080, TLS: "auto"}, + "public without tls mode": {Host: "blog.example.com", Port: 8080}, + "public with unknown tls mode": {Host: "blog.example.com", Port: 8080, TLS: "sometimes"}, + "public with out of range port": {Host: "blog.example.com", Port: 0, TLS: "auto"}, + "internal with out of range port": {Port: -1, Visibility: domain.AppVisibilityInternal}, + } + for name, iface := range reject { + t.Run(name, func(t *testing.T) { + err := visibilitySpec(iface).Validate() + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrInvalidAppSpec) + }) + } +} + +// TestAppService_InternallyOnlyPort proves a port is internal-only only +// when internal HTTP declares it and no externally backed interface does. +func TestAppService_InternallyOnlyPort(t *testing.T) { + internalOnly := domain.AppService{ + HTTP: []domain.AppHTTPInterface{{Port: 8080, Visibility: domain.AppVisibilityInternal}}, + } + assert.True(t, internalOnly.InternallyOnlyPort(8080)) + assert.False(t, internalOnly.InternallyOnlyPort(8081)) + assert.True(t, internalOnly.IsPublicHTTP() == false) + + mixed := domain.AppService{ + HTTP: []domain.AppHTTPInterface{ + {Host: "blog.example.com", Port: 8080, TLS: "auto"}, + {Port: 9090, Visibility: domain.AppVisibilityInternal}, + }, + } + assert.True(t, mixed.IsPublicHTTP()) + assert.True(t, mixed.InternallyOnlyPort(9090)) + assert.False(t, mixed.InternallyOnlyPort(8080)) + + // A TCP interface on the same port is externally backed: the port is + // never internal-only, even though an internal HTTP interface wants it. + // Validation rejects that combination, and the read helper must not + // treat it as internal-only either. + shadowed := domain.AppService{ + HTTP: []domain.AppHTTPInterface{{Port: 8080, Visibility: domain.AppVisibilityInternal}}, + TCP: []domain.AppTCPInterface{{Port: 8080}}, + } + assert.False(t, shadowed.InternallyOnlyPort(8080)) +} + +// TestAppSpec_ValidateInterfacePorts proves an internal HTTP port may not +// collide with an externally backed interface on the same container port, +// and that internal ports stay unique. +func TestAppSpec_ValidateInterfacePorts(t *testing.T) { + collide := visibilitySpec( + domain.AppHTTPInterface{Port: 8080, Visibility: domain.AppVisibilityInternal}, + ) + collide.Services[0].Readiness.Port = 8080 + collide.Services[0].TCP = []domain.AppTCPInterface{{Port: 8080}} + require.ErrorIs(t, collide.Validate(), domain.ErrInvalidAppSpec) + + duplicate := visibilitySpec( + domain.AppHTTPInterface{Port: 8080, Visibility: domain.AppVisibilityInternal}, + domain.AppHTTPInterface{Port: 8080, Visibility: domain.AppVisibilityInternal}, + ) + // An explicit port keeps this test on the duplicate-internal-port rule + // instead of short-circuiting on readiness-port ambiguity. + duplicate.Services[0].Readiness.Port = 8080 + require.ErrorIs(t, duplicate.Validate(), domain.ErrInvalidAppSpec) +} + +// TestAppSpec_ValidateMixedVisibilityReadiness proves one public HTTP +// interface plus internal HTTP interfaces is a valid readiness target with +// no explicit readiness.port, while interface sets that cannot select one +// port stay strict. +func TestAppSpec_ValidateMixedVisibilityReadiness(t *testing.T) { + require.NoError(t, visibilitySpec( + domain.AppHTTPInterface{Host: "blog.example.com", Port: 8080, TLS: "auto"}, + domain.AppHTTPInterface{Port: 9090, Visibility: domain.AppVisibilityInternal}, + ).Validate()) + + tcpReadiness := visibilitySpec( + domain.AppHTTPInterface{Host: "blog.example.com", Port: 8080, TLS: "auto"}, + domain.AppHTTPInterface{Port: 9090, Visibility: domain.AppVisibilityInternal}, + ) + tcpReadiness.Services[0].Readiness.Type = domain.AppReadinessTCP + require.NoError(t, tcpReadiness.Validate()) + + explicit := visibilitySpec( + domain.AppHTTPInterface{Host: "blog.example.com", Port: 8080, TLS: "auto"}, + domain.AppHTTPInterface{Port: 9090, Visibility: domain.AppVisibilityInternal}, + ) + explicit.Services[0].Readiness.Port = 9090 + require.NoError(t, explicit.Validate()) + + reject := map[string]func(spec domain.AppSpec){ + "several internal ports": func(spec domain.AppSpec) { + spec.Services[0].HTTP = []domain.AppHTTPInterface{ + {Port: 8080, Visibility: domain.AppVisibilityInternal}, + {Port: 9090, Visibility: domain.AppVisibilityInternal}, + } + }, + "public http plus tcp": func(spec domain.AppSpec) { + spec.Services[0].TCP = []domain.AppTCPInterface{{Entrypoint: "tcp", Port: 9000, Publish: "9000"}} + }, + "several public http ports": func(spec domain.AppSpec) { + spec.Services[0].HTTP = []domain.AppHTTPInterface{ + {Host: "blog.example.com", Port: 8080, TLS: "auto"}, + {Host: "www.example.com", Port: 8081, TLS: "auto"}, + } + }, + } + for name, mutate := range reject { + t.Run(name, func(t *testing.T) { + spec := visibilitySpec( + domain.AppHTTPInterface{Host: "blog.example.com", Port: 8080, TLS: "auto"}, + domain.AppHTTPInterface{Port: 9090, Visibility: domain.AppVisibilityInternal}, + ) + mutate(spec) + err := spec.Validate() + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrInvalidAppSpec) + assert.Contains(t, err.Error(), "readiness.port is required") + }) + } +} diff --git a/internal/domain/auth.go b/internal/domain/auth.go index 882b6654e..58053640a 100644 --- a/internal/domain/auth.go +++ b/internal/domain/auth.go @@ -77,12 +77,13 @@ const ( // Admin scope resource constants. const ( AdminResourceRoutes = "routes" - AdminResourceSecrets = "secrets" AdminResourceConfig = "config" AdminResourceStatus = "status" AdminResourceLogs = "logs" AdminResourceVolumes = "volumes" - AdminResourceAll = "*" + // AdminResourceApps gates the declarative app admin surface. + AdminResourceApps = "apps" + AdminResourceAll = "*" ) // Admin scope action constants. @@ -285,11 +286,6 @@ func AdminScopeRoutes(actions ...string) string { return fmt.Sprintf("%s:%s:%s", ScopeTypeAdmin, AdminResourceRoutes, strings.Join(actions, ",")) } -// AdminScopeSecrets creates an admin scope for secrets with the given actions. -func AdminScopeSecrets(actions ...string) string { - return fmt.Sprintf("%s:%s:%s", ScopeTypeAdmin, AdminResourceSecrets, strings.Join(actions, ",")) -} - // AdminScopeConfig creates an admin scope for config with the given actions. func AdminScopeConfig(actions ...string) string { return fmt.Sprintf("%s:%s:%s", ScopeTypeAdmin, AdminResourceConfig, strings.Join(actions, ",")) @@ -310,6 +306,11 @@ func AdminScopeVolumes(actions ...string) string { return fmt.Sprintf("%s:%s:%s", ScopeTypeAdmin, AdminResourceVolumes, strings.Join(actions, ",")) } +// AdminScopeApps creates an admin scope for apps with the given actions. +func AdminScopeApps(actions ...string) string { + return fmt.Sprintf("%s:%s:%s", ScopeTypeAdmin, AdminResourceApps, strings.Join(actions, ",")) +} + // AuthStatus represents status of an authentication session. type AuthStatus struct { Valid bool diff --git a/internal/domain/auto.go b/internal/domain/auto.go deleted file mode 100644 index 655894945..000000000 --- a/internal/domain/auto.go +++ /dev/null @@ -1,24 +0,0 @@ -package domain - -import "time" - -// AutoConfig holds the unified auto-route and auto-preview configuration. -type AutoConfig struct { - Enabled bool - AllowedDomains []string - Route AutoRouteConfig - Preview PreviewConfig -} - -// AutoRouteConfig holds auto-route specific settings (reserved for future use). -type AutoRouteConfig struct{} - -// PreviewConfig holds auto-preview specific settings. -type PreviewConfig struct { - Enabled bool - TTL time.Duration - Separator string - TagPatterns []string - DataCopy bool - EnvCopy bool -} diff --git a/internal/domain/backup.go b/internal/domain/backup.go index 9e00e06fd..87ac7a87a 100644 --- a/internal/domain/backup.go +++ b/internal/domain/backup.go @@ -49,37 +49,24 @@ const ( BackupStatusFailed BackupJobStatus = "failed" ) -// DBInfo holds detected database information from an attachment container. -type DBInfo struct { - Type DBType - Version string - Domain string - Name string - Host string - Port int +// DatabaseTarget is one declarative database backup target: the app name +// is the identity, the service and database select the exact declaration +// inside the app's ACTIVE record. Schedule is the declared backup +// schedule of that database. +type DatabaseTarget struct { + App string + Service string + Database string + Schedule BackupSchedule ContainerID string - ImageName string - // Credentials contains sensitive values (passwords/tokens). - // Never log or expose this map in API responses. - Credentials map[string]string `json:"-"` -} - -// ClearCredentials clears sensitive credential values from DBInfo. -func (d *DBInfo) ClearCredentials() { - if d == nil || d.Credentials == nil { - return - } - for k := range d.Credentials { - d.Credentials[k] = "" - delete(d.Credentials, k) - } - d.Credentials = nil } // BackupJob represents a scheduled or manual backup operation. type BackupJob struct { - ID string - Domain string + ID string + // App is the canonical backup identity: an app name, never a domain. + App string + Service string DBName string Schedule BackupSchedule Type BackupType @@ -158,29 +145,35 @@ type VolumeBackupConfig struct { // VolumeBackupJob represents a filesystem archive backup of a named volume. type VolumeBackupJob struct { - ID string - Domain string - ContainerName string - ContainerID string - VolumeName string - MountPath string - Type BackupType - Status BackupJobStatus - StartedAt time.Time - CompletedAt time.Time - SizeBytes int64 - ArtifactRef string - Error string - Metadata map[string]string + ID string + // App is the canonical backup identity: an app name, never a domain. + App string + Service string + ContainerName string + ContainerID string + VolumeName string + RuntimeVolumeName string + MountPath string + Type BackupType + Status BackupJobStatus + StartedAt time.Time + CompletedAt time.Time + SizeBytes int64 + ArtifactRef string + Error string + Metadata map[string]string } -// VolumeBackupTarget identifies one selected volume backup source. +// VolumeBackupTarget identifies one declared volume backup source: +// app, service, and the declared volume name. RuntimeVolumeName is the +// runtime volume the archive is exported from, and MountPath is the +// declared mount path inside that volume. type VolumeBackupTarget struct { - Domain string - ContainerName string - ContainerID string - VolumeName string - MountPath string + App string + Service string + VolumeName string + RuntimeVolumeName string + MountPath string } // VolumeArchiveRequest describes a volume archive export request. @@ -202,12 +195,3 @@ type VolumeArchiveResult struct { Stream io.ReadCloser Metadata VolumeArchiveMetadata } - -// Backup labels for container metadata. -const ( - LabelBackupEnabled = "gordon.backup" - LabelBackupType = "gordon.backup.type" - LabelBackupVersion = "gordon.backup.version" - LabelBackupSchedule = "gordon.backup.schedule" - LabelBackupSidecar = "gordon.backup.sidecar" -) diff --git a/internal/domain/backup_test.go b/internal/domain/backup_test.go deleted file mode 100644 index dedadb9d7..000000000 --- a/internal/domain/backup_test.go +++ /dev/null @@ -1,24 +0,0 @@ -package domain - -import ( - "encoding/json" - "testing" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" -) - -func TestDBInfoJSONOmitsCredentials(t *testing.T) { - info := DBInfo{ - Type: DBTypePostgreSQL, - Name: "postgres", - Credentials: map[string]string{ - "password": "secret", - }, - } - - payload, err := json.Marshal(info) - require.NoError(t, err) - assert.NotContains(t, string(payload), "credentials") - assert.NotContains(t, string(payload), "secret") -} diff --git a/internal/domain/container.go b/internal/domain/container.go index f06a84337..410e91d47 100644 --- a/internal/domain/container.go +++ b/internal/domain/container.go @@ -16,6 +16,14 @@ type Container struct { Labels map[string]string VolumeMounts []ContainerVolumeMount Created time.Time + // StartedAt is the start timestamp of the current execution. Zero + // when the container never started or the runtime reports none. + // Readiness scoping uses it so log markers from a previous + // execution of the same container ID cannot satisfy a probe. + StartedAt time.Time + // Env is the container's KEY=value environment as created. It may hold + // secret values: compare it in memory only, never persist or log it. + Env []string } // ContainerVolumeMount describes a mounted volume-like resource on a container. @@ -71,6 +79,30 @@ type ContainerPortPublish struct { Protocol NetworkProtocol } +// ContainerBackendPort identifies one protocol-specific container port +// for grouped backend-bind inspection. +type ContainerBackendPort struct { + ContainerPort int + Protocol NetworkProtocol +} + +// ContainerBackendBind is an observed protocol-specific host binding. +type ContainerBackendBind struct { + ContainerPort int + HostPort int + Protocol NetworkProtocol +} + +// ContainerBind is one ephemeral host bind mount for a container. Source is a +// resolved host path produced by administrative policy: it is never persisted +// and never logged by the runtime adapter. Destination is the container path. +type ContainerBind struct { + Name string + Source string + Destination string + ReadOnly bool +} + // ContainerConfig holds configuration for creating a container. type ContainerConfig struct { Image string @@ -81,20 +113,26 @@ type ContainerConfig struct { Labels map[string]string WorkingDir string Cmd []string + Entrypoint []string AutoRemove bool RestartPolicy string Volumes map[string]string // map[containerPath]volumeName ReadOnlyVolumes map[string]string // containerPath -> volumeName (mounted read-only) - NetworkMode string // Network to join - Hostname string // Container hostname for DNS - Aliases []string // Additional network aliases - MemoryLimit int64 // Memory limit in bytes (0 = no limit) - NanoCPUs int64 // CPU quota in nanoseconds (1e9 = 1 core, 0 = no limit) - PidsLimit int64 // Max number of PIDs (0 = no limit) - ReadOnlyRootFS bool // Mount container root filesystem read-only - User string // User to run as - CapDrop []string // Linux capabilities to drop; nil uses runtime compat defaults - CapAdd []string // Linux capabilities to add; nil uses runtime compat defaults + Binds []ContainerBind // ephemeral resolved host binds, keyed by destination + // CDIDevices holds explicit CDI device IDs resolved from administrative + // device policy at activation time. It is ephemeral like Binds: never + // persisted, never logged by the runtime adapter. + CDIDevices []string // ephemeral resolved CDI IDs, encoded as one native CDI DeviceRequest + NetworkMode string // Network to join + Hostname string // Container hostname for DNS + Aliases []string // Additional network aliases + MemoryLimit int64 // Memory limit in bytes (0 = no limit) + NanoCPUs int64 // CPU quota in nanoseconds (1e9 = 1 core, 0 = no limit) + PidsLimit int64 // Max number of PIDs (0 = no limit) + ReadOnlyRootFS bool // Mount container root filesystem read-only + User string // User to run as + CapDrop []string // Linux capabilities to drop; nil uses runtime compat defaults + CapAdd []string // Linux capabilities to add; nil uses runtime compat defaults } // ContainerStatus represents the current state of a container. diff --git a/internal/domain/container_network_probe.go b/internal/domain/container_network_probe.go new file mode 100644 index 000000000..ee4d5890c --- /dev/null +++ b/internal/domain/container_network_probe.go @@ -0,0 +1,89 @@ +package domain + +import ( + "fmt" + "strings" + "time" +) + +// ProbeProtocol is the transport one bounded container-network readiness +// session uses. +type ProbeProtocol string + +const ( + ProbeProtocolHTTP ProbeProtocol = "http" + ProbeProtocolTCP ProbeProtocol = "tcp" +) + +// ContainerNetworkProbeRequest is one bounded readiness session against a +// declared internal container port. It carries exact identity: the target +// container, the execution start that must still be current when the +// session completes, and the already-derived private network the helper +// joins. Callers never pass a network they did not derive from the app +// incarnation. +type ContainerNetworkProbeRequest struct { + // TargetContainerID is the exact container to probe. + TargetContainerID string + // ExpectedStartedAt is the execution start the adapter must observe on + // the target before and after the session. A mismatch means the + // generation changed under the probe and the result is discarded. + ExpectedStartedAt time.Time + // Network is the derived private network the helper joins. It is the + // only network the helper is attached to. + Network string + Protocol ProbeProtocol + Port int + // Path is the origin-form request path for HTTP probes; empty for TCP. + Path string + // Timeout bounds the operational session through outcome validation. + // Mandatory force-removal uses its own independent bounded context. + Timeout time.Duration +} + +// ContainerNetworkProbeResult reports one completed session. +type ContainerNetworkProbeResult struct { + // Ready is true when the target accepted a 2xx/3xx HTTP status, or an + // accepted TCP connection. + Ready bool + // Status is the observed HTTP status code when one was read, else 0. + Status int + // Diagnostic is a bounded, non-sensitive description of an unhealthy + // attempt. It never carries addresses, paths, or raw command output. + Diagnostic string +} + +// Validate checks the request shape. It rejects missing identity or +// network, an invalid protocol, port, path, or timeout, and a path on a +// TCP probe. +func (r ContainerNetworkProbeRequest) Validate() error { + if strings.TrimSpace(r.TargetContainerID) == "" { + return fmt.Errorf("%w: target container id is required", ErrInvalidNetworkProbe) + } + if strings.TrimSpace(r.Network) == "" { + return fmt.Errorf("%w: private network is required", ErrInvalidNetworkProbe) + } + if r.ExpectedStartedAt.IsZero() { + // Without an execution boundary the probe could report readiness + // for a generation that already restarted. Fail closed. + return fmt.Errorf("%w: expected execution start is required", ErrInvalidNetworkProbe) + } + if r.Port < 1 || r.Port > 65535 { + return fmt.Errorf("%w: port must be 1-65535", ErrInvalidNetworkProbe) + } + if r.Timeout <= 0 { + return fmt.Errorf("%w: timeout must be positive", ErrInvalidNetworkProbe) + } + switch r.Protocol { + case ProbeProtocolHTTP: + if err := ValidateReadinessPath(r.Path); err != nil { + return fmt.Errorf("%w: %v", ErrInvalidNetworkProbe, err) + } + case ProbeProtocolTCP: + if r.Path != "" { + return fmt.Errorf("%w: tcp probe takes no path", ErrInvalidNetworkProbe) + } + default: + return fmt.Errorf("%w: protocol must be http|tcp", ErrInvalidNetworkProbe) + } + return nil +} diff --git a/internal/domain/container_network_probe_test.go b/internal/domain/container_network_probe_test.go new file mode 100644 index 000000000..2ed40faf5 --- /dev/null +++ b/internal/domain/container_network_probe_test.go @@ -0,0 +1,69 @@ +package domain_test + +import ( + "errors" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +// TestContainerNetworkProbeRequest_Validate proves the bounded probe +// request fails closed on every missing or malformed field. +func TestContainerNetworkProbeRequest_Validate(t *testing.T) { + valid := domain.ContainerNetworkProbeRequest{ + TargetContainerID: "abc123", + ExpectedStartedAt: time.Now().UTC(), + Network: "gordon--app--net", + Protocol: domain.ProbeProtocolHTTP, + Port: 8080, + Path: "/healthz", + Timeout: time.Second, + } + require.NoError(t, valid.Validate()) + + tests := []struct { + name string + mutate func(*domain.ContainerNetworkProbeRequest) + }{ + {"missing target id", func(r *domain.ContainerNetworkProbeRequest) { r.TargetContainerID = " " }}, + {"missing network", func(r *domain.ContainerNetworkProbeRequest) { r.Network = "" }}, + {"missing execution start", func(r *domain.ContainerNetworkProbeRequest) { r.ExpectedStartedAt = time.Time{} }}, + {"zero port", func(r *domain.ContainerNetworkProbeRequest) { r.Port = 0 }}, + {"port too high", func(r *domain.ContainerNetworkProbeRequest) { r.Port = 70000 }}, + {"zero timeout", func(r *domain.ContainerNetworkProbeRequest) { r.Timeout = 0 }}, + {"unknown protocol", func(r *domain.ContainerNetworkProbeRequest) { r.Protocol = "grpc" }}, + {"http without path", func(r *domain.ContainerNetworkProbeRequest) { r.Path = "" }}, + {"http path not origin form", func(r *domain.ContainerNetworkProbeRequest) { r.Path = "healthz" }}, + {"http path with surrounding whitespace", func(r *domain.ContainerNetworkProbeRequest) { r.Path = " /healthz " }}, + {"tcp with path", func(r *domain.ContainerNetworkProbeRequest) { + r.Protocol = domain.ProbeProtocolTCP + }}, + } + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + request := valid + tc.mutate(&request) + err := request.Validate() + require.Error(t, err) + assert.True(t, errors.Is(err, domain.ErrInvalidNetworkProbe), "want ErrInvalidNetworkProbe, got %v", err) + }) + } +} + +// TestContainerNetworkProbeRequest_ValidateTCP proves a pathless TCP probe +// is valid. +func TestContainerNetworkProbeRequest_ValidateTCP(t *testing.T) { + request := domain.ContainerNetworkProbeRequest{ + TargetContainerID: "abc123", + ExpectedStartedAt: time.Now().UTC(), + Network: "gordon--app--net", + Protocol: domain.ProbeProtocolTCP, + Port: 5432, + Timeout: time.Second, + } + require.NoError(t, request.Validate()) +} diff --git a/internal/domain/env.go b/internal/domain/env.go index d99cb4663..3bcdef4bf 100644 --- a/internal/domain/env.go +++ b/internal/domain/env.go @@ -1,71 +1,15 @@ package domain import ( - "bufio" - "bytes" "regexp" "strings" ) var ( - envKeyRegex = regexp.MustCompile(`^[A-Za-z_][A-Za-z0-9_]*$`) - containerNameRegex = regexp.MustCompile(`^[A-Za-z][A-Za-z0-9_-]*$`) - secretRefRegex = regexp.MustCompile(`\$\{(pass|sops):[^}]+\}`) + envKeyRegex = regexp.MustCompile(`^[A-Za-z_][A-Za-z0-9_]*$`) + secretRefRegex = regexp.MustCompile(`\$\{(pass|sops):[^}]+\}`) ) -// SanitizeDomainForEnvFile validates and sanitizes a domain name for env file storage. -// Returns the collision-resistant storage key used for filenames. -func SanitizeDomainForEnvFile(domainName string) (string, error) { - safeDomain, err := NewEnvStorageKey(domainName) - if err != nil { - return "", err - } - - return string(safeDomain), nil -} - -// ParseEnvData parses env data into a key-value map. -func ParseEnvData(data []byte) (map[string]string, error) { - secrets := make(map[string]string) - scanner := bufio.NewScanner(bytes.NewReader(data)) - // Allow large env values (certs/keys) without hitting the default ~64KB limit. - scanner.Buffer(make([]byte, 0, 64*1024), 1024*1024) - for scanner.Scan() { - line := strings.TrimSpace(scanner.Text()) - if line == "" || strings.HasPrefix(line, "#") { - continue - } - - parts := strings.SplitN(line, "=", 2) - if len(parts) != 2 { - continue - } - - key := strings.TrimSpace(parts[0]) - rawValue := parts[1] - - trimmedValue := strings.TrimSpace(rawValue) - value := trimmedValue - - if len(trimmedValue) >= 2 { - if (strings.HasPrefix(trimmedValue, "\"") && strings.HasSuffix(trimmedValue, "\"")) || - (strings.HasPrefix(trimmedValue, "'") && strings.HasSuffix(trimmedValue, "'")) { - value = trimmedValue[1 : len(trimmedValue)-1] - } - } - - if key != "" { - secrets[key] = value - } - } - - if err := scanner.Err(); err != nil { - return nil, err - } - - return secrets, nil -} - // ValidateEnvKey validates an env key for storage. func ValidateEnvKey(key string) error { if key == "" { @@ -87,19 +31,3 @@ func ValidateEnvKey(key string) error { func ContainsSecretReference(value string) bool { return secretRefRegex.MatchString(value) } - -// ValidateContainerName validates a container name for attachment storage. -// Container names can contain alphanumeric characters, hyphens, and underscores, -// but must start with a letter. -func ValidateContainerName(name string) error { - if name == "" { - return ErrInvalidContainerName - } - if strings.Contains(name, "..") || strings.ContainsAny(name, "/\\") { - return ErrPathTraversal - } - if !containerNameRegex.MatchString(name) { - return ErrInvalidContainerName - } - return nil -} diff --git a/internal/domain/env_test.go b/internal/domain/env_test.go index 5ff53d526..67a0880a9 100644 --- a/internal/domain/env_test.go +++ b/internal/domain/env_test.go @@ -1,147 +1,12 @@ package domain import ( - "encoding/base64" - "strings" "testing" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) -func TestSanitizeDomainForEnvFile(t *testing.T) { - tests := []struct { - name string - domain string - want string - wantErr bool - }{ - {name: "simple domain", domain: "example.com", want: base64.RawURLEncoding.EncodeToString([]byte("example.com"))}, - {name: "subdomain", domain: "app.example.com", want: base64.RawURLEncoding.EncodeToString([]byte("app.example.com"))}, - {name: "domain with port", domain: "example.com:8080", want: base64.RawURLEncoding.EncodeToString([]byte("example.com:8080"))}, - {name: "domain with path", domain: "example.com/path", want: base64.RawURLEncoding.EncodeToString([]byte("example.com/path"))}, - {name: "single char", domain: "a", want: base64.RawURLEncoding.EncodeToString([]byte("a"))}, - {name: "hyphenated", domain: "my-app.example.com", want: base64.RawURLEncoding.EncodeToString([]byte("my-app.example.com"))}, - {name: "empty", domain: "", wantErr: true}, - {name: "path traversal", domain: "../etc/passwd", wantErr: true}, - {name: "double dots", domain: "foo..bar", wantErr: true}, - {name: "starts with dot", domain: ".hidden", wantErr: true}, - {name: "ends with dot", domain: "trailing.", wantErr: true}, - {name: "space", domain: "has space", wantErr: true}, - {name: "special chars", domain: "bad$domain", wantErr: true}, - {name: "underscore rejected as invalid domain character", domain: "example_com", wantErr: true}, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - got, err := SanitizeDomainForEnvFile(tt.domain) - if tt.wantErr { - assert.Error(t, err) - return - } - require.NoError(t, err) - assert.Equal(t, tt.want, got) - }) - } -} - -func TestSanitizeDomainForEnvFile_DistinguishesSeparators(t *testing.T) { - gotDot, err := SanitizeDomainForEnvFile("app.example.com") - require.NoError(t, err) - - gotPort, err := SanitizeDomainForEnvFile("app:example:com") - require.NoError(t, err) - - gotPath, err := SanitizeDomainForEnvFile("app/example/com") - require.NoError(t, err) - - assert.NotEqual(t, gotDot, gotPort) - assert.NotEqual(t, gotDot, gotPath) - assert.NotEqual(t, gotPort, gotPath) -} - -func TestParseEnvData(t *testing.T) { - tests := []struct { - name string - data string - want map[string]string - wantErr bool - }{ - { - name: "simple key-value", - data: "FOO=bar\nBAZ=qux", - want: map[string]string{"FOO": "bar", "BAZ": "qux"}, - }, - { - name: "double-quoted value", - data: `KEY="hello world"`, - want: map[string]string{"KEY": "hello world"}, - }, - { - name: "single-quoted value", - data: `KEY='hello world'`, - want: map[string]string{"KEY": "hello world"}, - }, - { - name: "comment lines", - data: "# this is a comment\nKEY=val", - want: map[string]string{"KEY": "val"}, - }, - { - name: "empty lines", - data: "\n\nKEY=val\n\n", - want: map[string]string{"KEY": "val"}, - }, - { - name: "value with equals", - data: "URL=postgres://host:5432/db?opt=1", - want: map[string]string{"URL": "postgres://host:5432/db?opt=1"}, - }, - { - name: "no value", - data: "NOVALUE", - want: map[string]string{}, - }, - { - name: "empty value", - data: "KEY=", - want: map[string]string{"KEY": ""}, - }, - { - name: "empty input", - data: "", - want: map[string]string{}, - }, - { - name: "large value within buffer", - data: "BIG=" + strings.Repeat("x", 100000), - want: map[string]string{"BIG": strings.Repeat("x", 100000)}, - }, - { - name: "quoted value with inner spaces", - data: `KEY=" hello "`, - want: map[string]string{"KEY": " hello "}, - }, - { - name: "unquoted value with outer spaces", - data: "KEY= hello ", - want: map[string]string{"KEY": "hello"}, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - got, err := ParseEnvData([]byte(tt.data)) - if tt.wantErr { - assert.Error(t, err) - return - } - require.NoError(t, err) - assert.Equal(t, tt.want, got) - }) - } -} - func TestValidateEnvKey(t *testing.T) { tests := []struct { name string @@ -209,44 +74,3 @@ func TestContainsSecretReference(t *testing.T) { }) } } - -func TestValidateContainerName(t *testing.T) { - tests := []struct { - name string - container string - wantErr bool - wantErrType error - }{ - {name: "simple", container: "postgres", wantErr: false}, - {name: "with hyphen", container: "gitea-postgres", wantErr: false}, - {name: "with underscore", container: "gitea_postgres", wantErr: false}, - {name: "complex real container", container: "gordon-git-example-com-gitea-postgres", wantErr: false}, - {name: "starts with letter", container: "a-b", wantErr: false}, - {name: "lowercase", container: "my-container", wantErr: false}, - {name: "mixed case", container: "MyContainer-123", wantErr: false}, - {name: "empty", container: "", wantErr: true, wantErrType: ErrInvalidContainerName}, - {name: "starts with number", container: "1container", wantErr: true, wantErrType: ErrInvalidContainerName}, - {name: "starts with hyphen", container: "-container", wantErr: true, wantErrType: ErrInvalidContainerName}, - {name: "starts with underscore", container: "_container", wantErr: true, wantErrType: ErrInvalidContainerName}, - {name: "contains slash", container: "container/name", wantErr: true, wantErrType: ErrPathTraversal}, - {name: "contains backslash", container: `container\name`, wantErr: true, wantErrType: ErrPathTraversal}, - {name: "path traversal", container: "..", wantErr: true, wantErrType: ErrPathTraversal}, - {name: "contains dot", container: "container.name", wantErr: true, wantErrType: ErrInvalidContainerName}, - {name: "contains space", container: "container name", wantErr: true, wantErrType: ErrInvalidContainerName}, - {name: "contains special char", container: "container$name", wantErr: true, wantErrType: ErrInvalidContainerName}, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - err := ValidateContainerName(tt.container) - if tt.wantErr { - require.Error(t, err) - if tt.wantErrType != nil { - assert.ErrorIs(t, err, tt.wantErrType) - } - } else { - assert.NoError(t, err) - } - }) - } -} diff --git a/internal/domain/envfile.go b/internal/domain/envfile.go deleted file mode 100644 index 66a7090e2..000000000 --- a/internal/domain/envfile.go +++ /dev/null @@ -1,47 +0,0 @@ -package domain - -import ( - "encoding/base64" - "regexp" - "strings" -) - -// envDomainRegex validates domain names used for env-backed secret storage. -// Allows: alphanumeric, dots, hyphens, colons (for ports), and forward slashes (for paths). -var envDomainRegex = regexp.MustCompile(`^[a-zA-Z0-9][a-zA-Z0-9.:/-]*[a-zA-Z0-9]$|^[a-zA-Z0-9]$`) - -// EnvStorageKey is the domain-level storage identifier used for env-backed secrets. -// It is deterministic, collision-resistant, and safe to use in file names. -type EnvStorageKey string - -// NewEnvStorageKey validates a route domain and returns the storage-safe identifier -// used by env/secrets adapters. The identifier is reversible and unique for the input. -func NewEnvStorageKey(domainName string) (EnvStorageKey, error) { - if err := ValidateEnvStorageDomain(domainName); err != nil { - return "", err - } - - return EnvStorageKey(base64.RawURLEncoding.EncodeToString([]byte(domainName))), nil -} - -// ValidateEnvStorageDomain validates domains used for env-backed secret storage. -func ValidateEnvStorageDomain(domainName string) error { - if domainName == "" { - return ErrPathTraversal - } - - if strings.Contains(domainName, "..") { - return ErrPathTraversal - } - - if !envDomainRegex.MatchString(domainName) { - return ErrPathTraversal - } - - return nil -} - -// FileName returns the on-disk env filename for this storage key. -func (k EnvStorageKey) FileName() string { - return string(k) + ".env" -} diff --git a/internal/domain/errors.go b/internal/domain/errors.go index 49e30cde1..ca122f6d3 100644 --- a/internal/domain/errors.go +++ b/internal/domain/errors.go @@ -11,6 +11,10 @@ var ( ErrContainerNotRunning = errors.New("container is not running") ErrContainerRunning = errors.New("container is already running") ErrContainerExited = errors.New("container exited") + // ErrRuntimeUnsupported wraps device-bearing create failures on + // engines that cannot serve CDI device requests. It fails closed: + // Gordon never retries without devices. + ErrRuntimeUnsupported = errors.New("runtime does not support CDI device requests") // Image errors ErrImageNotFound = errors.New("image not found") @@ -36,6 +40,9 @@ var ( ErrUnauthorized = errors.New("unauthorized") ErrBlobSizeExceeded = errors.New("blob size exceeds maximum") ErrExecOutputExceeded = errors.New("container exec output exceeds maximum") + // ErrManifestBlobUnknown marks a manifest that references config or + // layer content the repository never completed an upload for. + ErrManifestBlobUnknown = errors.New("manifest references a blob the repository does not own") // Network errors ErrNetworkNotFound = errors.New("network not found") @@ -59,12 +66,57 @@ var ( ErrInvalidDomainPattern = errors.New("invalid domain pattern") ErrRouteConflict = errors.New("route conflicts with existing configuration") + // App manifest errors + ErrInvalidAppSpec = errors.New("invalid app manifest") + // ErrInvalidNetworkProbe wraps malformed bounded container-network + // readiness requests. It is distinct from ErrInvalidAppSpec: the + // manifest may be valid while one probe request is not. + ErrInvalidNetworkProbe = errors.New("invalid network probe request") + // ErrNetworkProbeCleanup marks a helper that could not be force-removed. + // Callers must fail immediately rather than create another helper. + ErrNetworkProbeCleanup = errors.New("network probe helper cleanup failed") + // ErrBindPolicy wraps administrative bind policy violations. It is + // distinct from ErrInvalidAppSpec: the manifest may be valid while the + // installation policy refuses to serve it. + ErrBindPolicy = errors.New("bind policy violation") + // ErrDevicePolicy wraps administrative device policy violations. It is + // distinct from ErrInvalidAppSpec: the manifest may be valid while the + // installation policy refuses to serve it. + ErrDevicePolicy = errors.New("device policy violation") + + // App state errors + ErrAppStateIO = errors.New("app state storage failure") + ErrAppStateCorrupt = errors.New("app state is corrupt") + ErrAppStateIncompatible = errors.New("app state format is not supported by this binary") + ErrAppStateConflict = errors.New("app state conflict") + // ErrAppNotFound marks a lifecycle mutation of a name that has no + // live app identity at all. It is distinct from a missing revision + // or operation on a known app. + ErrAppNotFound = errors.New("app not found") + ErrAppRevisionNotFound = errors.New("app revision not found") + ErrAppIntentNotFound = errors.New("app apply intent not found") + ErrAppOperationNotFound = errors.New("app operation not found") + ErrAppReservationConflict = errors.New("listener reservation conflict") + ErrAppTrafficProjection = errors.New("app traffic projection failed") + ErrAppImageUnresolvable = errors.New("image reference unresolvable") + ErrAppImageNotAllowed = errors.New("image reference not allowed by installation policy") + ErrAppSecretMissing = errors.New("required app secret missing") + ErrAppUnmanagedImageVolume = errors.New("image declares unmanaged volume") + // ErrAppServiceScope marks an app-wide mutation of a multi-service app + // that named neither one service nor all of them. + ErrAppServiceScope = errors.New("app service scope required") + // ErrPruneDisabled means prune could not establish a safe scope for + // the requested operation at all (for example, its protection or + // runtime ports are not wired). It is never returned merely because + // app state exists: protected and unknown candidates are reported + // per candidate and do not fail the operation. + ErrPruneDisabled = errors.New("pruning unavailable: cannot establish a safe prune scope") + // Environment errors ErrEnvFileNotFound = errors.New("environment file not found") ErrSecretNotFound = errors.New("secret not found") ErrSecretsAlreadyExist = errors.New("secrets already exist") ErrProviderNotFound = errors.New("secret provider not found") - ErrInvalidContainerName = errors.New("invalid container name") ErrAttachmentOwnershipMismatch = errors.New("attachment ownership mismatch") ErrReadinessLogSizeExceeded = errors.New("readiness log exceeds maximum") diff --git a/internal/domain/event.go b/internal/domain/event.go index 3a9fa0ec8..9d02090d8 100644 --- a/internal/domain/event.go +++ b/internal/domain/event.go @@ -9,15 +9,9 @@ import ( type EventType string const ( - EventImagePushed EventType = "image.pushed" - EventImageDeleted EventType = "image.deleted" - EventConfigReload EventType = "config.reload" - EventManualDeploy EventType = "manual.deploy" - EventContainerStop EventType = "container.stop" - EventContainerStart EventType = "container.start" - EventContainerHealthCheck EventType = "container.health_check" - EventContainerDeployed EventType = "container.deployed" - EventSecretsChanged EventType = "secrets.changed" + EventImagePushed EventType = "image.pushed" + EventImageDeleted EventType = "image.deleted" + EventConfigReload EventType = "config.reload" ) // Event represents a domain event that occurred in the system. @@ -40,14 +34,6 @@ type ImagePushedPayload struct { Annotations map[string]string } -// ContainerEventPayload contains data for container events. -type ContainerEventPayload struct { - ContainerID string - Domain string - Image string - Action string -} - // ConfigReloadPayload contains data for config.reload events. type ConfigReloadPayload struct { Source string // "file" or "manual" @@ -56,39 +42,9 @@ type ConfigReloadPayload struct { UpdatedRoutes []string } -// ManualDeployPayload contains data for manual.deploy events. -type ManualDeployPayload struct { - Domain string `json:"domain"` -} - -// SecretsChangedPayload contains data for secrets.changed events. -type SecretsChangedPayload struct { - Domain string // Route domain whose secrets changed - Operation string // "set" or "delete" - Keys []string // Secret key names (not values) -} - // Context keys for domain-level concerns. type contextKey string -const ( - // ContextKeyInternalDeploy indicates the deployment is triggered internally - // (e.g., from our own registry's image.pushed event) and should use - // internal registry authentication for image pulls. - ContextKeyInternalDeploy contextKey = "internal_deploy" -) - -// IsInternalDeploy checks if the context indicates an internal deployment. -func IsInternalDeploy(ctx context.Context) bool { - v, ok := ctx.Value(ContextKeyInternalDeploy).(bool) - return ok && v -} - -// WithInternalDeploy returns a context marked as an internal deployment. -func WithInternalDeploy(ctx context.Context) context.Context { - return context.WithValue(ctx, ContextKeyInternalDeploy, true) -} - const ( // ContextKeySkipReadiness indicates that readiness checks should be skipped // (e.g., during AutoStart where the background monitor handles crash recovery). diff --git a/internal/domain/event_test.go b/internal/domain/event_test.go deleted file mode 100644 index ca8571e3f..000000000 --- a/internal/domain/event_test.go +++ /dev/null @@ -1,58 +0,0 @@ -package domain - -import ( - "context" - "testing" - - "github.com/stretchr/testify/assert" -) - -func TestIsInternalDeploy(t *testing.T) { - tests := []struct { - name string - ctx context.Context - expected bool - }{ - { - name: "returns false for plain context", - ctx: context.Background(), - expected: false, - }, - { - name: "returns true for internal deploy context", - ctx: WithInternalDeploy(context.Background()), - expected: true, - }, - { - name: "returns false for context with wrong type", - ctx: context.WithValue(context.Background(), ContextKeyInternalDeploy, "true"), - expected: false, - }, - { - name: "returns false for context with false value", - ctx: context.WithValue(context.Background(), ContextKeyInternalDeploy, false), - expected: false, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - result := IsInternalDeploy(tt.ctx) - assert.Equal(t, tt.expected, result) - }) - } -} - -func TestWithInternalDeploy(t *testing.T) { - ctx := context.Background() - - // Before marking - assert.False(t, IsInternalDeploy(ctx)) - - // After marking - internalCtx := WithInternalDeploy(ctx) - assert.True(t, IsInternalDeploy(internalCtx)) - - // Original context unchanged - assert.False(t, IsInternalDeploy(ctx)) -} diff --git a/internal/domain/image_policy.go b/internal/domain/image_policy.go new file mode 100644 index 000000000..e1c234d1c --- /dev/null +++ b/internal/domain/image_policy.go @@ -0,0 +1,219 @@ +package domain + +import ( + "fmt" + "net" + "strconv" + "strings" + + "github.com/bnema/gordon/pkg/validation" +) + +var defaultImageRegistries = []string{"docker.io", "ghcr.io", "quay.io"} + +type ImagePullTransport string + +const ( + ImagePullTransportVerifiedTLS ImagePullTransport = "verified-tls" + ImagePullTransportHTTP ImagePullTransport = "http" +) + +type ImagePullRequest struct { + Reference string + Username string + Password string + Transport ImagePullTransport +} + +// IsInstallationImage reports whether ref belongs to the configured installation registry. +func (p ImageSourcePolicy) IsInstallationImage(ref string) bool { + host, _, err := parseImageRegistry(strings.TrimSpace(ref)) + if err != nil || p.InstallationRegistry == "" { + return false + } + installation, err := canonicalRegistryHost(p.InstallationRegistry) + return err == nil && dockerRegistryAlias(host) == dockerRegistryAlias(installation) +} + +// ImageSourcePolicy is the installation policy for every image reference +// the daemon validates, resolves, pulls, or runs. +type ImageSourcePolicy struct { + // AllowedRegistries adds explicit hostname+port entries to the defaults. + AllowedRegistries []string + // RequireDigest rejects mutable tag references. + RequireDigest bool + // InstallationRegistry is Gordon's own registry host, always allowed. + InstallationRegistry string +} + +// Validate checks every configured registry authority. +func (p ImageSourcePolicy) Validate() error { + entries := append([]string{}, p.AllowedRegistries...) + if p.InstallationRegistry != "" { + entries = append(entries, p.InstallationRegistry) + } + for _, entry := range entries { + if _, err := canonicalRegistryHost(entry); err != nil { + return fmt.Errorf("%w: %s", ErrAppImageNotAllowed, err) + } + } + return nil +} + +// ValidateImageSource checks one reference against the installation registry +// allowlist. The allowlist controls names the runtime may contact; it does not +// constrain where DNS resolves or provide runtime network egress enforcement. +func (p ImageSourcePolicy) ValidateImageSource(ref string) error { + host, rest, err := parseImageRegistry(strings.TrimSpace(ref)) + if err != nil { + return fmt.Errorf("%w: %s", ErrAppImageNotAllowed, err) + } + repository, reference := splitRepositoryReference(rest) + if err := validation.ValidateRepositoryName(repository); err != nil { + return fmt.Errorf("%w: %s", ErrAppImageNotAllowed, err) + } + if strings.Contains(rest, "@") { + if err := validation.ValidateImageDigest(reference); err != nil { + return fmt.Errorf("%w: %s", ErrAppImageNotAllowed, err) + } + } else if p.RequireDigest { + return fmt.Errorf("%w: registry policy requires an immutable digest", ErrAppImageNotAllowed) + } + if p.registryAllowed(host) { + return nil + } + return fmt.Errorf("%w: registry %q is not allowlisted", ErrAppImageNotAllowed, host) +} + +func (p ImageSourcePolicy) registryAllowed(host string) bool { + entries := append(append([]string{}, defaultImageRegistries...), p.AllowedRegistries...) + if p.InstallationRegistry != "" { + entries = append(entries, p.InstallationRegistry) + } + for _, entry := range entries { + allowed, err := canonicalRegistryHost(entry) + if err == nil && dockerRegistryAlias(allowed) == dockerRegistryAlias(host) { + return true + } + } + return false +} + +func parseImageRegistry(ref string) (string, string, error) { + if ref == "" || strings.Contains(ref, "://") || strings.Contains(ref, "\\") { + return "", "", fmt.Errorf("malformed image reference") + } + first, rest, explicit := strings.Cut(ref, "/") + if !explicit || (!strings.ContainsAny(first, ".:") && !strings.EqualFold(first, "localhost")) { + first, rest = "docker.io", ref + } + if strings.Contains(first, "@") || rest == "" { + return "", "", fmt.Errorf("malformed registry host") + } + host, err := canonicalRegistryHost(first) + if err != nil { + return "", "", err + } + return host, rest, nil +} + +// canonicalRegistryHost safely normalizes a hostname or IP plus optional port. +// HTTPS's default port is omitted so host and host:443 are equivalent. +func canonicalRegistryHost(value string) (string, error) { + value = strings.TrimSpace(value) + if value == "" || strings.ContainsAny(value, "/@?#") { + return "", fmt.Errorf("malformed registry host %q", value) + } + host, port, err := splitRegistryAuthority(value) + if err != nil { + return "", err + } + host = strings.TrimSuffix(strings.ToLower(host), ".") + if host == "" || strings.Contains(host, "..") || (net.ParseIP(host) == nil && !validRegistryHostname(host)) { + return "", fmt.Errorf("malformed registry host %q", value) + } + if port != "" { + n, err := strconv.Atoi(port) + if err != nil || n < 1 || n > 65535 || strconv.Itoa(n) != port { + return "", fmt.Errorf("malformed registry port %q", port) + } + if n != 443 { + return net.JoinHostPort(host, port), nil + } + } + return host, nil +} + +func splitRegistryAuthority(value string) (string, string, error) { + if strings.HasPrefix(value, "[") { + if strings.HasSuffix(value, "]") { + return strings.TrimSuffix(strings.TrimPrefix(value, "["), "]"), "", nil + } + host, port, err := net.SplitHostPort(value) + if err != nil { + return "", "", fmt.Errorf("malformed registry host %q", value) + } + return host, port, nil + } + if strings.Count(value, ":") == 1 { + host, port, _ := strings.Cut(value, ":") + if host == "" || port == "" { + return "", "", fmt.Errorf("malformed registry host %q", value) + } + return host, port, nil + } + if strings.Contains(value, ":") && net.ParseIP(value) == nil { + return "", "", fmt.Errorf("ambiguous registry host %q", value) + } + return value, "", nil +} + +func validRegistryHostname(host string) bool { + for _, label := range strings.Split(host, ".") { + if label == "" || len(label) > 63 || label[0] == '-' || label[len(label)-1] == '-' { + return false + } + for _, r := range label { + if (r < 'a' || r > 'z') && (r < '0' || r > '9') && r != '-' { + return false + } + } + } + return len(host) <= 253 +} + +func dockerRegistryAlias(host string) string { + if host == "registry-1.docker.io" { + return "docker.io" + } + return host +} + +func splitRepositoryReference(rest string) (string, string) { + if base, digest, ok := strings.Cut(rest, "@"); ok { + return base, digest + } + if idx := strings.LastIndex(rest, ":"); idx >= 0 { + return rest[:idx], rest[idx+1:] + } + return rest, "" +} + +// IsLocalOrPrivateHost reports whether a literal registry host is local or +// non-public. It does not resolve DNS names and must not be treated as an +// egress guarantee. +func IsLocalOrPrivateHost(host string) bool { + hostname, err := canonicalRegistryHost(host) + if err != nil { + return true + } + if h, _, splitErr := net.SplitHostPort(hostname); splitErr == nil { + hostname = h + } + hostname = strings.Trim(hostname, "[]") + if hostname == "localhost" || strings.HasSuffix(hostname, ".localhost") { + return true + } + ip := net.ParseIP(hostname) + return ip != nil && (ip.IsLoopback() || ip.IsPrivate() || ip.IsLinkLocalUnicast() || ip.IsLinkLocalMulticast() || ip.IsUnspecified() || ip.IsMulticast()) +} diff --git a/internal/domain/image_policy_test.go b/internal/domain/image_policy_test.go new file mode 100644 index 000000000..aabc91a8d --- /dev/null +++ b/internal/domain/image_policy_test.go @@ -0,0 +1,127 @@ +package domain_test + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +const testDigest202 = "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + +func TestImageSourcePolicy_DefaultRegistries(t *testing.T) { + policy := domain.ImageSourcePolicy{} + allowed := []string{ + "nginx:1.25", + "docker.io/library/nginx:1.25", + "registry-1.docker.io/library/nginx@" + testDigest202, + "ghcr.io/example/app:1", + "quay.io/example/app:1", + } + for _, ref := range allowed { + require.NoError(t, policy.ValidateImageSource(ref), ref) + } + assert.ErrorIs(t, policy.ValidateImageSource("registry.example.com/team/app:1"), domain.ErrAppImageNotAllowed) +} + +func TestImageSourcePolicy_AllowsExplicitPrivateRegistry(t *testing.T) { + policy := domain.ImageSourcePolicy{AllowedRegistries: []string{"REGISTRY.INTERNAL.:5000"}} + require.NoError(t, policy.ValidateImageSource("registry.internal:5000/team/app:1")) + assert.ErrorIs(t, policy.ValidateImageSource("registry.internal/team/app:1"), domain.ErrAppImageNotAllowed) +} + +func TestImageSourcePolicy_EnforcesAllowlist(t *testing.T) { + policy := domain.ImageSourcePolicy{AllowedRegistries: []string{"registry.example.com"}} + require.NoError(t, policy.ValidateImageSource("registry.example.com/team/app:1.0")) + require.NoError(t, policy.ValidateImageSource("registry.example.com/team/app@"+testDigest202)) + require.NoError(t, policy.ValidateImageSource("docker.io/library/nginx:1.0")) + assert.ErrorIs(t, policy.ValidateImageSource("other.example.com/team/app:1.0"), domain.ErrAppImageNotAllowed) +} + +func TestImageSourcePolicy_AlwaysAllowsInstallationRegistry(t *testing.T) { + policy := domain.ImageSourcePolicy{ + AllowedRegistries: []string{"registry.example.com"}, + InstallationRegistry: "gordon.example.com", + } + require.NoError(t, policy.ValidateImageSource("gordon.example.com/blog/web:1.4.2")) + require.NoError(t, policy.ValidateImageSource("gordon.example.com/blog/web@"+testDigest202)) +} + +func TestImageSourcePolicy_RequireDigest(t *testing.T) { + policy := domain.ImageSourcePolicy{ + AllowedRegistries: []string{"registry.example.com"}, + RequireDigest: true, + InstallationRegistry: "gordon.example.com", + } + require.NoError(t, policy.ValidateImageSource("registry.example.com/team/app@"+testDigest202)) + require.NoError(t, policy.ValidateImageSource("gordon.example.com/team/app@"+testDigest202)) + assert.ErrorIs(t, policy.ValidateImageSource("registry.example.com/team/app:1.0"), domain.ErrAppImageNotAllowed) + assert.ErrorIs(t, policy.ValidateImageSource("gordon.example.com/team/app:1.0"), domain.ErrAppImageNotAllowed) + assert.ErrorIs(t, policy.ValidateImageSource("nginx"), domain.ErrAppImageNotAllowed) +} + +func TestImageSourcePolicy_RejectsMalformedDigests(t *testing.T) { + policy := domain.ImageSourcePolicy{InstallationRegistry: "gordon.example.com"} + for _, digest := range []string{ + "sha256:short", + "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaag", + "sha512:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + } { + assert.ErrorIs(t, policy.ValidateImageSource("gordon.example.com/team/app@"+digest), domain.ErrAppImageNotAllowed, digest) + } +} + +func TestImageSourcePolicy_CanonicalizesHostAndDefaultPort(t *testing.T) { + policy := domain.ImageSourcePolicy{AllowedRegistries: []string{"registry.example.com"}} + require.NoError(t, policy.ValidateImageSource("REGISTRY.EXAMPLE.COM./team/app:1")) + require.NoError(t, policy.ValidateImageSource("registry.example.com:443/team/app:1")) +} + +func TestImageSourcePolicy_RejectsMalformedAllowlistEntries(t *testing.T) { + for _, entry := range []string{"https://registry.example.com", "user@registry.example.com", "registry.example.com:", "registry.example.com:0443", "registry.example.com:443:80"} { + policy := domain.ImageSourcePolicy{AllowedRegistries: []string{entry}} + assert.ErrorIs(t, policy.Validate(), domain.ErrAppImageNotAllowed, entry) + } +} + +func TestImageSourcePolicy_RejectsMalformedReferences(t *testing.T) { + policy := domain.ImageSourcePolicy{AllowedRegistries: []string{"registry.example.com"}} + refs := []string{"", " ", "registry.example.com/BadRepo:1.0", "https://registry.example.com/app:1", "user@registry.example.com/app:1", "registry.example.com:/app:1", "registry.example.com:0443/app:1", "registry.example.com:443:80/app:1", "[::1/app:1"} + for _, ref := range refs { + assert.ErrorIs(t, policy.ValidateImageSource(ref), domain.ErrAppImageNotAllowed, ref) + } +} + +func TestImageSourcePolicy_IsInstallationImageDockerHubAlias(t *testing.T) { + cases := []struct { + name string + installation string + ref string + want bool + }{ + {"implicit docker.io ref matches docker.io installation", "docker.io", "library/nginx:1", true}, + {"registry-1 ref matches docker.io installation", "docker.io", "registry-1.docker.io/library/nginx:1", true}, + {"docker.io ref matches registry-1 installation", "registry-1.docker.io", "docker.io/library/nginx:1", true}, + {"installation registry is not docker.io", "gordon.example.com", "docker.io/library/nginx:1", false}, + {"foreign ref does not match docker.io installation", "docker.io", "ghcr.io/example/app:1", false}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + policy := domain.ImageSourcePolicy{InstallationRegistry: tc.installation} + assert.Equal(t, tc.want, policy.IsInstallationImage(tc.ref)) + }) + } +} + +func TestIsLocalOrPrivateHost(t *testing.T) { + local := []string{"localhost", "sub.localhost", "127.0.0.1", "127.0.0.1:5000", "[::1]:5000", "10.0.0.1", "192.168.0.1", "172.31.0.1", "fc00::1", "0.0.0.0", "169.254.169.254"} + for _, host := range local { + assert.True(t, domain.IsLocalOrPrivateHost(host), "host %q should be local/private", host) + } + remote := []string{"registry.example.com", "registry.example.com:5000", "8.8.8.8", "docker.io", "2001:4860:4860::8888"} + for _, host := range remote { + assert.False(t, domain.IsLocalOrPrivateHost(host), "host %q should be public", host) + } +} diff --git a/internal/domain/image_ref_repo_test.go b/internal/domain/image_ref_repo_test.go new file mode 100644 index 000000000..426df4746 --- /dev/null +++ b/internal/domain/image_ref_repo_test.go @@ -0,0 +1,47 @@ +package domain_test + +import ( + "testing" + + "github.com/stretchr/testify/assert" + + "github.com/bnema/gordon/internal/domain" +) + +func TestImageRefRepository(t *testing.T) { + cases := []struct { + ref string + want string + }{ + {"registry.example.com/blog/web:1.4.2", "blog/web"}, + {"registry.example.com/blog/web@sha256:" + testDigest202[7:], "blog/web"}, + {"blog/web:1.4.2", "blog/web"}, + {"localhost:5000/private/app:1", "private/app"}, + {"127.0.0.1:12345/private@sha256:" + testDigest202[7:], "private"}, + {"nginx:1.25", "nginx"}, + {"", ""}, + } + for _, tc := range cases { + t.Run(tc.ref, func(t *testing.T) { + assert.Equal(t, tc.want, domain.ImageRefRepository(tc.ref)) + }) + } +} + +// TestImageRefRoots_QualifiesDigestWithRepository proves a digest root keeps +// the repository so prune can traverse the manifest's full closure. +func TestImageRefRoots_QualifiesDigestWithRepository(t *testing.T) { + digest := testDigest202 + roots := domain.ImageRefRoots(domain.ProtectionActiveService, "registry.example.com/blog/web@"+digest, "blog@active.web") + + var digestRoot *domain.ProtectionRoot + for i := range roots { + if roots[i].Ref == digest { + digestRoot = &roots[i] + } + } + if digestRoot == nil { + t.Fatalf("expected a digest root in %+v", roots) + } + assert.Equal(t, "blog/web", digestRoot.Repository) +} diff --git a/internal/domain/images.go b/internal/domain/images.go index c40a21695..928140ad0 100644 --- a/internal/domain/images.go +++ b/internal/domain/images.go @@ -12,6 +12,8 @@ type ImagePruneOptions struct { PruneDangling bool // PruneRegistry enables registry tag retention and blob garbage collection. PruneRegistry bool + // DryRun plans everything and deletes nothing. + DryRun bool } // DefaultImagePruneOptions returns options that prune both scopes with the default retention. @@ -55,6 +57,9 @@ type RegistryPruneResult struct { type ImagePruneReport struct { Runtime RuntimePruneResult Registry RegistryPruneResult + // Plan carries every candidate's verdict, the inventory gaps, and + // the applied flag. Dry-run and execution share this shape. + Plan PruneReport } // ImageInfo describes an image/tag visible from runtime and registry data. diff --git a/internal/domain/labels.go b/internal/domain/labels.go index 3f77fa765..6878b0c78 100644 --- a/internal/domain/labels.go +++ b/internal/domain/labels.go @@ -3,18 +3,30 @@ package domain // Label keys used by Gordon for container and image metadata. const ( // Container labels - LabelDomain = "gordon.domain" - LabelImage = "gordon.image" - LabelManaged = "gordon.managed" - LabelRoute = "gordon.route" - LabelAttachment = "gordon.attachment" - LabelAttachedTo = "gordon.attached-to" - LabelCreated = "gordon.created" + LabelDomain = "gordon.domain" + LabelImage = "gordon.image" + LabelManaged = "gordon.managed" + LabelCreated = "gordon.created" // LabelEnvHash stores a SHA-256 hash of the effective environment // variables at deploy time, used to detect env drift without // exposing secret values. LabelEnvHash = "gordon.env-hash" + // App ownership labels are stamped by the v3 deploy engine on every + // container and volume it creates (03-deployment.md §4A/B, frozen here + // so prune/backup guards can consume them before the engine activates). + // The engine wiring that stamps them lands at cutover; until then no + // live resource carries them and guards treat their absence on a + // managed resource as legacy/unknown provenance (never prune-eligible + // once app state exists). + LabelApp = "gordon.app" + LabelAppService = "gordon.app.service" + LabelAppRevision = "gordon.app.revision" + // LabelAppID carries the app's stable internal UUID, so provenance + // survives a public-name reuse without ever inferring ownership + // from the name alone. + LabelAppID = "gordon.app.id" + // Standalone service labels identify Gordon-managed L4 service containers. LabelService = "gordon.service" LabelServiceName = "gordon.service.name" @@ -40,4 +52,15 @@ const ( LabelPort = "gordon.port" // LabelEnvFile specifies the path to .env file inside the image. LabelEnvFile = "gordon.env-file" + // LabelPurpose marks Gordon-owned helper containers that are not + // managed workloads (volume archives, bounded readiness probes). + // Lifecycle, reconcile and prune paths must treat them as helper + // infrastructure, never as apps or services. + LabelPurpose = "gordon.purpose" +) + +// Purpose values for Gordon-owned helper containers. +const ( + // PurposeNetworkProbe identifies one bounded readiness probe helper. + PurposeNetworkProbe = "network-probe" ) diff --git a/internal/domain/labels_test.go b/internal/domain/labels_test.go index 6d0c9c5e7..fa781da19 100644 --- a/internal/domain/labels_test.go +++ b/internal/domain/labels_test.go @@ -16,9 +16,6 @@ func TestLabelConstantsValues(t *testing.T) { {domain.LabelDomain, "gordon.domain"}, {domain.LabelImage, "gordon.image"}, {domain.LabelManaged, "gordon.managed"}, - {domain.LabelRoute, "gordon.route"}, - {domain.LabelAttachment, "gordon.attachment"}, - {domain.LabelAttachedTo, "gordon.attached-to"}, {domain.LabelCreated, "gordon.created"}, } for _, tt := range tests { diff --git a/internal/domain/logs.go b/internal/domain/logs.go new file mode 100644 index 000000000..53343ae86 --- /dev/null +++ b/internal/domain/logs.go @@ -0,0 +1,67 @@ +package domain + +import "time" + +// LogSource identifies the workload that emitted a log record. +// A zero LogSource is Gordon itself. +type LogSource struct { + App string + Service string +} + +// IsGordon reports whether the record belongs to Gordon itself. +func (s LogSource) IsGordon() bool { + return s.App == "" +} + +// ServiceName is the unique exported service name, ".". +// Service names alone collide across apps (many apps have "web"). +func (s LogSource) ServiceName() string { + if s.IsGordon() { + return "gordon" + } + return s.App + "." + s.Service +} + +// Log stream names for container output. +const ( + LogStreamStdout = "stdout" + LogStreamStderr = "stderr" +) + +// Log types distinguish record families sharing one source. +const ( + LogTypeContainer = "container" + LogTypeAccess = "access" +) + +// LogSeverity is a coarse, exporter-neutral severity. +type LogSeverity int + +// Severities. The zero value means unspecified. +const ( + LogSeverityUnspecified LogSeverity = iota + LogSeverityInfo + LogSeverityWarn + LogSeverityError +) + +// ContainerLogLine is one demultiplexed line of container output. +type ContainerLogLine struct { + Time time.Time + Stream string + Body string +} + +// LogRecord is one exported log record. +type LogRecord struct { + Time time.Time + Source LogSource + Type string + Severity LogSeverity + Body string + // Stream is stdout or stderr for container output, empty otherwise. + Stream string + // Attributes carry per-record metadata (status, path...). + Attributes map[string]string +} diff --git a/internal/domain/preview.go b/internal/domain/preview.go deleted file mode 100644 index d481fa3f0..000000000 --- a/internal/domain/preview.go +++ /dev/null @@ -1,36 +0,0 @@ -package domain - -import "time" - -// PreviewStatus represents the lifecycle state of a preview environment. -type PreviewStatus string - -const ( - PreviewStatusDeploying PreviewStatus = "deploying" - PreviewStatusRunning PreviewStatus = "running" - PreviewStatusFailed PreviewStatus = "failed" - PreviewStatusTimeout PreviewStatus = "timeout" - - // DefaultPreviewSeparator is the subdomain separator used to construct - // preview domain names (e.g. "myapp--feat.example.com"). - DefaultPreviewSeparator = "--" -) - -// PreviewRoute represents an ephemeral preview environment. -type PreviewRoute struct { - Domain string `json:"domain"` - Image string `json:"image"` - BaseRoute string `json:"base_route"` - Name string `json:"name"` - CreatedAt time.Time `json:"created_at"` - ExpiresAt time.Time `json:"expires_at"` - HTTPS bool `json:"https"` - Status PreviewStatus `json:"status"` - Volumes []string `json:"volumes"` - Containers []string `json:"containers"` -} - -// IsExpired returns true if the preview has exceeded its TTL. -func (p PreviewRoute) IsExpired(now time.Time) bool { - return now.After(p.ExpiresAt) -} diff --git a/internal/domain/preview_test.go b/internal/domain/preview_test.go deleted file mode 100644 index 9ef4d6fa9..000000000 --- a/internal/domain/preview_test.go +++ /dev/null @@ -1,27 +0,0 @@ -package domain - -import ( - "testing" - "time" - - "github.com/stretchr/testify/assert" -) - -func TestPreviewRoute_IsExpired(t *testing.T) { - now := time.Now() - tests := []struct { - name string - expiresAt time.Time - want bool - }{ - {"expired", now.Add(-1 * time.Hour), true}, - {"not expired", now.Add(1 * time.Hour), false}, - {"just expired", now.Add(-1 * time.Second), true}, - } - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - p := PreviewRoute{ExpiresAt: tt.expiresAt} - assert.Equal(t, tt.want, p.IsExpired(now)) - }) - } -} diff --git a/internal/domain/prune.go b/internal/domain/prune.go new file mode 100644 index 000000000..1e05bb55d --- /dev/null +++ b/internal/domain/prune.go @@ -0,0 +1,1378 @@ +package domain + +import ( + "fmt" + "sort" + "strings" + "time" +) + +// This file freezes the selective-prune data model: resource identities, +// protection roots, inventory completeness, verdicts, stable reason +// codes, and the aggregate plan/report shapes. It is pure data plus +// validation — no store reads, no runtime calls, no JSON concerns. +// +// Fail-closed rule: a candidate is only PruneVerdictEligible when every +// fact needed to prove it safe was read completely. Missing, unreadable, +// or unsupported information yields PruneVerdictUnknown for the +// dependent candidates; it never yields eligibility, and it is never an +// operation-wide shutdown. + +// PruneVerdict is the fail-closed classification of one prune candidate. +type PruneVerdict string + +const ( + // PruneVerdictEligible means every fact needed to prove the + // candidate safe was available and no durable claim covers it. + PruneVerdictEligible PruneVerdict = "eligible" + // PruneVerdictProtected means a durable fact positively claims the + // candidate, so it must survive this prune. + PruneVerdictProtected PruneVerdict = "protected" + // PruneVerdictUnknown means at least one fact needed to prove the + // candidate safe was missing, unreadable, or unsupported. + PruneVerdictUnknown PruneVerdict = "unknown" +) + +// Valid reports whether v is one of the defined verdicts. +func (v PruneVerdict) Valid() bool { + switch v { + case PruneVerdictEligible, PruneVerdictProtected, PruneVerdictUnknown: + return true + default: + return false + } +} + +// PruneReason is a stable machine-readable reason code. The wire forms +// are frozen: they appear in JSON and human output, so existing values +// must not be renamed or re-typed. +type PruneReason string + +// Eligible reason codes. +const ( + // PruneReasonEligibleRetention marks a registry tag beyond the + // latest + keep_last retention window with no protection root. + PruneReasonEligibleRetention PruneReason = "eligible-retention-window" + // PruneReasonEligibleUnreferencedBlob marks registry content no + // retained or protected manifest references. + PruneReasonEligibleUnreferencedBlob PruneReason = "eligible-unreferenced-blob" + // PruneReasonEligibleDanglingRuntimeImage marks a runtime image with + // no tag, no container use, and no protection root. + PruneReasonEligibleDanglingRuntimeImage PruneReason = "eligible-dangling-runtime-image" + // PruneReasonEligibleReleasedVolume marks an app-owned volume whose + // durable lifecycle state explicitly records released/abandoned, + // which no container uses. + PruneReasonEligibleReleasedVolume PruneReason = "eligible-released-volume" +) + +// Protected reason codes. +const ( + // PruneReasonProtectedLatest marks the latest tag of a repository. + PruneReasonProtectedLatest PruneReason = "protected-latest" + // PruneReasonProtectedRetentionWindow marks a tag inside the + // configured latest + keep_last window. + PruneReasonProtectedRetentionWindow PruneReason = "protected-retention-window" + // PruneReasonProtectedDesiredRevision marks content a desired + // revision references. + PruneReasonProtectedDesiredRevision PruneReason = "protected-desired-revision" + // PruneReasonProtectedActiveService marks content an ACTIVE service + // references, including stopped and partially converged services. + PruneReasonProtectedActiveService PruneReason = "protected-active-service" + // PruneReasonProtectedRecoveryInhibition marks content a durable + // recovery inhibition still claims. + PruneReasonProtectedRecoveryInhibition PruneReason = "protected-recovery-inhibition" + // PruneReasonProtectedApplyIntent marks content a staged or + // committed apply intent can still materialize. + PruneReasonProtectedApplyIntent PruneReason = "protected-apply-intent" + // PruneReasonProtectedOperation marks content an unfinished + // operation journal references. + PruneReasonProtectedOperation PruneReason = "protected-operation" + // PruneReasonProtectedOwnership marks content a durable ownership + // record claims as attached or retained. + PruneReasonProtectedOwnership PruneReason = "protected-ownership" + // PruneReasonProtectedContainerUse marks runtime content any + // container uses, running or stopped. + PruneReasonProtectedContainerUse PruneReason = "protected-container-use" + // PruneReasonProtectedPendingUpload marks registry content a recent + // upload may still complete into. + PruneReasonProtectedPendingUpload PruneReason = "protected-pending-upload" + // PruneReasonProtectedSharedContent marks content referenced by + // retained or protected roots through the OCI closure. + PruneReasonProtectedSharedContent PruneReason = "protected-shared-content" + // PruneReasonProtectedPinned marks explicitly pinned content. + PruneReasonProtectedPinned PruneReason = "protected-pinned" + // PruneReasonProtectedServiceManaged marks a standalone service + // container's own managed resource. + PruneReasonProtectedServiceManaged PruneReason = "protected-service-managed" + // PruneReasonProtectedManagedOnly marks a Gordon-managed resource + // with no app ownership record: the legacy/unknown provenance class. + PruneReasonProtectedManagedOnly PruneReason = "protected-managed-without-ownership" + // PruneReasonProtectedTagged marks a runtime image that repo tags + // still point at: tag retention owns it, not runtime prune. + PruneReasonProtectedTagged PruneReason = "protected-tagged-runtime-image" + // PruneReasonProtectedUnmanaged marks a resource with no Gordon + // provenance at all. Gordon never adopts or deletes it. + PruneReasonProtectedUnmanaged PruneReason = "protected-unmanaged" +) + +// Unknown reason codes. +const ( + // PruneReasonUnknownIdentity marks a candidate whose exact identity + // could not be established. + PruneReasonUnknownIdentity PruneReason = "unknown-identity" + // PruneReasonUnknownProvenance marks a resource whose ownership + // could not be established. + PruneReasonUnknownProvenance PruneReason = "unknown-provenance" + // PruneReasonUnknownInventory marks a candidate whose safety depends + // on an inventory that was read incompletely. + PruneReasonUnknownInventory PruneReason = "unknown-incomplete-inventory" + // PruneReasonUnknownManifest marks content reachable only through a + // missing, corrupt, or unreadable manifest. + PruneReasonUnknownManifest PruneReason = "unknown-manifest-unreadable" + // PruneReasonUnknownMediaType marks content behind an unsupported + // OCI media type or manifest shape. + PruneReasonUnknownMediaType PruneReason = "unknown-unsupported-media-type" + // PruneReasonUnknownContainerUse marks runtime content whose + // container usage could not be established. + PruneReasonUnknownContainerUse PruneReason = "unknown-container-use" +) + +// pruneReasons lists every defined reason code in canonical output +// order, grouped by verdict category. +var pruneReasons = []PruneReason{ + PruneReasonEligibleRetention, + PruneReasonEligibleUnreferencedBlob, + PruneReasonEligibleDanglingRuntimeImage, + PruneReasonEligibleReleasedVolume, + + PruneReasonProtectedLatest, + PruneReasonProtectedRetentionWindow, + PruneReasonProtectedDesiredRevision, + PruneReasonProtectedActiveService, + PruneReasonProtectedRecoveryInhibition, + PruneReasonProtectedApplyIntent, + PruneReasonProtectedOperation, + PruneReasonProtectedOwnership, + PruneReasonProtectedContainerUse, + PruneReasonProtectedPendingUpload, + PruneReasonProtectedSharedContent, + PruneReasonProtectedPinned, + PruneReasonProtectedServiceManaged, + PruneReasonProtectedManagedOnly, + PruneReasonProtectedTagged, + PruneReasonProtectedUnmanaged, + + PruneReasonUnknownIdentity, + PruneReasonUnknownProvenance, + PruneReasonUnknownInventory, + PruneReasonUnknownManifest, + PruneReasonUnknownMediaType, + PruneReasonUnknownContainerUse, +} + +// pruneReasonCategory maps every reason code to the verdict it justifies. +var pruneReasonCategory = map[PruneReason]PruneVerdict{ + PruneReasonEligibleRetention: PruneVerdictEligible, + PruneReasonEligibleUnreferencedBlob: PruneVerdictEligible, + PruneReasonEligibleDanglingRuntimeImage: PruneVerdictEligible, + PruneReasonEligibleReleasedVolume: PruneVerdictEligible, + PruneReasonProtectedLatest: PruneVerdictProtected, + PruneReasonProtectedRetentionWindow: PruneVerdictProtected, + PruneReasonProtectedDesiredRevision: PruneVerdictProtected, + PruneReasonProtectedActiveService: PruneVerdictProtected, + PruneReasonProtectedRecoveryInhibition: PruneVerdictProtected, + PruneReasonProtectedApplyIntent: PruneVerdictProtected, + PruneReasonProtectedOperation: PruneVerdictProtected, + PruneReasonProtectedOwnership: PruneVerdictProtected, + PruneReasonProtectedContainerUse: PruneVerdictProtected, + PruneReasonProtectedPendingUpload: PruneVerdictProtected, + PruneReasonProtectedSharedContent: PruneVerdictProtected, + PruneReasonProtectedPinned: PruneVerdictProtected, + PruneReasonProtectedServiceManaged: PruneVerdictProtected, + PruneReasonProtectedManagedOnly: PruneVerdictProtected, + PruneReasonProtectedTagged: PruneVerdictProtected, + PruneReasonProtectedUnmanaged: PruneVerdictProtected, + PruneReasonUnknownIdentity: PruneVerdictUnknown, + PruneReasonUnknownProvenance: PruneVerdictUnknown, + PruneReasonUnknownInventory: PruneVerdictUnknown, + PruneReasonUnknownManifest: PruneVerdictUnknown, + PruneReasonUnknownMediaType: PruneVerdictUnknown, + PruneReasonUnknownContainerUse: PruneVerdictUnknown, +} + +// Valid reports whether r is a defined reason code. +func (r PruneReason) Valid() bool { + _, ok := pruneReasonCategory[r] + return ok +} + +// Category returns the verdict r justifies. An undefined reason returns +// the empty verdict, which fails validation. +func (r PruneReason) Category() PruneVerdict { + return pruneReasonCategory[r] +} + +// PruneReasons returns every defined reason code in canonical order. +func PruneReasons() []PruneReason { + out := make([]PruneReason, len(pruneReasons)) + copy(out, pruneReasons) + return out +} + +// CanonicalReasons returns reasons deduplicated and ordered by the +// frozen canonical order, so verdict output is deterministic. +func CanonicalReasons(reasons ...PruneReason) []PruneReason { + if len(reasons) == 0 { + return nil + } + seen := make(map[PruneReason]struct{}, len(reasons)) + for _, reason := range reasons { + seen[reason] = struct{}{} + } + out := make([]PruneReason, 0, len(seen)) + for _, reason := range pruneReasons { + if _, ok := seen[reason]; ok { + out = append(out, reason) + } + } + // Undefined reasons still surface, ordered after the known ones, so + // a programming error is visible instead of silently dropped. + var unknown []PruneReason + for reason := range seen { + if !reason.Valid() { + unknown = append(unknown, reason) + } + } + sort.Slice(unknown, func(i, j int) bool { return unknown[i] < unknown[j] }) + return append(out, unknown...) +} + +// PruneResourceKind classifies the namespace a candidate identity lives in. +type PruneResourceKind string + +const ( + // PruneResourceRegistryTag is a repository tag in registry storage. + PruneResourceRegistryTag PruneResourceKind = "registry-tag" + // PruneResourceOCIBlob is digest-addressed registry content. + PruneResourceOCIBlob PruneResourceKind = "oci-blob" + // PruneResourceRuntimeImage is a runtime image identified by image ID. + PruneResourceRuntimeImage PruneResourceKind = "runtime-image" + // PruneResourceVolume is a runtime volume identified by name. + PruneResourceVolume PruneResourceKind = "volume" +) + +// RegistryTagRef identifies one registry tag by repository and tag. +type RegistryTagRef struct { + Repository string + Tag string +} + +// NewRegistryTagRef validates and returns a registry tag identity. +func NewRegistryTagRef(repository, tag string) (RegistryTagRef, error) { + ref := RegistryTagRef{Repository: strings.TrimSpace(repository), Tag: strings.TrimSpace(tag)} + if !ref.Valid() { + return RegistryTagRef{}, fmt.Errorf("invalid registry tag reference %q/%q", repository, tag) + } + return ref, nil +} + +// Valid reports whether the reference identifies one exact registry tag. +func (r RegistryTagRef) Valid() bool { + return r.Repository != "" && r.Tag != "" && !strings.ContainsAny(r.Repository, " \t\n") && !strings.ContainsAny(r.Tag, " \t\n") +} + +// String renders repository:tag for humans. The form is ambiguous for +// repositories that themselves contain a colon, so it must never be +// parsed back: Key is the canonical identity. +func (r RegistryTagRef) String() string { + if !r.Valid() { + return "" + } + return r.Repository + ":" + r.Tag +} + +// Key is the deterministic, unambiguous map key for this reference. +func (r RegistryTagRef) Key() string { + if !r.Valid() { + return "" + } + return r.Repository + "\x00" + r.Tag +} + +// OCIRef identifies digest-addressed registry content. Scope is the +// repository the content belongs to; an empty scope addresses the +// global blob store. +type OCIRef struct { + Scope string + Digest string +} + +// NewOCIRef validates and returns a digest identity. +func NewOCIRef(scope, digest string) (OCIRef, error) { + ref := OCIRef{Scope: strings.TrimSpace(scope), Digest: strings.TrimSpace(digest)} + if !ref.Valid() { + return OCIRef{}, fmt.Errorf("invalid OCI reference %q@%q", scope, digest) + } + return ref, nil +} + +// Valid reports whether the reference carries a plausible content digest. +func (r OCIRef) Valid() bool { + return ValidOCIDigest(r.Digest) && !strings.ContainsAny(r.Scope, " \t\n") +} + +// String renders scope@digest, or the bare digest for global content. +func (r OCIRef) String() string { + if r.Digest == "" { + return "" + } + if r.Scope == "" { + return r.Digest + } + return r.Scope + "@" + r.Digest +} + +// Key is the deterministic map key for this reference. +func (r OCIRef) Key() string { + if !r.Valid() { + return "" + } + return r.Scope + "\x00" + r.Digest +} + +// RuntimeImageRef identifies one runtime image by its exact image ID. +type RuntimeImageRef struct { + ID string +} + +// NewRuntimeImageRef validates and returns a runtime image identity. +func NewRuntimeImageRef(id string) (RuntimeImageRef, error) { + ref := RuntimeImageRef{ID: strings.TrimSpace(id)} + if !ref.Valid() { + return RuntimeImageRef{}, fmt.Errorf("invalid runtime image ID %q", id) + } + return ref, nil +} + +// Valid reports whether the image ID is usable as a deletion target. +// Runtime placeholders such as never are. +func (r RuntimeImageRef) Valid() bool { + if r.ID == "" || strings.ContainsAny(r.ID, " \t\n") { + return false + } + return !strings.HasPrefix(r.ID, "") +} + +// String renders the image ID. +func (r RuntimeImageRef) String() string { return r.ID } + +// RuntimeVolumeRef identifies one runtime volume by name. +type RuntimeVolumeRef struct { + Name string +} + +// NewRuntimeVolumeRef validates and returns a runtime volume identity. +func NewRuntimeVolumeRef(name string) (RuntimeVolumeRef, error) { + ref := RuntimeVolumeRef{Name: strings.TrimSpace(name)} + if !ref.Valid() { + return RuntimeVolumeRef{}, fmt.Errorf("invalid runtime volume name %q", name) + } + return ref, nil +} + +// Valid reports whether the volume name is usable as a deletion target. +func (r RuntimeVolumeRef) Valid() bool { + return r.Name != "" && !strings.ContainsAny(r.Name, " \t\n") +} + +// String renders the volume name. +func (r RuntimeVolumeRef) String() string { return r.Name } + +// ValidOCIDigest reports whether digest has the OCI +// algorithm:encoded form with a hex-encoded value. +func ValidOCIDigest(digest string) bool { + algorithm, encoded, ok := strings.Cut(digest, ":") + if !ok || algorithm == "" || len(encoded) < 32 { + return false + } + for _, r := range algorithm { + if !isDigestAlgorithmRune(r) { + return false + } + } + for _, r := range encoded { + if !isHexRune(r) { + return false + } + } + return true +} + +func isDigestAlgorithmRune(r rune) bool { + switch { + case r >= 'a' && r <= 'z', r >= '0' && r <= '9': + return true + case r == '.', r == '-', r == '_', r == '+': + return true + default: + return false + } +} + +func isHexRune(r rune) bool { + switch { + case r >= '0' && r <= '9', r >= 'a' && r <= 'f', r >= 'A' && r <= 'F': + return true + default: + return false + } +} + +// ProtectionRootKind classifies one durable fact that retains content. +type ProtectionRootKind string + +const ( + // ProtectionLatest is the latest tag of a repository. + ProtectionLatest ProtectionRootKind = "latest" + // ProtectionRetentionWindow is the configured latest + keep_last set. + ProtectionRetentionWindow ProtectionRootKind = "retention-window" + // ProtectionDesiredRevision is a desired revision's image reference. + ProtectionDesiredRevision ProtectionRootKind = "desired-revision" + // ProtectionActiveService is an ACTIVE service's image reference, + // including stopped and partially converged services. + ProtectionActiveService ProtectionRootKind = "active-service" + // ProtectionRecoveryInhibition is a generation-scoped inhibition. + ProtectionRecoveryInhibition ProtectionRootKind = "recovery-inhibition" + // ProtectionApplyIntent is a staged or committed apply intent. + ProtectionApplyIntent ProtectionRootKind = "apply-intent" + // ProtectionOperation is an unfinished operation journal record. + ProtectionOperation ProtectionRootKind = "operation" + // ProtectionOwnership is a durable ownership claim on a resource. + ProtectionOwnership ProtectionRootKind = "ownership" + // ProtectionContainerUse is runtime container usage. + ProtectionContainerUse ProtectionRootKind = "container-use" + // ProtectionPendingUpload is a recent registry upload. + ProtectionPendingUpload ProtectionRootKind = "pending-upload" + // ProtectionSharedContent is transitive OCI closure content. + ProtectionSharedContent ProtectionRootKind = "shared-content" + // ProtectionPinned is explicitly pinned content. + ProtectionPinned ProtectionRootKind = "pinned" + // ProtectionServiceManaged is standalone service ownership. + ProtectionServiceManaged ProtectionRootKind = "service-managed" +) + +// ProtectionRoot is one durable fact that must retain content. +type ProtectionRoot struct { + Kind ProtectionRootKind + // Ref is the canonical identity the root protects in the namespace + // of its resource kind: a RegistryTagRef.Key, an OCIRef digest, a + // runtime image ID, or a runtime volume name. + Ref string + // Repository qualifies a digest Ref with the repository whose + // manifest names it, so prune can traverse the full OCI closure + // (config, layers, child manifests, subject) instead of protecting + // only the digest itself. Empty when not repository-scoped. + Repository string + // Owner names the app, service, or container carrying the root when + // known; it is informational and never used for matching. + Owner string +} + +// Valid reports whether the root names a kind and a non-empty identity. +func (r ProtectionRoot) Valid() bool { + return r.Kind != "" && r.Ref != "" +} + +// ReasonForRootKind maps a protection root kind to the verdict reason it +// justifies. An unmapped kind protects nothing and is reported as absent. +func ReasonForRootKind(kind ProtectionRootKind) (PruneReason, bool) { + reason, ok := protectionRootReason[kind] + return reason, ok +} + +var protectionRootReason = map[ProtectionRootKind]PruneReason{ + ProtectionLatest: PruneReasonProtectedLatest, + ProtectionRetentionWindow: PruneReasonProtectedRetentionWindow, + ProtectionDesiredRevision: PruneReasonProtectedDesiredRevision, + ProtectionActiveService: PruneReasonProtectedActiveService, + ProtectionRecoveryInhibition: PruneReasonProtectedRecoveryInhibition, + ProtectionApplyIntent: PruneReasonProtectedApplyIntent, + ProtectionOperation: PruneReasonProtectedOperation, + ProtectionOwnership: PruneReasonProtectedOwnership, + ProtectionContainerUse: PruneReasonProtectedContainerUse, + ProtectionPendingUpload: PruneReasonProtectedPendingUpload, + ProtectionSharedContent: PruneReasonProtectedSharedContent, + ProtectionPinned: PruneReasonProtectedPinned, + ProtectionServiceManaged: PruneReasonProtectedServiceManaged, +} + +// VolumeClaimState is the durable lifecycle state of one app-owned volume. +type VolumeClaimState string + +const ( + // VolumeClaimAttached means a live app generation still uses it. + VolumeClaimAttached VolumeClaimState = "attached" + // VolumeClaimRetained means the app was removed and the volume is + // retained without an owner: never implicitly deletable. + VolumeClaimRetained VolumeClaimState = "retained" + // VolumeClaimReleased means the owning app explicitly released the + // volume; only this state is ever eligible for pruning. + VolumeClaimReleased VolumeClaimState = "released" +) + +// Valid reports whether s is a defined claim state. +func (s VolumeClaimState) Valid() bool { + switch s { + case VolumeClaimAttached, VolumeClaimRetained, VolumeClaimReleased: + return true + default: + return false + } +} + +// Protects reports whether the claim state forbids deletion. Unknown or +// empty states protect by default. +func (s VolumeClaimState) Protects() bool { return s != VolumeClaimReleased } + +// ImageClaim is one durable app ownership claim on a runtime image. The +// claim is matched by exact identity (the app's pinned reference and/or +// manifest digest); image labels are never proof on their own. +type ImageClaim struct { + // Reference is the app-pinned image reference (registry/repo:tag). + Reference string + // Digest is the pinned manifest digest, when known. + Digest string + // App is the owning app name, AppID its stable internal UUID. + App string + AppID string + Service string + State VolumeClaimState +} + +// Valid reports whether the claim is complete enough to act on. +func (c ImageClaim) Valid() bool { + return (c.Reference != "" || c.Digest != "") && c.App != "" && c.State.Valid() +} + +// Identities returns every identity the claim may be matched against, +// including the registry tag key of a tagged reference. +func (c ImageClaim) Identities() []string { + var out []string + if c.Reference != "" { + out = append(out, c.Reference) + if tagRef, ok := ParseRegistryTagRef(c.Reference); ok { + out = append(out, tagRef.Key()) + } + } + if c.Digest != "" { + out = append(out, c.Digest) + } + return out +} + +// VolumeClaim is one durable app ownership claim on a runtime volume. +type VolumeClaim struct { + // Name is the exact runtime volume name. + Name string + // App is the owning app name, AppID its stable internal UUID. They + // are informational for matching and must never be inferred from + // the volume name. + App string + AppID string + Service string + State VolumeClaimState +} + +// Valid reports whether the claim is complete enough to act on. +func (c VolumeClaim) Valid() bool { + return c.Name != "" && c.App != "" && c.State.Valid() +} + +// MatchesLabels reports whether the durable claim agrees with the +// runtime labels of its volume. Both sides must be present: a claim is +// never confirmed by absent labels, and labels never override a +// contradictory or missing durable record. +func (c VolumeClaim) MatchesLabels(labels map[string]string) bool { + if len(labels) == 0 || !c.Valid() { + return false + } + if labels[LabelManaged] != "true" { + return false + } + claimed := labels[LabelApp] + if claimed == "" { + return false + } + if claimed != c.App && (c.AppID == "" || claimed != c.AppID) { + return false + } + // A stamped incarnation UUID must agree with the durable record. An + // absent stamped ID is not a match either: without it the claim + // cannot prove which incarnation owns the volume, so a name-only + // match must never authorize deletion. + stampedID := labels[LabelAppID] + if stampedID == "" || c.AppID == "" || stampedID != c.AppID { + return false + } + if service := labels[LabelAppService]; service != "" && c.Service != "" && service != c.Service { + return false + } + return true +} + +// InventorySource identifies one fact source feeding prune planning. +type InventorySource string + +const ( + // InventorySourceAppState is desired/active/intent state. + InventorySourceAppState InventorySource = "app-state" + // InventorySourceOwnership is durable ownership records. + InventorySourceOwnership InventorySource = "ownership" + // InventorySourceOperation is the operation journal. + InventorySourceOperation InventorySource = "operation" + // InventorySourceRegistry is registry repositories and tags. + InventorySourceRegistry InventorySource = "registry" + // InventorySourceManifests is manifest content and closure. + InventorySourceManifests InventorySource = "manifests" + // InventorySourceRuntimeContainers is runtime container usage. + InventorySourceRuntimeContainers InventorySource = "runtime-containers" + // InventorySourceRuntimeImages is runtime image inventory. + InventorySourceRuntimeImages InventorySource = "runtime-images" + // InventorySourceRuntimeVolumes is runtime volume inventory. + InventorySourceRuntimeVolumes InventorySource = "runtime-volumes" +) + +// InventoryGap records one fact that could not be established. A gap +// never blocks an operation by itself: it makes the dependent +// candidates unknown. +type InventoryGap struct { + Source InventorySource + Reason PruneReason + Detail string +} + +// Valid reports whether the gap carries a source and an unknown reason. +func (g InventoryGap) Valid() bool { + return g.Source != "" && g.Reason.Valid() && g.Reason.Category() == PruneVerdictUnknown +} + +// PruneProtectionSnapshot is one coherent read of every durable fact +// that protects prune candidates. Build it in a single read transaction +// so no candidate is judged against facts from a different moment. +// The zero value is a complete snapshot that protects nothing. +type PruneProtectionSnapshot struct { + // Roots is every protection root. Refs are canonical identities. + Roots []ProtectionRoot + // VolumeClaims is every durable app volume claim, including + // released ones. + VolumeClaims []VolumeClaim + // ImageClaims is every durable app image claim, including released + // ones. Runtime image prune requires a matching released claim; + // image labels alone never authorize deletion. + ImageClaims []ImageClaim + // Gaps records every fact that could not be read completely. + Gaps []InventoryGap +} + +// Complete reports whether the snapshot was read without gaps. +func (s *PruneProtectionSnapshot) Complete() bool { + if s == nil { + return false + } + for _, gap := range s.Gaps { + if gap.Valid() { + return false + } + } + return true +} + +// Protects reports whether any root of the given kind claims ref. +// Canonical identity strings are matched exactly: a non-canonical or +// empty ref never matches, so an unparseable candidate stays unknown +// rather than silently eligible. +func (s *PruneProtectionSnapshot) Protects(kind ProtectionRootKind, ref string) bool { + if s == nil || ref == "" { + return false + } + for _, root := range s.Roots { + if root.Kind == kind && root.Ref == ref { + return true + } + } + return false +} + +// ProtectsDigest reports whether any root of any kind claims the digest. +// Digest identities are shared across registry and runtime namespaces, +// so protection is namespace-independent. +func (s *PruneProtectionSnapshot) ProtectsDigest(digest string) bool { + if s == nil || !ValidOCIDigest(digest) { + return false + } + for _, root := range s.Roots { + if root.Ref == digest || strings.HasSuffix(root.Ref, "@"+digest) { + return true + } + } + return false +} + +// ProtectsRuntimeImage reports whether any root claims the image ID or +// any of its repo digests. +func (s *PruneProtectionSnapshot) ProtectsRuntimeImage(imageID string, repoDigests []string) bool { + if s == nil { + return false + } + if imageID != "" && s.Protects(ProtectionContainerUse, imageID) { + return true + } + if s.ProtectsAnyRef(imageID) { + return true + } + for _, digest := range repoDigests { + if s.ProtectsAnyRef(digest) { + return true + } + } + return false +} + +// ProtectsAnyRef reports whether any root of any kind claims ref. +func (s *PruneProtectionSnapshot) ProtectsAnyRef(ref string) bool { + if s == nil || ref == "" { + return false + } + for _, root := range s.Roots { + if root.Ref == ref || strings.HasSuffix(root.Ref, "@"+ref) { + return true + } + } + return false +} + +// ImageClaimsFor returns every claim whose identities include any of the +// given identities, in record order. +func (s *PruneProtectionSnapshot) ImageClaimsFor(identities ...string) []ImageClaim { + if s == nil || len(identities) == 0 { + return nil + } + wanted := make(map[string]struct{}, len(identities)) + for _, identity := range identities { + if identity != "" { + wanted[identity] = struct{}{} + } + } + var out []ImageClaim + for _, claim := range s.ImageClaims { + for _, identity := range claim.Identities() { + if _, ok := wanted[identity]; ok { + out = append(out, claim) + break + } + } + } + return out +} + +// ImageClaimFor returns the last recorded claim matching any identity. +func (s *PruneProtectionSnapshot) ImageClaimFor(identities ...string) (ImageClaim, bool) { + claims := s.ImageClaimsFor(identities...) + if len(claims) == 0 { + return ImageClaim{}, false + } + return claims[len(claims)-1], true +} + +// ImageProtects reports whether any matching claim forbids deletion. +func (s *PruneProtectionSnapshot) ImageProtects(identities ...string) bool { + for _, claim := range s.ImageClaimsFor(identities...) { + if claim.State.Protects() { + return true + } + } + return false +} + +// VolumeClaimFor returns the last recorded claim for name. +func (s *PruneProtectionSnapshot) VolumeClaimFor(name string) (VolumeClaim, bool) { + if s == nil || name == "" { + return VolumeClaim{}, false + } + for index := len(s.VolumeClaims) - 1; index >= 0; index-- { + if s.VolumeClaims[index].Name == name { + return s.VolumeClaims[index], true + } + } + return VolumeClaim{}, false +} + +// VolumeClaimsFor returns every claim recorded for name, in record +// order. A volume may be claimed by more than one app incarnation when +// a name is reused, and every claim must be considered. +func (s *PruneProtectionSnapshot) VolumeClaimsFor(name string) []VolumeClaim { + if s == nil || name == "" { + return nil + } + var claims []VolumeClaim + for _, claim := range s.VolumeClaims { + if claim.Name == name { + claims = append(claims, claim) + } + } + return claims +} + +// VolumeProtects reports whether any durable claim on name forbids +// deletion. Any protecting claim wins: a new incarnation attaching a +// volume that an older incarnation released must still protect it. +func (s *PruneProtectionSnapshot) VolumeProtects(name string) bool { + for _, claim := range s.VolumeClaimsFor(name) { + if claim.State.Protects() { + return true + } + } + return false +} + +// ProtectsTagRef reports whether any root claims the tag identity. +func (s *PruneProtectionSnapshot) ProtectsTagRef(ref RegistryTagRef) bool { + if s == nil || !ref.Valid() { + return false + } + return s.ProtectsAnyRef(ref.Key()) || s.ProtectsAnyRef(ref.String()) +} + +// RootReasons returns the canonical, deduplicated verdict reasons of +// every root claiming any of refs. An unmapped root kind contributes +// nothing: it retains content for consumers that match it directly but +// cannot justify a verdict reason on its own. +func (s *PruneProtectionSnapshot) RootReasons(refs ...string) []PruneReason { + if s == nil { + return nil + } + set := make(map[string]struct{}, len(refs)) + for _, ref := range refs { + if ref != "" { + set[ref] = struct{}{} + } + } + if len(set) == 0 { + return nil + } + var reasons []PruneReason + for _, root := range s.Roots { + if _, ok := set[root.Ref]; !ok { + continue + } + reason, mapped := protectionRootReason[root.Kind] + if !mapped { + continue + } + reasons = append(reasons, reason) + } + return CanonicalReasons(reasons...) +} + +// HasGapFor reports whether any recorded gap came from source. A nil +// snapshot reports a gap: it proves nothing. +func (s *PruneProtectionSnapshot) HasGapFor(source InventorySource) bool { + if s == nil { + return true + } + for _, gap := range s.Gaps { + if gap.Source == source && gap.Valid() { + return true + } + } + return false +} + +// RefsOfKind returns every distinct root identity of one kind, in +// first-seen order. +func (s *PruneProtectionSnapshot) RefsOfKind(kind ProtectionRootKind) []string { + if s == nil { + return nil + } + seen := make(map[string]struct{}) + var refs []string + for _, root := range s.Roots { + if root.Kind != kind || root.Ref == "" { + continue + } + if _, dup := seen[root.Ref]; dup { + continue + } + seen[root.Ref] = struct{}{} + refs = append(refs, root.Ref) + } + return refs +} + +// WithRoots returns a copy of the snapshot with extra roots appended. +// It is how one planning stage layers facts it derived itself (for +// example the OCI closure of already-protected manifests) onto the +// durable snapshot without mutating it. +func (s *PruneProtectionSnapshot) WithRoots(roots ...ProtectionRoot) *PruneProtectionSnapshot { + if len(roots) == 0 { + return s + } + clone := &PruneProtectionSnapshot{Roots: make([]ProtectionRoot, 0, len(roots))} + if s != nil { + clone.Roots = make([]ProtectionRoot, 0, len(s.Roots)+len(roots)) + clone.Roots = append(clone.Roots, s.Roots...) + clone.VolumeClaims = append([]VolumeClaim(nil), s.VolumeClaims...) + clone.ImageClaims = append([]ImageClaim(nil), s.ImageClaims...) + clone.Gaps = append([]InventoryGap(nil), s.Gaps...) + } + for _, root := range roots { + if root.Valid() { + clone.Roots = append(clone.Roots, root) + } + } + return clone +} + +// ImageRefRoots returns the protection roots that retain one image +// reference. A tag reference yields a tag-keyed root; a +// digest-qualified reference yields both the tag root (when it has a +// tag) and a digest root, so a candidate matching either identity stays +// protected. An unparseable reference yields no root: it protects +// nothing here and is left to the caller to fail closed on. +func ImageRefRoots(kind ProtectionRootKind, imageRef, owner string) []ProtectionRoot { + ref := strings.TrimSpace(imageRef) + if ref == "" { + return nil + } + + var roots []ProtectionRoot + name, digest := splitImageDigest(ref) + if digest != "" { + roots = append(roots, ProtectionRoot{Kind: kind, Ref: digest, Repository: ImageRefRepository(name), Owner: owner}) + } + if name != "" { + if repository, tag, ok := splitImageTag(name); ok { + tagRef := RegistryTagRef{Repository: repository, Tag: tag} + if tagRef.Valid() { + roots = append(roots, ProtectionRoot{Kind: kind, Ref: tagRef.Key(), Owner: owner}) + } + } else if ValidOCIDigest(name) { + roots = append(roots, ProtectionRoot{Kind: kind, Ref: name, Owner: owner}) + } + } + return roots +} + +// ImageRefRepository returns the registry-relative repository of an image +// reference: registry host, tag, and digest are stripped. It reports "" +// when no repository can be derived. +func ImageRefRepository(ref string) string { + name, _ := splitImageDigest(ref) + name = strings.TrimSpace(name) + if name == "" { + return "" + } + if repository, _, ok := splitImageTag(name); ok { + name = repository + } + // Strip a registry host following Docker's rule: the first path + // component is a registry only when it contains a dot or colon, or + // is localhost. + if first, rest, ok := strings.Cut(name, "/"); ok { + if strings.ContainsAny(first, ".:") || strings.EqualFold(first, "localhost") { + name = rest + } + } + return name +} + +// splitImageDigest splits "name@digest" into its parts. A reference +// without an @ keeps its whole value as the name. +func splitImageDigest(ref string) (name, digest string) { + at := strings.Index(ref, "@") + if at < 0 { + return ref, "" + } + digest = ref[at+1:] + if !ValidOCIDigest(digest) { + return ref, "" + } + return ref[:at], digest +} + +// splitImageTag splits "registry/repo:tag" into repository and tag. A +// reference without a tag suffix reports ok=false. +func splitImageTag(ref string) (repository, tag string, ok bool) { + colon := strings.LastIndex(ref, ":") + slash := strings.LastIndex(ref, "/") + if colon <= slash || colon >= len(ref)-1 { + return "", "", false + } + return ref[:colon], ref[colon+1:], true +} + +// ParseRegistryTagRef parses a runtime repository tag such as +// "registry.example/shop:1" into a registry tag identity. A reference +// carrying no usable tag reports ok=false. +func ParseRegistryTagRef(repoTag string) (RegistryTagRef, bool) { + repo, tag, ok := splitImageTag(strings.TrimSpace(repoTag)) + if !ok { + return RegistryTagRef{}, false + } + ref := RegistryTagRef{Repository: repo, Tag: tag} + if !ref.Valid() { + return RegistryTagRef{}, false + } + return ref, true +} + +// RegistryTagVerdict is the decision for one registry tag. +type RegistryTagVerdict struct { + Ref RegistryTagRef + // Digest is the manifest digest the tag resolved to, when known. + Digest string + Verdict PruneVerdict + Reasons []PruneReason +} + +// OCIBlobVerdict is the decision for digest-addressed registry content. +type OCIBlobVerdict struct { + Ref OCIRef + Verdict PruneVerdict + Reasons []PruneReason +} + +// RuntimeImageVerdict is the decision for one runtime image. +type RuntimeImageVerdict struct { + Ref RuntimeImageRef + Verdict PruneVerdict + Reasons []PruneReason +} + +// VolumeVerdict is the decision for one runtime volume. +type VolumeVerdict struct { + Ref RuntimeVolumeRef + Verdict PruneVerdict + Reasons []PruneReason +} + +// PrunePlan is the complete, validated decision set of one prune run. +// Execution deletes exactly the eligible entries: no stage may delete +// before the whole plan is computed and validated. +type PrunePlan struct { + Tags []RegistryTagVerdict + Blobs []OCIBlobVerdict + Images []RuntimeImageVerdict + Volumes []VolumeVerdict + Gaps []InventoryGap +} + +// EligibleTags returns the registry tags safe to delete, in plan order. +func (p *PrunePlan) EligibleTags() []RegistryTagRef { + out := make([]RegistryTagRef, 0, len(p.Tags)) + for _, verdict := range p.Tags { + if verdict.Verdict == PruneVerdictEligible { + out = append(out, verdict.Ref) + } + } + return out +} + +// EligibleBlobs returns the OCI content safe to delete, in plan order. +func (p *PrunePlan) EligibleBlobs() []OCIRef { + out := make([]OCIRef, 0, len(p.Blobs)) + for _, verdict := range p.Blobs { + if verdict.Verdict == PruneVerdictEligible { + out = append(out, verdict.Ref) + } + } + return out +} + +// EligibleImages returns the runtime images safe to delete, in plan order. +func (p *PrunePlan) EligibleImages() []RuntimeImageRef { + out := make([]RuntimeImageRef, 0, len(p.Images)) + for _, verdict := range p.Images { + if verdict.Verdict == PruneVerdictEligible { + out = append(out, verdict.Ref) + } + } + return out +} + +// EligibleVolumes returns the runtime volumes safe to delete, in plan order. +func (p *PrunePlan) EligibleVolumes() []RuntimeVolumeRef { + out := make([]RuntimeVolumeRef, 0, len(p.Volumes)) + for _, verdict := range p.Volumes { + if verdict.Verdict == PruneVerdictEligible { + out = append(out, verdict.Ref) + } + } + return out +} + +// CountByVerdict counts entries of one verdict across every resource kind. +func (p *PrunePlan) CountByVerdict(verdict PruneVerdict) int { + count := 0 + for _, item := range p.Tags { + if item.Verdict == verdict { + count++ + } + } + for _, item := range p.Blobs { + if item.Verdict == verdict { + count++ + } + } + for _, item := range p.Images { + if item.Verdict == verdict { + count++ + } + } + for _, item := range p.Volumes { + if item.Verdict == verdict { + count++ + } + } + return count +} + +// Validate returns the first structural inconsistency of the +// plan. It never re-derives safety: it proves the plan is well formed +// enough to execute, so a malformed plan can never delete anything. +func (p *PrunePlan) Validate() error { + for _, item := range p.Tags { + if !item.Ref.Valid() { + return fmt.Errorf("registry tag verdict has invalid identity %q", item.Ref.String()) + } + if err := validateVerdictReasons(item.Verdict, item.Reasons); err != nil { + return fmt.Errorf("registry tag %q: %w", item.Ref.String(), err) + } + } + for _, item := range p.Blobs { + if err := validateOCIVerdict(item.Ref, item.Verdict, item.Reasons); err != nil { + return err + } + } + for _, item := range p.Images { + if err := validateRuntimeImageVerdict(item.Ref, item.Verdict, item.Reasons); err != nil { + return err + } + } + for _, item := range p.Volumes { + if err := validateVolumeVerdict(item.Ref, item.Verdict, item.Reasons); err != nil { + return err + } + } + for _, gap := range p.Gaps { + if !gap.Valid() { + return fmt.Errorf("inventory gap has invalid source or reason %q/%q", gap.Source, gap.Reason) + } + } + return nil +} + +// validateOCIVerdict proves one blob verdict is well formed. An eligible +// entry must name a valid digest, so an executable plan can never carry +// a deletion target the adapter would reject. +func validateOCIVerdict(ref OCIRef, verdict PruneVerdict, reasons []PruneReason) error { + if verdict == PruneVerdictEligible && !ref.Valid() { + return fmt.Errorf("OCI blob verdict has invalid identity %q", ref.String()) + } + if err := validateVerdictReasons(verdict, reasons); err != nil { + return fmt.Errorf("OCI blob %q: %w", ref.String(), err) + } + return nil +} + +// validateRuntimeImageVerdict proves one runtime image verdict is well formed. +func validateRuntimeImageVerdict(ref RuntimeImageRef, verdict PruneVerdict, reasons []PruneReason) error { + if verdict == PruneVerdictEligible && !ref.Valid() { + return fmt.Errorf("runtime image verdict has invalid identity %q", ref.String()) + } + if err := validateVerdictReasons(verdict, reasons); err != nil { + return fmt.Errorf("runtime image %q: %w", ref.String(), err) + } + return nil +} + +// validateVolumeVerdict proves one volume verdict is well formed. +func validateVolumeVerdict(ref RuntimeVolumeRef, verdict PruneVerdict, reasons []PruneReason) error { + if verdict == PruneVerdictEligible && !ref.Valid() { + return fmt.Errorf("volume verdict has invalid identity %q", ref.String()) + } + if err := validateVerdictReasons(verdict, reasons); err != nil { + return fmt.Errorf("volume %q: %w", ref.String(), err) + } + return nil +} + +// Candidates returns every planned candidate with its verdict, in a +// deterministic order: tags, blobs, runtime images, then volumes. +func (p *PrunePlan) Candidates() []PruneCandidateReport { + total := len(p.Tags) + len(p.Blobs) + len(p.Images) + len(p.Volumes) + out := make([]PruneCandidateReport, 0, total) + for _, item := range p.Tags { + out = append(out, PruneCandidateReport{ + Kind: PruneResourceRegistryTag, Ref: item.Ref.String(), Verdict: item.Verdict, Reasons: item.Reasons, + }) + } + for _, item := range p.Blobs { + out = append(out, PruneCandidateReport{ + Kind: PruneResourceOCIBlob, Ref: item.Ref.String(), Verdict: item.Verdict, Reasons: item.Reasons, + }) + } + for _, item := range p.Images { + out = append(out, PruneCandidateReport{ + Kind: PruneResourceRuntimeImage, Ref: item.Ref.String(), Verdict: item.Verdict, Reasons: item.Reasons, + }) + } + for _, item := range p.Volumes { + out = append(out, PruneCandidateReport{ + Kind: PruneResourceVolume, Ref: item.Ref.String(), Verdict: item.Verdict, Reasons: item.Reasons, + }) + } + return out +} + +// PruneCandidateReport is one candidate's verdict in a prune report. +type PruneCandidateReport struct { + Kind PruneResourceKind + Ref string + Verdict PruneVerdict + Reasons []PruneReason +} + +// PruneFailure records one exact deletion that failed. A failure never +// aborts the run and never aborts the other candidates. +type PruneFailure struct { + Kind PruneResourceKind + Ref string + Err string +} + +// PruneReport is the operator-facing result of one prune run. Dry-run +// and execution share this shape; Applied distinguishes them. +type PruneReport struct { + // Applied is false for a dry run, where nothing was deleted. + Applied bool + // Candidates lists every planned candidate with its verdict, so an + // operator can see why each resource was kept or skipped. + Candidates []PruneCandidateReport + // Gaps lists every inventory that could not be read in full. + Gaps []InventoryGap + // Deleted lists exactly the identities that were removed. + Deleted []PruneCandidateReport + // Failures lists exact deletions that failed without aborting the run. + Failures []PruneFailure + // ReclaimedBytes and ReclaimedKnown are set only when the adapter + // can attribute exact bytes to the deletions it performed. + ReclaimedBytes int64 + ReclaimedKnown bool +} + +// CountByVerdict counts every planned candidate with one verdict. +func (r *PruneReport) CountByVerdict(verdict PruneVerdict) int { + count := 0 + for _, candidate := range r.Candidates { + if candidate.Verdict == verdict { + count++ + } + } + return count +} + +// CandidatesOfKind returns the planned candidates of one resource kind +// with one verdict, in report order. +func (r *PruneReport) CandidatesOfKind(kind PruneResourceKind, verdict PruneVerdict) []PruneCandidateReport { + var out []PruneCandidateReport + for _, candidate := range r.Candidates { + if candidate.Kind == kind && candidate.Verdict == verdict { + out = append(out, candidate) + } + } + return out +} + +// NewPruneReport builds the shared report shape from a validated plan. +// applied reports whether the caller went on to execute it. +func NewPruneReport(plan *PrunePlan, applied bool) *PruneReport { + report := &PruneReport{Applied: applied} + if plan == nil { + return report + } + report.Candidates = plan.Candidates() + report.Gaps = plan.Gaps + return report +} + +// validateVerdictReasons proves a verdict is well formed: it is defined, +// it carries at least one reason, and every reason justifies it. +func validateVerdictReasons(verdict PruneVerdict, reasons []PruneReason) error { + if !verdict.Valid() { + return fmt.Errorf("undefined verdict %q", verdict) + } + if len(reasons) == 0 { + return fmt.Errorf("verdict %q carries no reason", verdict) + } + for _, reason := range reasons { + if !reason.Valid() { + return fmt.Errorf("undefined reason %q", reason) + } + if reason.Category() != verdict { + return fmt.Errorf("reason %q does not justify verdict %q", reason, verdict) + } + } + return nil +} + +// RuntimeImage is one runtime image as the runtime reports it. +type RuntimeImage struct { + // ID is the exact runtime image identifier. + ID string + // RepoTags are the repository tags pointing at this image. + RepoTags []string + // RepoDigests are the manifest digests the runtime knows for it. + RepoDigests []string + Labels map[string]string + // Size is reported in bytes. + Size int64 + Created time.Time +} + +// RuntimeContainerUse is one container's image usage. +type RuntimeContainerUse struct { + ContainerID string + // ImageID is the resolved image identity; empty when unknown. + ImageID string + // ImageRef is the reference the container was created from. + ImageRef string + Running bool + // ImageUnknown marks a container whose image could not be resolved. + // It protects every image candidate whose safety depends on it. + ImageUnknown bool +} + +// RuntimeInventory is one coherent read of runtime resources. It is +// what the planner consumes; adapters build it without inspecting +// each other. +type RuntimeInventory struct { + Images []RuntimeImage + Containers []RuntimeContainerUse + // Volumes are the runtime's named volumes with labels and usage. + Volumes []*VolumeInfo + Gaps []InventoryGap +} + +// Complete reports whether the inventory was read without gaps. +func (i *RuntimeInventory) Complete() bool { + if i == nil { + return false + } + for _, gap := range i.Gaps { + if gap.Valid() { + return false + } + } + return true +} diff --git a/internal/domain/prune_test.go b/internal/domain/prune_test.go new file mode 100644 index 000000000..a474cedae --- /dev/null +++ b/internal/domain/prune_test.go @@ -0,0 +1,379 @@ +package domain + +import ( + "strings" + "testing" +) + +func TestValidOCIDigest(t *testing.T) { + tests := []struct { + name string + digest string + want bool + }{ + {"sha256", "sha256:" + strings.Repeat("ab", 32), true}, + {"sha512", "sha512:" + strings.Repeat("cd", 64), true}, + {"empty", "", false}, + {"no algorithm", strings.Repeat("ab", 32), false}, + {"empty encoded", "sha256:", false}, + {"short encoded", "sha256:abcd", false}, + {"uppercase algorithm", "SHA256:" + strings.Repeat("ab", 32), false}, + {"non hex", "sha256:" + strings.Repeat("zz", 32), false}, + {"space inside", "sha256:" + strings.Repeat("ab", 31) + "a ", false}, + } + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + if got := ValidOCIDigest(tc.digest); got != tc.want { + t.Fatalf("ValidOCIDigest(%q) = %v, want %v", tc.digest, got, tc.want) + } + }) + } +} + +func TestRegistryTagRefValidation(t *testing.T) { + valid := RegistryTagRef{Repository: "app", Tag: "v1"} + if !valid.Valid() || valid.String() != "app:v1" || valid.Key() == "" { + t.Fatalf("valid registry tag ref not accepted: %+v", valid) + } + + for _, ref := range []RegistryTagRef{ + {}, + {Repository: "app"}, + {Tag: "v1"}, + {Repository: "app", Tag: " "}, + {Repository: "a pp", Tag: "v1"}, + } { + if ref.Valid() { + t.Fatalf("empty or whitespace identity accepted: %+v", ref) + } + if ref.Key() != "" { + t.Fatalf("invalid ref produced a map key: %+v", ref) + } + } + + if _, err := NewRegistryTagRef("", "v1"); err == nil { + t.Fatal("expected constructor to reject an empty repository") + } +} + +func TestIdentityKeysAreUnambiguous(t *testing.T) { + // A repository containing a colon must not collide with the + // repository/tag separator used by the map key. The human display + // form is documented as ambiguous and never parsed back. + first := RegistryTagRef{Repository: "a:b", Tag: "c"} + second := RegistryTagRef{Repository: "a", Tag: "b:c"} + if first.Key() == second.Key() { + t.Fatalf("distinct references collided on key: %q", first.Key()) + } + if first.Key() == "" || second.Key() == "" { + t.Fatalf("valid references must carry a canonical key: %q / %q", first.Key(), second.Key()) + } +} + +func TestRuntimeImageRefRejectsPlaceholders(t *testing.T) { + for _, id := range []string{"", " ", "", ":"} { + if ref := (RuntimeImageRef{ID: id}); ref.Valid() { + t.Fatalf("placeholder image ID %q accepted", id) + } + } + if !(RuntimeImageRef{ID: "sha256:abc"}).Valid() { + t.Fatal("exact image ID rejected") + } +} + +func TestRuntimeVolumeRefValidation(t *testing.T) { + if !(RuntimeVolumeRef{Name: "gordon-app-data"}).Valid() { + t.Fatal("exact volume name rejected") + } + for _, name := range []string{"", " ", "vol ume", "vol\tume"} { + if ref := (RuntimeVolumeRef{Name: name}); ref.Valid() { + t.Fatalf("invalid volume name %q accepted", name) + } + } +} + +func TestPruneReasonCoverage(t *testing.T) { + reasons := PruneReasons() + if len(reasons) == 0 { + t.Fatal("no reason codes defined") + } + seen := make(map[PruneReason]struct{}, len(reasons)) + for _, reason := range reasons { + if _, dup := seen[reason]; dup { + t.Fatalf("reason %q listed twice in canonical order", reason) + } + seen[reason] = struct{}{} + if !reason.Valid() { + t.Fatalf("reason %q has no verdict category", reason) + } + if reason.Category() == "" { + t.Fatalf("reason %q has an empty category", reason) + } + } + if len(seen) != len(pruneReasonCategory) { + t.Fatalf("canonical order covers %d reasons, category map has %d", len(seen), len(pruneReasonCategory)) + } +} + +func TestPruneReasonWireValuesAreStable(t *testing.T) { + // These strings appear in JSON and human output. If one changes, + // this test must change deliberately. + want := map[PruneReason]string{ + PruneReasonEligibleRetention: "eligible-retention-window", + PruneReasonEligibleUnreferencedBlob: "eligible-unreferenced-blob", + PruneReasonEligibleDanglingRuntimeImage: "eligible-dangling-runtime-image", + PruneReasonEligibleReleasedVolume: "eligible-released-volume", + PruneReasonProtectedLatest: "protected-latest", + PruneReasonProtectedRetentionWindow: "protected-retention-window", + PruneReasonProtectedDesiredRevision: "protected-desired-revision", + PruneReasonProtectedActiveService: "protected-active-service", + PruneReasonProtectedRecoveryInhibition: "protected-recovery-inhibition", + PruneReasonProtectedApplyIntent: "protected-apply-intent", + PruneReasonProtectedOperation: "protected-operation", + PruneReasonProtectedOwnership: "protected-ownership", + PruneReasonProtectedContainerUse: "protected-container-use", + PruneReasonProtectedPendingUpload: "protected-pending-upload", + PruneReasonProtectedSharedContent: "protected-shared-content", + PruneReasonProtectedPinned: "protected-pinned", + PruneReasonProtectedServiceManaged: "protected-service-managed", + PruneReasonProtectedManagedOnly: "protected-managed-without-ownership", + PruneReasonProtectedTagged: "protected-tagged-runtime-image", + PruneReasonProtectedUnmanaged: "protected-unmanaged", + PruneReasonUnknownIdentity: "unknown-identity", + PruneReasonUnknownProvenance: "unknown-provenance", + PruneReasonUnknownInventory: "unknown-incomplete-inventory", + PruneReasonUnknownManifest: "unknown-manifest-unreadable", + PruneReasonUnknownMediaType: "unknown-unsupported-media-type", + PruneReasonUnknownContainerUse: "unknown-container-use", + } + for reason, wire := range want { + if string(reason) != wire { + t.Fatalf("reason %q wire value changed to %q", wire, reason) + } + } + if len(want) != len(pruneReasonCategory) { + t.Fatalf("stability table covers %d reasons, %d defined", len(want), len(pruneReasonCategory)) + } +} + +func TestCanonicalReasonsIsDeterministic(t *testing.T) { + input := []PruneReason{ + PruneReasonUnknownInventory, + PruneReasonProtectedActiveService, + PruneReasonProtectedLatest, + PruneReasonProtectedActiveService, + } + got := CanonicalReasons(input...) + want := []PruneReason{ + PruneReasonProtectedLatest, + PruneReasonProtectedActiveService, + PruneReasonUnknownInventory, + } + if len(got) != len(want) { + t.Fatalf("CanonicalReasons returned %v, want %v", got, want) + } + for i := range want { + if got[i] != want[i] { + t.Fatalf("CanonicalReasons returned %v, want %v", got, want) + } + } + + // A permuted input must produce the identical output. + permuted := []PruneReason{input[2], input[0], input[1], input[3]} + again := CanonicalReasons(permuted...) + for i := range want { + if again[i] != want[i] { + t.Fatalf("permuted input produced %v, want %v", again, want) + } + } +} + +func TestValidateVerdictReasons(t *testing.T) { + if err := validateVerdictReasons(PruneVerdictEligible, []PruneReason{PruneReasonEligibleRetention}); err != nil { + t.Fatalf("well-formed eligible verdict rejected: %v", err) + } + cases := []struct { + name string + verdict PruneVerdict + reasons []PruneReason + }{ + {"undefined verdict", PruneVerdict("maybe"), []PruneReason{PruneReasonEligibleRetention}}, + {"no reasons", PruneVerdictEligible, nil}, + {"undefined reason", PruneVerdictEligible, []PruneReason{"made-up"}}, + {"reason contradicts verdict", PruneVerdictEligible, []PruneReason{PruneReasonProtectedLatest}}, + {"unknown reason under protected", PruneVerdictProtected, []PruneReason{PruneReasonUnknownIdentity}}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + if err := validateVerdictReasons(tc.verdict, tc.reasons); err == nil { + t.Fatal("expected validation failure") + } + }) + } +} + +func TestPrunePlanValidateRejectsMalformedEntries(t *testing.T) { + plan := &PrunePlan{ + Tags: []RegistryTagVerdict{{ + Ref: RegistryTagRef{Repository: "app", Tag: "v1"}, + Verdict: PruneVerdictEligible, + Reasons: []PruneReason{PruneReasonEligibleRetention}, + }}, + Volumes: []VolumeVerdict{{ + Ref: RuntimeVolumeRef{Name: "gordon-app-data"}, + Verdict: PruneVerdictProtected, + Reasons: []PruneReason{PruneReasonProtectedOwnership}, + }}, + Gaps: []InventoryGap{{ + Source: InventorySourceRegistry, + Reason: PruneReasonUnknownInventory, + Detail: "list tags failed", + }}, + } + if err := plan.Validate(); err != nil { + t.Fatalf("well-formed plan rejected: %v", err) + } + + bad := *plan + bad.Tags = []RegistryTagVerdict{{Verdict: PruneVerdictEligible, Reasons: []PruneReason{PruneReasonEligibleRetention}}} + if err := bad.Validate(); err == nil { + t.Fatal("plan with an invalid tag identity passed validation") + } + + badGap := *plan + badGap.Gaps = []InventoryGap{{Source: InventorySourceRegistry, Reason: PruneReasonProtectedLatest}} + if err := badGap.Validate(); err == nil { + t.Fatal("plan with a non-unknown gap reason passed validation") + } +} + +func TestPrunePlanEligibleAccessors(t *testing.T) { + plan := &PrunePlan{ + Tags: []RegistryTagVerdict{ + {Ref: RegistryTagRef{Repository: "app", Tag: "v1"}, Verdict: PruneVerdictEligible, Reasons: []PruneReason{PruneReasonEligibleRetention}}, + {Ref: RegistryTagRef{Repository: "app", Tag: "v2"}, Verdict: PruneVerdictProtected, Reasons: []PruneReason{PruneReasonProtectedActiveService}}, + }, + Images: []RuntimeImageVerdict{ + {Ref: RuntimeImageRef{ID: "sha256:dead"}, Verdict: PruneVerdictEligible, Reasons: []PruneReason{PruneReasonEligibleDanglingRuntimeImage}}, + {Ref: RuntimeImageRef{ID: "sha256:beef"}, Verdict: PruneVerdictUnknown, Reasons: []PruneReason{PruneReasonUnknownContainerUse}}, + }, + Volumes: []VolumeVerdict{ + {Ref: RuntimeVolumeRef{Name: "gordon-a-data"}, Verdict: PruneVerdictEligible, Reasons: []PruneReason{PruneReasonEligibleReleasedVolume}}, + }, + } + + eligibleTags := plan.EligibleTags() + if len(eligibleTags) != 1 || eligibleTags[0].Tag != "v1" { + t.Fatalf("EligibleTags() = %v, want only app:v1", eligibleTags) + } + eligibleImages := plan.EligibleImages() + if len(eligibleImages) != 1 || eligibleImages[0].ID != "sha256:dead" { + t.Fatalf("EligibleImages() = %v, want only sha256:dead", eligibleImages) + } + eligibleVolumes := plan.EligibleVolumes() + if len(eligibleVolumes) != 1 || eligibleVolumes[0].Name != "gordon-a-data" { + t.Fatalf("EligibleVolumes() = %v, want only gordon-a-data", eligibleVolumes) + } + if got := plan.CountByVerdict(PruneVerdictEligible); got != 3 { + t.Fatalf("CountByVerdict(eligible) = %d, want 3", got) + } + if got := plan.CountByVerdict(PruneVerdictUnknown); got != 1 { + t.Fatalf("CountByVerdict(unknown) = %d, want 1", got) + } +} + +func TestPruneProtectionSnapshotLookups(t *testing.T) { + digest := "sha256:" + strings.Repeat("ab", 32) + snapshot := &PruneProtectionSnapshot{ + Roots: []ProtectionRoot{ + {Kind: ProtectionLatest, Ref: "app\x00latest"}, + {Kind: ProtectionActiveService, Ref: digest, Owner: "app/web"}, + {Kind: ProtectionContainerUse, Ref: "sha256:imageid", Owner: "container-1"}, + {Kind: ProtectionOwnership, Ref: "gordon-app-data", Owner: "app"}, + }, + VolumeClaims: []VolumeClaim{ + {Name: "gordon-app-data", App: "app", Service: "web", State: VolumeClaimAttached}, + {Name: "gordon-app-data", App: "app", Service: "web", State: VolumeClaimRetained}, + }, + } + + if !snapshot.ProtectsDigest(digest) { + t.Fatal("digest root not matched") + } + if snapshot.ProtectsDigest("") { + t.Fatal("empty digest matched a root") + } + if snapshot.ProtectsDigest("not-a-digest") { + t.Fatal("invalid digest matched a root") + } + if !snapshot.ProtectsRuntimeImage("sha256:imageid", nil) { + t.Fatal("runtime image ID root not matched") + } + if !snapshot.ProtectsRuntimeImage("", []string{digest}) { + t.Fatal("repo digest root not matched") + } + if snapshot.Protects(ProtectionLatest, "other\x00latest") { + t.Fatal("non-matching tag root matched") + } + + claim, ok := snapshot.VolumeClaimFor("gordon-app-data") + if !ok || claim.State != VolumeClaimRetained { + t.Fatalf("VolumeClaimFor returned %+v/%v, want the last recorded claim", claim, ok) + } + if claim.State.Protects() != true { + t.Fatal("retained claim must protect") + } + if VolumeClaimReleased.Protects() { + t.Fatal("released claim must not protect") + } + if _, ok := snapshot.VolumeClaimFor("absent"); ok { + t.Fatal("absent volume reported a claim") + } + if snapshot.Complete() != true { + t.Fatal("snapshot without gaps must be complete") + } + + gapped := &PruneProtectionSnapshot{Gaps: []InventoryGap{{ + Source: InventorySourceAppState, + Reason: PruneReasonUnknownInventory, + Detail: "unreadable app record", + }}} + if gapped.Complete() { + t.Fatal("snapshot with a valid gap reported complete") + } + if (*PruneProtectionSnapshot)(nil).Complete() { + t.Fatal("nil snapshot must not report complete") + } +} + +func TestVolumeClaimValidation(t *testing.T) { + if !(VolumeClaim{Name: "vol", App: "app", State: VolumeClaimAttached}).Valid() { + t.Fatal("complete claim rejected") + } + for _, claim := range []VolumeClaim{ + {App: "app", State: VolumeClaimAttached}, + {Name: "vol", State: VolumeClaimAttached}, + {Name: "vol", App: "app"}, + {Name: "vol", App: "app", State: "gone"}, + } { + if claim.Valid() { + t.Fatalf("incomplete claim accepted: %+v", claim) + } + } +} + +func TestRuntimeInventoryCompleteness(t *testing.T) { + if !(&RuntimeInventory{}).Complete() { + t.Fatal("empty inventory must be complete") + } + if (*RuntimeInventory)(nil).Complete() { + t.Fatal("nil inventory must not be complete") + } + if (&RuntimeInventory{Gaps: []InventoryGap{{ + Source: InventorySourceRuntimeImages, + Reason: PruneReasonUnknownInventory, + Detail: "list images failed", + }}}).Complete() { + t.Fatal("inventory with a gap reported complete") + } +} diff --git a/internal/domain/route.go b/internal/domain/route.go index b70b20367..eb7069365 100644 --- a/internal/domain/route.go +++ b/internal/domain/route.go @@ -5,7 +5,7 @@ type Route struct { Domain string Image string HTTPS bool - Env []string // Pre-resolved env vars ("KEY=VALUE"); when set, Deploy skips EnvLoader lookup. + Env []string // Pre-resolved env vars ("KEY=VALUE"). } // ProxyTarget represents the destination for proxying requests. diff --git a/internal/domain/volumes.go b/internal/domain/volumes.go index 269f09896..8d47e744c 100644 --- a/internal/domain/volumes.go +++ b/internal/domain/volumes.go @@ -21,4 +21,7 @@ type VolumeInfo struct { type VolumePruneReport struct { VolumesRemoved int SpaceReclaimed int64 + // Plan carries every candidate's verdict, the inventory gaps, and + // the applied flag. Dry-run and execution share this shape. + Plan PruneReport } diff --git a/internal/testutils/fixtures/configs/basic.toml b/internal/testutils/fixtures/configs/basic.toml index d69bb9d1b..015ac3895 100644 --- a/internal/testutils/fixtures/configs/basic.toml +++ b/internal/testutils/fixtures/configs/basic.toml @@ -18,10 +18,6 @@ auto_create = true prefix = "gordon" preserve = true -[env] -dir = "/tmp/env" -providers = ["pass", "sops"] - [logging] enabled = true level = "info" \ No newline at end of file diff --git a/internal/testutils/helpers.go b/internal/testutils/helpers.go index 4e5717f5b..71a285232 100644 --- a/internal/testutils/helpers.go +++ b/internal/testutils/helpers.go @@ -59,10 +59,6 @@ auto_create = true prefix = "gordon" preserve = true -[env] -dir = "/tmp/env" -providers = ["pass", "sops"] - [logging] enabled = true level = "info"` diff --git a/internal/testutils/mocks/runtime_gen.go b/internal/testutils/mocks/runtime_gen.go index ce6270b08..37c646b90 100644 --- a/internal/testutils/mocks/runtime_gen.go +++ b/internal/testutils/mocks/runtime_gen.go @@ -11,10 +11,10 @@ package mocks import ( context "context" - runtime "github.com/bnema/gordon/pkg/runtime" io "io" reflect "reflect" + runtime "github.com/bnema/gordon/pkg/runtime" gomock "go.uber.org/mock/gomock" ) @@ -279,6 +279,21 @@ func (mr *MockRuntimeMockRecorder) ListImages(ctx any) *gomock.Call { return mr.mock.ctrl.RecordCallWithMethodType(mr.mock, "ListImages", reflect.TypeOf((*MockRuntime)(nil).ListImages), ctx) } +// ListImagesDetailed mocks base method. +func (m *MockRuntime) ListImagesDetailed(ctx context.Context) ([]runtime.ImageDetail, error) { + m.ctrl.T.Helper() + ret := m.ctrl.Call(m, "ListImagesDetailed", ctx) + ret0, _ := ret[0].([]runtime.ImageDetail) + ret1, _ := ret[1].(error) + return ret0, ret1 +} + +// ListImagesDetailed indicates an expected call of ListImagesDetailed. +func (mr *MockRuntimeMockRecorder) ListImagesDetailed(ctx any) *gomock.Call { + mr.mock.ctrl.T.Helper() + return mr.mock.ctrl.RecordCallWithMethodType(mr.mock, "ListImagesDetailed", reflect.TypeOf((*MockRuntime)(nil).ListImagesDetailed), ctx) +} + // ListNetworks mocks base method. func (m *MockRuntime) ListNetworks(ctx context.Context) ([]*runtime.NetworkInfo, error) { m.ctrl.T.Helper() @@ -407,20 +422,6 @@ func (mr *MockRuntimeMockRecorder) RemoveVolume(ctx, volumeName, force any) *gom return mr.mock.ctrl.RecordCallWithMethodType(mr.mock, "RemoveVolume", reflect.TypeOf((*MockRuntime)(nil).RemoveVolume), ctx, volumeName, force) } -// RestartContainer mocks base method. -func (m *MockRuntime) RestartContainer(ctx context.Context, containerID string) error { - m.ctrl.T.Helper() - ret := m.ctrl.Call(m, "RestartContainer", ctx, containerID) - ret0, _ := ret[0].(error) - return ret0 -} - -// RestartContainer indicates an expected call of RestartContainer. -func (mr *MockRuntimeMockRecorder) RestartContainer(ctx, containerID any) *gomock.Call { - mr.mock.ctrl.T.Helper() - return mr.mock.ctrl.RecordCallWithMethodType(mr.mock, "RestartContainer", reflect.TypeOf((*MockRuntime)(nil).RestartContainer), ctx, containerID) -} - // StartContainer mocks base method. func (m *MockRuntime) StartContainer(ctx context.Context, containerID string) error { m.ctrl.T.Helper() @@ -435,20 +436,6 @@ func (mr *MockRuntimeMockRecorder) StartContainer(ctx, containerID any) *gomock. return mr.mock.ctrl.RecordCallWithMethodType(mr.mock, "StartContainer", reflect.TypeOf((*MockRuntime)(nil).StartContainer), ctx, containerID) } -// StopContainer mocks base method. -func (m *MockRuntime) StopContainer(ctx context.Context, containerID string) error { - m.ctrl.T.Helper() - ret := m.ctrl.Call(m, "StopContainer", ctx, containerID) - ret0, _ := ret[0].(error) - return ret0 -} - -// StopContainer indicates an expected call of StopContainer. -func (mr *MockRuntimeMockRecorder) StopContainer(ctx, containerID any) *gomock.Call { - mr.mock.ctrl.T.Helper() - return mr.mock.ctrl.RecordCallWithMethodType(mr.mock, "StopContainer", reflect.TypeOf((*MockRuntime)(nil).StopContainer), ctx, containerID) -} - // Version mocks base method. func (m *MockRuntime) Version(ctx context.Context) (string, error) { m.ctrl.T.Helper() @@ -478,3 +465,17 @@ func (mr *MockRuntimeMockRecorder) VolumeExists(ctx, volumeName any) *gomock.Cal mr.mock.ctrl.T.Helper() return mr.mock.ctrl.RecordCallWithMethodType(mr.mock, "VolumeExists", reflect.TypeOf((*MockRuntime)(nil).VolumeExists), ctx, volumeName) } + +// WaitForContainer mocks base method. +func (m *MockRuntime) WaitForContainer(ctx context.Context, containerID string) error { + m.ctrl.T.Helper() + ret := m.ctrl.Call(m, "WaitForContainer", ctx, containerID) + ret0, _ := ret[0].(error) + return ret0 +} + +// WaitForContainer indicates an expected call of WaitForContainer. +func (mr *MockRuntimeMockRecorder) WaitForContainer(ctx, containerID any) *gomock.Call { + mr.mock.ctrl.T.Helper() + return mr.mock.ctrl.RecordCallWithMethodType(mr.mock, "WaitForContainer", reflect.TypeOf((*MockRuntime)(nil).WaitForContainer), ctx, containerID) +} diff --git a/internal/usecase/apps/app_read_model_test.go b/internal/usecase/apps/app_read_model_test.go new file mode 100644 index 000000000..c90be80e0 --- /dev/null +++ b/internal/usecase/apps/app_read_model_test.go @@ -0,0 +1,191 @@ +package apps_test + +import ( + "context" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/apps" +) + +func TestAppService_Show_FreshDesiredOnlyApp(t *testing.T) { + ctx := context.Background() + store := newMockAppState(t) + svc := apps.NewAppServiceImpl(store, newMockDeployEngine(t), newMockSecretWriter(t), zerowrap.Default()) + + store.EXPECT().AppExists(mock.Anything, "blog").Return(true, nil).Once() + store.EXPECT().LoadDesired(mock.Anything, "blog").Return(domain.AppDesiredRevision{ + App: "blog", Revision: "rev-1", Status: "pending", + }, true, nil).Once() + store.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{}, false, nil).Once() + store.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + store.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-1"}, nil).Once() + store.EXPECT().LoadLatestOperation(mock.Anything, "blog"). + Return(domain.AppOperation{}, false, nil).Once() + + detail, err := svc.Show(ctx, "blog") + require.NoError(t, err) + assert.Equal(t, "rev-1", detail.DesiredRevision) + assert.Equal(t, "pending", detail.DesiredStatus) + assert.True(t, detail.Pending, "desired state ACTIVE never reached is pending") + assert.False(t, detail.Converged) + assert.Empty(t, detail.ConvergedRevision) + assert.Empty(t, detail.Services) + assert.Empty(t, detail.LastOp) + assert.Empty(t, detail.Retained.Volumes) +} + +func TestAppService_Show_ConvergedAppWithRetainedResources(t *testing.T) { + ctx := context.Background() + store := newMockAppState(t) + svc := apps.NewAppServiceImpl(store, newMockDeployEngine(t), newMockSecretWriter(t), zerowrap.Default()) + started := time.Date(2026, 2, 3, 4, 5, 6, 0, time.UTC) + + store.EXPECT().AppExists(mock.Anything, "blog").Return(true, nil).Once() + store.EXPECT().LoadDesired(mock.Anything, "blog").Return(domain.AppDesiredRevision{ + App: "blog", Revision: "rev-1", Status: "active", + }, true, nil).Once() + store.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", Converged: true, ConvergedRevision: "rev-1", + Services: map[string]domain.AppEffectiveService{ + "web": {EffectiveRevision: "rev-1", Digest: "sha256:abc", Container: "ctr-1"}, + }, + }, true, nil).Once() + store.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + store.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{ + App: "blog", ID: "app-1", + Volumes: []domain.AppOwnedVolume{ + {Name: "data", Service: "web", RuntimeName: "gordon-blog--web--vol--data", State: domain.AppResourceAttached}, + }, + Secrets: []domain.AppOwnedSecret{ + {Service: "web", Env: "DATABASE_URL", Name: "database-url", Path: "gordon/apps/app-1/blog/web/database-url", State: domain.AppResourceAttached}, + }, + Images: []domain.AppOwnedImage{ + {Service: "web", Reference: "registry.example.com/blog/web:1.4.2", State: domain.AppResourceAttached}, + }, + Services: map[string]domain.AppServiceRecovery{"web": {RestartUnsafe: true}}, + }, nil).Once() + store.EXPECT().LoadLatestOperation(mock.Anything, "blog").Return(domain.AppOperation{ + Op: "op-9", Kind: "deploy", App: "blog", Outcome: domain.AppOutcomeSuccess, StartedAt: started, + }, true, nil).Once() + + detail, err := svc.Show(ctx, "blog") + require.NoError(t, err) + assert.False(t, detail.Pending, "desired == ACTIVE is not pending") + assert.True(t, detail.Converged) + assert.Equal(t, "rev-1", detail.ConvergedRevision) + require.Contains(t, detail.Services, "web") + assert.Equal(t, "ctr-1", detail.Services["web"].Container) + assert.Equal(t, "sha256:abc", detail.Services["web"].Digest) + assert.True(t, detail.Services["web"].RestartUnsafe) + assert.Equal(t, []string{"gordon-blog--web--vol--data"}, detail.Retained.Volumes) + assert.Equal(t, []string{"gordon/apps/app-1/blog/web/database-url"}, detail.Retained.Secrets) + assert.Equal(t, []string{"registry.example.com/blog/web:1.4.2"}, detail.Retained.Images) + assert.Equal(t, "op-9", detail.LastOp) + assert.Equal(t, "deploy", detail.LastOpKind) + assert.Equal(t, domain.AppOutcomeSuccess, detail.LastOutcome) + assert.Equal(t, started, detail.LastOpStartedAt) +} + +func TestAppService_Show_MixedRevisionsArePending(t *testing.T) { + ctx := context.Background() + store := newMockAppState(t) + svc := apps.NewAppServiceImpl(store, newMockDeployEngine(t), newMockSecretWriter(t), zerowrap.Default()) + + store.EXPECT().AppExists(mock.Anything, "blog").Return(true, nil).Once() + store.EXPECT().LoadDesired(mock.Anything, "blog").Return(domain.AppDesiredRevision{ + App: "blog", Revision: "rev-2", Status: "pending", + }, true, nil).Once() + store.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", Converged: true, ConvergedRevision: "rev-1", + Services: map[string]domain.AppEffectiveService{ + "web": {EffectiveRevision: "rev-1", Container: "ctr-1"}, + }, + }, true, nil).Once() + store.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + store.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-1"}, nil).Once() + store.EXPECT().LoadLatestOperation(mock.Anything, "blog").Return(domain.AppOperation{ + Op: "op-8", Kind: "deploy", App: "blog", Outcome: domain.AppOutcomeFailed, + }, true, nil).Once() + + detail, err := svc.Show(ctx, "blog") + require.NoError(t, err) + assert.True(t, detail.Pending, "a newer desired revision is pending") + assert.Equal(t, "rev-1", detail.ConvergedRevision, "ACTIVE stays authoritative") + assert.Equal(t, domain.AppOutcomeFailed, detail.LastOutcome, "the failed journal is reported") +} + +func TestAppService_Show_StoppedAppKeepsReadModel(t *testing.T) { + ctx := context.Background() + store := newMockAppState(t) + svc := apps.NewAppServiceImpl(store, newMockDeployEngine(t), newMockSecretWriter(t), zerowrap.Default()) + + store.EXPECT().AppExists(mock.Anything, "blog").Return(true, nil).Once() + store.EXPECT().LoadDesired(mock.Anything, "blog").Return(domain.AppDesiredRevision{ + App: "blog", Revision: "rev-1", Status: "active", + }, true, nil).Once() + store.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", Converged: true, ConvergedRevision: "rev-1", + Services: map[string]domain.AppEffectiveService{"web": {EffectiveRevision: "rev-1", Container: "ctr-1"}}, + }, true, nil).Once() + store.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog", Stopped: true}, nil).Once() + store.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-1"}, nil).Once() + store.EXPECT().LoadLatestOperation(mock.Anything, "blog").Return(domain.AppOperation{}, false, nil).Once() + + detail, err := svc.Show(ctx, "blog") + require.NoError(t, err) + assert.True(t, detail.Stopped) + assert.False(t, detail.Pending, "stopping does not make desired state pending") + require.Contains(t, detail.Services, "web") +} + +// TestAppService_Show_UnknownAppIsNotFound proves a name with no live app +// identity reports not-found instead of an empty detail. +func TestAppService_Show_UnknownAppIsNotFound(t *testing.T) { + ctx := context.Background() + store := newMockAppState(t) + svc := apps.NewAppServiceImpl(store, newMockDeployEngine(t), newMockSecretWriter(t), zerowrap.Default()) + + store.EXPECT().AppExists(mock.Anything, "ghost").Return(false, nil).Once() + + _, err := svc.Show(ctx, "ghost") + require.ErrorIs(t, err, domain.ErrAppNotFound) +} + +func TestAppService_List_SkipsRetiredNamesAndReportsReadModel(t *testing.T) { + ctx := context.Background() + store := newMockAppState(t) + svc := apps.NewAppServiceImpl(store, newMockDeployEngine(t), newMockSecretWriter(t), zerowrap.Default()) + + store.EXPECT().ListApps(mock.Anything).Return([]string{"blog", "gone"}, nil).Once() + // blog is live: desired rev-2 while ACTIVE holds rev-1, latest deploy partial. + store.EXPECT().AppExists(mock.Anything, "blog").Return(true, nil).Once() + store.EXPECT().LoadDesired(mock.Anything, "blog").Return(domain.AppDesiredRevision{ + App: "blog", Revision: "rev-2", Status: "pending", + }, true, nil).Once() + store.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", Converged: true, ConvergedRevision: "rev-1", + }, true, nil).Once() + store.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + store.EXPECT().LoadLatestOperation(mock.Anything, "blog").Return(domain.AppOperation{ + Op: "op-7", Kind: "deploy", App: "blog", Outcome: domain.AppOutcomePartial, + }, true, nil).Once() + // gone keeps only its retired journal and is not an app any more. + store.EXPECT().AppExists(mock.Anything, "gone").Return(false, nil).Once() + + summaries, err := svc.List(ctx) + require.NoError(t, err) + require.Len(t, summaries, 1) + assert.Equal(t, "blog", summaries[0].App) + assert.Equal(t, "rev-2", summaries[0].Desired) + assert.Equal(t, "pending", summaries[0].DesiredStatus) + assert.Equal(t, "rev-1", summaries[0].Active) + assert.True(t, summaries[0].Pending) + assert.Equal(t, domain.AppOutcomePartial, summaries[0].LastOutcome) +} diff --git a/internal/usecase/apps/app_service.go b/internal/usecase/apps/app_service.go new file mode 100644 index 000000000..fef271db8 --- /dev/null +++ b/internal/usecase/apps/app_service.go @@ -0,0 +1,784 @@ +package apps + +import ( + "context" + "errors" + "fmt" + "sort" + "strings" + "sync" + + "github.com/bnema/zerowrap" + + "github.com/bnema/gordon/internal/boundaries/in" + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/deployment" +) + +// AppServiceImpl implements in.AppService over the deployment engine. The +// daemon owns this implementation; CLI always reaches it via the daemon. +// All mutations run recovery-before-mutation through the engine. +type AppServiceImpl struct { + store out.AppCatalogReader + deploy deployEngine + secrets out.SecretWriter + log zerowrap.Logger + // core is the single configured apps service. It is built once and + // reconfigured in place, never reconstructed per request. + core *Service + + // lifecycleMu guards the daemon lifecycle. daemonCtx is the + // daemon-owned parent for background executions: it is never a request + // context, so request cancellation cannot abort an in-flight deploy. + // stopped gates new executions once Shutdown runs, handoffs tracks the + // Deploy request hand-offs (claim through scheduling or settling) so + // Shutdown never closes state under a claim it did not wait for, and + // executions tracks the in-flight background executions so Shutdown can + // wait for them. + lifecycleMu sync.Mutex + daemonCtx context.Context + daemonStop context.CancelFunc + stopped bool + handoffs sync.WaitGroup + executions sync.WaitGroup +} + +// deployEngine is the subset of the deployment engine the app service needs. +type deployEngine interface { + StartDeploy(ctx context.Context, input deployment.DeployInput) (*deployment.StartDeployResult, error) + AbandonDeploy(ctx context.Context, claim deployment.DeployClaim) error + ExecuteDeploy(ctx context.Context, claim deployment.DeployClaim) (*deployment.DeployResult, error) + Stop(ctx context.Context, app, opID string) (*deployment.LifecycleResult, error) + Start(ctx context.Context, app, opID string) (*deployment.LifecycleResult, error) + Restart(ctx context.Context, app, service, opID string) (*deployment.LifecycleResult, error) + Remove(ctx context.Context, app, opID string) (*deployment.LifecycleResult, error) +} + +// AppStore is the state the app administration service needs: the +// read model plus the apply path. +type AppStore interface { + out.AppCatalogReader + out.AppApplyStore +} + +// NewAppServiceImpl wires the driving-port implementation. +func NewAppServiceImpl(store AppStore, deploy deployEngine, secrets out.SecretWriter, log zerowrap.Logger) *AppServiceImpl { + daemonCtx, daemonStop := context.WithCancel(context.Background()) + return &AppServiceImpl{ + store: store, + deploy: deploy, + secrets: secrets, + log: log, + core: NewService(store, log), + daemonCtx: daemonCtx, + daemonStop: daemonStop, + } +} + +// WithDaemonContext sets the parent context that owns background deploy +// executions. It must be called during wiring, before the first Deploy. The +// derived context is cancelled by Shutdown; request contexts are never used +// as its parent. +func (s *AppServiceImpl) WithDaemonContext(parent context.Context) *AppServiceImpl { + if parent == nil { + parent = context.Background() + } + s.lifecycleMu.Lock() + defer s.lifecycleMu.Unlock() + if s.stopped { + return s + } + ctx, cancel := context.WithCancel(parent) + if s.daemonStop != nil { + s.daemonStop() + } + s.daemonCtx, s.daemonStop = ctx, cancel + return s +} + +// Shutdown stops scheduling background deploy executions, cancels the daemon +// context so in-flight executions abort with a recoverable non-terminal +// journal, and waits for the in-flight hand-offs and executions to unwind or +// for ctx to expire. Waiting for the hand-offs first means a Deploy that +// claimed an operation just before Shutdown always schedules or settles it +// before Shutdown returns. +func (s *AppServiceImpl) Shutdown(ctx context.Context) error { + s.lifecycleMu.Lock() + if !s.stopped { + s.stopped = true + s.daemonStop() + } + s.lifecycleMu.Unlock() + + done := make(chan struct{}) + go func() { + s.handoffs.Wait() + s.executions.Wait() + close(done) + }() + select { + case <-done: + return nil + case <-ctx.Done(): + return ctx.Err() + } +} + +// scheduleExecution runs the claimed operation's effects on the daemon +// context. Only the owner of a freshly claimed key reaches it, so a duplicate +// key can never spawn a second execution. The journal claimed by StartDeploy +// is already durable when the goroutine starts. When Shutdown won the race +// between the claim and scheduling, the claim is settled without starting: +// the live marker is released and the journal converged as interrupted, so it +// can neither execute nor stay marked live forever. +func (s *AppServiceImpl) scheduleExecution(claim deployment.DeployClaim) error { + opFields := map[string]any{ + zerowrap.FieldLayer: "usecase", + zerowrap.FieldUseCase: "Deploy", + "app": claim.App, + "op": claim.Op, + } + s.lifecycleMu.Lock() + if s.stopped { + settleCtx := zerowrap.CtxWithFields(context.WithoutCancel(s.daemonCtx), opFields) + s.lifecycleMu.Unlock() + settleLog := zerowrap.FromCtx(settleCtx) + settleLog.Warn().Msg("apps: deploy claimed during shutdown; settling the claim without executing") + if err := s.deploy.AbandonDeploy(settleCtx, claim); err != nil { + settleLog.Warn().Err(err).Msg("apps: failed to settle a deploy claimed during shutdown") + } + return fmt.Errorf("apps: deploy claimed during shutdown: %w", domain.ErrAppStateConflict) + } + ctx := zerowrap.CtxWithFields(s.daemonCtx, opFields) + s.executions.Add(1) + s.lifecycleMu.Unlock() + + go func() { + defer s.executions.Done() + execLog := zerowrap.FromCtx(ctx) + if _, err := s.deploy.ExecuteDeploy(ctx, claim); err != nil { + execLog.Warn().Err(err).Msg("apps: background deploy execution failed") + } + }() + return nil +} + +// WithEntrypoints supplies the installation entrypoint listeners used to +// validate L4 publish declarations at apply time. +func (s *AppServiceImpl) WithEntrypoints(listeners map[string]domain.EntryPointListener) *AppServiceImpl { + s.core.WithEntrypoints(listeners) + return s +} + +// WithGCBarrier supplies the process-wide GC barrier used to serialize an +// apply against destructive prune. +func (s *AppServiceImpl) WithGCBarrier(barrier out.GCBarrier) *AppServiceImpl { + s.core.WithGCBarrier(barrier) + return s +} + +// WithImagePolicy supplies the registry policy used for manifest validation. +func (s *AppServiceImpl) WithImagePolicy(policy domain.ImageSourcePolicy) *AppServiceImpl { + s.core.WithImagePolicy(policy) + return s +} + +// WithBindPolicies supplies the administrative bind policies used at apply time. +func (s *AppServiceImpl) WithBindPolicies(policies map[string]domain.AppBindPolicy) *AppServiceImpl { + s.core.SetBindPolicies(policies) + return s +} + +// SetBindPolicies atomically replaces the administrative bind policies. It is +// safe to call from a config reload while applies are in flight. +func (s *AppServiceImpl) SetBindPolicies(policies map[string]domain.AppBindPolicy) { + s.core.SetBindPolicies(policies) +} + +// WithDevicePolicies supplies the administrative device policies used at apply time. +func (s *AppServiceImpl) WithDevicePolicies(policies map[string]domain.AppDevicePolicy) *AppServiceImpl { + s.core.SetDevicePolicies(policies) + return s +} + +// SetDevicePolicies atomically replaces the administrative device policies. It is +// safe to call from a config reload while applies are in flight. +func (s *AppServiceImpl) SetDevicePolicies(policies map[string]domain.AppDevicePolicy) { + s.core.SetDevicePolicies(policies) +} + +var _ in.AppService = (*AppServiceImpl)(nil) + +// Apply implements in.AppService. +func (s *AppServiceImpl) Apply(ctx context.Context, spec domain.AppSpec, source []byte, dryRun bool) (*in.AppApplyResult, *in.AppDryRunResult, error) { + applyResult, dryResult, err := s.core.Apply(ctx, spec, source, dryRun) + if err != nil { + return nil, nil, err + } + if dryResult != nil { + return nil, &in.AppDryRunResult{ + App: spec.Name, + Valid: true, + Diff: dryResult.Diff, + }, nil + } + if applyResult == nil { + return nil, &in.AppDryRunResult{App: spec.Name, Valid: true}, nil + } + return &in.AppApplyResult{ + App: applyResult.App, + FormerRevision: applyResult.FormerRevision, + ResultingRevision: applyResult.ResultingRevision, + Noop: applyResult.Noop, + Pending: applyResult.Pending, + Diff: applyResult.Diff, + IntentID: applyResult.IntentID, + }, nil, nil +} + +// List implements in.AppService. Names whose only remaining record is a +// retired operation journal are not apps any more and are not listed. +func (s *AppServiceImpl) List(ctx context.Context) ([]in.AppSummary, error) { + apps, err := s.store.ListApps(ctx) + if err != nil { + return nil, err + } + sort.Strings(apps) + summaries := make([]in.AppSummary, 0, len(apps)) + for _, app := range apps { + live, err := s.store.AppExists(ctx, app) + if err != nil { + return nil, err + } + if !live { + continue + } + summary, err := s.summary(ctx, app) + if err != nil { + return nil, err + } + summaries = append(summaries, summary) + } + return summaries, nil +} + +// summary builds one list row from desired + ACTIVE + intent + the latest +// journal outcome. +func (s *AppServiceImpl) summary(ctx context.Context, app string) (in.AppSummary, error) { + desired, _, err := s.store.LoadDesired(ctx, app) + if err != nil { + return in.AppSummary{}, err + } + active, ok, err := s.store.LoadActive(ctx, app) + if err != nil { + return in.AppSummary{}, err + } + intent, err := s.store.LoadIntent(ctx, app) + if err != nil { + return in.AppSummary{}, err + } + summary := in.AppSummary{ + App: app, + Desired: desired.Revision, + DesiredStatus: desired.Status, + Pending: desiredPending(desired.Revision, active, ok), + Stopped: intent.Stopped, + } + if ok { + summary.Active = active.ConvergedRevision + summary.Converged = active.Converged + } + latest, hasOp, err := s.store.LoadLatestOperation(ctx, app) + if err != nil { + return in.AppSummary{}, err + } + if hasOp { + summary.LastOutcome = latest.Outcome + } + return summary, nil +} + +// Show implements in.AppService. +func (s *AppServiceImpl) Show(ctx context.Context, app string) (*in.AppDetail, error) { + live, err := s.store.AppExists(ctx, app) + if err != nil { + return nil, err + } + if !live { + return nil, fmt.Errorf("apps: app %q does not exist: %w", app, domain.ErrAppNotFound) + } + desired, _, err := s.store.LoadDesired(ctx, app) + if err != nil { + return nil, err + } + active, ok, err := s.store.LoadActive(ctx, app) + if err != nil { + return nil, err + } + intent, err := s.store.LoadIntent(ctx, app) + if err != nil { + return nil, err + } + detail := &in.AppDetail{ + App: app, + DesiredRevision: desired.Revision, + DesiredStatus: desired.Status, + Pending: desiredPending(desired.Revision, active, ok), + Converged: active.Converged, + ConvergedRevision: active.ConvergedRevision, + Services: map[string]in.AppServiceView{}, + Stopped: intent.Stopped, + } + ownership, err := s.store.LoadOwnership(ctx, app) + if err != nil { + return nil, err + } + detail.Retained = retainedView(ownership) + for name, svc := range active.Services { + detail.Services[name] = in.AppServiceView{ + EffectiveRevision: svc.EffectiveRevision, + Digest: svc.Digest, + Container: svc.Container, + RestartUnsafe: ownership.Services[name].RestartUnsafe, + } + } + latest, ok, err := s.store.LoadLatestOperation(ctx, app) + if err != nil { + return nil, err + } + if ok { + detail.LastOp = latest.Op + detail.LastOpKind = latest.Kind + detail.LastOutcome = latest.Outcome + detail.LastOpStartedAt = latest.StartedAt + } + return detail, nil +} + +// desiredPending reports desired state the active revision has not +// reached. It reads desired vs ACTIVE only: a journal outcome never +// decides convergence. +func desiredPending(desiredRevision string, active domain.AppActive, hasActive bool) bool { + if desiredRevision == "" { + return false + } + return !hasActive || !active.Converged || active.ConvergedRevision != desiredRevision +} + +// retainedView lists the resources the app owns and would retain on +// removal, sorted for stable output. +func retainedView(ownership domain.AppOwnership) in.AppRetainedView { + view := in.AppRetainedView{} + for _, volume := range ownership.Volumes { + name := volume.RuntimeName + if name == "" { + name = volume.Name + } + view.Volumes = append(view.Volumes, name) + } + for _, secret := range ownership.Secrets { + view.Secrets = append(view.Secrets, secret.Path) + } + for _, image := range ownership.Images { + view.Images = append(view.Images, image.Reference) + } + sort.Strings(view.Volumes) + sort.Strings(view.Secrets) + sort.Strings(view.Images) + return view +} + +// Diff implements in.AppService. +func (s *AppServiceImpl) Diff(ctx context.Context, app string) (domain.AppDiff, error) { + desired, ok, err := s.store.LoadDesired(ctx, app) + if err != nil { + return domain.AppDiff{}, err + } + if !ok { + return domain.AppDiff{}, fmt.Errorf("apps: app %q has no desired state: %w", app, domain.ErrAppRevisionNotFound) + } + active, _, err := s.store.LoadActive(ctx, app) + if err != nil { + return domain.AppDiff{}, err + } + return domain.DiffAppSpec(desired.Spec, activeSpec(active)), nil +} + +// Deploy implements in.AppService. It enrolls the request hand-off with the +// lifecycle first, so a deploy arriving after shutdown began is refused before +// any claim, and one already in flight is always scheduled or settled before +// Shutdown returns. StartDeploy then claims and persists the operation under +// the app lock, and the effects run in the background on the daemon context, +// so Deploy returns the still-running journal promptly and request +// cancellation cannot abort an in-flight replacement. A repeated idempotency +// key is never owned twice: its stored journal replays instead of spawning a +// second execution, and only the owner schedules any work. +func (s *AppServiceImpl) Deploy(ctx context.Context, app, revision, service string, all bool, idempotencyKey string) (*domain.AppOperation, error) { + if err := s.requireServiceScope(ctx, app, revision, service, all, true); err != nil { + return nil, err + } + done, err := s.beginDeploy() + if err != nil { + return nil, err + } + defer done() + + started, err := s.deploy.StartDeploy(ctx, deployment.DeployInput{App: app, Revision: revision, Service: service, Op: idempotencyKey}) + if err != nil { + return nil, err + } + if !started.Owned { + // The key already answered this request: its stored journal is the + // result and no workload is touched again. + return s.operationResult(ctx, app, started.Claim.Op, started.ReplayError()) + } + if err := s.scheduleExecution(started.Claim); err != nil { + return s.operationResult(ctx, app, started.Claim.Op, err) + } + return s.operationResult(ctx, app, started.Claim.Op, nil) +} + +// requireServiceScope refuses an app-wide deploy or restart of a +// multi-service app unless the caller asked for it with all, so an +// unrelated service (a database, say) is never replaced by accident. It +// runs before any claim or idempotency key is recorded. Deploy counts the +// selected revision — an explicit revision when one is given, the desired +// head otherwise — plus ACTIVE, so an older multi-service revision cannot +// slip through a single-service head, and a first deploy is covered. +// Restart counts ACTIVE only. +func (s *AppServiceImpl) requireServiceScope(ctx context.Context, app, revision, service string, all, includeDesired bool) error { + if all && service != "" { + return fmt.Errorf("apps: --all and --service are mutually exclusive: %w", domain.ErrAppServiceScope) + } + if all || service != "" { + return nil + } + names := map[string]struct{}{} + if includeDesired { + desired, err := s.loadScopeRevision(ctx, app, revision) + if err != nil { + return err + } + for _, svc := range desired.Spec.Services { + names[svc.Name] = struct{}{} + } + } + active, ok, err := s.store.LoadActive(ctx, app) + if err != nil { + return fmt.Errorf("apps: load active of app %q for service scope: %w", app, err) + } + if ok { + for name := range active.Services { + names[name] = struct{}{} + } + } + if len(names) <= 1 { + return nil + } + sorted := make([]string, 0, len(names)) + for name := range names { + sorted = append(sorted, name) + } + sort.Strings(sorted) + return fmt.Errorf("app %s has several services (%s): pass --service NAME or --all: %w", + app, strings.Join(sorted, ", "), domain.ErrAppServiceScope) +} + +// loadScopeRevision returns the revision a deploy would activate: the +// explicit revision when one is selected, the desired head otherwise. A +// missing revision counts no services: an unknown explicit revision is +// left for StartDeploy to report, and a missing desired head means a +// first deploy with nothing captured yet. +func (s *AppServiceImpl) loadScopeRevision(ctx context.Context, app, revision string) (domain.AppDesiredRevision, error) { + if revision != "" { + rev, err := s.store.LoadRevision(ctx, app, revision) + if err != nil { + if errors.Is(err, domain.ErrAppRevisionNotFound) { + return domain.AppDesiredRevision{}, nil + } + return domain.AppDesiredRevision{}, fmt.Errorf("apps: load revision %q of app %q for service scope: %w", revision, app, err) + } + return rev, nil + } + desired, ok, err := s.store.LoadDesired(ctx, app) + if err != nil { + return domain.AppDesiredRevision{}, fmt.Errorf("apps: load desired of app %q for service scope: %w", app, err) + } + if !ok { + return domain.AppDesiredRevision{}, nil + } + return desired, nil +} + +// beginDeploy enrolls one Deploy hand-off before the engine claims anything. +// It refuses the request once Shutdown began, and otherwise guarantees Shutdown +// waits for the hand-off (claim, schedule, or settle) to finish before state is +// torn down, so no claim is made or settled against closed state. The returned +// done func must be called exactly once. +func (s *AppServiceImpl) beginDeploy() (func(), error) { + s.lifecycleMu.Lock() + defer s.lifecycleMu.Unlock() + if s.stopped { + return nil, fmt.Errorf("apps: deploy refused: daemon is shutting down: %w", domain.ErrAppStateConflict) + } + s.handoffs.Add(1) + return s.handoffs.Done, nil +} + +// Stop implements in.AppService. +func (s *AppServiceImpl) Stop(ctx context.Context, app, idempotencyKey string) (*domain.AppOperation, error) { + result, err := s.deploy.Stop(ctx, app, idempotencyKey) + if result == nil { + return nil, err + } + return s.operationResult(ctx, app, result.Op, err) +} + +// Start implements in.AppService. +func (s *AppServiceImpl) Start(ctx context.Context, app, idempotencyKey string) (*domain.AppOperation, error) { + result, err := s.deploy.Start(ctx, app, idempotencyKey) + if result == nil { + return nil, err + } + return s.operationResult(ctx, app, result.Op, err) +} + +// Restart implements in.AppService. +func (s *AppServiceImpl) Restart(ctx context.Context, app, service string, all bool, idempotencyKey string) (*domain.AppOperation, error) { + if err := s.requireServiceScope(ctx, app, "", service, all, false); err != nil { + return nil, err + } + result, err := s.deploy.Restart(ctx, app, service, idempotencyKey) + if result == nil { + return nil, err + } + return s.operationResult(ctx, app, result.Op, err) +} + +// Remove implements in.AppService. +func (s *AppServiceImpl) Remove(ctx context.Context, app, idempotencyKey string) (*domain.AppOperation, error) { + result, err := s.deploy.Remove(ctx, app, idempotencyKey) + if result == nil { + return nil, err + } + return s.operationResult(ctx, app, result.Op, err) +} + +// operationResult loads the journal an engine call produced. A call that +// failed after its journal was claimed returns both, so callers can +// inspect the recorded outcome instead of losing it. +func (s *AppServiceImpl) operationResult(ctx context.Context, app, opID string, callErr error) (*domain.AppOperation, error) { + if opID == "" { + return nil, callErr + } + op, err := s.store.LoadOperation(ctx, app, opID) + if err != nil { + if callErr != nil { + return nil, callErr + } + return nil, err + } + if callErr != nil { + return &op, callErr + } + return &op, nil +} + +// OperationByKey implements in.AppService. +func (s *AppServiceImpl) OperationByKey(ctx context.Context, app, key string) (*domain.AppOperation, error) { + op, err := s.store.LoadOperation(ctx, app, key) + if err != nil { + return nil, err + } + return &op, nil +} + +// ListSecrets implements in.AppService without reading secret values. A +// name with no live identity is not found; ownership-only apps remain live +// and report their registrations (possibly none). +func (s *AppServiceImpl) ListSecrets(ctx context.Context, app, service string) ([]in.AppSecretMetadata, error) { + live, err := s.store.AppExists(ctx, app) + if err != nil { + return nil, fmt.Errorf("list app secrets: check app %q: %w", app, err) + } + if !live { + return nil, fmt.Errorf("apps: app %q does not exist: %w", app, domain.ErrAppNotFound) + } + entries := []in.AppSecretMetadata{} + desired, ok, err := s.store.LoadDesired(ctx, app) + if err != nil { + return nil, fmt.Errorf("list app secrets: load desired for %q: %w", app, err) + } + if ok { + for _, svc := range desired.Spec.Services { + entries = appendSecretMetadata(entries, service, svc.Name, svc.Secrets, "desired") + } + } + active, ok, err := s.store.LoadActive(ctx, app) + if err != nil { + return nil, fmt.Errorf("list app secrets: load active for %q: %w", app, err) + } + if ok { + for svcName, svc := range active.Services { + entries = appendSecretMetadata(entries, service, svcName, svc.Spec.Secrets, "active") + } + } + sort.Slice(entries, func(i, j int) bool { + a, b := entries[i], entries[j] + if a.Service != b.Service { + return a.Service < b.Service + } + if a.Key != b.Key { + return a.Key < b.Key + } + if a.Source != b.Source { + return a.Source < b.Source + } + return a.Name < b.Name + }) + return entries, nil +} + +func appendSecretMetadata(entries []in.AppSecretMetadata, filter, service string, secrets map[string]string, source string) []in.AppSecretMetadata { + if filter != "" && filter != service { + return entries + } + for key, name := range secrets { + entries = append(entries, in.AppSecretMetadata{Service: service, Key: key, Name: name, Source: source, Presence: "unknown"}) + } + return entries +} + +// SetSecrets implements in.AppService. +func (s *AppServiceImpl) SetSecrets(ctx context.Context, app, service string, values map[string]string) error { + if service == "" { + return fmt.Errorf("apps: service is required: %w", domain.ErrInvalidAppSpec) + } + registered, err := s.secretNames(ctx, app, service) + if err != nil { + return err + } + for key, value := range values { + name, ok := registered[key] + if !ok { + return fmt.Errorf("apps: secret %q not registered for %s/%s: %w", key, app, service, domain.ErrAppSecretMissing) + } + path, err := s.secretPath(ctx, app, service, name) + if err != nil { + return err + } + if err := s.secrets.SetSecret(ctx, path, value); err != nil { + return fmt.Errorf("apps: write secret %q: %w", key, err) + } + } + return nil +} + +// DeleteSecret implements in.AppService. +func (s *AppServiceImpl) DeleteSecret(ctx context.Context, app, service, key string) error { + if service == "" || key == "" { + return fmt.Errorf("apps: service and key are required: %w", domain.ErrInvalidAppSpec) + } + registered, err := s.secretNames(ctx, app, service) + if err != nil { + return err + } + name, ok := registered[key] + if !ok { + return fmt.Errorf("apps: secret %q not registered for %s/%s: %w", key, app, service, domain.ErrAppSecretMissing) + } + // Refused while the name is still referenced by desired or active. + if referenced, err := s.secretReferenced(ctx, app, service, name); err != nil { + return err + } else if referenced { + return fmt.Errorf("apps: secret %q still referenced: %w", key, domain.ErrAppStateConflict) + } + path, err := s.secretPath(ctx, app, service, name) + if err != nil { + return err + } + return s.secrets.DeleteSecret(ctx, path) +} + +// secretPath returns the UUID-keyed pass path for one secret name. +// Empty ID means legacy name-keyed paths (records predating UUIDs). +func (s *AppServiceImpl) secretPath(ctx context.Context, app, service, name string) (string, error) { + ownership, err := s.store.LoadOwnership(ctx, app) + if err != nil { + return "", err + } + return domain.AppSecretPathForID(ownership.ID, app, service, name), nil +} + +// secretNames returns env-key → secret-name registrations. +func (s *AppServiceImpl) secretNames(ctx context.Context, app, service string) (map[string]string, error) { + merged := map[string]string{} + desired, ok, err := s.store.LoadDesired(ctx, app) + if err != nil { + return nil, err + } + if ok { + for _, svc := range desired.Spec.Services { + if svc.Name == service { + for envKey, name := range svc.Secrets { + merged[envKey] = name + } + } + } + } + active, ok, err := s.store.LoadActive(ctx, app) + if err != nil { + return nil, err + } + if ok { + for name, svc := range active.Services { + if name == service { + for envKey, secretName := range svc.Spec.Secrets { + merged[envKey] = secretName + } + } + } + } + if len(merged) == 0 { + return nil, fmt.Errorf("apps: service %q has no registered secrets: %w", service, domain.ErrAppSecretMissing) + } + return merged, nil +} + +// secretReferenced reports desired/active references to a secret name. +func (s *AppServiceImpl) secretReferenced(ctx context.Context, app, service, name string) (bool, error) { + desired, ok, err := s.store.LoadDesired(ctx, app) + if err != nil { + return false, err + } + if ok { + for _, svc := range desired.Spec.Services { + if svc.Name != service { + continue + } + for _, secretName := range svc.Secrets { + if secretName == name { + return true, nil + } + } + } + } + active, ok, err := s.store.LoadActive(ctx, app) + if err != nil { + return false, err + } + if ok { + for svcName, svc := range active.Services { + if svcName != service { + continue + } + for _, secretName := range svc.Spec.Secrets { + if secretName == name { + return true, nil + } + } + } + } + return false, nil +} diff --git a/internal/usecase/apps/app_service_test.go b/internal/usecase/apps/app_service_test.go new file mode 100644 index 000000000..8b47fc17b --- /dev/null +++ b/internal/usecase/apps/app_service_test.go @@ -0,0 +1,123 @@ +package apps_test + +import ( + "context" + "errors" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/apps" +) + +func secretSpec() domain.AppSpec { + return domain.AppSpec{ + Name: "blog", + Services: []domain.AppService{ + { + Name: "web", + Image: "img:1", + StopGrace: 10 * time.Second, + Readiness: domain.AppReadiness{Type: "none", Timeout: 30 * time.Second}, + HTTP: []domain.AppHTTPInterface{{Host: "blog.example.com", Port: 8080, TLS: "auto"}}, + Secrets: map[string]string{"DATABASE_URL": "database-url"}, + }, + }, + } +} + +func TestAppServiceImpl_SecretsRoundTrip(t *testing.T) { + ctx := context.Background() + store := newMockAppState(t) + deploy := newMockDeployEngine(t) + secrets := newMockSecretWriter(t) + + // Apply persists desired state through the store mock. + store.EXPECT().Recover(mock.Anything).Return(nil).Once() + store.EXPECT().LoadCheckpoint(mock.Anything).Return(domain.AppStoreCheckpoint{}, nil).Once() + store.EXPECT().LoadDesired(mock.Anything, "blog").Return(domain.AppDesiredRevision{}, false, nil).Once() + store.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{}, false, nil).Once() + store.EXPECT().AcceptApply(mock.Anything, mock.Anything).Return(nil).Once() + + svc := apps.NewAppServiceImpl(store, deploy, secrets, zerowrap.Default()) + _, _, err := svc.Apply(ctx, secretSpec(), []byte("m"), false) + require.NoError(t, err) + + // SetSecrets resolves registered names from desired state. + desired := domain.AppDesiredRevision{App: "blog", Revision: "rev-1", Spec: secretSpec()} + store.EXPECT().LoadDesired(mock.Anything, "blog").Return(desired, true, nil).Once() + store.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{}, false, nil).Once() + store.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-uuid-1"}, nil).Once() + secrets.EXPECT().SetSecret(mock.Anything, "gordon/apps/app-uuid-1/web/database-url", "v").Return(nil).Once() + require.NoError(t, svc.SetSecrets(ctx, "blog", "web", map[string]string{"DATABASE_URL": "v"})) + + // Delete refused while still referenced (names + reference check). + desiredWithSecret := desired + store.EXPECT().LoadDesired(mock.Anything, "blog").Return(desiredWithSecret, true, nil).Once() + store.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{}, false, nil).Once() + store.EXPECT().LoadDesired(mock.Anything, "blog").Return(desiredWithSecret, true, nil).Once() + require.ErrorIs(t, svc.DeleteSecret(ctx, "blog", "web", "DATABASE_URL"), domain.ErrAppStateConflict) + + // Unknown key refused. + store.EXPECT().LoadDesired(mock.Anything, "blog").Return(desired, true, nil).Once() + store.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{}, false, nil).Once() + require.ErrorIs(t, svc.SetSecrets(ctx, "blog", "web", map[string]string{"NOPE": "v"}), domain.ErrAppSecretMissing) +} + +func TestAppServiceImpl_OperationByKey(t *testing.T) { + ctx := context.Background() + store := newMockAppState(t) + deploy := newMockDeployEngine(t) + secrets := newMockSecretWriter(t) + svc := apps.NewAppServiceImpl(store, deploy, secrets, zerowrap.Default()) + + op := domain.AppOperation{Op: "op-1", Kind: "deploy", App: "blog", StartedAt: time.Now().UTC()} + store.EXPECT().LoadOperation(mock.Anything, "blog", "op-1").Return(op, nil).Once() + loaded, err := svc.OperationByKey(ctx, "blog", "op-1") + require.NoError(t, err) + assert.Equal(t, "op-1", loaded.Op) + + store.EXPECT().LoadOperation(mock.Anything, "blog", "missing").Return(domain.AppOperation{}, domain.ErrAppOperationNotFound).Once() + _, err = svc.OperationByKey(ctx, "blog", "missing") + require.ErrorIs(t, err, domain.ErrAppOperationNotFound) +} + +func TestAppServiceImpl_ListSecrets_MissingAppNotFound(t *testing.T) { + ctx := context.Background() + store := newMockAppState(t) + svc := apps.NewAppServiceImpl(store, newMockDeployEngine(t), newMockSecretWriter(t), zerowrap.Default()) + + store.EXPECT().AppExists(mock.Anything, "blog").Return(false, nil).Once() + _, err := svc.ListSecrets(ctx, "blog", "") + require.ErrorIs(t, err, domain.ErrAppNotFound) +} + +func TestAppServiceImpl_ListSecrets_OwnershipOnlyAppStaysLive(t *testing.T) { + ctx := context.Background() + store := newMockAppState(t) + svc := apps.NewAppServiceImpl(store, newMockDeployEngine(t), newMockSecretWriter(t), zerowrap.Default()) + + store.EXPECT().AppExists(mock.Anything, "blog").Return(true, nil).Once() + store.EXPECT().LoadDesired(mock.Anything, "blog").Return(domain.AppDesiredRevision{}, false, nil).Once() + store.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{}, false, nil).Once() + entries, err := svc.ListSecrets(ctx, "blog", "") + require.NoError(t, err) + assert.Empty(t, entries) +} + +func TestAppServiceImpl_ListSecrets_WrapsLoadErrors(t *testing.T) { + ctx := context.Background() + store := newMockAppState(t) + svc := apps.NewAppServiceImpl(store, newMockDeployEngine(t), newMockSecretWriter(t), zerowrap.Default()) + + store.EXPECT().AppExists(mock.Anything, "blog").Return(true, nil).Once() + store.EXPECT().LoadDesired(mock.Anything, "blog").Return(domain.AppDesiredRevision{}, false, errors.New("boom")).Once() + _, err := svc.ListSecrets(ctx, "blog", "") + require.ErrorContains(t, err, "load desired") + require.ErrorContains(t, err, "boom") +} diff --git a/internal/usecase/apps/bind_policy_test.go b/internal/usecase/apps/bind_policy_test.go new file mode 100644 index 000000000..97281dafe --- /dev/null +++ b/internal/usecase/apps/bind_policy_test.go @@ -0,0 +1,133 @@ +package apps_test + +import ( + "context" + "os" + "path/filepath" + "testing" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/apps" +) + +func bindPolicy(root, source string, apps, services []string) domain.AppBindPolicy { + return domain.AppBindPolicy{ + Name: "config", + Source: source, + Root: root, + AllowedApps: apps, + AllowedServices: services, + } +} + +func TestApply_AcceptsBindsUnderAdministrativePolicy(t *testing.T) { + root := t.TempDir() + source := filepath.Join(root, "binds") + require.NoError(t, os.MkdirAll(source, 0o755)) + + store := outmocks.NewMockAppState(t) + spec := testSpec("blog") + spec.Services[0].Binds = []domain.AppBind{{Name: "config", Path: "/etc/app.conf"}} + applySuccess(store, spec) + + svc := apps.NewService(store, zerowrap.Default()). + WithBindPolicies(map[string]domain.AppBindPolicy{ + "config": bindPolicy(root, source, []string{"blog"}, []string{"web"}), + }) + _, _, err := svc.Apply(context.Background(), spec, []byte("manifest"), false) + require.NoError(t, err) +} + +func TestApply_RejectsBindWithoutPolicyBeforeStoreAccess(t *testing.T) { + store := outmocks.NewMockAppState(t) + spec := testSpec("blog") + spec.Services[0].Binds = []domain.AppBind{{Name: "config", Path: "/etc/app.conf"}} + svc := apps.NewService(store, zerowrap.Default()) + + _, _, err := svc.Apply(context.Background(), spec, []byte("manifest"), false) + + require.ErrorIs(t, err, domain.ErrBindPolicy) + assert.Contains(t, err.Error(), "blog") + assert.Contains(t, err.Error(), "web") + assert.Contains(t, err.Error(), "config") +} + +func TestApply_RejectsBindOutsideAllowlists(t *testing.T) { + root := t.TempDir() + source := filepath.Join(root, "binds") + require.NoError(t, os.MkdirAll(source, 0o755)) + spec := testSpec("blog") + spec.Services[0].Binds = []domain.AppBind{{Name: "config", Path: "/etc/app.conf"}} + + t.Run("app not allowed", func(t *testing.T) { + store := outmocks.NewMockAppState(t) + svc := apps.NewService(store, zerowrap.Default()). + WithBindPolicies(map[string]domain.AppBindPolicy{ + "config": bindPolicy(root, source, []string{"other"}, []string{"web"}), + }) + _, _, err := svc.Apply(context.Background(), spec, []byte("manifest"), false) + require.ErrorIs(t, err, domain.ErrBindPolicy) + assert.NotContains(t, err.Error(), source) + }) + + t.Run("service not allowed", func(t *testing.T) { + store := outmocks.NewMockAppState(t) + svc := apps.NewService(store, zerowrap.Default()). + WithBindPolicies(map[string]domain.AppBindPolicy{ + "config": bindPolicy(root, source, []string{"blog"}, []string{"other"}), + }) + _, _, err := svc.Apply(context.Background(), spec, []byte("manifest"), false) + require.ErrorIs(t, err, domain.ErrBindPolicy) + assert.NotContains(t, err.Error(), source) + }) +} + +func TestApply_RejectsBindSourceOutsideRoot(t *testing.T) { + root := t.TempDir() + outside := t.TempDir() + source := filepath.Join(outside, "binds") + require.NoError(t, os.MkdirAll(source, 0o755)) + store := outmocks.NewMockAppState(t) + spec := testSpec("blog") + spec.Services[0].Binds = []domain.AppBind{{Name: "config", Path: "/etc/app.conf"}} + svc := apps.NewService(store, zerowrap.Default()). + WithBindPolicies(map[string]domain.AppBindPolicy{ + "config": bindPolicy(root, source, []string{"blog"}, []string{"web"}), + }) + + _, _, err := svc.Apply(context.Background(), spec, []byte("manifest"), false) + + require.ErrorIs(t, err, domain.ErrBindPolicy) + assert.NotContains(t, err.Error(), source, "apply errors must never leak the policy source path") + assert.NotContains(t, err.Error(), outside) +} + +// TestApply_BindPoliciesReplaceOnReload proves a reload-style replacement takes +// effect immediately: the second apply is refused before any store access. +func TestApply_BindPoliciesReplaceOnReload(t *testing.T) { + root := t.TempDir() + source := filepath.Join(root, "binds") + require.NoError(t, os.MkdirAll(source, 0o755)) + spec := testSpec("blog") + spec.Services[0].Binds = []domain.AppBind{{Name: "config", Path: "/etc/app.conf"}} + + store := outmocks.NewMockAppState(t) + applySuccess(store, spec) + svc := apps.NewService(store, zerowrap.Default()). + WithBindPolicies(map[string]domain.AppBindPolicy{ + "config": bindPolicy(root, source, []string{"blog"}, []string{"web"}), + }) + _, _, err := svc.Apply(context.Background(), spec, []byte("manifest"), false) + require.NoError(t, err) + + svc.SetBindPolicies(map[string]domain.AppBindPolicy{ + "config": bindPolicy(root, source, []string{"other"}, []string{"web"}), + }) + _, _, err = svc.Apply(context.Background(), spec, []byte("manifest"), false) + require.ErrorIs(t, err, domain.ErrBindPolicy) +} diff --git a/internal/usecase/apps/deploy_lifecycle_test.go b/internal/usecase/apps/deploy_lifecycle_test.go new file mode 100644 index 000000000..746f20ce2 --- /dev/null +++ b/internal/usecase/apps/deploy_lifecycle_test.go @@ -0,0 +1,387 @@ +package apps_test + +import ( + "context" + "errors" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/apps" + "github.com/bnema/gordon/internal/usecase/deployment" +) + +type deployCall struct { + op *domain.AppOperation + err error +} + +// callDeploy runs Deploy off the test goroutine so a blocking execution can +// never hang the test: only a prompt return is awaited. +func callDeploy(t *testing.T, svc *apps.AppServiceImpl, ctx context.Context, key string) <-chan deployCall { + t.Helper() + out := make(chan deployCall, 1) + go func() { + op, err := svc.Deploy(ctx, "blog", "rev-1", "web", false, key) + out <- deployCall{op: op, err: err} + }() + return out +} + +func awaitDeploy(t *testing.T, ch <-chan deployCall) deployCall { + t.Helper() + select { + case res := <-ch: + return res + case <-time.After(3 * time.Second): + t.Fatal("Deploy did not return promptly") + return deployCall{} + } +} + +// gate returns a release channel and an idempotent closer, so a blocking test +// execution is always unblocked before Shutdown waits on it. +func gate() (chan struct{}, func()) { + ch := make(chan struct{}) + var once sync.Once + return ch, func() { once.Do(func() { close(ch) }) } +} + +func runningDeployOp() domain.AppOperation { + return domain.AppOperation{Op: "op-1", Kind: "deploy", App: "blog", InputRevision: "rev-1", StartedAt: time.Now().UTC()} +} + +func ownedStart(op domain.AppOperation) *deployment.StartDeployResult { + return &deployment.StartDeployResult{ + Claim: deployment.DeployClaim{App: op.App, Op: op.Op, Service: "web", Revision: "rev-1", Journal: op}, + Owned: true, + } +} + +func replayStart(op domain.AppOperation) *deployment.StartDeployResult { + return &deployment.StartDeployResult{ + Claim: deployment.DeployClaim{App: op.App, Op: op.Op, Service: "web", Revision: "rev-1", Journal: op}, + Owned: false, + } +} + +// TestAppServiceImpl_Deploy_ReturnsRunningJournalAfterPersistence proves the +// request path is two-phase: StartDeploy has already journaled the operation +// when the background execution starts, and Deploy returns that non-terminal +// journal without waiting for the replacement. +func TestAppServiceImpl_Deploy_ReturnsRunningJournalAfterPersistence(t *testing.T) { + ctx := context.Background() + store := newMockAppState(t) + deploy := newMockDeployEngine(t) + svc := apps.NewAppServiceImpl(store, deploy, newMockSecretWriter(t), zerowrap.Default()) + t.Cleanup(func() { require.NoError(t, svc.Shutdown(context.Background())) }) + + op := runningDeployOp() + store.EXPECT().LoadOperation(mock.Anything, "blog", "op-1").Return(op, nil) + + startInputs := make(chan deployment.DeployInput, 1) + deploy.startDeployFn = func(_ context.Context, input deployment.DeployInput) (*deployment.StartDeployResult, error) { + startInputs <- input + return ownedStart(op), nil + } + + persistedBeforeExecution := make(chan error, 1) + release, releaseExecution := gate() + t.Cleanup(releaseExecution) + deploy.executeDeployFn = func(execCtx context.Context, claim deployment.DeployClaim) (*deployment.DeployResult, error) { + // The claimed journal must be durable and queryable before any effect. + stored, err := store.LoadOperation(execCtx, claim.App, claim.Op) + if err == nil && (stored.Op != claim.Op || stored.Terminal()) { + err = errors.New("claimed journal is not the persisted running operation") + } + persistedBeforeExecution <- err + <-release + return &deployment.DeployResult{Op: claim.Op, App: claim.App}, nil + } + + res := awaitDeploy(t, callDeploy(t, svc, ctx, "key-1")) + require.NoError(t, res.err) + require.NotNil(t, res.op) + assert.Equal(t, "op-1", res.op.Op) + assert.False(t, res.op.Terminal(), "Deploy returns the still-running journal") + assert.Equal(t, deployment.DeployInput{App: "blog", Revision: "rev-1", Service: "web", Op: "key-1"}, <-startInputs) + + select { + case err := <-persistedBeforeExecution: + require.NoError(t, err, "execution must start only after the journal is persisted") + case <-time.After(3 * time.Second): + t.Fatal("background execution never started") + } + releaseExecution() +} + +// TestAppServiceImpl_Deploy_RequestCancellationDoesNotCancelExecution proves +// the execution runs on the daemon context, never on the request context. +func TestAppServiceImpl_Deploy_RequestCancellationDoesNotCancelExecution(t *testing.T) { + requestCtx, cancelRequest := context.WithCancel(context.Background()) + defer cancelRequest() + store := newMockAppState(t) + deploy := newMockDeployEngine(t) + svc := apps.NewAppServiceImpl(store, deploy, newMockSecretWriter(t), zerowrap.Default()) + t.Cleanup(func() { require.NoError(t, svc.Shutdown(context.Background())) }) + + op := runningDeployOp() + store.EXPECT().LoadOperation(mock.Anything, "blog", "op-1").Return(op, nil) + deploy.startDeployFn = func(context.Context, deployment.DeployInput) (*deployment.StartDeployResult, error) { + return ownedStart(op), nil + } + + execCtxCh := make(chan context.Context, 1) + release, releaseExecution := gate() + t.Cleanup(releaseExecution) + deploy.executeDeployFn = func(execCtx context.Context, _ deployment.DeployClaim) (*deployment.DeployResult, error) { + execCtxCh <- execCtx + <-release + return nil, nil + } + + res := awaitDeploy(t, callDeploy(t, svc, requestCtx, "key-1")) + require.NoError(t, res.err) + require.NotNil(t, res.op) + + execCtx := <-execCtxCh + cancelRequest() + select { + case <-requestCtx.Done(): + default: + t.Fatal("request context was not cancelled") + } + require.NotEqual(t, requestCtx, execCtx, "execution must not run on the request context") + select { + case <-execCtx.Done(): + t.Fatal("request cancellation cancelled the in-flight execution") + case <-time.After(50 * time.Millisecond): + } + releaseExecution() +} + +// TestAppServiceImpl_Deploy_DuplicateKeyNeverSpawnsSecondExecution proves the +// claimed key is the only execution: a replay returns the stored journal, an +// in-flight replay reports a conflict, and ExecuteDeploy runs exactly once. +func TestAppServiceImpl_Deploy_DuplicateKeyNeverSpawnsSecondExecution(t *testing.T) { + ctx := context.Background() + store := newMockAppState(t) + deploy := newMockDeployEngine(t) + svc := apps.NewAppServiceImpl(store, deploy, newMockSecretWriter(t), zerowrap.Default()) + t.Cleanup(func() { require.NoError(t, svc.Shutdown(context.Background())) }) + + op := runningDeployOp() + store.EXPECT().LoadOperation(mock.Anything, "blog", "op-1").Return(op, nil) + + var startCalls atomic.Int32 + deploy.startDeployFn = func(context.Context, deployment.DeployInput) (*deployment.StartDeployResult, error) { + if startCalls.Add(1) == 1 { + return ownedStart(op), nil + } + return replayStart(op), nil + } + var execCalls atomic.Int32 + execStarted := make(chan struct{}) + release, releaseExecution := gate() + t.Cleanup(releaseExecution) + deploy.executeDeployFn = func(context.Context, deployment.DeployClaim) (*deployment.DeployResult, error) { + execCalls.Add(1) + close(execStarted) + <-release + return &deployment.DeployResult{Op: "op-1"}, nil + } + + first := awaitDeploy(t, callDeploy(t, svc, ctx, "key-1")) + require.NoError(t, first.err) + require.NotNil(t, first.op) + select { + case <-execStarted: + case <-time.After(3 * time.Second): + t.Fatal("first execution never started") + } + + second := awaitDeploy(t, callDeploy(t, svc, ctx, "key-1")) + require.ErrorIs(t, second.err, domain.ErrAppStateConflict, "an in-flight replay is never a second execution") + require.NotNil(t, second.op) + assert.Equal(t, "op-1", second.op.Op) + assert.Equal(t, int32(1), execCalls.Load(), "a duplicate key never spawns a second execution") + releaseExecution() +} + +// TestAppServiceImpl_Deploy_TerminalFailureRemainsQueryable proves a failed +// execution is recorded once, stays readable by key, and its replay reports +// the stored failure without executing again. +func TestAppServiceImpl_Deploy_TerminalFailureRemainsQueryable(t *testing.T) { + ctx := context.Background() + store := newMockAppState(t) + deploy := newMockDeployEngine(t) + svc := apps.NewAppServiceImpl(store, deploy, newMockSecretWriter(t), zerowrap.Default()) + + failed := runningDeployOp() + failed.Outcome = domain.AppOutcomeFailed + store.EXPECT().LoadOperation(mock.Anything, "blog", "op-1").Return(failed, nil) + + var startCalls atomic.Int32 + deploy.startDeployFn = func(context.Context, deployment.DeployInput) (*deployment.StartDeployResult, error) { + if startCalls.Add(1) == 1 { + return ownedStart(runningDeployOp()), nil + } + return replayStart(failed), nil + } + var execCalls atomic.Int32 + executed := make(chan struct{}, 1) + deploy.executeDeployFn = func(context.Context, deployment.DeployClaim) (*deployment.DeployResult, error) { + execCalls.Add(1) + executed <- struct{}{} + return nil, errors.New("replacement failed") + } + + res := awaitDeploy(t, callDeploy(t, svc, ctx, "key-1")) + require.NoError(t, res.err, "a failed execution is reported through the journal, not the start call") + require.NotNil(t, res.op) + assert.True(t, res.op.Terminal()) + // Join the background execution before asserting the journal, so the + // count is deterministic rather than a race with the goroutine. + <-executed + assert.Equal(t, int32(1), execCalls.Load()) + + // A terminal replay stays queryable and never executes again, even + // though it is answered as a conflict. + retry := awaitDeploy(t, callDeploy(t, svc, ctx, "key-1")) + require.ErrorIs(t, retry.err, domain.ErrAppStateConflict) + require.NotNil(t, retry.op) + assert.Equal(t, domain.AppOutcomeFailed, retry.op.Outcome) + assert.Equal(t, int32(1), execCalls.Load(), "a terminal replay never executes again") + + require.NoError(t, svc.Shutdown(context.Background())) + + got, err := svc.OperationByKey(ctx, "blog", "op-1") + require.NoError(t, err) + assert.Equal(t, domain.AppOutcomeFailed, got.Outcome, "terminal failure remains queryable by key") +} + +// TestAppServiceImpl_Deploy_ShutdownLeavesRecoverableNonTerminalJournal proves +// graceful shutdown cancels the execution and leaves the journal persisted +// non-terminal for reconciliation, instead of losing or finalizing it. +func TestAppServiceImpl_Deploy_ShutdownLeavesRecoverableNonTerminalJournal(t *testing.T) { + ctx := context.Background() + store := newMockAppState(t) + deploy := newMockDeployEngine(t) + svc := apps.NewAppServiceImpl(store, deploy, newMockSecretWriter(t), zerowrap.Default()). + WithDaemonContext(context.Background()) + + op := runningDeployOp() + store.EXPECT().LoadOperation(mock.Anything, "blog", "op-1").Return(op, nil) + deploy.startDeployFn = func(context.Context, deployment.DeployInput) (*deployment.StartDeployResult, error) { + return ownedStart(op), nil + } + + execCtxCh := make(chan context.Context, 1) + deploy.executeDeployFn = func(execCtx context.Context, _ deployment.DeployClaim) (*deployment.DeployResult, error) { + execCtxCh <- execCtx + <-execCtx.Done() + return nil, execCtx.Err() + } + + res := awaitDeploy(t, callDeploy(t, svc, ctx, "key-1")) + require.NoError(t, res.err) + require.NotNil(t, res.op) + + execCtx := <-execCtxCh + require.NoError(t, svc.Shutdown(context.Background()), "Shutdown waits for the execution to unwind") + require.ErrorIs(t, execCtx.Err(), context.Canceled, "shutdown cancels the in-flight execution") + + got, err := svc.OperationByKey(ctx, "blog", "op-1") + require.NoError(t, err) + assert.False(t, got.Terminal(), "shutdown must leave the non-terminal journal for reconciliation") + assert.Equal(t, "op-1", got.Op) +} + +// TestAppServiceImpl_Deploy_RefusesAfterShutdownBeforeClaim proves a deploy +// that arrives after shutdown began is refused with a state conflict before +// the engine can claim it, so neither a durable 202-running journal nor a +// workload effect can be produced. The engine mock has no expectations, so +// any StartDeploy or ExecuteDeploy call fails the test. +func TestAppServiceImpl_Deploy_RefusesAfterShutdownBeforeClaim(t *testing.T) { + ctx := context.Background() + store := newMockAppState(t) + deploy := newMockDeployEngine(t) + svc := apps.NewAppServiceImpl(store, deploy, newMockSecretWriter(t), zerowrap.Default()) + + require.NoError(t, svc.Shutdown(context.Background())) + + res := awaitDeploy(t, callDeploy(t, svc, ctx, "key-1")) + require.ErrorIs(t, res.err, domain.ErrAppStateConflict, "a deploy after shutdown is refused with a state conflict") + require.Nil(t, res.op, "a refused deploy never produces a journal to answer 202") +} + +// TestAppServiceImpl_Deploy_ShutdownRacingClaimSettlesWithoutExecuting proves +// the claim/schedule hand-off is closed against shutdown: once Shutdown starts +// after StartDeploy claimed the operation but before it was scheduled, the +// owned claim is settled without starting, Shutdown waits for that settle, and +// the request answers a conflict instead of a false 202-running orphan. +func TestAppServiceImpl_Deploy_ShutdownRacingClaimSettlesWithoutExecuting(t *testing.T) { + ctx := context.Background() + store := newMockAppState(t) + deploy := newMockDeployEngine(t) + svc := apps.NewAppServiceImpl(store, deploy, newMockSecretWriter(t), zerowrap.Default()) + + op := runningDeployOp() + store.EXPECT().LoadOperation(mock.Anything, "blog", "op-1").Return(op, nil) + + claimed := make(chan struct{}) + releaseClaim := make(chan struct{}) + deploy.startDeployFn = func(context.Context, deployment.DeployInput) (*deployment.StartDeployResult, error) { + close(claimed) + <-releaseClaim + return ownedStart(op), nil + } + var execCalls atomic.Int32 + deploy.executeDeployFn = func(context.Context, deployment.DeployClaim) (*deployment.DeployResult, error) { + execCalls.Add(1) + return nil, nil + } + abandoned := make(chan deployment.DeployClaim, 1) + deploy.abandonDeployFn = func(_ context.Context, claim deployment.DeployClaim) error { + abandoned <- claim + return nil + } + + resCh := callDeploy(t, svc, ctx, "key-1") + <-claimed + + shutdownDone := make(chan error, 1) + go func() { shutdownDone <- svc.Shutdown(context.Background()) }() + + // Shutdown must wait for the hand-off rather than return while the claim + // is still being made. + select { + case <-shutdownDone: + t.Fatal("Shutdown returned before the in-flight claim hand-off settled") + case <-time.After(50 * time.Millisecond): + } + + // The claim completes now; shutdown is already set, so it must be settled + // rather than executed. + close(releaseClaim) + + select { + case claim := <-abandoned: + assert.Equal(t, "op-1", claim.Op) + case <-time.After(3 * time.Second): + t.Fatal("the raced claim was not settled") + } + require.NoError(t, <-shutdownDone, "Shutdown waits for the settle and returns cleanly") + + res := awaitDeploy(t, resCh) + require.ErrorIs(t, res.err, domain.ErrAppStateConflict, "a settled claim answers a conflict, never 202-running") + require.NotNil(t, res.op) + assert.Equal(t, int32(0), execCalls.Load(), "a claim settled during shutdown never executes") +} diff --git a/internal/usecase/apps/device_policy_test.go b/internal/usecase/apps/device_policy_test.go new file mode 100644 index 000000000..56b4d4f24 --- /dev/null +++ b/internal/usecase/apps/device_policy_test.go @@ -0,0 +1,130 @@ +package apps_test + +import ( + "context" + "testing" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/apps" +) + +func devicePolicy(apps, services []string) domain.AppDevicePolicy { + return domain.AppDevicePolicy{ + Name: "test_gpu", + CDI: []string{"example.com/gpu=GPU-test-uuid"}, + AllowedApps: apps, + AllowedServices: services, + } +} + +func TestApply_AcceptsDevicesUnderAdministrativePolicy(t *testing.T) { + store := outmocks.NewMockAppState(t) + spec := testSpec("blog") + spec.Services[0].Devices = []string{"test_gpu"} + applySuccess(store, spec) + + svc := apps.NewService(store, zerowrap.Default()). + WithDevicePolicies(map[string]domain.AppDevicePolicy{ + "test_gpu": devicePolicy([]string{"blog"}, []string{"web"}), + }) + _, _, err := svc.Apply(context.Background(), spec, []byte("manifest"), false) + require.NoError(t, err) +} + +func TestApply_RejectsDeviceWithoutPolicyBeforeStoreAccess(t *testing.T) { + store := outmocks.NewMockAppState(t) + spec := testSpec("blog") + spec.Services[0].Devices = []string{"test_gpu"} + svc := apps.NewService(store, zerowrap.Default()) + + _, _, err := svc.Apply(context.Background(), spec, []byte("manifest"), false) + + require.ErrorIs(t, err, domain.ErrDevicePolicy) + assert.Contains(t, err.Error(), "blog") + assert.Contains(t, err.Error(), "web") + assert.Contains(t, err.Error(), "test_gpu") +} + +func TestApply_RejectsDeviceOutsideAllowlists(t *testing.T) { + spec := testSpec("blog") + spec.Services[0].Devices = []string{"test_gpu"} + + t.Run("app not allowed", func(t *testing.T) { + store := outmocks.NewMockAppState(t) + svc := apps.NewService(store, zerowrap.Default()). + WithDevicePolicies(map[string]domain.AppDevicePolicy{ + "test_gpu": devicePolicy([]string{"other"}, []string{"web"}), + }) + _, _, err := svc.Apply(context.Background(), spec, []byte("manifest"), false) + require.ErrorIs(t, err, domain.ErrDevicePolicy) + assert.NotContains(t, err.Error(), "GPU-test-uuid") + }) + + t.Run("service not allowed", func(t *testing.T) { + store := outmocks.NewMockAppState(t) + svc := apps.NewService(store, zerowrap.Default()). + WithDevicePolicies(map[string]domain.AppDevicePolicy{ + "test_gpu": devicePolicy([]string{"blog"}, []string{"other"}), + }) + _, _, err := svc.Apply(context.Background(), spec, []byte("manifest"), false) + require.ErrorIs(t, err, domain.ErrDevicePolicy) + assert.NotContains(t, err.Error(), "GPU-test-uuid") + }) +} + +// TestApply_DevicePoliciesReplaceOnReload proves a reload-style replacement +// takes effect immediately: the second apply is refused before any store +// access. +func TestApply_DevicePoliciesReplaceOnReload(t *testing.T) { + spec := testSpec("blog") + spec.Services[0].Devices = []string{"test_gpu"} + + store := outmocks.NewMockAppState(t) + applySuccess(store, spec) + svc := apps.NewService(store, zerowrap.Default()). + WithDevicePolicies(map[string]domain.AppDevicePolicy{ + "test_gpu": devicePolicy([]string{"blog"}, []string{"web"}), + }) + _, _, err := svc.Apply(context.Background(), spec, []byte("manifest"), false) + require.NoError(t, err) + + svc.SetDevicePolicies(map[string]domain.AppDevicePolicy{ + "test_gpu": devicePolicy([]string{"other"}, []string{"web"}), + }) + _, _, err = svc.Apply(context.Background(), spec, []byte("manifest"), false) + require.ErrorIs(t, err, domain.ErrDevicePolicy) +} + +func TestApply_DeviceDryRunRefusedWithoutMutation(t *testing.T) { + store := outmocks.NewMockAppState(t) + spec := testSpec("blog") + spec.Services[0].Devices = []string{"test_gpu"} + svc := apps.NewService(store, zerowrap.Default()) + + _, _, err := svc.Apply(context.Background(), spec, []byte("manifest"), true) + require.ErrorIs(t, err, domain.ErrDevicePolicy) + store.AssertNotCalled(t, "AcceptApply", mock.Anything, mock.Anything) +} + +func TestApply_RejectsDuplicateCDIIDsAcrossDevices(t *testing.T) { + shared := "example.com/gpu=GPU-shared" + store := outmocks.NewMockAppState(t) + spec := testSpec("blog") + spec.Services[0].Devices = []string{"gpu-a", "gpu-b"} + svc := apps.NewService(store, zerowrap.Default()). + WithDevicePolicies(map[string]domain.AppDevicePolicy{ + "gpu-a": {Name: "gpu-a", CDI: []string{shared}, AllowedApps: []string{"blog"}, AllowedServices: []string{"web"}}, + "gpu-b": {Name: "gpu-b", CDI: []string{shared}, AllowedApps: []string{"blog"}, AllowedServices: []string{"web"}}, + }) + + _, _, err := svc.Apply(context.Background(), spec, []byte("manifest"), false) + + require.ErrorIs(t, err, domain.ErrDevicePolicy) + assert.NotContains(t, err.Error(), shared, "apply errors must never leak CDI IDs") +} diff --git a/internal/usecase/apps/managed_count.go b/internal/usecase/apps/managed_count.go new file mode 100644 index 000000000..2a66a81f5 --- /dev/null +++ b/internal/usecase/apps/managed_count.go @@ -0,0 +1,42 @@ +package apps + +import ( + "context" + "fmt" + + "github.com/bnema/gordon/internal/boundaries/out" +) + +// CountManagedContainers returns the number of containers Gordon currently +// manages: ACTIVE app services bound to a container, excluding apps whose +// durable intent is stopped. It reads persisted state only, so the result is +// correct after boot and after every lifecycle operation, including failures. +func CountManagedContainers(ctx context.Context, state out.AppStateReader) (int64, error) { + names, err := state.ListApps(ctx) + if err != nil { + return 0, fmt.Errorf("failed to list apps: %w", err) + } + var total int64 + for _, name := range names { + intent, err := state.LoadIntent(ctx, name) + if err != nil { + return 0, fmt.Errorf("failed to load intent for app %q: %w", name, err) + } + if intent.Stopped { + continue + } + active, ok, err := state.LoadActive(ctx, name) + if err != nil { + return 0, fmt.Errorf("failed to load active state for app %q: %w", name, err) + } + if !ok { + continue + } + for _, svc := range active.Services { + if svc.Container != "" { + total++ + } + } + } + return total, nil +} diff --git a/internal/usecase/apps/managed_count_test.go b/internal/usecase/apps/managed_count_test.go new file mode 100644 index 000000000..3471e8a77 --- /dev/null +++ b/internal/usecase/apps/managed_count_test.go @@ -0,0 +1,92 @@ +package apps_test + +import ( + "context" + "errors" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/apps" +) + +// fakeAppStateReader is an in-memory AppStateReader for lifecycle scenarios. +type fakeAppStateReader struct { + active map[string]domain.AppActive + stopped map[string]bool + listErr error +} + +func newFakeAppStateReader() *fakeAppStateReader { + return &fakeAppStateReader{active: map[string]domain.AppActive{}, stopped: map[string]bool{}} +} + +func (f *fakeAppStateReader) ListApps(context.Context) ([]string, error) { + if f.listErr != nil { + return nil, f.listErr + } + names := make([]string, 0, len(f.active)) + for name := range f.active { + names = append(names, name) + } + return names, nil +} + +func (f *fakeAppStateReader) LoadActive(_ context.Context, app string) (domain.AppActive, bool, error) { + a, ok := f.active[app] + return a, ok, nil +} + +func (f *fakeAppStateReader) LoadIntent(_ context.Context, app string) (domain.AppStopIntent, error) { + return domain.AppStopIntent{App: app, Stopped: f.stopped[app]}, nil +} + +func activeWith(app string, containers ...string) domain.AppActive { + services := map[string]domain.AppEffectiveService{} + for i, c := range containers { + services[string(rune('a'+i))] = domain.AppEffectiveService{Container: c} + } + return domain.AppActive{App: app, Services: services} +} + +func TestCountManagedContainers_Lifecycle(t *testing.T) { + ctx := context.Background() + state := newFakeAppStateReader() + count := func() int64 { + t.Helper() + n, err := apps.CountManagedContainers(ctx, state) + require.NoError(t, err) + return n + } + + // Boot with no apps. + assert.Equal(t, int64(0), count()) + + // Boot hydration: persisted ACTIVE state is counted immediately. + state.active["web"] = activeWith("web", "c1", "c2") + state.active["db"] = activeWith("db", "c3") + assert.Equal(t, int64(3), count()) + + // Failed deploy of a new service leaves it without a container. + state.active["web"] = activeWith("web", "c1", "c2", "") + assert.Equal(t, int64(3), count()) + + // Stop excludes the app; start restores it. + state.stopped["web"] = true + assert.Equal(t, int64(1), count()) + state.stopped["web"] = false + assert.Equal(t, int64(3), count()) + + // Remove retires the app state. + delete(state.active, "db") + assert.Equal(t, int64(2), count()) +} + +func TestCountManagedContainers_Error(t *testing.T) { + state := newFakeAppStateReader() + state.listErr = errors.New("boom") + _, err := apps.CountManagedContainers(context.Background(), state) + require.ErrorContains(t, err, "boom") +} diff --git a/internal/usecase/apps/mocks_test.go b/internal/usecase/apps/mocks_test.go new file mode 100644 index 000000000..433491368 --- /dev/null +++ b/internal/usecase/apps/mocks_test.go @@ -0,0 +1,106 @@ +package apps_test + +import ( + "context" + "testing" + + "github.com/stretchr/testify/mock" + + "github.com/bnema/gordon/internal/boundaries/out" + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/usecase/deployment" +) + +// mockDeployEngine is a mockery-style test double for the unexported +// apps.deployEngine interface. mockery cannot generate it (unexported), +// so this hand-written mock follows the same EXPECT()/Return() pattern. +type mockDeployEngine struct { + mock.Mock + + // startDeployFn, abandonDeployFn and executeDeployFn override the testify + // expectations when set, for tests that script per-call results or block + // execution. + startDeployFn func(ctx context.Context, input deployment.DeployInput) (*deployment.StartDeployResult, error) + abandonDeployFn func(ctx context.Context, claim deployment.DeployClaim) error + executeDeployFn func(ctx context.Context, claim deployment.DeployClaim) (*deployment.DeployResult, error) +} + +func newMockDeployEngine(t *testing.T) *mockDeployEngine { + m := &mockDeployEngine{} + m.Test(t) + t.Cleanup(func() { m.AssertExpectations(t) }) + return m +} + +func (m *mockDeployEngine) StartDeploy(ctx context.Context, input deployment.DeployInput) (*deployment.StartDeployResult, error) { + if m.startDeployFn != nil { + return m.startDeployFn(ctx, input) + } + args := m.Called(ctx, input) + if args.Get(0) == nil { + return nil, args.Error(1) + } + return args.Get(0).(*deployment.StartDeployResult), args.Error(1) +} + +func (m *mockDeployEngine) AbandonDeploy(ctx context.Context, claim deployment.DeployClaim) error { + if m.abandonDeployFn != nil { + return m.abandonDeployFn(ctx, claim) + } + args := m.Called(ctx, claim) + return args.Error(0) +} + +func (m *mockDeployEngine) ExecuteDeploy(ctx context.Context, claim deployment.DeployClaim) (*deployment.DeployResult, error) { + if m.executeDeployFn != nil { + return m.executeDeployFn(ctx, claim) + } + args := m.Called(ctx, claim) + if args.Get(0) == nil { + return nil, args.Error(1) + } + return args.Get(0).(*deployment.DeployResult), args.Error(1) +} + +func (m *mockDeployEngine) Stop(ctx context.Context, app, opID string) (*deployment.LifecycleResult, error) { + args := m.Called(ctx, app, opID) + if args.Get(0) == nil { + return nil, args.Error(1) + } + return args.Get(0).(*deployment.LifecycleResult), args.Error(1) +} + +func (m *mockDeployEngine) Start(ctx context.Context, app, opID string) (*deployment.LifecycleResult, error) { + args := m.Called(ctx, app, opID) + if args.Get(0) == nil { + return nil, args.Error(1) + } + return args.Get(0).(*deployment.LifecycleResult), args.Error(1) +} + +func (m *mockDeployEngine) Restart(ctx context.Context, app, service, opID string) (*deployment.LifecycleResult, error) { + args := m.Called(ctx, app, service, opID) + if args.Get(0) == nil { + return nil, args.Error(1) + } + return args.Get(0).(*deployment.LifecycleResult), args.Error(1) +} + +func (m *mockDeployEngine) Remove(ctx context.Context, app, opID string) (*deployment.LifecycleResult, error) { + args := m.Called(ctx, app, opID) + if args.Get(0) == nil { + return nil, args.Error(1) + } + return args.Get(0).(*deployment.LifecycleResult), args.Error(1) +} + +func newMockAppState(t *testing.T) *outmocks.MockAppState { + return outmocks.NewMockAppState(t) +} + +func newMockSecretWriter(t *testing.T) *outmocks.MockSecretWriter { + return outmocks.NewMockSecretWriter(t) +} + +var _ out.AppState = (*outmocks.MockAppState)(nil) +var _ out.SecretWriter = (*outmocks.MockSecretWriter)(nil) diff --git a/internal/usecase/apps/service.go b/internal/usecase/apps/service.go new file mode 100644 index 000000000..dd56b337e --- /dev/null +++ b/internal/usecase/apps/service.go @@ -0,0 +1,413 @@ +// Package apps implements the desired-state use case: manifest apply, +// dry-run validation, desired/active inspection, and reservation +// management. It performs zero workload, pull, or secret-value effects; +// all persistence goes through the out.AppState boundary. +package apps + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "fmt" + "sync" + "time" + + "github.com/bnema/zerowrap" + "github.com/google/uuid" + + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/domain" +) + +// Service orchestrates app desired-state transitions. +type Service struct { + store out.AppApplyStore + log zerowrap.Logger + listeners map[string]domain.EntryPointListener + // barrier is the process-wide GC barrier. An apply holds the shared + // lease from validation through durable publication, so prune can + // never snapshot between a staged intent and its materialization. + barrier out.GCBarrier + imagePolicy domain.ImageSourcePolicy + // bindMu guards bindPolicies. Reload replaces the whole map + // atomically; applies read the current map without holding the lock. + bindMu sync.RWMutex + bindPolicies map[string]domain.AppBindPolicy + // deviceMu guards devicePolicies. Reload replaces the whole map + // atomically; applies read the current map without holding the lock. + deviceMu sync.RWMutex + devicePolicies map[string]domain.AppDevicePolicy +} + +// NewService creates the apps use case over an AppState store. +func NewService(store out.AppApplyStore, log zerowrap.Logger) *Service { + return &Service{store: store, log: log} +} + +// WithGCBarrier wires the process-wide GC barrier. Nil is valid for tests +// that never run prune. +func (s *Service) WithGCBarrier(barrier out.GCBarrier) *Service { + s.barrier = barrier + return s +} + +// WithImagePolicy supplies the installation registry policy enforced while +// validating manifests, before any desired state is persisted. +func (s *Service) WithImagePolicy(policy domain.ImageSourcePolicy) *Service { + s.imagePolicy = policy + return s +} + +// WithBindPolicies supplies the initial administrative bind policies. +func (s *Service) WithBindPolicies(policies map[string]domain.AppBindPolicy) *Service { + s.SetBindPolicies(policies) + return s +} + +// SetBindPolicies atomically replaces the administrative bind policies used +// to authorize manifest binds. The map is copied so later caller mutation +// cannot race an in-flight apply. +func (s *Service) SetBindPolicies(policies map[string]domain.AppBindPolicy) { + copied := make(map[string]domain.AppBindPolicy, len(policies)) + for name, policy := range policies { + copied[name] = policy + } + s.bindMu.Lock() + s.bindPolicies = copied + s.bindMu.Unlock() +} + +// snapshotBindPolicies returns the current policy map. SetBindPolicies never +// mutates a published map, so the snapshot stays safe after the lock drops. +func (s *Service) snapshotBindPolicies() map[string]domain.AppBindPolicy { + s.bindMu.RLock() + defer s.bindMu.RUnlock() + return s.bindPolicies +} + +// WithDevicePolicies supplies the initial administrative device policies. +func (s *Service) WithDevicePolicies(policies map[string]domain.AppDevicePolicy) *Service { + s.SetDevicePolicies(policies) + return s +} + +// SetDevicePolicies atomically replaces the administrative device policies +// used to authorize manifest devices. The map and its slices are copied so +// later caller mutation cannot race an in-flight apply. +func (s *Service) SetDevicePolicies(policies map[string]domain.AppDevicePolicy) { + copied := make(map[string]domain.AppDevicePolicy, len(policies)) + for name, policy := range policies { + policy.CDI = append([]string(nil), policy.CDI...) + policy.AllowedApps = append([]string(nil), policy.AllowedApps...) + policy.AllowedServices = append([]string(nil), policy.AllowedServices...) + copied[name] = policy + } + s.deviceMu.Lock() + s.devicePolicies = copied + s.deviceMu.Unlock() +} + +// snapshotDevicePolicies returns the current policy map. SetDevicePolicies +// never mutates a published map, so the snapshot stays safe after the +// lock drops. +func (s *Service) snapshotDevicePolicies() map[string]domain.AppDevicePolicy { + s.deviceMu.RLock() + defer s.deviceMu.RUnlock() + return s.devicePolicies +} + +// noopLease is a GC lease for an unwired barrier. +type noopLease struct{} + +func (noopLease) Release() {} + +// acquireSharedLease takes the shared GC lease for one apply. It fails +// closed: an apply must not publish desired state while prune holds the +// exclusive lease. +func (s *Service) acquireSharedLease(ctx context.Context) (out.GCLease, error) { + if s.barrier == nil { + return noopLease{}, nil + } + lease, err := s.barrier.AcquireShared(ctx) + if err != nil { + return nil, fmt.Errorf("apps: acquire GC lease: %w", err) + } + return lease, nil +} + +// WithEntrypoints supplies the installation entrypoint listeners so Apply +// can reject an L4 publish declaration that does not match the listener +// that will actually bind it. A nil map leaves the check to projection. +func (s *Service) WithEntrypoints(listeners map[string]domain.EntryPointListener) *Service { + s.listeners = listeners + return s +} + +// ApplyResult describes one accepted (or no-op) apply. +type ApplyResult struct { + App string + FormerRevision string + ResultingRevision string + Noop bool + Pending bool + Diff domain.AppDiff + IntentID string +} + +// DryRunResult describes validation without persistence. +type DryRunResult struct { + App string + Valid bool + Diff domain.AppDiff +} + +// newRevisionID allocates a time-ordered rev- identifier (uuid v7). +func newRevisionID() string { + id, err := uuid.NewV7() + if err != nil { + id = uuid.New() + } + return "rev-" + hexNoDashes(id) +} + +// newIntentID allocates a time-ordered apply- identifier (uuid v7). +func newIntentID() string { + id, err := uuid.NewV7() + if err != nil { + id = uuid.New() + } + return "apply-" + hexNoDashes(id) +} + +// hexNoDashes renders a UUID without dashes (32 hex chars). +func hexNoDashes(id uuid.UUID) string { + var buf [32]byte + hex.Encode(buf[:], id[:]) + return string(buf[:]) +} + +// Apply validates a normalized spec and persists it as desired state. +// source is the raw manifest bytes (hashed for audit, never re-read). +// dryRun validates and previews without persistence or effects. +func (s *Service) Apply(ctx context.Context, spec domain.AppSpec, source []byte, dryRun bool) (*ApplyResult, *DryRunResult, error) { + ctx = zerowrap.CtxWithFields(ctx, map[string]any{ + zerowrap.FieldLayer: "usecase", + zerowrap.FieldUseCase: "Apply", + "app": spec.Name, + "dry_run": dryRun, + }) + log := zerowrap.FromCtx(ctx) + + if err := spec.Validate(); err != nil { + return nil, nil, err + } + for _, service := range spec.Services { + if err := s.imagePolicy.ValidateImageSource(service.Image); err != nil { + return nil, nil, fmt.Errorf("apps: service %q image %q: %w", service.Name, service.Image, err) + } + } + if err := s.validateBindPolicies(spec); err != nil { + return nil, nil, err + } + if err := s.validateDevicePolicies(spec); err != nil { + return nil, nil, err + } + if err := s.validateEntrypointCompatibility(spec); err != nil { + return nil, nil, err + } + // Hold the shared GC lease from validation through publication: prune + // must never observe a staged-but-unmaterialized apply. + lease, err := s.acquireSharedLease(ctx) + if err != nil { + return nil, nil, err + } + defer lease.Release() + prepared, err := s.prepareApply(ctx, spec) + if err != nil { + return nil, nil, err + } + if prepared.noop { + log.Info().Str("revision", prepared.desired.Revision).Msg("apps: no-change apply, idempotent") + return &ApplyResult{ + App: spec.Name, + FormerRevision: prepared.desired.Revision, + ResultingRevision: prepared.desired.Revision, + Noop: true, + Diff: prepared.diff, + }, nil, nil + } + if dryRun { + return nil, &DryRunResult{ + App: spec.Name, + Valid: true, + Diff: prepared.diff, + }, nil + } + return s.persistApply(ctx, spec, source, prepared) +} + +// validateBindPolicies resolves every declared bind against the +// administrative bind policies before any desired state is persisted. +// Errors name only the mount, app, and service: policy source paths never +// appear in apply errors. +func (s *Service) validateBindPolicies(spec domain.AppSpec) error { + policies := s.snapshotBindPolicies() + for _, svc := range spec.Services { + for _, bind := range svc.Binds { + policy, ok := policies[bind.Name] + if !ok { + return fmt.Errorf("%w: app %q service %q mount %q: no administrative mount policy configured", domain.ErrBindPolicy, spec.Name, svc.Name, bind.Name) + } + if _, err := policy.ResolveAppBind(spec.Name, svc.Name, bind); err != nil { + return fmt.Errorf("%w: app %q service %q mount %q: refused by administrative mount policy", domain.ErrBindPolicy, spec.Name, svc.Name, bind.Name) + } + } + } + return nil +} + +// validateDevicePolicies resolves every declared device against the +// administrative device policies before any desired state is persisted. +// It runs the same union resolution as activation: duplicate CDI IDs +// across logical resources fail here, not later at preflight. Errors +// name only the device, app, and service with an actionable hint: +// host inventory never appears in apply errors. +func (s *Service) validateDevicePolicies(spec domain.AppSpec) error { + policies := s.snapshotDevicePolicies() + for _, svc := range spec.Services { + for _, device := range svc.Devices { + policy, ok := policies[device] + if !ok { + return fmt.Errorf("%w: app %q service %q device %q: no administrative device policy configured (declare [app_devices.%s] with allowed_apps/allowed_services)", domain.ErrDevicePolicy, spec.Name, svc.Name, device, device) + } + if _, err := policy.ResolveAppDevice(spec.Name, svc.Name); err != nil { + return fmt.Errorf("%w: app %q service %q device %q: refused by administrative device policy (check allowed_apps/allowed_services)", domain.ErrDevicePolicy, spec.Name, svc.Name, device) + } + } + // The union check catches duplicate CDI IDs granted under two + // logical names; single-device errors above already named the + // device, so this wraps only the cross-device conflict. + if _, err := domain.ResolveAppDevices(spec.Name, svc.Name, svc.Devices, policies); err != nil { + return fmt.Errorf("apps: %w", err) + } + } + return nil +} + +// validateEntrypointCompatibility checks every declared TCP/UDP interface +// against the installation entrypoint listeners before any persistence or +// workload effect. The runtime binds the entrypoint address, so a publish +// bind that differs from it (for example loopback) must be rejected +// rather than silently exposed on the public listener. +func (s *Service) validateEntrypointCompatibility(spec domain.AppSpec) error { + if s.listeners == nil { + return nil + } + for _, svc := range spec.Services { + for _, t := range svc.TCP { + if err := s.checkPublish(svc.Name, "tcp", t.Entrypoint, t.Publish, domain.NetworkProtocolTCP); err != nil { + return err + } + } + for _, u := range svc.UDP { + if err := s.checkPublish(svc.Name, "udp", u.Entrypoint, u.Publish, domain.NetworkProtocolUDP); err != nil { + return err + } + } + } + return nil +} + +func (s *Service) checkPublish(service, kind, entrypoint, publish string, transport domain.NetworkProtocol) error { + listener, ok := s.listeners[entrypoint] + if !ok { + return fmt.Errorf("%w: service %q %s interface references unknown entrypoint %q", domain.ErrInvalidAppSpec, service, kind, entrypoint) + } + if err := domain.ValidatePublishForListener(publish, transport, listener, entrypoint); err != nil { + return fmt.Errorf("apps: service %q: %w", service, err) + } + return nil +} + +// applyPrepared carries validated apply inputs through to persistence. +type applyPrepared struct { + desired domain.AppDesiredRevision + diff domain.AppDiff + reservations []domain.AppListenerReservation + noop bool +} + +// prepareApply validates, recovers, checks conflicts, detects no-ops. +// It performs no writes. +func (s *Service) prepareApply(ctx context.Context, spec domain.AppSpec) (*applyPrepared, error) { + if err := s.store.Recover(ctx); err != nil { + return nil, fmt.Errorf("apps: recover before mutation: %w", err) + } + checkpoint, err := s.store.LoadCheckpoint(ctx) + if err != nil { + return nil, err + } + reservations := domain.ReservationsFor(spec) + desired, hasDesired, err := s.store.LoadDesired(ctx, spec.Name) + if err != nil { + return nil, err + } + active, _, err := s.store.LoadActive(ctx, spec.Name) + if err != nil { + return nil, err + } + if err := domain.CheckReservations(checkpoint.Reservations, reservations, spec.Name); err != nil { + return nil, err + } + diff := domain.DiffAppSpec(spec, activeSpec(active)) + if hasDesired && specsEqual(desired.Spec, spec) { + return &applyPrepared{desired: desired, diff: diff, reservations: reservations, noop: true}, nil + } + return &applyPrepared{desired: desired, diff: diff, reservations: reservations}, nil +} + +// persistApply durably accepts one apply. +func (s *Service) persistApply(ctx context.Context, spec domain.AppSpec, source []byte, prepared *applyPrepared) (*ApplyResult, *DryRunResult, error) { + log := zerowrap.FromCtx(ctx) + sourceHash := sha256.Sum256(source) + supersedes := prepared.desired.Revision + intent := domain.AppApplyIntent{ + Intent: newIntentID(), + App: spec.Name, + Revision: newRevisionID(), + Supersedes: supersedes, + SourceSHA256: hex.EncodeToString(sourceHash[:]), + Spec: spec, + Reservations: prepared.reservations, + CreatedAt: time.Now().UTC(), + } + if err := s.store.AcceptApply(ctx, intent); err != nil { + return nil, nil, fmt.Errorf("apps: %w", err) + } + log.Info().Str("revision", intent.Revision).Str("intent", intent.Intent).Msg("apps: apply accepted") + return &ApplyResult{ + App: spec.Name, + FormerRevision: supersedes, + ResultingRevision: intent.Revision, + Pending: true, + Diff: prepared.diff, + IntentID: intent.Intent, + }, nil, nil +} + +// activeSpec rebuilds an AppSpec view from effective definitions for diffing. +func activeSpec(active domain.AppActive) domain.AppSpec { + spec := domain.AppSpec{Name: active.App, Env: map[string]string{}, Networks: append([]domain.AppSharedNetwork(nil), active.Networks...)} + for name, svc := range active.Services { + svcSpec := svc.Spec + svcSpec.Name = name + spec.Services = append(spec.Services, svcSpec) + } + return spec +} + +// specsEqual compares normalized specs for no-op detection. +func specsEqual(a, b domain.AppSpec) bool { + diff := domain.DiffAppSpec(a, b) + return len(diff.Added) == 0 && len(diff.Removed) == 0 && len(diff.Changed) == 0 +} diff --git a/internal/usecase/apps/service_scope_test.go b/internal/usecase/apps/service_scope_test.go new file mode 100644 index 000000000..00072279d --- /dev/null +++ b/internal/usecase/apps/service_scope_test.go @@ -0,0 +1,145 @@ +package apps_test + +import ( + "context" + "testing" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/apps" + "github.com/bnema/gordon/internal/usecase/deployment" +) + +func scopeDesired(names ...string) domain.AppDesiredRevision { + rev := domain.AppDesiredRevision{Revision: "rev-1", App: "blog"} + for _, name := range names { + rev.Spec.Services = append(rev.Spec.Services, domain.AppService{Name: name}) + } + return rev +} + +func scopeRevision(revision string, names ...string) domain.AppDesiredRevision { + rev := scopeDesired(names...) + rev.Revision = revision + return rev +} + +func scopeActive(names ...string) domain.AppActive { + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{}} + for _, name := range names { + active.Services[name] = domain.AppEffectiveService{} + } + return active +} + +func TestAppServiceImpl_Deploy_MultiServiceRequiresScope(t *testing.T) { + store := newMockAppState(t) + deploy := newMockDeployEngine(t) + svc := apps.NewAppServiceImpl(store, deploy, newMockSecretWriter(t), zerowrap.Default()) + t.Cleanup(func() { require.NoError(t, svc.Shutdown(context.Background())) }) + + // A first deploy has no ACTIVE yet: desired services still count. + store.EXPECT().LoadDesired(mock.Anything, "blog").Return(scopeDesired("web", "db"), true, nil) + store.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{}, false, nil) + deploy.startDeployFn = func(context.Context, deployment.DeployInput) (*deployment.StartDeployResult, error) { + t.Fatal("no claim may be made without a service scope") + return nil, nil + } + + op, err := svc.Deploy(context.Background(), "blog", "", "", false, "key-1") + require.ErrorIs(t, err, domain.ErrAppServiceScope) + assert.Nil(t, op) + assert.Contains(t, err.Error(), "app blog has several services (db, web): pass --service NAME or --all") +} + +func TestAppServiceImpl_Deploy_OldRevisionWithMultipleServicesRequiresScope(t *testing.T) { + store := newMockAppState(t) + deploy := newMockDeployEngine(t) + svc := apps.NewAppServiceImpl(store, deploy, newMockSecretWriter(t), zerowrap.Default()) + t.Cleanup(func() { require.NoError(t, svc.Shutdown(context.Background())) }) + + // The head and ACTIVE know only one service, but the revision the + // deploy would activate declares two: the scope check must count the + // selected revision, not the head. + store.EXPECT().LoadRevision(mock.Anything, "blog", "rev-2").Return(scopeRevision("rev-2", "web", "db"), nil) + store.EXPECT().LoadActive(mock.Anything, "blog").Return(scopeActive("web"), true, nil) + deploy.startDeployFn = func(context.Context, deployment.DeployInput) (*deployment.StartDeployResult, error) { + t.Fatal("no claim may be made without a service scope") + return nil, nil + } + + op, err := svc.Deploy(context.Background(), "blog", "rev-2", "", false, "key-1") + require.ErrorIs(t, err, domain.ErrAppServiceScope) + assert.Nil(t, op) + assert.Contains(t, err.Error(), "app blog has several services (db, web): pass --service NAME or --all") +} + +func TestAppServiceImpl_Deploy_SingleServiceNeedsNoScope(t *testing.T) { + store := newMockAppState(t) + deploy := newMockDeployEngine(t) + svc := apps.NewAppServiceImpl(store, deploy, newMockSecretWriter(t), zerowrap.Default()) + t.Cleanup(func() { require.NoError(t, svc.Shutdown(context.Background())) }) + + store.EXPECT().LoadDesired(mock.Anything, "blog").Return(scopeDesired("web"), true, nil) + store.EXPECT().LoadActive(mock.Anything, "blog").Return(scopeActive("web"), true, nil) + started := make(chan struct{}, 1) + deploy.startDeployFn = func(context.Context, deployment.DeployInput) (*deployment.StartDeployResult, error) { + started <- struct{}{} + return nil, domain.ErrAppStateConflict + } + + _, err := svc.Deploy(context.Background(), "blog", "", "", false, "key-1") + require.ErrorIs(t, err, domain.ErrAppStateConflict) + require.Len(t, started, 1, "a single-service app deploys without --service or --all") +} + +func TestAppServiceImpl_Deploy_AllSkipsScopeCheck(t *testing.T) { + store := newMockAppState(t) + deploy := newMockDeployEngine(t) + svc := apps.NewAppServiceImpl(store, deploy, newMockSecretWriter(t), zerowrap.Default()) + t.Cleanup(func() { require.NoError(t, svc.Shutdown(context.Background())) }) + + var got deployment.DeployInput + deploy.startDeployFn = func(_ context.Context, input deployment.DeployInput) (*deployment.StartDeployResult, error) { + got = input + return nil, domain.ErrAppStateConflict + } + + _, err := svc.Deploy(context.Background(), "blog", "", "", true, "key-1") + require.ErrorIs(t, err, domain.ErrAppStateConflict) + assert.Equal(t, deployment.DeployInput{App: "blog", Op: "key-1"}, got, "--all deploys every service") +} + +func TestAppServiceImpl_Deploy_AllAndServiceAreExclusive(t *testing.T) { + svc := apps.NewAppServiceImpl(newMockAppState(t), newMockDeployEngine(t), newMockSecretWriter(t), zerowrap.Default()) + t.Cleanup(func() { require.NoError(t, svc.Shutdown(context.Background())) }) + + _, err := svc.Deploy(context.Background(), "blog", "", "web", true, "key-1") + require.ErrorIs(t, err, domain.ErrAppServiceScope) +} + +func TestAppServiceImpl_Restart_MultiServiceRequiresScope(t *testing.T) { + store := newMockAppState(t) + svc := apps.NewAppServiceImpl(store, newMockDeployEngine(t), newMockSecretWriter(t), zerowrap.Default()) + t.Cleanup(func() { require.NoError(t, svc.Shutdown(context.Background())) }) + + store.EXPECT().LoadActive(mock.Anything, "blog").Return(scopeActive("web", "db"), true, nil) + + // The engine mock has no Restart expectation: reaching it fails the test. + op, err := svc.Restart(context.Background(), "blog", "", false, "key-1") + require.ErrorIs(t, err, domain.ErrAppServiceScope) + assert.Nil(t, op) + assert.Contains(t, err.Error(), "(db, web)") +} + +func TestAppServiceImpl_Restart_AllAndServiceAreExclusive(t *testing.T) { + svc := apps.NewAppServiceImpl(newMockAppState(t), newMockDeployEngine(t), newMockSecretWriter(t), zerowrap.Default()) + t.Cleanup(func() { require.NoError(t, svc.Shutdown(context.Background())) }) + + _, err := svc.Restart(context.Background(), "blog", "web", true, "key-1") + require.ErrorIs(t, err, domain.ErrAppServiceScope) +} diff --git a/internal/usecase/apps/service_test.go b/internal/usecase/apps/service_test.go new file mode 100644 index 000000000..7b24e4b5d --- /dev/null +++ b/internal/usecase/apps/service_test.go @@ -0,0 +1,375 @@ +package apps_test + +import ( + "context" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/boundaries/out" + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/apps" +) + +func testSpec(name string) domain.AppSpec { + return domain.AppSpec{ + Name: name, + Services: []domain.AppService{ + { + Name: "web", Image: "img:1", StopGrace: 10 * time.Second, + Readiness: domain.AppReadiness{Type: "none", Timeout: 30 * time.Second}, + HTTP: []domain.AppHTTPInterface{{Host: name + ".example.com", Port: 8080, TLS: "auto"}}, + }, + }, + } +} + +// applySuccess wires a full accept path on the store mock. +func applySuccess(store *outmocks.MockAppState, spec domain.AppSpec) { + store.EXPECT().Recover(mock.Anything).Return(nil).Once() + store.EXPECT().LoadCheckpoint(mock.Anything).Return(domain.AppStoreCheckpoint{}, nil).Once() + store.EXPECT().LoadDesired(mock.Anything, spec.Name).Return(domain.AppDesiredRevision{}, false, nil).Once() + store.EXPECT().LoadActive(mock.Anything, spec.Name).Return(domain.AppActive{}, false, nil).Once() + store.EXPECT().AcceptApply(mock.Anything, mock.Anything).Return(nil).Once() +} + +// recordingBarrier records shared-lease acquisition for apply tests. +type recordingBarrier struct { + shared int + released int + err error +} + +func (b *recordingBarrier) AcquireShared(context.Context) (out.GCLease, error) { + if b.err != nil { + return nil, b.err + } + b.shared++ + return &recordingLease{barrier: b}, nil +} + +func (b *recordingBarrier) AcquireExclusive(context.Context) (out.GCLease, error) { + return nil, assert.AnError +} + +type recordingLease struct{ barrier *recordingBarrier } + +func (l *recordingLease) Release() { l.barrier.released++ } + +func TestApply_RejectsImageOutsideRegistryAllowlistBeforeStoreAccess(t *testing.T) { + store := outmocks.NewMockAppState(t) + spec := testSpec("blog") + spec.Services[0].Image = "registry.example.com/team/app:1" + svc := apps.NewService(store, zerowrap.Default()).WithImagePolicy(domain.ImageSourcePolicy{}) + + _, _, err := svc.Apply(context.Background(), spec, []byte("manifest"), false) + + assert.ErrorIs(t, err, domain.ErrAppImageNotAllowed) +} + +func TestApply_RequireDigestIncludesInstallationRegistry(t *testing.T) { + store := outmocks.NewMockAppState(t) + spec := testSpec("blog") + spec.Services[0].Image = "gordon.example.com/team/app:1" + svc := apps.NewService(store, zerowrap.Default()).WithImagePolicy(domain.ImageSourcePolicy{ + InstallationRegistry: "gordon.example.com", + RequireDigest: true, + }) + + _, _, err := svc.Apply(context.Background(), spec, []byte("manifest"), false) + assert.ErrorIs(t, err, domain.ErrAppImageNotAllowed) +} + +// TestApply_HoldsSharedGCLease proves apply holds the shared GC lease for +// its whole mutation, so prune cannot snapshot a half-published apply. +func TestApply_HoldsSharedGCLease(t *testing.T) { + ctx := context.Background() + store := outmocks.NewMockAppState(t) + spec := testSpec("blog") + applySuccess(store, spec) + barrier := &recordingBarrier{} + svc := apps.NewService(store, zerowrap.Default()).WithGCBarrier(barrier) + + _, _, err := svc.Apply(ctx, spec, []byte("manifest"), false) + + require.NoError(t, err) + assert.Equal(t, 1, barrier.shared, "apply must take the shared GC lease") + assert.Equal(t, 1, barrier.released, "apply must release the shared GC lease") +} + +// TestApply_FailsClosedWhenGCBarrierRefuses proves an apply never publishes +// desired state while it cannot hold the shared lease (e.g. prune holds the +// exclusive lease). +func TestApply_FailsClosedWhenGCBarrierRefuses(t *testing.T) { + ctx := context.Background() + barrier := &recordingBarrier{err: assert.AnError} + svc := apps.NewService(outmocks.NewMockAppState(t), zerowrap.Default()).WithGCBarrier(barrier) + + _, _, err := svc.Apply(ctx, testSpec("blog"), []byte("manifest"), false) + + require.Error(t, err) + assert.Equal(t, 0, barrier.shared) +} + +func TestApply_AcceptsAndPersists(t *testing.T) { + ctx := context.Background() + store := outmocks.NewMockAppState(t) + svc := apps.NewService(store, zerowrap.Default()) + spec := testSpec("blog") + + applySuccess(store, spec) + result, dry, err := svc.Apply(ctx, spec, []byte("manifest"), false) + require.NoError(t, err) + assert.Nil(t, dry) + require.NotNil(t, result) + assert.True(t, result.Pending) + assert.False(t, result.Noop) + assert.NotEmpty(t, result.ResultingRevision) + assert.NotEmpty(t, result.IntentID) +} + +func TestApply_NoopIsIdempotent(t *testing.T) { + ctx := context.Background() + store := outmocks.NewMockAppState(t) + svc := apps.NewService(store, zerowrap.Default()) + spec := testSpec("blog") + + // First apply persists. + applySuccess(store, spec) + first, _, err := svc.Apply(ctx, spec, []byte("manifest"), false) + require.NoError(t, err) + + // Second apply with identical content is a no-op: no writes. + desired := domain.AppDesiredRevision{Revision: first.ResultingRevision, App: "blog", Spec: spec} + store.EXPECT().Recover(mock.Anything).Return(nil).Once() + store.EXPECT().LoadCheckpoint(mock.Anything).Return(domain.AppStoreCheckpoint{}, nil).Once() + store.EXPECT().LoadDesired(mock.Anything, "blog").Return(desired, true, nil).Once() + store.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{}, false, nil).Once() + second, _, err := svc.Apply(ctx, spec, []byte("manifest"), false) + require.NoError(t, err) + assert.True(t, second.Noop) + assert.Equal(t, first.ResultingRevision, second.ResultingRevision) +} + +func TestApply_DryRunWritesNothing(t *testing.T) { + ctx := context.Background() + store := outmocks.NewMockAppState(t) + svc := apps.NewService(store, zerowrap.Default()) + spec := testSpec("blog") + + store.EXPECT().Recover(mock.Anything).Return(nil).Once() + store.EXPECT().LoadCheckpoint(mock.Anything).Return(domain.AppStoreCheckpoint{}, nil).Once() + store.EXPECT().LoadDesired(mock.Anything, "blog").Return(domain.AppDesiredRevision{}, false, nil).Once() + store.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{}, false, nil).Once() + result, dry, err := svc.Apply(ctx, spec, []byte("manifest"), true) + require.NoError(t, err) + assert.Nil(t, result) + require.NotNil(t, dry) + assert.True(t, dry.Valid) + store.AssertNotCalled(t, "AcceptApply", mock.Anything, mock.Anything) +} + +func TestApply_ImageOnlyDiffPreservesActiveNetworks(t *testing.T) { + ctx := context.Background() + store := outmocks.NewMockAppState(t) + svc := apps.NewService(store, zerowrap.Default()) + spec := testSpec("blog") + spec.Networks = []domain.AppSharedNetwork{{Network: "database", Services: []string{"web"}, Aliases: []string{"db"}}} + activeService := spec.Services[0] + activeService.Image = "img:1" + spec.Services[0].Image = "img:2" + + store.EXPECT().Recover(mock.Anything).Return(nil).Once() + store.EXPECT().LoadCheckpoint(mock.Anything).Return(domain.AppStoreCheckpoint{}, nil).Once() + previousDesired := spec + previousDesired.Services = append([]domain.AppService(nil), spec.Services...) + previousDesired.Services[0].Image = "img:1" + store.EXPECT().LoadDesired(mock.Anything, "blog").Return(domain.AppDesiredRevision{Revision: "rev-1", App: "blog", Spec: previousDesired}, true, nil).Once() + store.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", Networks: spec.Networks, + Services: map[string]domain.AppEffectiveService{"web": {Spec: activeService}}, + }, true, nil).Once() + + _, dry, err := svc.Apply(ctx, spec, []byte("manifest"), true) + require.NoError(t, err) + require.NotNil(t, dry) + assert.Equal(t, []string{"service/web/image"}, dry.Diff.Changed) +} + +func TestApply_RejectsCrossAppConflict(t *testing.T) { + ctx := context.Background() + store := outmocks.NewMockAppState(t) + svc := apps.NewService(store, zerowrap.Default()) + + applySuccess(store, testSpec("blog")) + _, _, err := svc.Apply(ctx, testSpec("blog"), []byte("m"), false) + require.NoError(t, err) + + // Same host, different app → conflict before any write. + other := testSpec("shop") + other.Services[0].HTTP[0].Host = "blog.example.com" + store.EXPECT().Recover(mock.Anything).Return(nil).Once() + store.EXPECT().LoadCheckpoint(mock.Anything).Return(domain.AppStoreCheckpoint{ + Reservations: []domain.AppListenerReservation{{Proto: "http", Host: "blog.example.com", Service: "web", App: "blog"}}, + }, nil).Once() + store.EXPECT().LoadDesired(mock.Anything, "shop").Return(domain.AppDesiredRevision{}, false, nil).Once() + store.EXPECT().LoadActive(mock.Anything, "shop").Return(domain.AppActive{}, false, nil).Once() + _, _, err = svc.Apply(ctx, other, []byte("m"), false) + require.ErrorIs(t, err, domain.ErrAppReservationConflict) + store.AssertNotCalled(t, "AcceptApply", mock.Anything, mock.MatchedBy(func(intent domain.AppApplyIntent) bool { + return intent.App == "shop" + })) +} + +func TestApply_RejectsDuplicateClaimInCandidate(t *testing.T) { + ctx := context.Background() + store := outmocks.NewMockAppState(t) + svc := apps.NewService(store, zerowrap.Default()) + + spec := testSpec("blog") + dup := spec.Services[0] + dup.Name = "web2" + spec.Services = append(spec.Services, dup) + + store.EXPECT().Recover(mock.Anything).Return(nil).Once() + store.EXPECT().LoadCheckpoint(mock.Anything).Return(domain.AppStoreCheckpoint{}, nil).Once() + store.EXPECT().LoadDesired(mock.Anything, "blog").Return(domain.AppDesiredRevision{}, false, nil).Once() + store.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{}, false, nil).Once() + _, _, err := svc.Apply(ctx, spec, []byte("m"), false) + require.ErrorIs(t, err, domain.ErrAppReservationConflict) +} + +func TestApply_RejectsOverlappingWildcardClaimInCandidate(t *testing.T) { + ctx := context.Background() + store := outmocks.NewMockAppState(t) + svc := apps.NewService(store, zerowrap.Default()) + + spec := domain.AppSpec{ + Name: "tcp-overlap", + Services: []domain.AppService{ + { + Name: "one", Image: "img:1", StopGrace: 10 * time.Second, + Readiness: domain.AppReadiness{Type: "none", Timeout: 30 * time.Second}, + TCP: []domain.AppTCPInterface{{Entrypoint: "tcp", Port: 9000, Publish: "0.0.0.0:19090"}}, + }, + { + Name: "two", Image: "img:1", StopGrace: 10 * time.Second, + Readiness: domain.AppReadiness{Type: "none", Timeout: 30 * time.Second}, + TCP: []domain.AppTCPInterface{{Entrypoint: "tcp", Port: 9000, Publish: "127.0.0.1:19090"}}, + }, + }, + } + + store.EXPECT().Recover(mock.Anything).Return(nil).Once() + store.EXPECT().LoadCheckpoint(mock.Anything).Return(domain.AppStoreCheckpoint{}, nil).Once() + store.EXPECT().LoadDesired(mock.Anything, spec.Name).Return(domain.AppDesiredRevision{}, false, nil).Once() + store.EXPECT().LoadActive(mock.Anything, spec.Name).Return(domain.AppActive{}, false, nil).Once() + _, _, err := svc.Apply(ctx, spec, []byte("m"), false) + require.ErrorIs(t, err, domain.ErrAppReservationConflict) + store.AssertNotCalled(t, "AcceptApply", mock.Anything, mock.Anything) +} + +func TestApply_InvalidSpecFailsBeforeStore(t *testing.T) { + ctx := context.Background() + store := outmocks.NewMockAppState(t) + svc := apps.NewService(store, zerowrap.Default()) + + spec := testSpec("Bad!") + _, _, err := svc.Apply(ctx, spec, []byte("m"), false) + require.ErrorIs(t, err, domain.ErrInvalidAppSpec) + store.AssertNotCalled(t, "Recover", mock.Anything) +} + +// TestApply_RejectsPublishMismatchBeforeStore proves an L4 publish that does +// not match the entrypoint listener is refused before any persistence or +// workload effect, so the runtime can never widen a narrower declaration. +func TestApply_RejectsPublishMismatchBeforeStore(t *testing.T) { + ctx := context.Background() + store := outmocks.NewMockAppState(t) + svc := apps.NewService(store, zerowrap.Default()).WithEntrypoints(map[string]domain.EntryPointListener{ + "tcp": {Address: "0.0.0.0:25432", Protocol: domain.EntryPointProtocolTCP}, + }) + + spec := testSpec("game") + spec.Services[0].HTTP = nil + spec.Services[0].TCP = []domain.AppTCPInterface{{Entrypoint: "tcp", Port: 25432, Publish: "127.0.0.1:25432"}} + + _, _, err := svc.Apply(ctx, spec, []byte("m"), false) + require.ErrorIs(t, err, domain.ErrInvalidAppSpec) + store.AssertNotCalled(t, "Recover", mock.Anything) +} + +// TestApply_RejectsUnknownEntrypointBeforeStore proves an L4 interface that +// references a missing entrypoint fails closed at apply time. +func TestApply_RejectsUnknownEntrypointBeforeStore(t *testing.T) { + ctx := context.Background() + store := outmocks.NewMockAppState(t) + svc := apps.NewService(store, zerowrap.Default()).WithEntrypoints(map[string]domain.EntryPointListener{}) + + spec := testSpec("game") + spec.Services[0].HTTP = nil + spec.Services[0].TCP = []domain.AppTCPInterface{{Entrypoint: "tcp", Port: 25432, Publish: "0.0.0.0:25432"}} + + _, _, err := svc.Apply(ctx, spec, []byte("m"), false) + require.ErrorIs(t, err, domain.ErrInvalidAppSpec) + store.AssertNotCalled(t, "Recover", mock.Anything) +} + +func TestApply_SamePortDifferentProtoCoexists(t *testing.T) { + ctx := context.Background() + store := outmocks.NewMockAppState(t) + svc := apps.NewService(store, zerowrap.Default()) + + tcpApp := domain.AppSpec{ + Name: "tcpapp", + Services: []domain.AppService{ + { + Name: "s", Image: "img:1", StopGrace: 10 * time.Second, + Readiness: domain.AppReadiness{Type: "none", Timeout: 30 * time.Second}, + TCP: []domain.AppTCPInterface{{Entrypoint: "tcp", Port: 9000, Publish: "0.0.0.0:9000"}}, + }, + }, + } + applySuccess(store, tcpApp) + _, _, err := svc.Apply(ctx, tcpApp, []byte("m"), false) + require.NoError(t, err) + + udpApp := domain.AppSpec{ + Name: "udpapp", + Services: []domain.AppService{ + { + Name: "s", Image: "img:1", StopGrace: 10 * time.Second, + Readiness: domain.AppReadiness{Type: "none", Timeout: 30 * time.Second}, + UDP: []domain.AppUDPInterface{{Entrypoint: "udp", Port: 9000, Publish: "0.0.0.0:9000"}}, + }, + }, + } + applySuccess(store, udpApp) + _, _, err = svc.Apply(ctx, udpApp, []byte("m"), false) + require.NoError(t, err) +} + +func TestReservationsFor_DelegatesToDomain(t *testing.T) { + spec := domain.AppSpec{ + Name: "blog", + Services: []domain.AppService{ + { + Name: "web", Image: "img:1", + HTTP: []domain.AppHTTPInterface{{Host: "b.example.com", Port: 8080, TLS: "auto"}}, + TCP: []domain.AppTCPInterface{{Entrypoint: "tcp", Port: 9000, Publish: "9000"}}, + }, + }, + } + // Apply derives reservations through the domain owner; derivation + // policy itself is covered in the domain package. + reservations := domain.ReservationsFor(spec) + require.Len(t, reservations, 2) + assert.Equal(t, "http", reservations[0].Proto) + assert.Equal(t, "tcp", reservations[1].Proto) + assert.Equal(t, "dual", reservations[1].IP) +} diff --git a/internal/usecase/apptraffic/activate.go b/internal/usecase/apptraffic/activate.go new file mode 100644 index 000000000..7faa3df8c --- /dev/null +++ b/internal/usecase/apptraffic/activate.go @@ -0,0 +1,220 @@ +package apptraffic + +import ( + "context" + "fmt" + "sort" + "sync" + + "github.com/bnema/zerowrap" + + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/domain" +) + +// HostIndex is the proxy-facing read model: canonical HTTP host to the +// resolved loopback backend. It derives from ACTIVE app state (bbolt +// stays authoritative); the daemon rebuilds it after every activation +// and at boot. Lookups never touch container IPs. +type HostIndex struct { + mu sync.RWMutex + byHost map[string]RouteEntry + entries []RouteEntry +} + +// NewHostIndex creates an empty host index. +func NewHostIndex() *HostIndex { + return &HostIndex{byHost: map[string]RouteEntry{}} +} + +// Replace installs a rebuilt host map. +func (h *HostIndex) Replace(entries []RouteEntry) { + byHost := make(map[string]RouteEntry, len(entries)) + for _, entry := range entries { + if entry.Kind != "http" || entry.Host == "" { + continue + } + byHost[entry.Host] = entry + } + h.mu.Lock() + defer h.mu.Unlock() + h.byHost = byHost + h.entries = append([]RouteEntry(nil), entries...) +} + +// Entries returns a copy of the projected entries, for a caller that +// builds a candidate index and publishes it through another one. +func (h *HostIndex) Entries() []RouteEntry { + if h == nil { + return nil + } + h.mu.RLock() + defer h.mu.RUnlock() + return append([]RouteEntry(nil), h.entries...) +} + +// Lookup returns the projected entry for a canonical host. +func (h *HostIndex) Lookup(host string) (RouteEntry, bool) { + if h == nil { + return RouteEntry{}, false + } + h.mu.RLock() + defer h.mu.RUnlock() + entry, ok := h.byHost[host] + return entry, ok +} + +// HostOwner returns the app and service serving a canonical host. +func (h *HostIndex) HostOwner(host string) (app, service string, ok bool) { + entry, found := h.Lookup(host) + if !found { + return "", "", false + } + return entry.App, entry.Service, true +} + +// LookupHost implements proxy.TargetProvider over the host index. +// The proxy package owns the interface; apptraffic implements it to +// keep app-state interpretation out of the proxy. +func (h *HostIndex) LookupHost(host string) (domain.AppBackend, bool) { + if h == nil { + return domain.AppBackend{}, false + } + entry, ok := h.Lookup(host) + if !ok { + return domain.AppBackend{}, false + } + return entry.Backend, true +} + +// AppHosts returns the sorted served hosts with resolved backends. +// Unresolved entries (no recorded bind) are excluded: fail closed. +// It implements out.AppHostSource: the shared read model for traffic, +// PKI and public-TLS consumers. +func (h *HostIndex) AppHosts() []out.AppHost { + if h == nil { + return nil + } + h.mu.RLock() + defer h.mu.RUnlock() + hosts := make([]out.AppHost, 0, len(h.byHost)) + for host, entry := range h.byHost { + if !entry.Backend.Resolved() { + continue + } + hosts = append(hosts, out.AppHost{ + Host: host, + App: entry.App, + Service: entry.Service, + TLSMode: entry.TLSMode, + }) + } + sort.Slice(hosts, func(i, j int) bool { return hosts[i].Host < hosts[j].Host }) + return hosts +} + +// L4Entries returns resolved TCP and UDP projections. +func (h *HostIndex) L4Entries() []RouteEntry { + if h == nil { + return nil + } + h.mu.RLock() + defer h.mu.RUnlock() + entries := make([]RouteEntry, 0, len(h.entries)) + for _, entry := range h.entries { + if (entry.Kind == "tcp" || entry.Kind == "udp") && entry.Backend.Resolved() { + entries = append(entries, entry) + } + } + return entries +} + +// Activator re-projects active app state into the host index. +// It is the coordinated traffic activation for the deployment engine: +// after each deployment the engine rebuilds the host index from ACTIVE +// state so the proxy only serves current backends. +type Activator struct { + log zerowrap.Logger +} + +// NewActivator wires activation with its logger. +func NewActivator(log zerowrap.Logger) *Activator { + return &Activator{log: log} +} + +// RebuildHostIndex re-projects every app's ACTIVE state into the host +// index. The daemon calls it after each activation and at boot. +// STOPPED-intent apps are skipped: their backends must not stay +// routable merely because ACTIVE retains their effective spec. +func (a *Activator) RebuildHostIndex( + ctx context.Context, + index *HostIndex, + state out.AppStateReader, + entrypoints map[string]EntrypointPolicy, +) error { + apps, err := state.ListApps(ctx) + if err != nil { + return fmt.Errorf("apptraffic: list apps for host index: %w", err) + } + actives := make(map[string]domain.AppActive, len(apps)) + for _, app := range apps { + intent, err := state.LoadIntent(ctx, app) + if err != nil { + return fmt.Errorf("apptraffic: load intent for host index: %w", err) + } + if intent.Stopped { + continue + } + active, ok, err := state.LoadActive(ctx, app) + if err != nil { + return fmt.Errorf("apptraffic: load active for host index: %w", err) + } + if !ok { + continue + } + actives[app] = active + } + projections, err := ProjectAll(actives, entrypoints) + if err != nil { + return err + } + var entries []RouteEntry + for _, app := range sortedKeys(projections) { + entries = append(entries, projections[app]...) + } + index.Replace(entries) + a.log.Info().Int("hosts", len(entries)).Msg("apptraffic: host index rebuilt") + return nil +} + +// AppStateReader is the ACTIVE/intent subset read paths need. +// Narrower than out.AppState so read paths never gain write access. +// (Re-exported alias: the canonical definition lives in out.) + +func sortedKeys(projections map[string][]RouteEntry) []string { + apps := make([]string, 0, len(projections)) + for app := range projections { + apps = append(apps, app) + } + sort.Strings(apps) + return apps +} +func ProjectAll( + apps map[string]domain.AppActive, + entrypoints map[string]EntrypointPolicy, +) (map[string][]RouteEntry, error) { + names := make([]string, 0, len(apps)) + for app := range apps { + names = append(names, app) + } + sort.Strings(names) + projections := make(map[string][]RouteEntry, len(apps)) + for _, app := range names { + entries, err := Project(app, apps[app], entrypoints) + if err != nil { + return nil, err + } + projections[app] = entries + } + return projections, nil +} diff --git a/internal/usecase/apptraffic/activate_test.go b/internal/usecase/apptraffic/activate_test.go new file mode 100644 index 000000000..ad961dd06 --- /dev/null +++ b/internal/usecase/apptraffic/activate_test.go @@ -0,0 +1,85 @@ +package apptraffic_test + +import ( + "context" + "testing" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/apptraffic" +) + +type stubState struct { + intents map[string]domain.AppStopIntent + actives map[string]domain.AppActive +} + +func (s stubState) ListApps(_ context.Context) ([]string, error) { + apps := make([]string, 0, len(s.actives)+len(s.intents)) + seen := map[string]bool{} + for app := range s.actives { + if !seen[app] { + seen[app] = true + apps = append(apps, app) + } + } + for app := range s.intents { + if !seen[app] { + seen[app] = true + apps = append(apps, app) + } + } + return apps, nil +} + +func (s stubState) LoadIntent(_ context.Context, app string) (domain.AppStopIntent, error) { + return s.intents[app], nil +} + +func (s stubState) LoadActive(_ context.Context, app string) (domain.AppActive, bool, error) { + active, ok := s.actives[app] + return active, ok, nil +} + +func TestActivator_RebuildHostIndex(t *testing.T) { + ctx := context.Background() + activator := apptraffic.NewActivator(zerowrap.Default()) + index := apptraffic.NewHostIndex() + + served := domain.AppActive{ + App: "blog", + Services: map[string]domain.AppEffectiveService{ + "web": { + EffectiveRevision: "rev-1", Image: "img:1", + Spec: domain.AppService{ + HTTP: []domain.AppHTTPInterface{{Host: "blog.example.com", Port: 8080, TLS: "auto"}}, + }, + }, + }, + } + stopped := domain.AppActive{ + App: "old", + Services: map[string]domain.AppEffectiveService{ + "web": { + EffectiveRevision: "rev-1", Image: "img:1", + Spec: domain.AppService{ + HTTP: []domain.AppHTTPInterface{{Host: "old.example.com", Port: 8080, TLS: "auto"}}, + }, + }, + }, + } + state := stubState{ + actives: map[string]domain.AppActive{"blog": served, "old": stopped}, + intents: map[string]domain.AppStopIntent{"old": {Stopped: true}}, + } + + require.NoError(t, activator.RebuildHostIndex(ctx, index, state, nil)) + + _, ok := index.Lookup("blog.example.com") + assert.True(t, ok) + _, ok = index.Lookup("old.example.com") + assert.False(t, ok, "STOPPED-intent app must not stay routable") +} diff --git a/internal/usecase/apptraffic/projection.go b/internal/usecase/apptraffic/projection.go new file mode 100644 index 000000000..243b215fb --- /dev/null +++ b/internal/usecase/apptraffic/projection.go @@ -0,0 +1,179 @@ +// Package apptraffic projects active app definitions into traffic +// snapshot entries consumed by the traffic manager. +package apptraffic + +import ( + "fmt" + "sort" + "strconv" + "strings" + + "github.com/bnema/gordon/internal/domain" +) + +// EntrypointPolicy is the installation policy inherited by generated +// listeners: the canonical public bind, trusted CIDRs, raw-fallback +// behavior, and transport limits. The code never reads installation +// config; the caller supplies it. +type EntrypointPolicy struct { + Name string + Address string + Protocol domain.EntryPointProtocol + TrustedCIDRs []string + RawFallback string + RawFallbackTrustedCIDRs []string + AllowPublicRawFallback bool +} + +// listener projects the policy to the domain listener contract used for +// exact publish compatibility checks. +func (p EntrypointPolicy) listener() domain.EntryPointListener { + return domain.EntryPointListener{Address: p.Address, Protocol: p.Protocol} +} + +// RouteEntry is one projected traffic entry. +type RouteEntry struct { + // Kind is http, tcp, or udp. + Kind string + // RouterName is the deterministic generated router name. + RouterName string + // Entrypoint names the installation entrypoint (empty for http). + Entrypoint string + // Host is the canonical HTTP host (http only). + Host string + // BindIP and BindPort are the literal public bind (tcp/udp only). + BindIP string + BindPort int + // TLSMode is auto, always, or never (http only). + TLSMode string + // App and Service identify the owning workload. + App string + Service string + // Backend is the resolved loopback endpoint for the interface's + // container port. Zero when unbound (proxy fails closed). + Backend domain.AppBackend +} + +// Project maps one app's active definitions to traffic entries. +// Entrypoints supplies installation policy by name; unknown entrypoint references are validation errors. +// Loopback endpoints resolve from each service's recorded backend binds; +// services without a recorded bind for an interface port project with an +// empty BackendHost (proxy fails closed). +func Project(app string, active domain.AppActive, entrypoints map[string]EntrypointPolicy) ([]RouteEntry, error) { + if active.App != "" && active.App != app { + return nil, fmt.Errorf("%w: projection for %q got active state of %q", domain.ErrAppTrafficProjection, app, active.App) + } + var entries []RouteEntry + for name, svc := range active.Services { + svcEntries, err := projectService(app, name, svc, entrypoints) + if err != nil { + return nil, err + } + entries = append(entries, svcEntries...) + } + sort.Slice(entries, func(i, j int) bool { + if entries[i].RouterName != entries[j].RouterName { + return entries[i].RouterName < entries[j].RouterName + } + return entries[i].Backend.ContainerPort < entries[j].Backend.ContainerPort + }) + return entries, nil +} + +// projectService maps one effective service to entries. Endpoints +// resolve from the recorded backend binds; missing binds stay empty +// (proxy fails closed, never falls back to container IPs). +func projectService(app, name string, eff domain.AppEffectiveService, entrypoints map[string]EntrypointPolicy) ([]RouteEntry, error) { + spec := eff.Spec + spec.Name = name + var entries []RouteEntry + for _, h := range spec.HTTP { + // Internal HTTP is reachable only from the app private network: it + // is never routed, so it must not project a traffic entry or a + // certificate host. Fail closed by skipping it entirely. + if !h.IsPublic() { + continue + } + entries = append(entries, RouteEntry{ + Kind: "http", + RouterName: "app-" + app + "--" + spec.Name + "--http-" + sanitizeHost(h.Host), + Host: h.Host, + TLSMode: h.TLS, + App: app, + Service: spec.Name, + Backend: eff.BackendFor(h.Port), + }) + } + for _, t := range spec.TCP { + policy, err := lookupEntrypoint(entrypoints, t.Entrypoint, app, spec.Name, "tcp") + if err != nil { + return nil, err + } + host, port, err := domain.ParsePublish(t.Publish) + if err != nil { + return nil, err + } + if err := domain.ValidatePublishForListener(t.Publish, domain.NetworkProtocolTCP, policy.listener(), t.Entrypoint); err != nil { + return nil, fmt.Errorf("%w: service %q tcp interface: %s", domain.ErrAppTrafficProjection, spec.Name, err) + } + entries = append(entries, RouteEntry{ + Kind: "tcp", + RouterName: "app-" + app + "--" + spec.Name + "--tcp-" + itoa(t.Port), + Entrypoint: t.Entrypoint, + BindIP: host, + BindPort: port, + App: app, + Service: spec.Name, + Backend: eff.BackendFor(t.Port), + }) + } + for _, u := range spec.UDP { + policy, err := lookupEntrypoint(entrypoints, u.Entrypoint, app, spec.Name, "udp") + if err != nil { + return nil, err + } + host, port, err := domain.ParsePublish(u.Publish) + if err != nil { + return nil, err + } + if err := domain.ValidatePublishForListener(u.Publish, domain.NetworkProtocolUDP, policy.listener(), u.Entrypoint); err != nil { + return nil, fmt.Errorf("%w: service %q udp interface: %s", domain.ErrAppTrafficProjection, spec.Name, err) + } + entries = append(entries, RouteEntry{ + Kind: "udp", + RouterName: "app-" + app + "--" + spec.Name + "--udp-" + itoa(u.Port), + Entrypoint: u.Entrypoint, + BindIP: host, + BindPort: port, + App: app, + Service: spec.Name, + Backend: eff.BackendForProtocol(u.Port, domain.NetworkProtocolUDP), + }) + } + return entries, nil +} + +// lookupEntrypoint resolves a named installation entrypoint. +func lookupEntrypoint(entrypoints map[string]EntrypointPolicy, name, app, service, kind string) (EntrypointPolicy, error) { + policy, ok := entrypoints[name] + if !ok { + return EntrypointPolicy{}, fmt.Errorf( + "%w: service %q %s references unknown entrypoint %q", + domain.ErrAppTrafficProjection, service, kind, name, + ) + } + _ = app + return policy, nil +} + +// sanitizeHost maps a hostname to a router-name fragment. +func sanitizeHost(host string) string { + fragment := strings.ReplaceAll(host, ".", "-") + fragment = strings.ReplaceAll(fragment, ":", "-") + return strings.ReplaceAll(fragment, "/", "-") +} + +// itoa formats a port number. +func itoa(n int) string { + return strconv.Itoa(n) +} diff --git a/internal/usecase/apptraffic/projection_test.go b/internal/usecase/apptraffic/projection_test.go new file mode 100644 index 000000000..8dda512e0 --- /dev/null +++ b/internal/usecase/apptraffic/projection_test.go @@ -0,0 +1,225 @@ +package apptraffic_test + +import ( + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/apptraffic" +) + +// l4 builds installation entrypoint policies. An empty address omits the +// entrypoint, so tests exercise only the transports they configure. +func l4(tcpAddress, udpAddress string) map[string]apptraffic.EntrypointPolicy { + policies := map[string]apptraffic.EntrypointPolicy{} + if tcpAddress != "" { + policies["tcp"] = apptraffic.EntrypointPolicy{ + Name: "tcp", Address: tcpAddress, Protocol: domain.EntryPointProtocolTCP, + TrustedCIDRs: []string{"10.0.0.0/8"}, + } + } + if udpAddress != "" { + policies["udp"] = apptraffic.EntrypointPolicy{ + Name: "udp", Address: udpAddress, Protocol: domain.EntryPointProtocolUDP, + } + } + return policies +} + +func webActive() domain.AppActive { + return domain.AppActive{ + App: "blog", + Services: map[string]domain.AppEffectiveService{ + "web": { + EffectiveRevision: "rev-1", + Image: "img:1", + Container: "c-web", + BackendBinds: map[int]int{8080: 18080}, + Spec: domain.AppService{ + HTTP: []domain.AppHTTPInterface{{Host: "blog.example.com", Port: 8080, TLS: "auto"}}, + }, + }, + }, + } +} + +func TestProject_HTTP(t *testing.T) { + entries, err := apptraffic.Project("blog", webActive(), nil) + require.NoError(t, err) + require.Len(t, entries, 1) + assert.Equal(t, "http", entries[0].Kind) + assert.Equal(t, "blog.example.com", entries[0].Host) + assert.Equal(t, "auto", entries[0].TLSMode) + assert.Equal(t, 8080, entries[0].Backend.ContainerPort) + assert.True(t, entries[0].Backend.Resolved()) + assert.Equal(t, "127.0.0.1", entries[0].Backend.Host) + assert.Equal(t, 18080, entries[0].Backend.Port) + assert.Equal(t, "c-web", entries[0].Backend.ContainerID) + assert.Empty(t, entries[0].Entrypoint) +} + +func TestProject_TCPUDP(t *testing.T) { + active := domain.AppActive{ + App: "game", + Services: map[string]domain.AppEffectiveService{ + "server": { + EffectiveRevision: "rev-1", + Image: "img:1", + Spec: domain.AppService{ + TCP: []domain.AppTCPInterface{{Entrypoint: "tcp", Port: 25565, Publish: "0.0.0.0:25565"}}, + UDP: []domain.AppUDPInterface{{Entrypoint: "udp", Port: 28015, Publish: "0.0.0.0:28015"}}, + }, + }, + }, + } + entries, err := apptraffic.Project("game", active, l4("0.0.0.0:25565", "0.0.0.0:28015")) + require.NoError(t, err) + require.Len(t, entries, 2) + assert.Equal(t, "tcp", entries[0].Kind) + assert.Equal(t, "udp", entries[1].Kind) + assert.Equal(t, 25565, entries[0].BindPort) + assert.Equal(t, "tcp", entries[0].Entrypoint) +} + +// TestProject_SamePortTCPUDPResolvesDistinctBackends proves the same +// container port number on TCP and UDP resolves to two distinct loopback +// backends from their respective bind maps. +func TestProject_SamePortTCPUDPResolvesDistinctBackends(t *testing.T) { + active := domain.AppActive{ + App: "game", + Services: map[string]domain.AppEffectiveService{ + "server": { + EffectiveRevision: "rev-1", + Image: "img:1", + Container: "c-game", + BackendBinds: map[int]int{9000: 19000}, + UDPBackendBinds: map[int]int{9000: 19001}, + Spec: domain.AppService{ + TCP: []domain.AppTCPInterface{{Entrypoint: "tcp", Port: 9000, Publish: "0.0.0.0:9000"}}, + UDP: []domain.AppUDPInterface{{Entrypoint: "udp", Port: 9000, Publish: "0.0.0.0:9000"}}, + }, + }, + }, + } + entries, err := apptraffic.Project("game", active, l4("0.0.0.0:9000", "0.0.0.0:9000")) + require.NoError(t, err) + require.Len(t, entries, 2) + byKind := map[string]apptraffic.RouteEntry{} + for _, entry := range entries { + byKind[entry.Kind] = entry + } + require.Contains(t, byKind, "tcp") + require.Contains(t, byKind, "udp") + assert.True(t, byKind["tcp"].Backend.Resolved()) + assert.True(t, byKind["udp"].Backend.Resolved()) + assert.Equal(t, 9000, byKind["tcp"].Backend.ContainerPort) + assert.Equal(t, 9000, byKind["udp"].Backend.ContainerPort) + assert.Equal(t, 19000, byKind["tcp"].Backend.Port) + assert.Equal(t, 19001, byKind["udp"].Backend.Port) + assert.NotEqual(t, byKind["tcp"].Backend.Port, byKind["udp"].Backend.Port) +} + +func TestProject_TCPAsOrdinaryRCON(t *testing.T) { + // RCON is ordinary TCP: no special kind or policy tag. + active := domain.AppActive{ + App: "rust", + Services: map[string]domain.AppEffectiveService{ + "server": { + EffectiveRevision: "rev-1", + Image: "img:1", + Spec: domain.AppService{ + TCP: []domain.AppTCPInterface{{Entrypoint: "tcp", Port: 28016, Publish: "0.0.0.0:28016"}}, + }, + }, + }, + } + entries, err := apptraffic.Project("rust", active, l4("0.0.0.0:28016", "")) + require.NoError(t, err) + require.Len(t, entries, 1) + assert.Equal(t, "tcp", entries[0].Kind) + assert.Equal(t, 28016, entries[0].Backend.ContainerPort) + assert.False(t, entries[0].Backend.Resolved(), "no recorded bind: fail closed") +} + +// TestProject_RejectsPublishMismatch proves a declared bind that differs +// from the entrypoint listener is refused rather than silently widened to +// the entrypoint address. +func TestProject_RejectsPublishMismatch(t *testing.T) { + cases := []struct { + name string + entrypoint apptraffic.EntrypointPolicy + publish string + }{ + { + name: "loopback-vs-wildcard", + entrypoint: apptraffic.EntrypointPolicy{Name: "tcp", Address: "0.0.0.0:25432", Protocol: domain.EntryPointProtocolTCP}, + publish: "127.0.0.1:15432", + }, + { + name: "same-host-different-port", + entrypoint: apptraffic.EntrypointPolicy{Name: "tcp", Address: "0.0.0.0:25432", Protocol: domain.EntryPointProtocolTCP}, + publish: "0.0.0.0:15432", + }, + { + name: "protocol-mismatch", + entrypoint: apptraffic.EntrypointPolicy{Name: "tcp", Address: "0.0.0.0:25432", Protocol: domain.EntryPointProtocolUDP}, + publish: "0.0.0.0:25432", + }, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + active := domain.AppActive{ + App: "game", + Services: map[string]domain.AppEffectiveService{ + "server": { + EffectiveRevision: "rev-1", Image: "img:1", + Spec: domain.AppService{ + TCP: []domain.AppTCPInterface{{Entrypoint: "tcp", Port: 25432, Publish: tc.publish}}, + }, + }, + }, + } + _, err := apptraffic.Project("game", active, map[string]apptraffic.EntrypointPolicy{"tcp": tc.entrypoint}) + require.ErrorIs(t, err, domain.ErrAppTrafficProjection) + }) + } +} + +func TestProject_UnknownEntrypoint(t *testing.T) { + active := domain.AppActive{ + App: "game", + Services: map[string]domain.AppEffectiveService{ + "server": { + EffectiveRevision: "rev-1", + Image: "img:1", + Spec: domain.AppService{ + TCP: []domain.AppTCPInterface{{Entrypoint: "nope", Port: 1, Publish: "1"}}, + }, + }, + }, + } + _, err := apptraffic.Project("game", active, l4("0.0.0.0:25565", "")) + require.ErrorIs(t, err, domain.ErrAppTrafficProjection) +} + +func TestProject_GameExample(t *testing.T) { + active := domain.AppActive{ + App: "rust", + Services: map[string]domain.AppEffectiveService{ + "server": { + EffectiveRevision: "rev-1", Image: "img:1", + Spec: domain.AppService{ + Readiness: domain.AppReadiness{Type: "log", Path: "/data/logs/server.log", Contains: "ready", Timeout: 5 * time.Minute}, + UDP: []domain.AppUDPInterface{{Entrypoint: "udp", Port: 28015, Publish: "0.0.0.0:28015"}}, + TCP: []domain.AppTCPInterface{{Entrypoint: "tcp", Port: 28016, Publish: "0.0.0.0:28016"}}, + }, + }, + }, + } + entries, err := apptraffic.Project("rust", active, l4("0.0.0.0:28016", "0.0.0.0:28015")) + require.NoError(t, err) + require.Len(t, entries, 2) +} diff --git a/internal/usecase/apptraffic/projection_visibility_test.go b/internal/usecase/apptraffic/projection_visibility_test.go new file mode 100644 index 000000000..9d72d1369 --- /dev/null +++ b/internal/usecase/apptraffic/projection_visibility_test.go @@ -0,0 +1,56 @@ +package apptraffic_test + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/apptraffic" +) + +// internalActive is one app whose only HTTP interface is internal. +func internalActive() domain.AppActive { + return domain.AppActive{ + App: "blog", + Services: map[string]domain.AppEffectiveService{ + "api": { + EffectiveRevision: "rev-1", + Image: "img:1", + Container: "c-api", + Spec: domain.AppService{ + HTTP: []domain.AppHTTPInterface{{Port: 8080, Visibility: domain.AppVisibilityInternal}}, + }, + }, + }, + } +} + +// TestProject_InternalHTTPExcluded proves an internal HTTP interface +// projects no route: it can never appear in the host index, so it also +// yields no certificate host and no TLS mode. +func TestProject_InternalHTTPExcluded(t *testing.T) { + entries, err := apptraffic.Project("blog", internalActive(), nil) + require.NoError(t, err) + assert.Empty(t, entries) +} + +// TestProject_MixedVisibilityProjectsOnlyPublic proves a service with both +// planes projects only the public interface. +func TestProject_MixedVisibilityProjectsOnlyPublic(t *testing.T) { + active := internalActive() + svc := active.Services["api"] + svc.BackendBinds = map[int]int{8080: 18080} + svc.Spec.HTTP = []domain.AppHTTPInterface{ + {Host: "blog.example.com", Port: 8080, TLS: "auto", Visibility: domain.AppVisibilityPublic}, + {Port: 9090, Visibility: domain.AppVisibilityInternal}, + } + active.Services["api"] = svc + + entries, err := apptraffic.Project("blog", active, nil) + require.NoError(t, err) + require.Len(t, entries, 1) + assert.Equal(t, "blog.example.com", entries[0].Host) + assert.Equal(t, 8080, entries[0].Backend.ContainerPort) +} diff --git a/internal/usecase/auto/dispatcher.go b/internal/usecase/auto/dispatcher.go deleted file mode 100644 index f92a6a685..000000000 --- a/internal/usecase/auto/dispatcher.go +++ /dev/null @@ -1,77 +0,0 @@ -package auto - -import ( - "context" - "path/filepath" - - "github.com/bnema/zerowrap" - - "github.com/bnema/gordon/internal/boundaries/out" - "github.com/bnema/gordon/internal/domain" -) - -// AutoConfigProvider exposes unified auto config to the dispatcher. -type AutoConfigProvider interface { - IsAutoEnabled() bool - IsPreviewEnabled() bool - GetPreviewTagPatterns() []string - GetPreviewConfig() domain.PreviewConfig - GetAllowedDomains() []string - FindRoutesByImage(ctx context.Context, imageName string) []domain.Route -} - -// ImagePushDispatcher classifies image push events and delegates -// to either the auto-route handler or the auto-preview handler. -type ImagePushDispatcher struct { - config AutoConfigProvider - routeHandler out.EventHandler - previewHandler out.EventHandler -} - -func NewImagePushDispatcher(config AutoConfigProvider, routeHandler, previewHandler out.EventHandler) *ImagePushDispatcher { - return &ImagePushDispatcher{ - config: config, - routeHandler: routeHandler, - previewHandler: previewHandler, - } -} - -func (d *ImagePushDispatcher) CanHandle(t domain.EventType) bool { - return t == domain.EventImagePushed -} - -func (d *ImagePushDispatcher) Handle(ctx context.Context, event domain.Event) error { - log := zerowrap.FromCtx(ctx) - - payload, ok := event.Data.(domain.ImagePushedPayload) - if !ok { - log.Warn().Str("event_type", string(event.Type)).Msg("dispatcher received event with unexpected payload type; forwarding to route handler") - return d.routeHandler.Handle(ctx, event) - } - - previewEnabled := d.config.IsPreviewEnabled() - tagMatch := d.matchesTagPatterns(payload.Reference) - log.Debug(). - Bool("preview_enabled", previewEnabled). - Bool("tag_match", tagMatch). - Str("reference", payload.Reference). - Msg("dispatcher classifying push event") - - if previewEnabled && tagMatch { - return d.previewHandler.Handle(ctx, event) - } - return d.routeHandler.Handle(ctx, event) -} - -func (d *ImagePushDispatcher) matchesTagPatterns(reference string) bool { - for _, pattern := range d.config.GetPreviewTagPatterns() { - matched, err := filepath.Match(pattern, reference) - if err != nil { - continue // skip malformed patterns - } - if matched { - return true - } - } - return false -} diff --git a/internal/usecase/auto/dispatcher_test.go b/internal/usecase/auto/dispatcher_test.go deleted file mode 100644 index 9d1d6042e..000000000 --- a/internal/usecase/auto/dispatcher_test.go +++ /dev/null @@ -1,94 +0,0 @@ -package auto - -import ( - "context" - "testing" - - "github.com/stretchr/testify/assert" - - "github.com/bnema/gordon/internal/domain" -) - -type mockHandler struct { - called bool -} - -func (m *mockHandler) Handle(_ context.Context, _ domain.Event) error { - m.called = true - return nil -} - -func (m *mockHandler) CanHandle(t domain.EventType) bool { - return t == domain.EventImagePushed -} - -type mockAutoConfig struct { - previewEnabled bool - tagPatterns []string -} - -func (m *mockAutoConfig) IsAutoEnabled() bool { return true } -func (m *mockAutoConfig) IsPreviewEnabled() bool { return m.previewEnabled } -func (m *mockAutoConfig) GetPreviewTagPatterns() []string { return m.tagPatterns } -func (m *mockAutoConfig) GetPreviewConfig() domain.PreviewConfig { return domain.PreviewConfig{} } -func (m *mockAutoConfig) GetAllowedDomains() []string { return nil } -func (m *mockAutoConfig) FindRoutesByImage(_ context.Context, _ string) []domain.Route { - return nil -} - -func TestDispatcher_PreviewTag_DelegatesToPreview(t *testing.T) { - route := &mockHandler{} - preview := &mockHandler{} - d := NewImagePushDispatcher( - &mockAutoConfig{previewEnabled: true, tagPatterns: []string{"preview-*"}}, - route, preview, - ) - event := domain.Event{ - Type: domain.EventImagePushed, - Data: domain.ImagePushedPayload{Name: "myapp", Reference: "preview-feat"}, - } - err := d.Handle(context.Background(), event) - assert.NoError(t, err) - assert.True(t, preview.called) - assert.False(t, route.called) -} - -func TestDispatcher_NormalTag_DelegatesToRoute(t *testing.T) { - route := &mockHandler{} - preview := &mockHandler{} - d := NewImagePushDispatcher( - &mockAutoConfig{previewEnabled: true, tagPatterns: []string{"preview-*"}}, - route, preview, - ) - event := domain.Event{ - Type: domain.EventImagePushed, - Data: domain.ImagePushedPayload{Name: "myapp", Reference: "v1.0.0"}, - } - err := d.Handle(context.Background(), event) - assert.NoError(t, err) - assert.True(t, route.called) - assert.False(t, preview.called) -} - -func TestDispatcher_PreviewDisabled_AlwaysRoute(t *testing.T) { - route := &mockHandler{} - preview := &mockHandler{} - d := NewImagePushDispatcher( - &mockAutoConfig{previewEnabled: false, tagPatterns: []string{"preview-*"}}, - route, preview, - ) - event := domain.Event{ - Type: domain.EventImagePushed, - Data: domain.ImagePushedPayload{Name: "myapp", Reference: "preview-feat"}, - } - err := d.Handle(context.Background(), event) - assert.NoError(t, err) - assert.True(t, route.called) - assert.False(t, preview.called) -} - -func TestDispatcher_CanHandle(t *testing.T) { - d := &ImagePushDispatcher{} - assert.True(t, d.CanHandle(domain.EventImagePushed)) - assert.False(t, d.CanHandle(domain.EventConfigReload)) -} diff --git a/internal/usecase/auto/labels.go b/internal/usecase/auto/labels.go deleted file mode 100644 index cbfb79f96..000000000 --- a/internal/usecase/auto/labels.go +++ /dev/null @@ -1,124 +0,0 @@ -package auto - -import ( - "context" - "encoding/json" - "fmt" - "io" - "strings" - - "github.com/bnema/gordon/internal/boundaries/out" - "github.com/bnema/gordon/internal/domain" -) - -// manifestSchema represents the relevant parts of an OCI/Docker manifest. -type manifestSchema struct { - SchemaVersion int `json:"schemaVersion"` - MediaType string `json:"mediaType"` - Config struct { - MediaType string `json:"mediaType"` - Digest string `json:"digest"` - Size int64 `json:"size"` - } `json:"config"` -} - -// ParseConfigDigest extracts the config digest from a manifest. -func ParseConfigDigest(manifestData []byte) (string, error) { - var manifest manifestSchema - if err := json.Unmarshal(manifestData, &manifest); err != nil { - return "", err - } - - return manifest.Config.Digest, nil -} - -// imageConfig represents the relevant parts of an OCI/Docker image config. -type imageConfig struct { - Config struct { - Labels map[string]string `json:"Labels"` - } `json:"config"` -} - -// ParseImageLabels extracts Gordon labels from an image config blob. -func ParseImageLabels(configData []byte) (*domain.ImageLabels, error) { - var config imageConfig - if err := json.Unmarshal(configData, &config); err != nil { - return nil, err - } - - labels := &domain.ImageLabels{} - - if config.Config.Labels == nil { - return labels, nil - } - - // Extract gordon.* labels - if v, ok := config.Config.Labels[domain.LabelDomain]; ok { - labels.Domain = strings.TrimSpace(v) - } - - if v, ok := config.Config.Labels[domain.LabelDomains]; ok { - // Parse comma-separated domains - for _, d := range strings.Split(v, ",") { - d = strings.TrimSpace(d) - if d != "" { - labels.Domains = append(labels.Domains, d) - } - } - } - - if v, ok := config.Config.Labels[domain.LabelHealth]; ok { - labels.Health = strings.TrimSpace(v) - } - - for _, key := range []string{domain.LabelProxyPort, domain.LabelPort} { - if v, ok := config.Config.Labels[key]; ok { - labels.Port = strings.TrimSpace(v) - break - } - } - - if v, ok := config.Config.Labels[domain.LabelEnvFile]; ok { - labels.EnvFile = strings.TrimSpace(v) - } - - return labels, nil -} - -// ExtractLabels extracts Gordon labels from an image manifest by fetching the -// config blob from blobStorage and parsing it. It is the shared implementation -// used by both AutoRouteHandler and AutoPreviewHandler. -// ctx is accepted for future use when BlobStorage gains context support. -func ExtractLabels(ctx context.Context, manifestData []byte, blobStorage out.BlobStorage) (*domain.ImageLabels, error) { - configDigest, err := ParseConfigDigest(manifestData) - if err != nil { - return nil, fmt.Errorf("failed to parse manifest: %w", err) - } - - if configDigest == "" { - return nil, fmt.Errorf("no config digest found in manifest") - } - - reader, err := blobStorage.GetBlob(configDigest) - if err != nil { - return nil, fmt.Errorf("failed to get config blob: %w", err) - } - defer reader.Close() - - const maxConfigBlobSize = 1 << 20 // 1 MB - limited := io.LimitReader(reader, maxConfigBlobSize+1) - configData, err := io.ReadAll(limited) - if err != nil { - return nil, fmt.Errorf("failed to read config blob: %w", err) - } - if len(configData) > maxConfigBlobSize { - return nil, fmt.Errorf("config blob exceeds maximum size of %d bytes", maxConfigBlobSize) - } - - labels, err := ParseImageLabels(configData) - if err != nil { - return nil, fmt.Errorf("failed to parse config: %w", err) - } - - return labels, nil -} diff --git a/internal/usecase/auto/labels_test.go b/internal/usecase/auto/labels_test.go deleted file mode 100644 index 2f87741b4..000000000 --- a/internal/usecase/auto/labels_test.go +++ /dev/null @@ -1,277 +0,0 @@ -package auto - -import ( - "context" - "encoding/json" - "io" - "strings" - "testing" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" - - "github.com/bnema/gordon/internal/boundaries/out/mocks" - "github.com/bnema/gordon/internal/domain" -) - -func TestParseConfigDigest(t *testing.T) { - tests := []struct { - name string - manifest map[string]any - expected string - wantErr bool - }{ - { - name: "valid manifest v2", - manifest: map[string]any{ - "schemaVersion": 2, - "mediaType": "application/vnd.docker.distribution.manifest.v2+json", - "config": map[string]any{ - "mediaType": "application/vnd.docker.container.image.v1+json", - "digest": "sha256:abc123", - "size": 1234, - }, - }, - expected: "sha256:abc123", - }, - { - name: "OCI manifest", - manifest: map[string]any{ - "schemaVersion": 2, - "mediaType": "application/vnd.oci.image.manifest.v1+json", - "config": map[string]any{ - "mediaType": "application/vnd.oci.image.config.v1+json", - "digest": "sha256:def456", - "size": 5678, - }, - }, - expected: "sha256:def456", - }, - { - name: "empty config digest", - manifest: map[string]any{ - "schemaVersion": 2, - "config": map[string]any{ - "digest": "", - }, - }, - expected: "", - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - data, err := json.Marshal(tt.manifest) - require.NoError(t, err) - - result, err := ParseConfigDigest(data) - - if tt.wantErr { - assert.Error(t, err) - return - } - - require.NoError(t, err) - assert.Equal(t, tt.expected, result) - }) - } -} - -func TestParseConfigDigest_InvalidJSON(t *testing.T) { - _, err := ParseConfigDigest([]byte("invalid json")) - assert.Error(t, err) -} - -func TestParseImageLabels(t *testing.T) { - tests := []struct { - name string - config map[string]any - expected *domain.ImageLabels - }{ - { - name: "single domain label", - config: map[string]any{ - "config": map[string]any{ - "Labels": map[string]any{ - "gordon.domain": "app.example.com", - }, - }, - }, - expected: &domain.ImageLabels{ - Domain: "app.example.com", - }, - }, - { - name: "multiple domains label", - config: map[string]any{ - "config": map[string]any{ - "Labels": map[string]any{ - "gordon.domains": "app1.example.com, app2.example.com", - }, - }, - }, - expected: &domain.ImageLabels{ - Domains: []string{"app1.example.com", "app2.example.com"}, - }, - }, - { - name: "all gordon labels", - config: map[string]any{ - "config": map[string]any{ - "Labels": map[string]any{ - "gordon.domain": "app.example.com", - "gordon.health": "/healthz", - "gordon.proxy.port": "8080", - "gordon.env-file": ".env.prod", - }, - }, - }, - expected: &domain.ImageLabels{ - Domain: "app.example.com", - Health: "/healthz", - Port: "8080", - EnvFile: ".env.prod", - }, - }, - { - name: "no labels", - config: map[string]any{ - "config": map[string]any{ - "Labels": nil, - }, - }, - expected: &domain.ImageLabels{}, - }, - { - name: "no gordon labels", - config: map[string]any{ - "config": map[string]any{ - "Labels": map[string]any{ - "other.label": "value", - }, - }, - }, - expected: &domain.ImageLabels{}, - }, - { - name: "whitespace in labels", - config: map[string]any{ - "config": map[string]any{ - "Labels": map[string]any{ - "gordon.domain": " app.example.com ", - }, - }, - }, - expected: &domain.ImageLabels{ - Domain: "app.example.com", - }, - }, - { - name: "deprecated port label fallback", - config: map[string]any{ - "config": map[string]any{ - "Labels": map[string]any{ - "gordon.port": "9090", - }, - }, - }, - expected: &domain.ImageLabels{ - Port: "9090", - }, - }, - { - name: "proxy.port takes precedence over port", - config: map[string]any{ - "config": map[string]any{ - "Labels": map[string]any{ - "gordon.proxy.port": "8080", - "gordon.port": "9090", - }, - }, - }, - expected: &domain.ImageLabels{ - Port: "8080", - }, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - data, err := json.Marshal(tt.config) - require.NoError(t, err) - - result, err := ParseImageLabels(data) - require.NoError(t, err) - - assert.Equal(t, tt.expected.Domain, result.Domain) - assert.Equal(t, tt.expected.Domains, result.Domains) - assert.Equal(t, tt.expected.Health, result.Health) - assert.Equal(t, tt.expected.Port, result.Port) - assert.Equal(t, tt.expected.EnvFile, result.EnvFile) - }) - } -} - -func TestParseImageLabels_InvalidJSON(t *testing.T) { - _, err := ParseImageLabels([]byte("invalid json")) - assert.Error(t, err) -} - -func TestExtractLabels(t *testing.T) { - ctx := context.Background() - - manifest := map[string]any{ - "schemaVersion": 2, - "config": map[string]any{ - "digest": "sha256:configdigest123", - }, - } - manifestData, err := json.Marshal(manifest) - require.NoError(t, err) - - imageConf := map[string]any{ - "config": map[string]any{ - "Labels": map[string]any{ - "gordon.domain": "app.example.com", - "gordon.health": "/healthz", - }, - }, - } - configData, err := json.Marshal(imageConf) - require.NoError(t, err) - - blobStorage := mocks.NewMockBlobStorage(t) - blobStorage.EXPECT().GetBlob("sha256:configdigest123").Return(io.NopCloser(strings.NewReader(string(configData))), nil) - - labels, err := ExtractLabels(ctx, manifestData, blobStorage) - require.NoError(t, err) - assert.Equal(t, "app.example.com", labels.Domain) - assert.Equal(t, "/healthz", labels.Health) -} - -func TestExtractLabels_InvalidManifest(t *testing.T) { - ctx := context.Background() - blobStorage := mocks.NewMockBlobStorage(t) - - _, err := ExtractLabels(ctx, []byte("bad json"), blobStorage) - assert.Error(t, err) -} - -func TestExtractLabels_EmptyConfigDigest(t *testing.T) { - ctx := context.Background() - - manifest := map[string]any{ - "schemaVersion": 2, - "config": map[string]any{ - "digest": "", - }, - } - manifestData, err := json.Marshal(manifest) - require.NoError(t, err) - - blobStorage := mocks.NewMockBlobStorage(t) - - _, err = ExtractLabels(ctx, manifestData, blobStorage) - assert.Error(t, err) - assert.Contains(t, err.Error(), "no config digest") -} diff --git a/internal/usecase/auto/preview/handler.go b/internal/usecase/auto/preview/handler.go deleted file mode 100644 index 2a0c3b188..000000000 --- a/internal/usecase/auto/preview/handler.go +++ /dev/null @@ -1,157 +0,0 @@ -package preview - -import ( - "context" - "fmt" - "strings" - - "github.com/bnema/zerowrap" - - "github.com/bnema/gordon/internal/domain" - "github.com/bnema/gordon/internal/usecase/auto" -) - -// AutoPreviewHandler processes EventImagePushed for preview tag patterns. -type AutoPreviewHandler struct { - serviceCtx context.Context - config auto.AutoConfigProvider - previewService *Service - jobs chan struct{} -} - -func NewAutoPreviewHandler( - serviceCtx context.Context, - config auto.AutoConfigProvider, - previewService *Service, -) *AutoPreviewHandler { - return &AutoPreviewHandler{ - serviceCtx: serviceCtx, - config: config, - previewService: previewService, - jobs: make(chan struct{}, 16), - } -} - -func (h *AutoPreviewHandler) CanHandle(t domain.EventType) bool { - return t == domain.EventImagePushed -} - -func (h *AutoPreviewHandler) Handle(ctx context.Context, event domain.Event) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldHandler: "AutoPreviewHandler", - zerowrap.FieldEvent: string(event.Type), - "event_id": event.ID, - }) - log := zerowrap.FromCtx(ctx) - - payload, ok := event.Data.(domain.ImagePushedPayload) - if !ok { - log.Debug().Str("actual_type", fmt.Sprintf("%T", event.Data)).Msg("unexpected event data type, skipping") - return nil - } - - previewConfig := h.config.GetPreviewConfig() - previewName := ExtractPreviewName(payload.Reference, previewConfig.TagPatterns) - - routes := h.config.FindRoutesByImage(ctx, payload.Name) - baseRoutes := resolveBaseRoutes(routes, previewConfig) - if len(baseRoutes) == 0 { - log.Debug().Str("image", payload.Name).Msg("no trusted route found for preview base, skipping") - return nil - } - - allowedDomains := h.config.GetAllowedDomains() - - imageName := payload.Name + ":" + payload.Reference - for _, baseRoute := range baseRoutes { - previewDomain, err := GeneratePreviewDomain(baseRoute.Domain, previewName, previewConfig.Separator) - if err != nil { - log.Debug().Err(err).Str("base_domain", baseRoute.Domain).Msg("failed to generate preview domain") - continue - } - if !auto.MatchesDomainAllowlist(previewDomain, allowedDomains) { - log.Debug().Str("domain", previewDomain).Strs("patterns", allowedDomains).Msg("preview domain not in allowlist, skipping") - continue - } - - // Launch async — volume cloning may exceed event bus timeout. The bounded - // semaphore prevents a push burst from creating an unbounded goroutine backlog. - select { - case h.jobs <- struct{}{}: - case <-ctx.Done(): - return fmt.Errorf("schedule preview %s for %s: %w", previewName, baseRoute.Domain, ctx.Err()) - case <-h.serviceCtx.Done(): - return h.serviceCtx.Err() - } - go func(baseRoute baseRouteInfo, previewDomain string) { - defer func() { <-h.jobs }() - if err := h.previewService.CreatePreview(h.serviceCtx, CreatePreviewRequest{ - Name: previewName, - Domain: previewDomain, - BaseRoute: baseRoute.Domain, - Image: imageName, - HTTPS: baseRoute.HTTPS, - PreviewConfig: previewConfig, - }); err != nil { - log.Error().Err(err).Str("preview", previewName).Str("base_route", baseRoute.Domain).Msg("failed to create preview") - } - }(baseRoute, previewDomain) - } - - return nil -} - -// resolveBaseRoutes determines the eligible base routes for preview env/data inheritance. -type baseRouteInfo struct { - Domain string - HTTPS bool -} - -func resolveBaseRoutes(routes []domain.Route, previewConfig domain.PreviewConfig) []baseRouteInfo { - domainSet := make(map[string]struct{}, len(routes)) - for _, r := range routes { - domainSet[r.Domain] = struct{}{} - } - - baseRoutes := make([]baseRouteInfo, 0, len(routes)) - for _, r := range routes { - if !isPreviewDomain(r.Domain, previewConfig.Separator, domainSet) { - baseRoutes = append(baseRoutes, baseRouteInfo{Domain: r.Domain, HTTPS: r.HTTPS}) - } - } - return baseRoutes -} - -// isPreviewDomain checks if a domain is a preview domain by examining the first -// DNS label for the separator AND verifying that the implied base domain exists -// in the route set. This avoids false positives like "my--app.example.com" when -// no route "my.example.com" exists — that domain just happens to contain the -// separator in its name. -// If separator is empty, no domain is treated as a preview domain. -func isPreviewDomain(d, separator string, routeSet map[string]struct{}) bool { - if separator == "" { - return false - } - // Extract the first label (subdomain) before the first dot - firstLabel := d - rest := "" - if idx := strings.Index(d, "."); idx != -1 { - firstLabel = d[:idx] - rest = d[idx:] // includes leading dot - } - // Check if the first label contains the separator - sepIdx := strings.Index(firstLabel, separator) - if sepIdx < 0 { - return false - } - // Reconstruct the potential base domain: everything before the separator + rest - baseLabel := firstLabel[:sepIdx] - if baseLabel == "" { - return false - } - baseDomain := baseLabel + rest - // Only treat as preview if the base domain actually exists in the route set - _, exists := routeSet[baseDomain] - return exists -} diff --git a/internal/usecase/auto/preview/handler_test.go b/internal/usecase/auto/preview/handler_test.go deleted file mode 100644 index bce18a58c..000000000 --- a/internal/usecase/auto/preview/handler_test.go +++ /dev/null @@ -1,147 +0,0 @@ -package preview - -import ( - "context" - "sync" - "testing" - "time" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" - - "github.com/bnema/gordon/internal/domain" -) - -type captureStore struct { - mu sync.Mutex - previews []domain.PreviewRoute - history [][]domain.PreviewRoute -} - -func (s *captureStore) Load(_ context.Context) ([]domain.PreviewRoute, error) { - s.mu.Lock() - defer s.mu.Unlock() - cp := make([]domain.PreviewRoute, len(s.previews)) - copy(cp, s.previews) - return cp, nil -} - -func (s *captureStore) Save(_ context.Context, previews []domain.PreviewRoute) error { - s.mu.Lock() - defer s.mu.Unlock() - s.previews = make([]domain.PreviewRoute, len(previews)) - copy(s.previews, previews) - snapshot := make([]domain.PreviewRoute, len(previews)) - copy(snapshot, previews) - s.history = append(s.history, snapshot) - return nil -} - -func (s *captureStore) saves() [][]domain.PreviewRoute { - s.mu.Lock() - defer s.mu.Unlock() - cp := make([][]domain.PreviewRoute, len(s.history)) - for i := range s.history { - cp[i] = make([]domain.PreviewRoute, len(s.history[i])) - copy(cp[i], s.history[i]) - } - return cp -} - -type fakeAutoConfigProvider struct { - previewConfig domain.PreviewConfig - routesByImage map[string][]domain.Route - allowedDomains []string -} - -func (f *fakeAutoConfigProvider) IsAutoEnabled() bool { return false } -func (f *fakeAutoConfigProvider) IsPreviewEnabled() bool { return true } -func (f *fakeAutoConfigProvider) GetPreviewTagPatterns() []string { return f.previewConfig.TagPatterns } -func (f *fakeAutoConfigProvider) GetPreviewConfig() domain.PreviewConfig { return f.previewConfig } -func (f *fakeAutoConfigProvider) GetAllowedDomains() []string { return f.allowedDomains } -func (f *fakeAutoConfigProvider) FindRoutesByImage(_ context.Context, imageName string) []domain.Route { - return f.routesByImage[imageName] -} - -func TestAutoPreviewHandler_CanHandle(t *testing.T) { - h := &AutoPreviewHandler{} - assert.True(t, h.CanHandle(domain.EventImagePushed)) - assert.False(t, h.CanHandle(domain.EventConfigReload)) -} - -func TestResolveBaseRoutes_ReturnsAllEligibleBases(t *testing.T) { - routes := []domain.Route{ - {Domain: "app-preview-old.example.com", Image: "myapp", HTTPS: true}, - {Domain: "app.example.com", Image: "myapp", HTTPS: true}, - {Domain: "alias.example.com", Image: "myapp", HTTPS: false}, - } - - baseRoutes := resolveBaseRoutes(routes, domain.PreviewConfig{Separator: "-preview-"}) - assert.ElementsMatch(t, []baseRouteInfo{ - {Domain: "app.example.com", HTTPS: true}, - {Domain: "alias.example.com", HTTPS: false}, - }, baseRoutes) -} - -func TestResolveBaseRoutes_TreatsSeparatorWithoutBaseAsEligible(t *testing.T) { - routes := []domain.Route{ - {Domain: "my--app.example.com", Image: "myapp", HTTPS: false}, - {Domain: "other.example.com", Image: "myapp", HTTPS: true}, - } - - baseRoutes := resolveBaseRoutes(routes, domain.PreviewConfig{Separator: "--"}) - assert.ElementsMatch(t, []baseRouteInfo{ - {Domain: "my--app.example.com", HTTPS: false}, - {Domain: "other.example.com", HTTPS: true}, - }, baseRoutes) -} - -func TestAutoPreviewHandler_Handle_CreatesPreviewPerBaseRoute(t *testing.T) { - store := &captureStore{} - svc := NewService(store, time.Hour) - h := NewAutoPreviewHandler(t.Context(), &fakeAutoConfigProvider{ - previewConfig: domain.PreviewConfig{ - Separator: "--", - TagPatterns: []string{"preview-*"}, - TTL: time.Hour, - }, - allowedDomains: []string{"*"}, - routesByImage: map[string][]domain.Route{ - "myapp": { - {Domain: "app.example.com", Image: "myapp", HTTPS: true}, - {Domain: "alias.example.com", Image: "myapp", HTTPS: false}, - }, - }, - }, svc) - - err := h.Handle(t.Context(), domain.Event{ - ID: "evt-1", - Type: domain.EventImagePushed, - Data: domain.ImagePushedPayload{Name: "myapp", Reference: "preview-login"}, - }) - require.NoError(t, err) - - require.Eventually(t, func() bool { - for _, save := range store.saves() { - if len(save) != 2 { - continue - } - - pairs := map[string]domain.PreviewRoute{} - for _, preview := range save { - pairs[preview.Domain] = preview - } - - app, ok := pairs["app--login.example.com"] - if !ok || app.BaseRoute != "app.example.com" || !app.HTTPS { - continue - } - - alias, ok := pairs["alias--login.example.com"] - if ok && alias.BaseRoute == "alias.example.com" && !alias.HTTPS { - return true - } - } - return false - }, time.Second, 10*time.Millisecond) -} diff --git a/internal/usecase/auto/preview/integration_test.go b/internal/usecase/auto/preview/integration_test.go deleted file mode 100644 index 4b06ee038..000000000 --- a/internal/usecase/auto/preview/integration_test.go +++ /dev/null @@ -1,240 +0,0 @@ -package preview - -import ( - "context" - "fmt" - "testing" - "time" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" - - "github.com/bnema/gordon/internal/domain" -) - -type fakeDeployer struct { - deployFn func(ctx context.Context, route domain.Route) (*domain.Container, error) -} - -func (f *fakeDeployer) Deploy(ctx context.Context, route domain.Route) (*domain.Container, error) { - return f.deployFn(ctx, route) -} - -type fakeRouteManager struct { - addRouteFn func(ctx context.Context, route domain.Route) error - removeRouteFn func(ctx context.Context, domain string) error - getRouteFn func(ctx context.Context, domain string) (*domain.Route, error) -} - -func (f *fakeRouteManager) AddRoute(ctx context.Context, route domain.Route) error { - if f.addRouteFn != nil { - return f.addRouteFn(ctx, route) - } - return nil -} - -func (f *fakeRouteManager) RemoveRoute(ctx context.Context, d string) error { - if f.removeRouteFn != nil { - return f.removeRouteFn(ctx, d) - } - return nil -} - -func (f *fakeRouteManager) GetRoute(ctx context.Context, d string) (*domain.Route, error) { - if f.getRouteFn != nil { - return f.getRouteFn(ctx, d) - } - return nil, fmt.Errorf("route %q: %w", d, domain.ErrRouteNotFound) -} - -func (f *fakeRouteManager) GetVolumeConfig() (bool, string, bool) { - return true, "gordon", false -} - -func TestPreviewLifecycle_Integration(t *testing.T) { - store := &fakeStore{} - svc := NewService(store, 2*time.Second) - - ctx := t.Context() - - // Create - err := svc.CreatePreview(ctx, CreatePreviewRequest{ - Name: "test-feat", - Domain: "myapp--test-feat.example.com", - BaseRoute: "myapp.example.com", - Image: "myapp:preview-test-feat", - HTTPS: true, - PreviewConfig: domain.PreviewConfig{ - TTL: 2 * time.Second, - Separator: "--", - DataCopy: false, - }, - }) - require.NoError(t, err) - - // Verify created - all, err := svc.List(ctx) - require.NoError(t, err) - require.Len(t, all, 1) - assert.Equal(t, "test-feat", all[0].Name) - assert.Equal(t, "myapp--test-feat.example.com", all[0].Domain) - assert.Equal(t, "myapp.example.com", all[0].BaseRoute) - - // Extend - err = svc.Extend(ctx, "test-feat", 1*time.Hour) - require.NoError(t, err) - - p, err := svc.Get(ctx, "test-feat") - require.NoError(t, err) - assert.True(t, p.ExpiresAt.After(time.Now().Add(55*time.Minute))) - - // Delete - err = svc.Delete(ctx, "test-feat") - require.NoError(t, err) - - all, err = svc.List(ctx) - require.NoError(t, err) - assert.Empty(t, all) -} - -func TestPreviewLifecycle_CleanupExpired(t *testing.T) { - store := &fakeStore{} - svc := NewService(store, 1*time.Millisecond) - - ctx := t.Context() - - // Create with very short TTL - err := svc.CreatePreview(ctx, CreatePreviewRequest{ - Name: "short-lived", - Domain: "myapp--short-lived.example.com", - BaseRoute: "myapp.example.com", - Image: "myapp:preview-short-lived", - HTTPS: true, - PreviewConfig: domain.PreviewConfig{ - TTL: 1 * time.Millisecond, - Separator: "--", - DataCopy: false, - }, - }) - require.NoError(t, err) - - // Wait for expiry - time.Sleep(10 * time.Millisecond) - - // Cleanup should find it - expired := svc.CleanupExpired(ctx) - assert.Len(t, expired, 1) - assert.Equal(t, "short-lived", expired[0].Name) - - // Should be gone - all, err := svc.List(ctx) - require.NoError(t, err) - assert.Empty(t, all) -} - -type fakeEnvLoader struct { - loadEnvFn func(ctx context.Context, domain string) ([]string, error) -} - -func (f *fakeEnvLoader) LoadEnv(ctx context.Context, domain string) ([]string, error) { - if f.loadEnvFn != nil { - return f.loadEnvFn(ctx, domain) - } - return nil, nil -} - -func (f *fakeEnvLoader) CreateEnvFile(_ context.Context, _ string) error { return nil } -func (f *fakeEnvLoader) EnvFileExists(_ string) (bool, error) { return false, nil } - -func TestPreviewLifecycle_WithEnvInheritance(t *testing.T) { - store := &fakeStore{} - var deployedRoute domain.Route - - mockDeployer := &fakeDeployer{ - deployFn: func(_ context.Context, route domain.Route) (*domain.Container, error) { - deployedRoute = route - return &domain.Container{Name: "gordon-" + route.Domain}, nil - }, - } - mockRouteManager := &fakeRouteManager{} - mockEnvLoader := &fakeEnvLoader{ - loadEnvFn: func(_ context.Context, d string) ([]string, error) { - if d == "myapp.example.com" { - return []string{"DB_HOST=prod-db", "API_KEY=secret123"}, nil - } - return nil, nil - }, - } - - svc := NewService(store, 2*time.Second). - WithDeployer(mockDeployer). - WithRouteManager(mockRouteManager). - WithEnvLoader(mockEnvLoader). - WithRegistryDomain("reg.example.com") - - ctx := t.Context() - err := svc.CreatePreview(ctx, CreatePreviewRequest{ - Name: "test-feat", - Domain: "myapp--test-feat.example.com", - BaseRoute: "myapp.example.com", - Image: "myapp:preview-test-feat", - HTTPS: true, - PreviewConfig: domain.PreviewConfig{ - TTL: 2 * time.Second, - Separator: "--", - EnvCopy: true, - }, - }) - require.NoError(t, err) - - // Deployer should receive base route's env vars via Route.Env - assert.Equal(t, []string{"DB_HOST=prod-db", "API_KEY=secret123"}, deployedRoute.Env) -} - -func TestPreviewLifecycle_WithDeploy(t *testing.T) { - store := &fakeStore{} - deployed := false - routeAdded := "" - - mockDeployer := &fakeDeployer{ - deployFn: func(_ context.Context, route domain.Route) (*domain.Container, error) { - deployed = true - return &domain.Container{Name: "gordon-" + route.Domain}, nil - }, - } - mockRouteManager := &fakeRouteManager{ - addRouteFn: func(_ context.Context, route domain.Route) error { - routeAdded = route.Domain - return nil - }, - } - - svc := NewService(store, 2*time.Second). - WithDeployer(mockDeployer). - WithRouteManager(mockRouteManager). - WithRegistryDomain("reg.example.com") - - ctx := t.Context() - err := svc.CreatePreview(ctx, CreatePreviewRequest{ - Name: "test-feat", - Domain: "myapp--test-feat.example.com", - BaseRoute: "myapp.example.com", - Image: "myapp:preview-test-feat", - HTTPS: true, - PreviewConfig: domain.PreviewConfig{ - TTL: 2 * time.Second, - Separator: "--", - DataCopy: false, - }, - }) - require.NoError(t, err) - - assert.True(t, deployed, "deployer should have been called") - assert.Equal(t, "myapp--test-feat.example.com", routeAdded) - - all, err := svc.List(ctx) - require.NoError(t, err) - require.Len(t, all, 1) - assert.Equal(t, domain.PreviewStatusRunning, all[0].Status) - assert.Equal(t, []string{"gordon-myapp--test-feat.example.com"}, all[0].Containers) -} diff --git a/internal/usecase/auto/preview/naming.go b/internal/usecase/auto/preview/naming.go deleted file mode 100644 index 067e5cd67..000000000 --- a/internal/usecase/auto/preview/naming.go +++ /dev/null @@ -1,75 +0,0 @@ -package preview - -import ( - "fmt" - "path/filepath" - "strings" -) - -// ExtractPreviewName strips the matched pattern prefix from a tag. -// Only simple trailing-wildcard patterns like "preview-*" or "pr-*" are supported. -// Complex globs (e.g., "pr-*-test", "*-preview") are not supported. -// "preview-login-redesign" with pattern "preview-*" → "login-redesign" -func ExtractPreviewName(tag string, patterns []string) string { - for _, pattern := range patterns { - matched, err := filepath.Match(pattern, tag) - if err != nil { - continue // skip malformed patterns - } - if matched { - prefix := strings.TrimSuffix(pattern, "*") - return strings.TrimPrefix(tag, prefix) - } - } - return tag -} - -// GeneratePreviewDomain creates the preview domain from a base route domain. -// "myapp.example.com" + "login" + "--" → "myapp--login.example.com" -func GeneratePreviewDomain(baseRoute, previewName, separator string) (string, error) { - dot := strings.Index(baseRoute, ".") - if dot < 0 { - return "", fmt.Errorf("invalid base route domain: %s", baseRoute) - } - app := baseRoute[:dot] - rest := baseRoute[dot:] // includes leading dot - return app + separator + previewName + rest, nil -} - -// SanitizeBranchName converts a git branch name to a valid DNS-safe preview name. -// "feat/login-redesign" → "login-redesign" -func SanitizeBranchName(branch string) string { - branch = strings.ToLower(branch) - for _, prefix := range []string{"feat/", "fix/", "feature/", "hotfix/", "release/", "chore/"} { - if strings.HasPrefix(branch, prefix) { - branch = strings.TrimPrefix(branch, prefix) - break - } - } - // Replace disallowed characters with dashes - branch = strings.ReplaceAll(branch, "/", "-") - branch = strings.ReplaceAll(branch, "_", "-") - branch = strings.ReplaceAll(branch, ".", "-") - - // Remove any remaining non DNS-safe characters - var result strings.Builder - for _, c := range branch { - if (c >= 'a' && c <= 'z') || (c >= '0' && c <= '9') || c == '-' { - result.WriteRune(c) - } - } - branch = result.String() - - // Collapse consecutive dashes and trim - for strings.Contains(branch, "--") { - branch = strings.ReplaceAll(branch, "--", "-") - } - branch = strings.Trim(branch, "-") - - // Truncate to 63 chars (DNS label limit) - if len(branch) > 63 { - branch = branch[:63] - branch = strings.TrimRight(branch, "-") - } - return branch -} diff --git a/internal/usecase/auto/preview/naming_test.go b/internal/usecase/auto/preview/naming_test.go deleted file mode 100644 index 50e97a2d8..000000000 --- a/internal/usecase/auto/preview/naming_test.go +++ /dev/null @@ -1,80 +0,0 @@ -package preview - -import ( - "strings" - "testing" - - "github.com/stretchr/testify/assert" -) - -func TestExtractPreviewName(t *testing.T) { - tests := []struct { - name string - tag string - patterns []string - want string - }{ - {"simple prefix", "preview-login", []string{"preview-*"}, "login"}, - {"pr prefix", "pr-42", []string{"pr-*"}, "42"}, - {"multi-word", "preview-login-redesign", []string{"preview-*"}, "login-redesign"}, - {"second pattern matches", "pr-99", []string{"preview-*", "pr-*"}, "99"}, - {"no match", "v1.0.0", []string{"preview-*"}, "v1.0.0"}, - } - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - assert.Equal(t, tt.want, ExtractPreviewName(tt.tag, tt.patterns)) - }) - } -} - -func TestGeneratePreviewDomain(t *testing.T) { - tests := []struct { - name string - baseRoute string - previewName string - separator string - want string - wantErr bool - }{ - {"flat default", "myapp.example.com", "login", "--", "myapp--login.example.com", false}, - {"flat with dashes", "my-app.example.com", "feat-123", "--", "my-app--feat-123.example.com", false}, - {"multi-level base", "myapp.sub.example.com", "feat", "--", "myapp--feat.sub.example.com", false}, - {"invalid domain", "nodot", "feat", "--", "", true}, - } - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - got, err := GeneratePreviewDomain(tt.baseRoute, tt.previewName, tt.separator) - if tt.wantErr { - assert.Error(t, err) - } else { - assert.NoError(t, err) - assert.Equal(t, tt.want, got) - } - }) - } -} - -func TestSanitizeBranchName(t *testing.T) { - tests := []struct { - name string - branch string - want string - }{ - {"feature branch", "feat/login-redesign", "login-redesign"}, - {"fix branch", "fix/bug-123", "bug-123"}, - {"simple", "main", "main"}, - {"nested", "feat/ui/button", "ui-button"}, - {"uppercase", "Feat/MyBranch", "mybranch"}, - {"dots and underscores", "feat/my_branch.name", "my-branch-name"}, - {"special chars", "feat/hello@world!", "helloworld"}, - {"long branch", strings.Repeat("a", 70), strings.Repeat("a", 63)}, - {"consecutive dashes", "feat/a--b--c", "a-b-c"}, - {"empty branch", "", ""}, - {"whitespace branch", " ", ""}, - } - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - assert.Equal(t, tt.want, SanitizeBranchName(tt.branch)) - }) - } -} diff --git a/internal/usecase/auto/preview/orphans.go b/internal/usecase/auto/preview/orphans.go deleted file mode 100644 index e7710fec8..000000000 --- a/internal/usecase/auto/preview/orphans.go +++ /dev/null @@ -1,80 +0,0 @@ -package preview - -import ( - "path/filepath" - "strings" - "time" - - "github.com/bnema/gordon/internal/domain" -) - -const orphanGracePeriod = 5 * time.Minute - -// IsOrphanPreview returns true if the container looks like a preview (by image -// tag pattern and domain separator) but is not in the tracked preview list. -// Containers younger than 5 minutes are skipped to avoid racing with -// CreatePreview. -func IsOrphanPreview(c *domain.Container, tracked []domain.PreviewRoute, tagPatterns []string, separator string) bool { - imgLabel := c.Labels[domain.LabelImage] - domLabel := c.Labels[domain.LabelDomain] - if imgLabel == "" || domLabel == "" { - return false - } - - // Grace period: skip very recent containers. - if !c.Created.IsZero() && time.Since(c.Created) < orphanGracePeriod { - return false - } - - // Condition 1: image tag matches a preview pattern. - tag := tagFromImage(imgLabel) - if !matchesAnyPattern(tag, tagPatterns) { - return false - } - - // Condition 2: domain contains the separator. - if !strings.Contains(domLabel, separator) { - return false - } - - // Not an orphan if tracked. - for _, p := range tracked { - if p.Domain == domLabel { - return false - } - } - return true -} - -// IsExpiredOrphan returns true if the container has exceeded the TTL based on -// its creation time. -func IsExpiredOrphan(c *domain.Container, ttl time.Duration) bool { - if c.Created.IsZero() { - return false // unknown creation time — don't delete, play it safe - } - return time.Since(c.Created) > ttl -} - -// tagFromImage extracts the tag portion from an image reference. -// "myapp:preview-feat" -> "preview-feat" -// "registry.example.com/myapp:preview-feat" -> "preview-feat" -// "myapp" -> "" (no tag) -func tagFromImage(image string) string { - if idx := strings.Index(image, "@"); idx != -1 { - return "" - } - if idx := strings.LastIndex(image, ":"); idx != -1 { - return image[idx+1:] - } - return "" -} - -// matchesAnyPattern checks if the tag matches any of the glob patterns. -func matchesAnyPattern(tag string, patterns []string) bool { - for _, p := range patterns { - if matched, _ := filepath.Match(p, tag); matched { - return true - } - } - return false -} diff --git a/internal/usecase/auto/preview/orphans_test.go b/internal/usecase/auto/preview/orphans_test.go deleted file mode 100644 index 84d834ed9..000000000 --- a/internal/usecase/auto/preview/orphans_test.go +++ /dev/null @@ -1,165 +0,0 @@ -package preview - -import ( - "testing" - "time" - - "github.com/stretchr/testify/assert" - - "github.com/bnema/gordon/internal/domain" -) - -func TestIsOrphanPreview(t *testing.T) { - patterns := []string{"preview-*"} - separator := "--" - - tests := []struct { - name string - c *domain.Container - tracked []domain.PreviewRoute - want bool - }{ - { - name: "orphan preview container", - c: &domain.Container{ - Created: time.Now().Add(-1 * time.Hour), - Labels: map[string]string{ - domain.LabelImage: "myapp:preview-feat", - domain.LabelDomain: "myapp--feat.example.com", - }, - }, - tracked: nil, - want: true, - }, - { - name: "tracked preview is not orphan", - c: &domain.Container{ - Created: time.Now().Add(-1 * time.Hour), - Labels: map[string]string{ - domain.LabelImage: "myapp:preview-feat", - domain.LabelDomain: "myapp--feat.example.com", - }, - }, - tracked: []domain.PreviewRoute{ - {Domain: "myapp--feat.example.com"}, - }, - want: false, - }, - { - name: "production container — no separator in domain", - c: &domain.Container{ - Created: time.Now().Add(-1 * time.Hour), - Labels: map[string]string{ - domain.LabelImage: "myapp:preview-release", - domain.LabelDomain: "myapp.example.com", - }, - }, - tracked: nil, - want: false, - }, - { - name: "production container — no preview tag", - c: &domain.Container{ - Created: time.Now().Add(-1 * time.Hour), - Labels: map[string]string{ - domain.LabelImage: "myapp:latest", - domain.LabelDomain: "myapp--feat.example.com", - }, - }, - tracked: nil, - want: false, - }, - { - name: "container without gordon labels", - c: &domain.Container{ - Labels: map[string]string{}, - }, - tracked: nil, - want: false, - }, - { - name: "container too young — grace period", - c: &domain.Container{ - Created: time.Now().Add(-2 * time.Minute), - Labels: map[string]string{ - domain.LabelImage: "myapp:preview-feat", - domain.LabelDomain: "myapp--feat.example.com", - }, - }, - tracked: nil, - want: false, - }, - { - name: "digest-only image — no tag to match", - c: &domain.Container{ - Created: time.Now().Add(-1 * time.Hour), - Labels: map[string]string{ - domain.LabelImage: "myapp@sha256:abc123", - domain.LabelDomain: "myapp--feat.example.com", - }, - }, - tracked: nil, - want: false, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - got := IsOrphanPreview(tt.c, tt.tracked, patterns, separator) - assert.Equal(t, tt.want, got) - }) - } -} - -func TestIsExpiredOrphan(t *testing.T) { - ttl := 48 * time.Hour - - tests := []struct { - name string - created time.Time - want bool - }{ - { - name: "expired — older than TTL", - created: time.Now().Add(-72 * time.Hour), - want: true, - }, - { - name: "not expired — within TTL", - created: time.Now().Add(-12 * time.Hour), - want: false, - }, - { - name: "zero time — unknown, skip to be safe", - created: time.Time{}, - want: false, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - c := &domain.Container{Created: tt.created} - got := IsExpiredOrphan(c, ttl) - assert.Equal(t, tt.want, got) - }) - } -} - -func TestTagFromImage(t *testing.T) { - tests := []struct { - image string - want string - }{ - {"myapp:preview-feat", "preview-feat"}, - {"registry.example.com/myapp:preview-feat", "preview-feat"}, - {"myapp:latest", "latest"}, - {"myapp", ""}, - {"myapp@sha256:abc123", ""}, - } - - for _, tt := range tests { - t.Run(tt.image, func(t *testing.T) { - assert.Equal(t, tt.want, tagFromImage(tt.image)) - }) - } -} diff --git a/internal/usecase/auto/preview/service.go b/internal/usecase/auto/preview/service.go deleted file mode 100644 index 6a299bb98..000000000 --- a/internal/usecase/auto/preview/service.go +++ /dev/null @@ -1,477 +0,0 @@ -package preview - -import ( - "context" - "fmt" - "strings" - "sync" - "time" - - "github.com/bnema/zerowrap" - - "github.com/bnema/gordon/internal/boundaries/out" - "github.com/bnema/gordon/internal/domain" -) - -type nameLockEntry struct { - mu *sync.Mutex - refs int -} - -// Deployer abstracts the container service deploy method. -type Deployer interface { - Deploy(ctx context.Context, route domain.Route) (*domain.Container, error) -} - -// RouteManager abstracts config service methods needed for preview route lifecycle. -type RouteManager interface { - AddRoute(ctx context.Context, route domain.Route) error - RemoveRoute(ctx context.Context, domain string) error - GetRoute(ctx context.Context, domain string) (*domain.Route, error) - GetVolumeConfig() (autoCreate bool, prefix string, preserve bool) -} - -// Service manages preview lifecycle: CRUD, TTL, persistence. -type Service struct { - store out.PreviewStore - defaultTTL time.Duration - deployer Deployer - routeManager RouteManager - volumeCloner VolumeCloner - envLoader out.EnvLoader - registryDomain string - previews []domain.PreviewRoute - mu sync.RWMutex - nameLocks map[string]*nameLockEntry - nameLockMu sync.Mutex -} - -func NewService(store out.PreviewStore, defaultTTL time.Duration) *Service { - return &Service{ - store: store, - defaultTTL: defaultTTL, - nameLocks: make(map[string]*nameLockEntry), - } -} - -func (s *Service) WithDeployer(d Deployer) *Service { - s.deployer = d - return s -} - -func (s *Service) WithRouteManager(rm RouteManager) *Service { - s.routeManager = rm - return s -} - -func (s *Service) WithVolumeCloner(vc VolumeCloner) *Service { - s.volumeCloner = vc - return s -} - -func (s *Service) WithRegistryDomain(d string) *Service { - s.registryDomain = d - return s -} - -func (s *Service) WithEnvLoader(el out.EnvLoader) *Service { - s.envLoader = el - return s -} - -func (s *Service) Load(ctx context.Context) error { - previews, err := s.store.Load(ctx) - if err != nil { - return err - } - s.mu.Lock() - s.previews = previews - s.mu.Unlock() - return nil -} - -func (s *Service) Add(ctx context.Context, p domain.PreviewRoute) error { - s.mu.Lock() - replaced := false - for i := range s.previews { - if s.previews[i].Domain == p.Domain { - s.previews[i] = p - replaced = true - break - } - } - if !replaced { - s.previews = append(s.previews, p) - } - cp := make([]domain.PreviewRoute, len(s.previews)) - copy(cp, s.previews) - s.mu.Unlock() - return s.store.Save(ctx, cp) -} - -func (s *Service) Update(ctx context.Context, p domain.PreviewRoute) error { - s.mu.Lock() - for i := range s.previews { - if s.previews[i].Domain == p.Domain { - s.previews[i] = p - cp := make([]domain.PreviewRoute, len(s.previews)) - copy(cp, s.previews) - s.mu.Unlock() - return s.store.Save(ctx, cp) - } - } - s.mu.Unlock() - return fmt.Errorf("preview %q: %w", p.Name, domain.ErrPreviewNotFound) -} - -func (s *Service) Delete(ctx context.Context, name string) error { - s.mu.Lock() - var deleted domain.PreviewRoute - found := false - filtered := make([]domain.PreviewRoute, 0, len(s.previews)) - for _, p := range s.previews { - if p.Name == name { - found = true - deleted = p - } else { - filtered = append(filtered, p) - } - } - if !found { - s.mu.Unlock() - return fmt.Errorf("preview %q: %w", name, domain.ErrPreviewNotFound) - } - s.previews = filtered - cp := make([]domain.PreviewRoute, len(s.previews)) - copy(cp, s.previews) - s.mu.Unlock() - - if err := s.store.Save(ctx, cp); err != nil { - return err - } - - // Clean up the route from config so proxy stops routing to this domain. - if s.routeManager != nil && deleted.Domain != "" { - log := zerowrap.FromCtx(ctx) - if err := s.routeManager.RemoveRoute(ctx, deleted.Domain); err != nil { - log.Debug().Err(err).Str("domain", deleted.Domain).Msg("preview route already removed from config") - } - } - return nil -} - -func (s *Service) Get(_ context.Context, name string) (*domain.PreviewRoute, error) { - s.mu.RLock() - defer s.mu.RUnlock() - for i := range s.previews { - if s.previews[i].Name == name { - p := s.previews[i] - return &p, nil - } - } - return nil, fmt.Errorf("preview %q: %w", name, domain.ErrPreviewNotFound) -} - -func (s *Service) List(_ context.Context) ([]domain.PreviewRoute, error) { - s.mu.RLock() - defer s.mu.RUnlock() - cp := make([]domain.PreviewRoute, len(s.previews)) - copy(cp, s.previews) - return cp, nil -} - -func (s *Service) Extend(ctx context.Context, name string, ttl time.Duration) error { - s.mu.Lock() - for i := range s.previews { - if s.previews[i].Name == name { - s.previews[i].ExpiresAt = time.Now().Add(ttl) - cp := make([]domain.PreviewRoute, len(s.previews)) - copy(cp, s.previews) - s.mu.Unlock() - return s.store.Save(ctx, cp) - } - } - s.mu.Unlock() - return fmt.Errorf("preview %q: %w", name, domain.ErrPreviewNotFound) -} - -func (s *Service) GetExpired() []domain.PreviewRoute { - s.mu.RLock() - defer s.mu.RUnlock() - var expired []domain.PreviewRoute - for _, p := range s.previews { - if p.IsExpired(time.Now()) { - expired = append(expired, p) - } - } - return expired -} - -// CleanupExpired removes expired previews from the service and persists. -// Returns the removed previews so the caller can teardown their resources. -func (s *Service) CleanupExpired(ctx context.Context) []domain.PreviewRoute { - s.mu.Lock() - var expired []domain.PreviewRoute - var active []domain.PreviewRoute - for _, p := range s.previews { - if p.IsExpired(time.Now()) { - expired = append(expired, p) - } else { - active = append(active, p) - } - } - s.previews = active - cp := make([]domain.PreviewRoute, len(active)) - copy(cp, active) - s.mu.Unlock() - - if len(expired) > 0 { - if err := s.store.Save(ctx, cp); err != nil { - // Log but don't fail — expired previews are already removed from memory - log := zerowrap.FromCtx(ctx) - log.Warn().Err(err).Msg("failed to persist preview state after cleanup") - } - } - return expired -} - -// CollectOrphans lists Docker containers and returns those that look like -// expired preview orphans (not tracked, matching tag pattern + separator, -// past TTL based on container creation time). -func (s *Service) CollectOrphans(ctx context.Context, lister out.ContainerLister, tagPatterns []string, separator string) []*domain.Container { - containers, err := lister.ListContainers(ctx, true) - if err != nil { - log := zerowrap.FromCtx(ctx) - log.Warn().Err(err).Msg("orphan GC: failed to list containers") - return nil - } - - s.mu.RLock() - tracked := make([]domain.PreviewRoute, len(s.previews)) - copy(tracked, s.previews) - s.mu.RUnlock() - - var orphans []*domain.Container - for _, c := range containers { - if IsOrphanPreview(c, tracked, tagPatterns, separator) && IsExpiredOrphan(c, s.defaultTTL) { - orphans = append(orphans, c) - } - } - return orphans -} - -// StartTicker starts a background goroutine that checks for expired previews -// and orphaned preview containers. The teardownFn is called for each expired -// tracked preview. The orphanGCFn, if non-nil, is called on each tick to -// clean up orphaned containers. -func (s *Service) StartTicker(ctx context.Context, interval time.Duration, teardownFn func(context.Context, domain.PreviewRoute), orphanGCFn func(context.Context)) { - go func() { - ticker := time.NewTicker(interval) - defer ticker.Stop() - for { - select { - case <-ctx.Done(): - return - case <-ticker.C: - expired := s.CleanupExpired(ctx) - for _, p := range expired { - teardownFn(ctx, p) - } - if orphanGCFn != nil { - orphanGCFn(ctx) - } - } - } - }() -} - -func (s *Service) AcquireNameLock(name string) *sync.Mutex { - s.nameLockMu.Lock() - defer s.nameLockMu.Unlock() - entry, ok := s.nameLocks[name] - if !ok { - entry = &nameLockEntry{mu: &sync.Mutex{}} - s.nameLocks[name] = entry - } - entry.refs++ - return entry.mu -} - -// ReleaseNameLock decrements the refcount and removes the per-name mutex when no longer in use. -func (s *Service) ReleaseNameLock(name string) { - s.nameLockMu.Lock() - defer s.nameLockMu.Unlock() - entry, ok := s.nameLocks[name] - if !ok { - return - } - entry.refs-- - if entry.refs <= 0 { - delete(s.nameLocks, name) - } -} - -func (s *Service) qualifyImage(image string) string { - if s.registryDomain != "" && !strings.Contains(image, "/") { - return s.registryDomain + "/" + image - } - return image -} - -// generateVolumeName mirrors the naming convention in container/service.go. -// This MUST stay in sync with the container service's generateVolumeName. -func generateVolumeName(prefix, domainName, volumePath string) string { - return fmt.Sprintf("%s-%s-%s", - prefix, - domain.StableResourceName(domainName), - domain.StableResourceName(strings.Trim(volumePath, "/"))) -} - -func (s *Service) cloneBaseRouteVolumes(ctx context.Context, baseRoute, previewDomain string) ([]string, error) { - if s.volumeCloner == nil || s.routeManager == nil { - return nil, nil - } - - _, prefix, _ := s.routeManager.GetVolumeConfig() - if prefix == "" { - prefix = "gordon" - } - - log := zerowrap.FromCtx(ctx) - basePrefix := prefix + "-" + domain.StableResourceName(baseRoute) + "-" - legacyBasePrefix := prefix + "-" + strings.ReplaceAll(baseRoute, ".", "-") + "-" - - vols, err := s.volumeCloner.ListVolumes(ctx) - if err != nil { - return nil, fmt.Errorf("list volumes: %w", err) - } - - var sourceVols []string - for _, v := range vols { - if strings.HasPrefix(v.Name, basePrefix) || strings.HasPrefix(v.Name, legacyBasePrefix) { - sourceVols = append(sourceVols, v.Name) - } - } - - if len(sourceVols) == 0 { - return nil, nil - } - - namer := func(sourceVolName string) string { - pathSuffix := strings.TrimPrefix(sourceVolName, basePrefix) - if pathSuffix == sourceVolName { - pathSuffix = strings.TrimPrefix(sourceVolName, legacyBasePrefix) - } - return generateVolumeName(prefix, previewDomain, pathSuffix) - } - - for _, src := range sourceVols { - log.Debug().Str("source", src).Str("target", namer(src)).Msg("will clone volume for preview") - } - - return CloneVolumes(ctx, s.volumeCloner, namer, sourceVols) -} - -// CreatePreviewRequest contains parameters for creating a preview. -type CreatePreviewRequest struct { - Name string - Domain string - BaseRoute string - Image string - HTTPS bool - PreviewConfig domain.PreviewConfig -} - -// CreatePreview orchestrates the full preview creation: lock, clone volumes, start containers, register. -func (s *Service) CreatePreview(ctx context.Context, req CreatePreviewRequest) error { - lock := s.AcquireNameLock(req.Domain) - lock.Lock() - defer func() { - lock.Unlock() - s.ReleaseNameLock(req.Domain) - }() - - log := zerowrap.FromCtx(ctx) - now := time.Now() - - preview := domain.PreviewRoute{ - Domain: req.Domain, - Image: req.Image, - BaseRoute: req.BaseRoute, - Name: req.Name, - CreatedAt: now, - ExpiresAt: now.Add(req.PreviewConfig.TTL), - HTTPS: req.HTTPS, - Status: domain.PreviewStatusDeploying, - } - - if err := s.Add(ctx, preview); err != nil { - return fmt.Errorf("register preview %q: %w", req.Name, err) - } - - if req.PreviewConfig.DataCopy { - cloned, err := s.cloneBaseRouteVolumes(ctx, req.BaseRoute, req.Domain) - if err != nil { - log.Warn().Err(err).Str("base_route", req.BaseRoute).Msg("failed to clone volumes, deploying with empty volumes") - } else if len(cloned) > 0 { - preview.Volumes = cloned - } - } - - if s.routeManager != nil { - imageRef := s.qualifyImage(req.Image) - if err := s.routeManager.AddRoute(ctx, domain.Route{ - Domain: req.Domain, - Image: imageRef, - HTTPS: req.HTTPS, - }); err != nil { - preview.Status = domain.PreviewStatusFailed - if updateErr := s.Update(ctx, preview); updateErr != nil { - log.Warn().Err(updateErr).Str("preview", req.Name).Msg("failed to update preview status") - } - return fmt.Errorf("register preview route %q: %w", req.Domain, err) - } - } - - var baseEnv []string - if s.envLoader != nil && req.PreviewConfig.EnvCopy { - var envErr error - baseEnv, envErr = s.envLoader.LoadEnv(ctx, req.BaseRoute) - if envErr != nil { - log.Warn().Err(envErr).Str("base_route", req.BaseRoute).Msg("failed to load base route env, deploying without env vars") - } - } - - if s.deployer != nil { - imageRef := s.qualifyImage(req.Image) - deployCtx := domain.WithInternalDeploy(ctx) - container, err := s.deployer.Deploy(deployCtx, domain.Route{ - Domain: req.Domain, - Image: imageRef, - HTTPS: req.HTTPS, - Env: baseEnv, - }) - if err != nil { - preview.Status = domain.PreviewStatusFailed - if updateErr := s.Update(ctx, preview); updateErr != nil { - log.Warn().Err(updateErr).Str("preview", req.Name).Msg("failed to update preview status") - } - return fmt.Errorf("deploy preview %q: %w", req.Name, err) - } - preview.Containers = []string{container.Name} - } - - preview.Status = domain.PreviewStatusRunning - if err := s.Update(ctx, preview); err != nil { - return fmt.Errorf("finalize preview %q: %w", req.Name, err) - } - - log.Info(). - Str("name", req.Name). - Str("domain", req.Domain). - Str("image", req.Image). - Str("base_route", req.BaseRoute). - Msg("preview environment created") - - return nil -} diff --git a/internal/usecase/auto/preview/service_test.go b/internal/usecase/auto/preview/service_test.go deleted file mode 100644 index ec5b51c7b..000000000 --- a/internal/usecase/auto/preview/service_test.go +++ /dev/null @@ -1,215 +0,0 @@ -package preview - -import ( - "context" - "testing" - "time" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" - - "github.com/bnema/gordon/internal/domain" -) - -type fakeStore struct { - previews []domain.PreviewRoute -} - -func (f *fakeStore) Load(_ context.Context) ([]domain.PreviewRoute, error) { - return f.previews, nil -} -func (f *fakeStore) Save(_ context.Context, p []domain.PreviewRoute) error { - f.previews = p - return nil -} - -func TestPreviewService_Add(t *testing.T) { - store := &fakeStore{} - svc := NewService(store, 48*time.Hour) - - p := domain.PreviewRoute{ - Domain: "myapp--feat.example.com", - Name: "feat", - BaseRoute: "myapp.example.com", - Image: "myapp:preview-feat", - } - err := svc.Add(t.Context(), p) - require.NoError(t, err) - - all, err := svc.List(t.Context()) - require.NoError(t, err) - assert.Len(t, all, 1) - assert.Equal(t, "feat", all[0].Name) -} - -func TestPreviewService_Delete(t *testing.T) { - store := &fakeStore{ - previews: []domain.PreviewRoute{ - {Name: "feat", Domain: "myapp--feat.example.com"}, - }, - } - svc := NewService(store, 48*time.Hour) - require.NoError(t, svc.Load(t.Context())) - - err := svc.Delete(t.Context(), "feat") - require.NoError(t, err) - - all, err := svc.List(t.Context()) - require.NoError(t, err) - assert.Empty(t, all) -} - -func TestPreviewService_Get(t *testing.T) { - store := &fakeStore{ - previews: []domain.PreviewRoute{ - {Name: "feat", Domain: "myapp--feat.example.com"}, - }, - } - svc := NewService(store, 48*time.Hour) - require.NoError(t, svc.Load(t.Context())) - - p, err := svc.Get(t.Context(), "feat") - require.NoError(t, err) - assert.Equal(t, "myapp--feat.example.com", p.Domain) - - _, err = svc.Get(t.Context(), "nonexistent") - assert.Error(t, err) -} - -func TestPreviewService_Extend(t *testing.T) { - now := time.Now() - store := &fakeStore{ - previews: []domain.PreviewRoute{ - {Name: "feat", ExpiresAt: now.Add(1 * time.Hour)}, - }, - } - svc := NewService(store, 48*time.Hour) - require.NoError(t, svc.Load(t.Context())) - - err := svc.Extend(t.Context(), "feat", 24*time.Hour) - require.NoError(t, err) - - p, err := svc.Get(t.Context(), "feat") - require.NoError(t, err) - assert.True(t, p.ExpiresAt.After(now.Add(23*time.Hour))) -} - -func TestPreviewService_GetExpired(t *testing.T) { - store := &fakeStore{ - previews: []domain.PreviewRoute{ - {Name: "expired", ExpiresAt: time.Now().Add(-1 * time.Hour)}, - {Name: "active", ExpiresAt: time.Now().Add(1 * time.Hour)}, - }, - } - svc := NewService(store, 48*time.Hour) - require.NoError(t, svc.Load(t.Context())) - - expired := svc.GetExpired() - assert.Len(t, expired, 1) - assert.Equal(t, "expired", expired[0].Name) -} - -func TestPreviewService_Update(t *testing.T) { - store := &fakeStore{ - previews: []domain.PreviewRoute{ - {Name: "feat", Domain: "myapp--feat.example.com", Status: domain.PreviewStatusDeploying}, - }, - } - svc := NewService(store, 48*time.Hour) - require.NoError(t, svc.Load(t.Context())) - - updated := domain.PreviewRoute{ - Name: "feat", - Domain: "myapp--feat.example.com", - Status: domain.PreviewStatusRunning, - Containers: []string{"gordon-myapp--feat.example.com"}, - } - err := svc.Update(t.Context(), updated) - require.NoError(t, err) - - p, err := svc.Get(t.Context(), "feat") - require.NoError(t, err) - assert.Equal(t, domain.PreviewStatusRunning, p.Status) - assert.Equal(t, []string{"gordon-myapp--feat.example.com"}, p.Containers) -} - -func TestPreviewService_Update_NotFound(t *testing.T) { - store := &fakeStore{} - svc := NewService(store, 48*time.Hour) - - err := svc.Update(t.Context(), domain.PreviewRoute{Name: "nonexistent"}) - assert.ErrorIs(t, err, domain.ErrPreviewNotFound) -} - -type fakeRuntime struct { - containers []*domain.Container -} - -func (f *fakeRuntime) ListContainers(_ context.Context, _ bool) ([]*domain.Container, error) { - return f.containers, nil -} - -func TestPreviewService_CollectOrphans(t *testing.T) { - store := &fakeStore{ - previews: []domain.PreviewRoute{ - {Name: "tracked", Domain: "myapp--tracked.example.com"}, - }, - } - svc := NewService(store, 48*time.Hour) - require.NoError(t, svc.Load(t.Context())) - - runtime := &fakeRuntime{ - containers: []*domain.Container{ - { - Name: "gordon-myapp--orphan.example.com", - Created: time.Now().Add(-72 * time.Hour), - Labels: map[string]string{ - domain.LabelImage: "myapp:preview-orphan", - domain.LabelDomain: "myapp--orphan.example.com", - }, - }, - { - Name: "gordon-myapp--tracked.example.com", - Created: time.Now().Add(-72 * time.Hour), - Labels: map[string]string{ - domain.LabelImage: "myapp:preview-tracked", - domain.LabelDomain: "myapp--tracked.example.com", - }, - }, - { - Name: "gordon-prod.example.com", - Created: time.Now().Add(-72 * time.Hour), - Labels: map[string]string{ - domain.LabelImage: "myapp:latest", - domain.LabelDomain: "prod.example.com", - }, - }, - }, - } - - orphans := svc.CollectOrphans(t.Context(), runtime, []string{"preview-*"}, "--") - assert.Len(t, orphans, 1) - assert.Equal(t, "gordon-myapp--orphan.example.com", orphans[0].Name) -} - -func TestPreviewService_CollectOrphans_NotExpired(t *testing.T) { - store := &fakeStore{} - svc := NewService(store, 48*time.Hour) - require.NoError(t, svc.Load(t.Context())) - - runtime := &fakeRuntime{ - containers: []*domain.Container{ - { - Name: "gordon-myapp--recent.example.com", - Created: time.Now().Add(-12 * time.Hour), - Labels: map[string]string{ - domain.LabelImage: "myapp:preview-recent", - domain.LabelDomain: "myapp--recent.example.com", - }, - }, - }, - } - - orphans := svc.CollectOrphans(t.Context(), runtime, []string{"preview-*"}, "--") - assert.Empty(t, orphans) -} diff --git a/internal/usecase/auto/preview/ticker_test.go b/internal/usecase/auto/preview/ticker_test.go deleted file mode 100644 index d24ea8b26..000000000 --- a/internal/usecase/auto/preview/ticker_test.go +++ /dev/null @@ -1,31 +0,0 @@ -package preview - -import ( - "testing" - "time" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" - - "github.com/bnema/gordon/internal/domain" -) - -func TestPreviewService_CleanupExpired(t *testing.T) { - store := &fakeStore{ - previews: []domain.PreviewRoute{ - {Name: "expired", Domain: "a--expired.example.com", ExpiresAt: time.Now().Add(-1 * time.Hour)}, - {Name: "active", Domain: "a--active.example.com", ExpiresAt: time.Now().Add(24 * time.Hour)}, - }, - } - svc := NewService(store, 48*time.Hour) - require.NoError(t, svc.Load(t.Context())) - - expired := svc.CleanupExpired(t.Context()) - assert.Len(t, expired, 1) - assert.Equal(t, "expired", expired[0].Name) - - all, err := svc.List(t.Context()) - require.NoError(t, err) - assert.Len(t, all, 1) - assert.Equal(t, "active", all[0].Name) -} diff --git a/internal/usecase/auto/preview/volumes.go b/internal/usecase/auto/preview/volumes.go deleted file mode 100644 index d4420cdd2..000000000 --- a/internal/usecase/auto/preview/volumes.go +++ /dev/null @@ -1,104 +0,0 @@ -package preview - -import ( - "context" - "fmt" - "strings" - - "github.com/bnema/gordon/internal/domain" -) - -// VolumeCloner abstracts container runtime operations needed for volume cloning. -type VolumeCloner interface { - CreateVolume(ctx context.Context, name string) error - RemoveVolume(ctx context.Context, name string, force bool) error - CreateContainer(ctx context.Context, config *domain.ContainerConfig) (*domain.Container, error) - StartContainer(ctx context.Context, containerID string) error - WaitForContainer(ctx context.Context, containerID string) error - RemoveContainer(ctx context.Context, containerID string, force bool) error - ListVolumes(ctx context.Context) ([]*domain.VolumeInfo, error) -} - -// VolumeNamer is a function that maps a source volume name to a destination volume name. -type VolumeNamer func(sourceVolName string) string - -// DefaultNamer returns a VolumeNamer that uses BuildCloneVolumeName with the given preview name. -func DefaultNamer(previewName string) VolumeNamer { - return func(sourceVolName string) string { - return BuildCloneVolumeName(previewName, sourceVolName) - } -} - -// BuildCloneVolumeName generates a preview volume name. -func BuildCloneVolumeName(previewName, volumeName string) string { - return "preview-" + previewName + "-" + volumeName -} - -// BuildCloneContainerName generates a preview attachment container name from an image reference. -func BuildCloneContainerName(previewName, image string) string { - name := image - if i := strings.LastIndex(name, "/"); i >= 0 { - name = name[i+1:] - } - if i := strings.Index(name, ":"); i >= 0 { - name = name[:i] - } - return "preview-" + previewName + "-" + name -} - -// CloneVolumes copies volumes from source using a read-only helper container. -// sourceVolumes is a list of actual Docker volume names to clone. -func CloneVolumes(ctx context.Context, cloner VolumeCloner, namer VolumeNamer, sourceVolumes []string) ([]string, error) { - var clonedNames []string - for _, sourceVolName := range sourceVolumes { - destName := namer(sourceVolName) - - if err := cloner.CreateVolume(ctx, destName); err != nil { - cleanupVolumes(ctx, cloner, clonedNames) - return nil, fmt.Errorf("create volume %s: %w", destName, err) - } - - helperConfig := &domain.ContainerConfig{ - Image: "busybox:1.37", - Name: "gordon-vol-copy-" + destName, - Cmd: []string{"cp", "-a", "/src/.", "/dst/"}, - Volumes: map[string]string{ - "/dst": destName, - }, - ReadOnlyVolumes: map[string]string{ - "/src": sourceVolName, - }, - } - - created, err := cloner.CreateContainer(ctx, helperConfig) - if err != nil { - _ = cloner.RemoveVolume(ctx, destName, true) - cleanupVolumes(ctx, cloner, clonedNames) - return nil, fmt.Errorf("create copy helper for %s: %w", destName, err) - } - - if err := cloner.StartContainer(ctx, created.ID); err != nil { - _ = cloner.RemoveContainer(ctx, created.ID, true) - _ = cloner.RemoveVolume(ctx, destName, true) - cleanupVolumes(ctx, cloner, clonedNames) - return nil, fmt.Errorf("start copy helper for %s: %w", destName, err) - } - - if err := cloner.WaitForContainer(ctx, created.ID); err != nil { - _ = cloner.RemoveContainer(ctx, created.ID, true) - _ = cloner.RemoveVolume(ctx, destName, true) - cleanupVolumes(ctx, cloner, clonedNames) - return nil, fmt.Errorf("wait for copy helper %s: %w", destName, err) - } - - _ = cloner.RemoveContainer(ctx, created.ID, true) - clonedNames = append(clonedNames, destName) - } - return clonedNames, nil -} - -func cleanupVolumes(ctx context.Context, cloner VolumeCloner, names []string) { - for _, name := range names { - _ = cloner.RemoveVolume(ctx, name, true) - } -} diff --git a/internal/usecase/auto/preview/volumes_test.go b/internal/usecase/auto/preview/volumes_test.go deleted file mode 100644 index dc2f3c5c6..000000000 --- a/internal/usecase/auto/preview/volumes_test.go +++ /dev/null @@ -1,18 +0,0 @@ -package preview - -import ( - "testing" - - "github.com/stretchr/testify/assert" -) - -func TestBuildCloneVolumeName(t *testing.T) { - assert.Equal(t, "preview-feat-pgdata", BuildCloneVolumeName("feat", "pgdata")) - assert.Equal(t, "preview-pr-42-redis-data", BuildCloneVolumeName("pr-42", "redis-data")) -} - -func TestBuildCloneContainerName(t *testing.T) { - assert.Equal(t, "preview-feat-postgres", BuildCloneContainerName("feat", "postgres:17")) - assert.Equal(t, "preview-feat-redis", BuildCloneContainerName("feat", "redis:7")) - assert.Equal(t, "preview-feat-myapp", BuildCloneContainerName("feat", "org/myapp:latest")) -} diff --git a/internal/usecase/auto/validation.go b/internal/usecase/auto/validation.go deleted file mode 100644 index 6d72336f6..000000000 --- a/internal/usecase/auto/validation.go +++ /dev/null @@ -1,48 +0,0 @@ -package auto - -import ( - "strings" - - "github.com/bnema/gordon/internal/domain" -) - -// MatchesDomainAllowlist reports whether domain matches any of the given patterns. -// Patterns may be exact domains, wildcard subdomains (*.example.com), or "*" to -// allow everything. Wildcard patterns match a single subdomain level only, per -// DNS/TLS conventions. -func MatchesDomainAllowlist(domain string, patterns []string) bool { - domain = strings.ToLower(domain) - for _, pattern := range patterns { - pattern = strings.ToLower(strings.TrimSpace(pattern)) - if pattern == "" { - continue - } - if pattern == "*" { - return true - } - if pattern == domain { - return true - } - if !strings.HasPrefix(pattern, "*.") { - continue - } - // Wildcard patterns match single-level subdomains only (per DNS/TLS conventions). - // e.g., "*.example.com" matches "foo.example.com" but not "bar.foo.example.com" - suffix := strings.TrimPrefix(pattern, "*.") - if !strings.HasSuffix(domain, "."+suffix) { - continue - } - prefix := strings.TrimSuffix(domain, "."+suffix) - if prefix != "" && !strings.Contains(prefix, ".") { - return true - } - } - return false -} - -// ExtractRepoName strips the current or legacy Gordon registry domain, tag, -// and digest from an image reference and returns the bare repository name -// (e.g. "org/myapp"). -func ExtractRepoName(imageRef, registryDomain string, legacyRegistryDomains ...string) string { - return strings.ToLower(domain.ExtractGordonRepoName(imageRef, registryDomain, legacyRegistryDomains)) -} diff --git a/internal/usecase/auto/validation_test.go b/internal/usecase/auto/validation_test.go deleted file mode 100644 index 8cf36a6f3..000000000 --- a/internal/usecase/auto/validation_test.go +++ /dev/null @@ -1,75 +0,0 @@ -package auto - -import ( - "testing" - - "github.com/stretchr/testify/assert" -) - -func TestMatchesDomainAllowlist(t *testing.T) { - tests := []struct { - name string - domain string - patterns []string - want bool - }{ - {name: "exact match", domain: "app.example.com", patterns: []string{"app.example.com"}, want: true}, - {name: "no match", domain: "app.example.com", patterns: []string{"api.example.com"}, want: false}, - {name: "wildcard match", domain: "app.example.com", patterns: []string{"*.example.com"}, want: true}, - {name: "wildcard no match on root", domain: "example.com", patterns: []string{"*.example.com"}, want: false}, - {name: "wildcard matches one level only", domain: "api.app.example.com", patterns: []string{"*.example.com"}, want: false}, - {name: "empty allowlist", domain: "app.example.com", patterns: nil, want: false}, - {name: "case insensitive", domain: "App.Example.Com", patterns: []string{"APP.EXAMPLE.COM"}, want: true}, - {name: "multiple patterns", domain: "api.example.com", patterns: []string{"foo.com", "*.example.com"}, want: true}, - {name: "star allows all", domain: "anything.example.com", patterns: []string{"*"}, want: true}, - {name: "empty pattern skipped", domain: "app.example.com", patterns: []string{"", "app.example.com"}, want: true}, - {name: "whitespace trimmed in pattern", domain: "app.example.com", patterns: []string{" app.example.com "}, want: true}, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - assert.Equal(t, tt.want, MatchesDomainAllowlist(tt.domain, tt.patterns)) - }) - } -} - -func TestExtractRepoName(t *testing.T) { - tests := []struct { - name string - imageRef string - registryDomain string - want string - }{ - {name: "simple tag", imageRef: "myapp:latest", want: "myapp"}, - {name: "version tag", imageRef: "myapp:v2.0.0", want: "myapp"}, - {name: "digest", imageRef: "myapp@sha256:abc123", want: "myapp"}, - {name: "strip registry", imageRef: "registry.example.com/myapp:latest", registryDomain: "registry.example.com", want: "myapp"}, - {name: "org image", imageRef: "org/myapp:latest", want: "org/myapp"}, - {name: "registry org image", imageRef: "registry.example.com/org/myapp:latest", registryDomain: "registry.example.com", want: "org/myapp"}, - {name: "lowercase", imageRef: "MyApp:Latest", want: "myapp"}, - {name: "no tag no digest", imageRef: "myapp", want: "myapp"}, - {name: "empty registry domain", imageRef: "myapp:latest", registryDomain: "", want: "myapp"}, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - assert.Equal(t, tt.want, ExtractRepoName(tt.imageRef, tt.registryDomain)) - }) - } -} - -func TestExtractRepoName_TreatsLegacyAndCurrentGordonRegistryHostsAsSameRepo(t *testing.T) { - current := "new-registry.example.com" - legacy := []string{"old-registry.example.com"} - - oldRepo := ExtractRepoName("old-registry.example.com/app:latest", current, legacy...) - newRepo := ExtractRepoName("new-registry.example.com/app:latest", current, legacy...) - - assert.Equal(t, "app", oldRepo) - assert.Equal(t, oldRepo, newRepo) -} - -func TestExtractRepoName_HandlesPortsAndDigests(t *testing.T) { - assert.Equal(t, "org/app", ExtractRepoName("new-registry.example.com:5000/org/app:latest", "new-registry.example.com:5000")) - assert.Equal(t, "org/app", ExtractRepoName("old-registry.example.com:5001/org/app@sha256:deadbeef", "new-registry.example.com:5000", "old-registry.example.com:5001")) -} diff --git a/internal/usecase/backup/detector.go b/internal/usecase/backup/detector.go deleted file mode 100644 index 8299b689d..000000000 --- a/internal/usecase/backup/detector.go +++ /dev/null @@ -1,89 +0,0 @@ -package backup - -import ( - "slices" - "strings" - - "github.com/bnema/gordon/internal/domain" -) - -// postgresAliases contains substrings that identify a PostgreSQL image or service. -var postgresAliases = []string{"postgres", "postgresql", "pgsql", "postgis"} - -// detectDatabaseFromAttachment maps a container attachment to DB metadata. -func detectDatabaseFromAttachment(domainName string, a domain.Attachment) (domain.DBInfo, bool) { - image := strings.ToLower(a.Image) - name := strings.ToLower(a.Name) - - switch { - case looksLikePostgres(image) || looksLikePostgres(name): - return buildPostgresInfo(domainName, a), true - case hasPort(a.Ports, 5432): - // Port 5432 fallback: likely PostgreSQL even with non-standard naming. - return buildPostgresInfo(domainName, a), true - default: - return domain.DBInfo{}, false - } -} - -// looksLikePostgres returns true if s contains any known PostgreSQL alias. -func looksLikePostgres(s string) bool { - for _, alias := range postgresAliases { - if strings.Contains(s, alias) { - return true - } - } - return false -} - -// hasPort returns true if the port list contains p. -func hasPort(ports []int, p int) bool { - return slices.Contains(ports, p) -} - -// buildPostgresInfo constructs a DBInfo for a PostgreSQL attachment. -func buildPostgresInfo(domainName string, a domain.Attachment) domain.DBInfo { - port := 5432 - for _, p := range a.Ports { - if p == 5432 { - port = 5432 - break - } - if p > 0 && port == 5432 { - port = p - } - } - - return domain.DBInfo{ - Type: domain.DBTypePostgreSQL, - Version: postgresVersionFromImage(a.Image), - Domain: domainName, - Name: a.Name, - Host: a.Name, - Port: port, - ContainerID: a.ContainerID, - ImageName: a.Image, - } -} - -func postgresVersionFromImage(image string) string { - lastColon := strings.LastIndex(image, ":") - if lastColon == -1 { - return "" - } - tag := strings.TrimSpace(image[lastColon+1:]) - if tag == "" { - return "" - } - - for i := 0; i < len(tag); i++ { - if tag[i] < '0' || tag[i] > '9' { - if i == 0 { - return "" - } - return tag[:i] - } - } - - return tag -} diff --git a/internal/usecase/backup/detector_test.go b/internal/usecase/backup/detector_test.go deleted file mode 100644 index d0aba65f3..000000000 --- a/internal/usecase/backup/detector_test.go +++ /dev/null @@ -1,180 +0,0 @@ -package backup - -import ( - "testing" - - "github.com/bnema/gordon/internal/domain" - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" -) - -func TestDetectDatabaseFromAttachment_ExtractsPostgresVersion(t *testing.T) { - db, ok := detectDatabaseFromAttachment("app.example.com", domain.Attachment{ - Name: "postgres", - Image: "postgres:17.4-alpine", - ContainerID: "abc123", - Ports: []int{55432}, - }) - - require.True(t, ok) - assert.Equal(t, domain.DBTypePostgreSQL, db.Type) - assert.Equal(t, "17", db.Version) - assert.Equal(t, 55432, db.Port) -} - -func TestDetectDatabaseFromAttachment_DefaultPort(t *testing.T) { - db, ok := detectDatabaseFromAttachment("app.example.com", domain.Attachment{ - Name: "postgres", - Image: "postgres:16", - ContainerID: "xyz789", - Ports: []int{}, - }) - - require.True(t, ok) - assert.Equal(t, 5432, db.Port) - assert.Equal(t, "16", db.Version) -} - -func TestDetectDatabaseFromAttachment_PrefersPostgresPort(t *testing.T) { - db, ok := detectDatabaseFromAttachment("app.example.com", domain.Attachment{ - Name: "postgres", - Image: "postgres:16", - ContainerID: "xyz789", - Ports: []int{15432, 5432, 25432}, - }) - - require.True(t, ok) - assert.Equal(t, 5432, db.Port) -} - -func TestDetectDatabaseFromAttachment_UsesFirstPositivePortWhen5432Missing(t *testing.T) { - db, ok := detectDatabaseFromAttachment("app.example.com", domain.Attachment{ - Name: "postgres", - Image: "postgres:16", - ContainerID: "xyz789", - Ports: []int{0, -1, 15432, 25432}, - }) - - require.True(t, ok) - assert.Equal(t, 15432, db.Port) -} - -func TestPostgresVersionFromImage(t *testing.T) { - assert.Equal(t, "18", postgresVersionFromImage("postgres:18")) - assert.Equal(t, "17", postgresVersionFromImage("postgres:17.4-alpine")) - assert.Equal(t, "", postgresVersionFromImage("postgres:latest")) - assert.Equal(t, "", postgresVersionFromImage("postgres:alpine")) - assert.Equal(t, "", postgresVersionFromImage("postgres")) -} - -func TestDetectDatabaseFromAttachment_NonPostgres(t *testing.T) { - _, ok := detectDatabaseFromAttachment("app.example.com", domain.Attachment{ - Name: "mysql", - Image: "mysql:8.0", - ContainerID: "def456", - Ports: []int{3306}, - }) - assert.False(t, ok) -} - -func TestDetectDatabaseFromAttachment_PgsqlImage(t *testing.T) { - db, ok := detectDatabaseFromAttachment("app.example.com", domain.Attachment{ - Name: "pgsql", - Image: "pgsql:16", - ContainerID: "pgsql-1", - Ports: []int{5432}, - }) - - require.True(t, ok) - assert.Equal(t, domain.DBTypePostgreSQL, db.Type) - assert.Equal(t, "16", db.Version) - assert.Equal(t, 5432, db.Port) -} - -func TestDetectDatabaseFromAttachment_PgsqlInCompositeName(t *testing.T) { - db, ok := detectDatabaseFromAttachment("mon-fauteuil.fr", domain.Attachment{ - Name: "mon-fauteuil-prod-pgsql", - Image: "mon-fauteuil-prod-pgsql:v18", - ContainerID: "pgsql-2", - Ports: []int{5432}, - }) - - require.True(t, ok) - assert.Equal(t, domain.DBTypePostgreSQL, db.Type) - assert.Equal(t, "", db.Version) // v18 tag starts with non-digit, no version extracted -} - -func TestDetectDatabaseFromAttachment_PostgresqlFullName(t *testing.T) { - db, ok := detectDatabaseFromAttachment("app.example.com", domain.Attachment{ - Name: "postgresql", - Image: "postgresql:15", - ContainerID: "pg-full-1", - Ports: []int{5432}, - }) - - require.True(t, ok) - assert.Equal(t, domain.DBTypePostgreSQL, db.Type) - assert.Equal(t, "15", db.Version) -} - -func TestDetectDatabaseFromAttachment_PostgisImage(t *testing.T) { - db, ok := detectDatabaseFromAttachment("app.example.com", domain.Attachment{ - Name: "postgis", - Image: "postgis/postgis:16-3.4", - ContainerID: "postgis-1", - Ports: []int{5432}, - }) - - require.True(t, ok) - assert.Equal(t, domain.DBTypePostgreSQL, db.Type) - assert.Equal(t, "16", db.Version) -} - -func TestDetectDatabaseFromAttachment_Port5432Fallback(t *testing.T) { - db, ok := detectDatabaseFromAttachment("app.example.com", domain.Attachment{ - Name: "custom-db", - Image: "custom-db:latest", - ContainerID: "fallback-1", - Ports: []int{5432}, - }) - - require.True(t, ok, "port 5432 should trigger PostgreSQL fallback detection") - assert.Equal(t, domain.DBTypePostgreSQL, db.Type) - assert.Equal(t, 5432, db.Port) -} - -func TestDetectDatabaseFromAttachment_UnknownImageNoPostgresPort(t *testing.T) { - _, ok := detectDatabaseFromAttachment("app.example.com", domain.Attachment{ - Name: "custom-db", - Image: "custom-db:latest", - ContainerID: "unknown-1", - Ports: []int{3306}, - }) - assert.False(t, ok, "non-postgres port with unknown image should not match") -} - -func TestLooksLikePostgres(t *testing.T) { - tests := []struct { - name string - image string - want bool - }{ - {name: "postgres base image", image: "postgres", want: true}, - {name: "postgres with version tag", image: "postgres:16", want: true}, - {name: "name contains postgres", image: "my-postgres-db", want: true}, - {name: "postgresql alias", image: "postgresql", want: true}, - {name: "pgsql alias", image: "pgsql", want: true}, - {name: "project specific pgsql alias", image: "mon-fauteuil-prod-pgsql", want: true}, - {name: "postgis alias", image: "postgis/postgis", want: true}, - {name: "mysql negative", image: "mysql", want: false}, - {name: "redis negative", image: "redis", want: false}, - {name: "custom negative", image: "custom-db", want: false}, - {name: "empty negative", image: "", want: false}, - } - - for _, tc := range tests { - t.Run(tc.name, func(t *testing.T) { - assert.Equal(t, tc.want, looksLikePostgres(tc.image)) - }) - } -} diff --git a/internal/usecase/backup/integration_test.go b/internal/usecase/backup/integration_test.go index 1784f4e6b..6b5c66020 100644 --- a/internal/usecase/backup/integration_test.go +++ b/internal/usecase/backup/integration_test.go @@ -11,6 +11,7 @@ import ( "testing" "time" + "github.com/bnema/gordon/internal/adapters/out/appstate" "github.com/bnema/gordon/internal/adapters/out/docker" "github.com/bnema/gordon/internal/adapters/out/filesystem" "github.com/bnema/gordon/internal/domain" @@ -38,7 +39,7 @@ func TestBackupService_Integration_Postgres17And18(t *testing.T) { func runPostgresBackupFlow(t *testing.T, ctx context.Context, runtime *docker.Runtime, version string) { image := fmt.Sprintf("postgres:%s", version) - domainName := fmt.Sprintf("backup-it-%s.example.com", version) + appName := fmt.Sprintf("backup-it-%s", version) networkName := fmt.Sprintf("gordon-backup-it-%s-%d", version, time.Now().UnixNano()) containerName := fmt.Sprintf("gordon-backup-it-%s-%d", version, time.Now().UnixNano()) @@ -65,10 +66,10 @@ func runPostgresBackupFlow(t *testing.T, ctx context.Context, runtime *docker.Ru "POSTGRES_DB=appdb", }, Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: domainName, - domain.LabelImage: image, + domain.LabelManaged: "true", + domain.LabelApp: appName, + domain.LabelService: "postgres", + domain.LabelImage: image, }, } @@ -78,7 +79,7 @@ func runPostgresBackupFlow(t *testing.T, ctx context.Context, runtime *docker.Ru require.NoError(t, err) t.Cleanup(func() { stopCtx, cancelStop := context.WithTimeout(context.Background(), 20*time.Second) - _ = runtime.StopContainer(stopCtx, container.ID) + _ = runtime.StopContainer(stopCtx, container.ID, domain.AppDefaultStopGrace) cancelStop() removeCtx, cancelRemove := context.WithTimeout(context.Background(), 20*time.Second) @@ -95,32 +96,33 @@ func runPostgresBackupFlow(t *testing.T, ctx context.Context, runtime *docker.Ru storage, err := filesystem.NewBackupStorage(t.TempDir(), zerowrap.Default()) require.NoError(t, err) - containerSvc := &integrationContainerService{ - routes: map[string]*domain.Container{ - domainName: container, - }, - attachments: map[string][]domain.Attachment{ - domainName: { - { - Name: "postgres", - Image: image, - ContainerID: container.ID, - Status: container.Status, - Network: networkName, - Ports: []int{5432}, + state, err := appstate.NewStore(t.TempDir(), zerowrap.Default()) + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, state.Close()) }) + require.NoError(t, state.SaveActive(ctx, domain.AppActive{ + App: appName, + Converged: true, + Services: map[string]domain.AppEffectiveService{ + "postgres": { + Image: image, + Container: container.ID, + Spec: domain.AppService{ + Name: "postgres", + Image: image, + Databases: []domain.AppDatabase{{Name: "appdb", Type: domain.AppDBPostgres, Schedule: "daily"}}, + Backup: domain.AppBackup{Postgres: []string{"appdb"}}, }, }, }, - } + })) - svc := backup.NewService(runtime, storage, containerSvc, domain.BackupConfig{Enabled: true}, zerowrap.Default()) - - detected, err := svc.DetectDatabases(ctx, domainName) + svc := backup.NewService(runtime, storage, domain.BackupConfig{Enabled: true}, zerowrap.Default()).WithAppState(state) + targets, err := svc.Targets(ctx, appName) require.NoError(t, err) - require.Len(t, detected, 1) - assert.Equal(t, domain.DBTypePostgreSQL, detected[0].Type) + require.Len(t, targets, 1) + assert.Equal(t, "postgres", targets[0].Service) - result, err := svc.RunBackup(ctx, domainName, "postgres") + result, err := svc.RunBackup(ctx, appName, "postgres", "appdb") require.NoError(t, err) require.NotNil(t, result) assert.Equal(t, domain.BackupStatusCompleted, result.Job.Status) @@ -130,7 +132,7 @@ func runPostgresBackupFlow(t *testing.T, ctx context.Context, runtime *docker.Ru assert.Greater(t, len(backupBytes), 32) assert.True(t, bytes.HasPrefix(backupBytes, []byte("PGDMP")), "expected pg_dump custom format header") - jobs, err := svc.ListBackups(ctx, domainName) + jobs, err := svc.ListBackups(ctx, appName) require.NoError(t, err) assert.Len(t, jobs, 1) } @@ -181,71 +183,3 @@ func seedPostgresData(ctx context.Context, runtime *docker.Runtime, containerID } return nil } - -type integrationContainerService struct { - routes map[string]*domain.Container - attachments map[string][]domain.Attachment -} - -func (s *integrationContainerService) Deploy(context.Context, domain.Route) (*domain.Container, error) { - return nil, fmt.Errorf("not implemented") -} - -func (s *integrationContainerService) Stop(context.Context, string) error { - return fmt.Errorf("not implemented") -} - -func (s *integrationContainerService) Remove(context.Context, string, bool) error { - return fmt.Errorf("not implemented") -} - -func (s *integrationContainerService) Get(_ context.Context, domainName string) (*domain.Container, bool) { - c, ok := s.routes[domainName] - return c, ok -} - -func (s *integrationContainerService) Restart(context.Context, string, bool) error { - return fmt.Errorf("not implemented") -} - -func (s *integrationContainerService) List(context.Context) map[string]*domain.Container { - out := make(map[string]*domain.Container, len(s.routes)) - for k, v := range s.routes { - out[k] = v - } - return out -} - -func (s *integrationContainerService) ListRoutesWithDetails(context.Context) []domain.RouteInfo { - return nil -} - -func (s *integrationContainerService) ListAttachments(_ context.Context, domainName string) []domain.Attachment { - attachments := s.attachments[domainName] - out := make([]domain.Attachment, len(attachments)) - copy(out, attachments) - return out -} - -func (s *integrationContainerService) ListNetworks(context.Context) ([]*domain.NetworkInfo, error) { - return nil, fmt.Errorf("not implemented") -} - -func (s *integrationContainerService) HealthCheck(context.Context) map[string]bool { - return map[string]bool{} -} - -func (s *integrationContainerService) SyncContainers(context.Context) error { - return nil -} - -func (s *integrationContainerService) UpdateAttachments(map[string][]string) { -} - -func (s *integrationContainerService) AutoStart(context.Context, []domain.Route) error { - return nil -} - -func (s *integrationContainerService) Shutdown(context.Context) error { - return nil -} diff --git a/internal/usecase/backup/service.go b/internal/usecase/backup/service.go index f6bd91fce..21c53f0a2 100644 --- a/internal/usecase/backup/service.go +++ b/internal/usecase/backup/service.go @@ -9,137 +9,221 @@ import ( "io" "sort" "strings" - "sync" "time" "github.com/bnema/zerowrap" - "github.com/bnema/gordon/internal/boundaries/in" "github.com/bnema/gordon/internal/boundaries/out" "github.com/bnema/gordon/internal/domain" ) const backupExecTimeout = 30 * time.Minute -// Service orchestrates backup operations. +// Service orchestrates declarative PostgreSQL app backups. Targets come +// from the app's ACTIVE record: an app declares its databases and which +// of them are backed up, and the app name is the backup identity. type Service struct { - runtime out.ContainerRuntime - storage out.BackupStorage - containerSvc in.ContainerService - config domain.BackupConfig - log zerowrap.Logger + runtime out.ContainerRuntime + storage out.BackupStorage + config domain.BackupConfig + log zerowrap.Logger + state out.AppStateReader } // NewService creates a backup service. func NewService( runtime out.ContainerRuntime, storage out.BackupStorage, - containerSvc in.ContainerService, config domain.BackupConfig, log zerowrap.Logger, ) *Service { return &Service{ - runtime: runtime, - storage: storage, - containerSvc: containerSvc, - config: config, - log: log, + runtime: runtime, + storage: storage, + config: config, + log: log, } } -// ListBackups returns backups for a domain. -func (s *Service) ListBackups(ctx context.Context, domainName string) ([]domain.BackupJob, error) { - return s.storage.List(ctx, domainName, nil) +// WithAppState wires the ACTIVE app state target resolution reads. +// Without it no target can be resolved. +func (s *Service) WithAppState(state out.AppStateReader) *Service { + s.state = state + return s } -// DetectDatabases inspects attachments and returns detected DBs. -func (s *Service) DetectDatabases(ctx context.Context, domainName string) ([]domain.DBInfo, error) { - attachments := s.containerSvc.ListAttachments(ctx, domainName) - dbs := make([]domain.DBInfo, 0, len(attachments)) - for _, attachment := range attachments { - if db, ok := detectDatabaseFromAttachment(domainName, attachment); ok { - dbs = append(dbs, db) +// Targets returns every declared database target of one app, ordered by +// service then database. +func (s *Service) Targets(ctx context.Context, app string) ([]domain.DatabaseTarget, error) { + if s.state == nil { + return nil, fmt.Errorf("backup: app state is not wired") + } + targets, _, err := declaredTargets(ctx, s.state, app) + if err != nil { + return nil, err + } + return targets, nil +} + +// allTargets returns the declared database targets of every app. +func (s *Service) allTargets(ctx context.Context) ([]domain.DatabaseTarget, error) { + if s.state == nil { + return nil, fmt.Errorf("backup: app state is not wired") + } + apps, err := appNames(ctx, s.state) + if err != nil { + return nil, err + } + var targets []domain.DatabaseTarget + for _, app := range apps { + appTargets, err := s.Targets(ctx, app) + if err != nil { + return nil, err } + targets = append(targets, appTargets...) } - return dbs, nil + return targets, nil } -// RunBackup triggers a logical PostgreSQL backup for a detected DB. -func (s *Service) RunBackup(ctx context.Context, domainName, dbName string) (*domain.BackupResult, error) { - return s.runBackup(ctx, domainName, dbName, domain.BackupSchedule("")) +// RunBackup runs one declared database backup. The service and database +// selectors are explicit; an omitted selector only succeeds when exactly +// one compatible target exists. +func (s *Service) RunBackup(ctx context.Context, app, service, database string) (*domain.BackupResult, error) { + targets, err := s.Targets(ctx, app) + if err != nil { + return nil, err + } + target, err := selectTarget(targets, "database", service, database, databaseTargetKey) + if err != nil { + return nil, err + } + return s.runTarget(ctx, target, "") } -// RunForSchedule executes backups for all detected databases and applies retention. +// RunForSchedule runs every declared database whose own schedule matches, +// then applies retention under the canonical app identity. Apps and +// databases that do not declare this schedule are never touched. func (s *Service) RunForSchedule(ctx context.Context, schedule domain.BackupSchedule) error { if !isValidBackupSchedule(schedule) { return fmt.Errorf("invalid backup schedule: %q", schedule) } - - routes := s.containerSvc.List(ctx) - domainNames := make([]string, 0, len(routes)) - for domainName := range routes { - domainNames = append(domainNames, domainName) + targets, err := s.allTargets(ctx) + if err != nil { + return err } - sort.Strings(domainNames) - var firstErr error - for _, domainName := range domainNames { - dbs, err := s.DetectDatabases(ctx, domainName) - if err != nil { - s.log.Error().Err(err).Str("domain", domainName).Msg("scheduled backup database detection failed") - if firstErr == nil { - firstErr = fmt.Errorf("detect databases for %s: %w", domainName, err) - } + apps := map[string]struct{}{} + for _, target := range targets { + if target.Schedule != schedule { continue } - - for _, db := range dbs { - if _, err := s.runBackupForDB(ctx, domainName, db, schedule); err != nil { - s.log.Error().Err(err).Str("domain", domainName).Str("db", db.Name).Msg("scheduled backup failed") - if firstErr == nil { - firstErr = fmt.Errorf("run backup for %s/%s: %w", domainName, db.Name, err) - } + if target.ContainerID == "" { + // Nothing to dump: the declared database has no live + // container. Reported per target, never guessed. + s.log.Warn().Str("app", target.App).Str("service", target.Service). + Str("database", target.Database).Msg("scheduled backup skipped: no active container") + continue + } + if _, err := s.runTarget(ctx, target, schedule); err != nil { + s.log.Error().Err(err).Str("app", target.App).Str("service", target.Service). + Str("database", target.Database).Msg("scheduled backup failed") + if firstErr == nil { + firstErr = fmt.Errorf("backup %s/%s/%s: %w", target.App, target.Service, target.Database, err) } + continue } - - if _, err := s.storage.ApplyRetention(ctx, domainName, s.config.Retention); err != nil { - s.log.Error().Err(err).Str("domain", domainName).Msg("scheduled backup retention failed") + apps[target.App] = struct{}{} + } + for _, app := range sortedSet(apps) { + if _, err := s.storage.ApplyRetention(ctx, app, s.config.Retention); err != nil { + s.log.Error().Err(err).Str("app", app).Msg("scheduled backup retention failed") if firstErr == nil { - firstErr = fmt.Errorf("apply retention for %s: %w", domainName, err) + firstErr = fmt.Errorf("apply retention for %s: %w", app, err) } } } - return firstErr } -func (s *Service) runBackup(ctx context.Context, domainName, dbName string, schedule domain.BackupSchedule) (*domain.BackupResult, error) { - dbs, err := s.DetectDatabases(ctx, domainName) +// ListBackups lists stored backups of one app, or of every app when app +// is empty. Storage is keyed by the canonical app name. +func (s *Service) ListBackups(ctx context.Context, app string) ([]domain.BackupJob, error) { + if app != "" { + return s.storage.List(ctx, app, nil) + } + if s.state == nil { + return nil, fmt.Errorf("backup: app state is not wired") + } + apps, err := appNames(ctx, s.state) if err != nil { return nil, err } + var jobs []domain.BackupJob + for _, name := range apps { + appJobs, err := s.storage.List(ctx, name, nil) + if err != nil { + return nil, err + } + jobs = append(jobs, appJobs...) + } + sort.Slice(jobs, func(i, j int) bool { return jobs[i].StartedAt.After(jobs[j].StartedAt) }) + return jobs, nil +} - db, err := selectDatabase(dbs, dbName) +// Status returns stored backups plus the declared targets that have no +// completed backup yet, so an operator can see what is configured and +// what actually ran. +func (s *Service) Status(ctx context.Context) ([]domain.BackupJob, error) { + jobs, err := s.ListBackups(ctx, "") if err != nil { return nil, err } - - return s.runBackupForDB(ctx, domainName, db, schedule) + completed := make(map[string]struct{}, len(jobs)) + for _, job := range jobs { + completed[job.App+"\x00"+job.Service+"\x00"+job.DBName] = struct{}{} + } + targets, err := s.allTargets(ctx) + if err != nil { + return nil, err + } + for _, target := range targets { + if _, ok := completed[target.App+"\x00"+target.Service+"\x00"+target.Database]; ok { + continue + } + jobs = append(jobs, domain.BackupJob{ + ID: "target:" + target.App + "/" + target.Service + "/" + target.Database, + App: target.App, + Service: target.Service, + DBName: target.Database, + Schedule: target.Schedule, + Type: domain.BackupTypeLogical, + Status: domain.BackupStatusPending, + Metadata: map[string]string{"declared_schedule": string(target.Schedule)}, + }) + } + sort.Slice(jobs, func(i, j int) bool { + if !jobs[i].StartedAt.Equal(jobs[j].StartedAt) { + return jobs[i].StartedAt.After(jobs[j].StartedAt) + } + return jobs[i].App < jobs[j].App + }) + return jobs, nil } -func (s *Service) runBackupForDB(ctx context.Context, domainName string, db domain.DBInfo, schedule domain.BackupSchedule) (*domain.BackupResult, error) { - started := time.Now().UTC() - - if db.Type != domain.DBTypePostgreSQL { - return nil, fmt.Errorf("unsupported database type: %s", db.Type) +// runTarget runs one declared database backup through the service +// container that hosts it. +func (s *Service) runTarget(ctx context.Context, target domain.DatabaseTarget, schedule domain.BackupSchedule) (*domain.BackupResult, error) { + if target.ContainerID == "" { + return nil, fmt.Errorf("backup: app %q service %q has no active container", target.App, target.Service) } + started := time.Now().UTC() execCtx, cancelExec := context.WithTimeout(ctx, backupExecTimeout) defer cancelExec() dumpPath := fmt.Sprintf("/tmp/gordon-backup-%d.bak", started.UnixNano()) - defer s.cleanupDumpFile(db.ContainerID, dumpPath) + defer s.cleanupDumpFile(target.ContainerID, dumpPath) - execResult, err := s.runtime.ExecInContainer(execCtx, db.ContainerID, []string{"sh", "-c", pgDumpToPathCommand(dumpPath, db.Name)}) + execResult, err := s.runtime.ExecInContainer(execCtx, target.ContainerID, []string{"sh", "-c", pgDumpToPathCommand(dumpPath, target.Database)}) if err != nil { if errors.Is(execCtx.Err(), context.DeadlineExceeded) { return nil, fmt.Errorf("pg_dump timed out after %s", backupExecTimeout) @@ -150,7 +234,7 @@ func (s *Service) runBackupForDB(ctx context.Context, domainName string, db doma return nil, fmt.Errorf("pg_dump failed with exit code %d: %s", execResult.ExitCode, string(execResult.Stderr)) } - dumpStream, err := s.runtime.CopyFromContainer(execCtx, db.ContainerID, dumpPath) + dumpStream, err := s.runtime.CopyFromContainer(execCtx, target.ContainerID, dumpPath) if err != nil { if errors.Is(execCtx.Err(), context.DeadlineExceeded) { return nil, fmt.Errorf("backup copy timed out after %s", backupExecTimeout) @@ -160,15 +244,17 @@ func (s *Service) runBackupForDB(ctx context.Context, domainName string, db doma defer dumpStream.Close() counter := &byteCounter{} - path, err := s.storage.Store(ctx, domainName, db.Name, schedule, started, io.TeeReader(dumpStream, counter)) + path, err := s.storage.Store(ctx, target.App, target.Service, target.Database, schedule, started, io.TeeReader(dumpStream, counter)) if err != nil { return nil, err } job := domain.BackupJob{ ID: newBackupJobID(started), - Domain: domainName, - DBName: db.Name, + App: target.App, + Service: target.Service, + DBName: target.Database, + Schedule: schedule, Type: domain.BackupTypeLogical, Status: domain.BackupStatusCompleted, StartedAt: started, @@ -183,6 +269,12 @@ func (s *Service) runBackupForDB(ctx context.Context, domainName string, db doma }, nil } +// databaseTargetKey reports the (service, database) identity of one +// target for selector matching and ambiguity reporting. +func databaseTargetKey(target domain.DatabaseTarget) (string, string) { + return target.Service, target.Database +} + func isValidBackupSchedule(schedule domain.BackupSchedule) bool { switch schedule { case domain.ScheduleHourly, domain.ScheduleDaily, domain.ScheduleWeekly, domain.ScheduleMonthly: @@ -202,66 +294,6 @@ func (s *Service) RestorePITR(context.Context, string, time.Time) error { return fmt.Errorf("pitr restore not implemented yet") } -// Status returns aggregate backup status for all managed domains. -func (s *Service) Status(ctx context.Context) ([]domain.BackupJob, error) { - routes := s.containerSvc.List(ctx) - domainNames := make([]string, 0, len(routes)) - for domainName := range routes { - domainNames = append(domainNames, domainName) - } - sort.Strings(domainNames) - - const maxWorkers = 4 - sem := make(chan struct{}, maxWorkers) - results := make([][]domain.BackupJob, len(domainNames)) - errCh := make(chan error, 1) - var wg sync.WaitGroup - - for i, domainName := range domainNames { - select { - case <-ctx.Done(): - return nil, ctx.Err() - default: - } - - select { - case sem <- struct{}{}: - case <-ctx.Done(): - return nil, ctx.Err() - } - - wg.Add(1) - go func(idx int, name string) { - defer wg.Done() - defer func() { <-sem }() - - domainJobs, err := s.ListBackups(ctx, name) - if err != nil { - select { - case errCh <- err: - default: - } - return - } - results[idx] = domainJobs - }(i, domainName) - } - - wg.Wait() - select { - case err := <-errCh: - return nil, err - default: - } - - jobs := make([]domain.BackupJob, 0) - for _, domainJobs := range results { - jobs = append(jobs, domainJobs...) - } - - return jobs, nil -} - func newBackupJobID(started time.Time) string { random := make([]byte, 4) if _, err := rand.Read(random); err != nil { @@ -270,33 +302,22 @@ func newBackupJobID(started time.Time) string { return fmt.Sprintf("%s-%s", started.Format(time.RFC3339Nano), hex.EncodeToString(random)) } -func selectDatabase(dbs []domain.DBInfo, requested string) (domain.DBInfo, error) { - if len(dbs) == 0 { - return domain.DBInfo{}, fmt.Errorf("no supported database attachments detected") - } - - if requested == "" { - if len(dbs) == 1 { - return dbs[0], nil - } - return domain.DBInfo{}, fmt.Errorf("multiple database attachments detected; please specify one") - } - - for _, db := range dbs { - if strings.EqualFold(db.Name, requested) { - return db, nil - } - } - - return domain.DBInfo{}, fmt.Errorf("database %q not found for domain", requested) -} - -func postgresDumpCommand(_ string) string { - return "pg_dump -Fc --dbname=\"${POSTGRES_DB:-postgres}\" --username=\"${POSTGRES_USER:-postgres}\"" +// pgDumpToPathCommand runs pg_dump inside the declared service container. +// The database name comes from the app declaration (validated as a name), +// while credentials keep coming from the container's own environment: the +// manifest never carries secret values. +func pgDumpToPathCommand(path, dbName string) string { + return fmt.Sprintf( + "pg_dump -Fc --dbname=%s --username=\"${POSTGRES_USER:-postgres}\" > %q", + shellSingleQuote(dbName), path, + ) } -func pgDumpToPathCommand(path, dbName string) string { - return fmt.Sprintf("%s > %q", postgresDumpCommand(dbName), path) +// shellSingleQuote quotes one validated identifier for a single-quoted +// shell word. Declared names are already restricted to safe characters; +// this keeps that guarantee explicit across shells. +func shellSingleQuote(value string) string { + return "'" + strings.ReplaceAll(value, "'", `'\''`) + "'" } func (s *Service) cleanupDumpFile(containerID, dumpPath string) { @@ -306,6 +327,7 @@ func (s *Service) cleanupDumpFile(containerID, dumpPath string) { _, _ = s.runtime.ExecInContainer(cleanupCtx, containerID, []string{"sh", "-c", fmt.Sprintf("rm -f %q", dumpPath)}) } +// byteCounter counts the bytes of one stream. type byteCounter struct { n int64 } @@ -314,3 +336,13 @@ func (c *byteCounter) Write(p []byte) (int, error) { c.n += int64(len(p)) return len(p), nil } + +// sortedSet returns the keys of a set in sorted order. +func sortedSet(set map[string]struct{}) []string { + keys := make([]string, 0, len(set)) + for key := range set { + keys = append(keys, key) + } + sort.Strings(keys) + return keys +} diff --git a/internal/usecase/backup/service_test.go b/internal/usecase/backup/service_test.go index 8766bd93c..8f7bb7d7c 100644 --- a/internal/usecase/backup/service_test.go +++ b/internal/usecase/backup/service_test.go @@ -1,236 +1,231 @@ package backup import ( - "bytes" "context" "io" - "sync/atomic" + "strings" "testing" "time" - inmocks "github.com/bnema/gordon/internal/boundaries/in/mocks" - outiface "github.com/bnema/gordon/internal/boundaries/out" - outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" - "github.com/bnema/gordon/internal/domain" "github.com/bnema/zerowrap" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/mock" "github.com/stretchr/testify/require" -) - -func TestService_DetectDatabases_PostgresAttachment(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - storage := outmocks.NewMockBackupStorage(t) - containerSvc := inmocks.NewMockContainerService(t) - - containerSvc.EXPECT().ListAttachments(mock.Anything, "app.example.com").Return([]domain.Attachment{ - {Name: "postgres", Image: "postgres:17", ContainerID: "db1", Status: "running"}, - }) - svc := NewService(runtime, storage, containerSvc, domain.BackupConfig{}, zerowrap.Default()) + "github.com/bnema/gordon/internal/boundaries/out" + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" +) - dbs, err := svc.DetectDatabases(context.Background(), "app.example.com") - require.NoError(t, err) - require.Len(t, dbs, 1) - assert.Equal(t, domain.DBTypePostgreSQL, dbs[0].Type) - assert.Equal(t, "db1", dbs[0].ContainerID) +// appSpecWithDatabase builds one service that declares a database and +// references it from its backup declaration. +func appSpecWithDatabase(service, database, schedule string) domain.AppService { + return domain.AppService{ + Name: service, + Image: "postgres:17", + Databases: []domain.AppDatabase{{Name: database, Type: domain.AppDBPostgres, Schedule: schedule}}, + Backup: domain.AppBackup{Postgres: []string{database}}, + } } -func TestService_RunBackup_Postgres(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - storage := outmocks.NewMockBackupStorage(t) - containerSvc := inmocks.NewMockContainerService(t) - - containerSvc.EXPECT().ListAttachments(mock.Anything, "app.example.com").Return([]domain.Attachment{ - {Name: "postgres", Image: "postgres:17", ContainerID: "db123", Status: "running"}, - }) - - runtime.EXPECT().ExecInContainer(mock.Anything, "db123", mock.MatchedBy(func(cmd []string) bool { - if len(cmd) != 3 || cmd[0] != "sh" || cmd[1] != "-c" { - return false +// activeWith builds an ACTIVE record holding one effective service. +func activeWith(app string, services map[string]domain.AppService, container string) domain.AppActive { + effective := make(map[string]domain.AppEffectiveService, len(services)) + for name, spec := range services { + effective[name] = domain.AppEffectiveService{ + EffectiveRevision: "rev-1", + Container: container, + Spec: spec, } - return bytes.Contains([]byte(cmd[2]), []byte("pg_dump -Fc")) && - bytes.Contains([]byte(cmd[2]), []byte("${POSTGRES_DB:-postgres}")) && - bytes.Contains([]byte(cmd[2]), []byte(" > ")) - })).Return(&outiface.ExecResult{ExitCode: 0, Stdout: []byte("backup-data")}, nil) - - runtime.EXPECT().CopyFromContainer(mock.Anything, "db123", mock.MatchedBy(func(path string) bool { - return path != "" - })).Return(io.NopCloser(bytes.NewReader([]byte("backup-data"))), nil) - - runtime.EXPECT().ExecInContainer(mock.Anything, "db123", mock.MatchedBy(func(cmd []string) bool { - return len(cmd) == 3 && cmd[0] == "sh" && cmd[1] == "-c" && bytes.Contains([]byte(cmd[2]), []byte("rm -f")) - })).Return(&outiface.ExecResult{ExitCode: 0}, nil) - - storage.EXPECT().Store( - mock.Anything, - "app.example.com", - "postgres", - domain.BackupSchedule(""), - mock.Anything, - mock.MatchedBy(func(r io.Reader) bool { - data, _ := io.ReadAll(r) - return string(data) == "backup-data" - }), - ).Return("/tmp/backup.bak", nil) - - svc := NewService(runtime, storage, containerSvc, domain.BackupConfig{}, zerowrap.Default()) - - result, err := svc.RunBackup(context.Background(), "app.example.com", "postgres") - require.NoError(t, err) - require.NotNil(t, result) - assert.Equal(t, domain.BackupStatusCompleted, result.Job.Status) - assert.Equal(t, "/tmp/backup.bak", result.Job.FilePath) - assert.Equal(t, int64(len("backup-data")), result.Job.SizeBytes) + } + return domain.AppActive{App: app, Converged: true, ConvergedRevision: "rev-1", Services: effective} } -func TestService_RunBackup_CleansUpDumpWhenPgDumpFails(t *testing.T) { +func backupTestService(t *testing.T, state *outmocks.MockAppStateReader) (*outmocks.MockContainerRuntime, *outmocks.MockBackupStorage, *Service) { + t.Helper() runtime := outmocks.NewMockContainerRuntime(t) storage := outmocks.NewMockBackupStorage(t) - containerSvc := inmocks.NewMockContainerService(t) - - containerSvc.EXPECT().ListAttachments(mock.Anything, "app.example.com").Return([]domain.Attachment{ - {Name: "postgres", Image: "postgres:17", ContainerID: "db123", Status: "running"}, - }) - - runtime.EXPECT().ExecInContainer(mock.Anything, "db123", mock.MatchedBy(func(cmd []string) bool { - return len(cmd) == 3 && cmd[0] == "sh" && cmd[1] == "-c" && bytes.Contains([]byte(cmd[2]), []byte("pg_dump -Fc")) - })).Return(&outiface.ExecResult{ExitCode: 2, Stderr: []byte("dump failed")}, nil) - - runtime.EXPECT().ExecInContainer(mock.Anything, "db123", mock.MatchedBy(func(cmd []string) bool { - return len(cmd) == 3 && cmd[0] == "sh" && cmd[1] == "-c" && bytes.Contains([]byte(cmd[2]), []byte("rm -f")) - })).Return(&outiface.ExecResult{ExitCode: 0}, nil) - - svc := NewService(runtime, storage, containerSvc, domain.BackupConfig{}, zerowrap.Default()) - - result, err := svc.RunBackup(context.Background(), "app.example.com", "postgres") - require.Error(t, err) - assert.Nil(t, result) - assert.Contains(t, err.Error(), "pg_dump failed") + svc := &Service{runtime: runtime, storage: storage, log: zerowrap.Default(), state: state} + return runtime, storage, svc } -func TestSelectDatabaseRequiresExplicitNameWhenMultipleDetected(t *testing.T) { - db, err := selectDatabase([]domain.DBInfo{ - {Name: "postgres"}, - {Name: "analytics"}, - }, "") - require.Error(t, err) - assert.Contains(t, err.Error(), "multiple database attachments detected") - assert.Equal(t, domain.DBInfo{}, db) +// TestService_TargetsResolveFromDeclarations proves a target exists only +// because the ACTIVE record declares it: no image, port, or label is +// inspected. +func TestService_TargetsResolveFromDeclarations(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppStateReader(t) + state.EXPECT().LoadIntent(mock.Anything, "shop").Return(domain.AppStopIntent{App: "shop"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "shop").Return(activeWith("shop", map[string]domain.AppService{ + "api": appSpecWithDatabase("api", "orders", "daily"), + "web": appSpecWithDatabase("web", "users", "hourly"), + }, "ctr-api"), true, nil).Once() + + _, _, svc := backupTestService(t, state) + targets, err := svc.Targets(ctx, "shop") + require.NoError(t, err) + require.Len(t, targets, 2) + assert.Equal(t, domain.DatabaseTarget{ + App: "shop", Service: "api", Database: "orders", + Schedule: domain.ScheduleDaily, ContainerID: "ctr-api", + }, targets[0]) + assert.Equal(t, "users", targets[1].Database) + assert.Equal(t, domain.ScheduleHourly, targets[1].Schedule) } -func TestSelectDatabaseAutoSelectsOnlyDatabaseWhenUnspecified(t *testing.T) { - db, err := selectDatabase([]domain.DBInfo{{Name: "postgres"}}, "") +// TestService_TargetsIgnoreUnreferencedDatabases proves a declared +// database the backup declaration does not reference is not a target. +func TestService_TargetsIgnoreUnreferencedDatabases(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppStateReader(t) + spec := appSpecWithDatabase("api", "orders", "daily") + spec.Databases = append(spec.Databases, domain.AppDatabase{Name: "metrics", Type: domain.AppDBPostgres, Schedule: "daily"}) + state.EXPECT().LoadIntent(mock.Anything, "shop").Return(domain.AppStopIntent{App: "shop"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "shop").Return(activeWith("shop", map[string]domain.AppService{"api": spec}, "ctr-api"), true, nil).Once() + + _, _, svc := backupTestService(t, state) + targets, err := svc.Targets(ctx, "shop") require.NoError(t, err) - assert.Equal(t, "postgres", db.Name) + require.Len(t, targets, 1) + assert.Equal(t, "orders", targets[0].Database) } -func TestPostgresDumpCommandUsesEnvVar(t *testing.T) { - cmd := postgresDumpCommand("ignored") - assert.Contains(t, cmd, "${POSTGRES_DB:-postgres}") - assert.Contains(t, cmd, "${POSTGRES_USER:-postgres}") +// TestService_TargetsSkipStoppedApps proves a stopped app has no target: +// its workloads are intentionally down. +func TestService_TargetsSkipStoppedApps(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppStateReader(t) + state.EXPECT().LoadIntent(mock.Anything, "shop"). + Return(domain.AppStopIntent{App: "shop", Stopped: true}, nil).Once() + + _, _, svc := backupTestService(t, state) + targets, err := svc.Targets(ctx, "shop") + require.NoError(t, err) + assert.Empty(t, targets) } -func TestServiceStatusReturnsWhenContextCancelledDuringSemaphoreAcquire(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - storage := outmocks.NewMockBackupStorage(t) - containerSvc := inmocks.NewMockContainerService(t) - - containerSvc.EXPECT().List(mock.Anything).Return(map[string]*domain.Container{ - "a.example.com": {}, - "b.example.com": {}, - "c.example.com": {}, - "d.example.com": {}, - "e.example.com": {}, - }) - - unblock := make(chan struct{}) - defer close(unblock) - - var started int32 - startedFour := make(chan struct{}) - storage.EXPECT().List(mock.Anything, mock.Anything, (*domain.BackupSchedule)(nil)).RunAndReturn( - func(context.Context, string, *domain.BackupSchedule) ([]domain.BackupJob, error) { - if atomic.AddInt32(&started, 1) == 4 { - close(startedFour) +// TestService_RunBackup_StoresUnderAppIdentity proves a run dumps from the +// declared service container and stores the artifact under the app name. +func TestService_RunBackup_StoresUnderAppIdentity(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppStateReader(t) + state.EXPECT().LoadIntent(mock.Anything, "shop").Return(domain.AppStopIntent{App: "shop"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "shop").Return(activeWith("shop", map[string]domain.AppService{ + "api": appSpecWithDatabase("api", "orders", "daily"), + }, "ctr-api"), true, nil).Once() + + runtime, storage, svc := backupTestService(t, state) + runtime.EXPECT().ExecInContainer(mock.Anything, "ctr-api", mock.MatchedBy(func(cmd []string) bool { + return len(cmd) == 3 && strings.Contains(cmd[2], "'orders'") && !strings.Contains(cmd[2], "POSTGRES_DB") + })).Return(&out.ExecResult{ExitCode: 0}, nil).Once() + runtime.EXPECT().CopyFromContainer(mock.Anything, "ctr-api", mock.Anything). + Return(io.NopCloser(strings.NewReader("dump-bytes")), nil).Once() + runtime.EXPECT().ExecInContainer(mock.Anything, "ctr-api", mock.MatchedBy(func(cmd []string) bool { + return len(cmd) == 3 && strings.HasPrefix(cmd[2], "rm -f ") + })).Return(&out.ExecResult{ExitCode: 0}, nil).Once() + storage.EXPECT().Store(mock.Anything, "shop", "api", "orders", domain.BackupSchedule(""), mock.Anything, mock.Anything). + RunAndReturn(func(_ context.Context, _, _, _ string, _ domain.BackupSchedule, _ time.Time, data io.Reader) (string, error) { + if _, err := io.Copy(io.Discard, data); err != nil { + return "", err } - <-unblock - return nil, nil - }, - ).Times(4) - - svc := NewService(runtime, storage, containerSvc, domain.BackupConfig{}, zerowrap.Default()) - - ctx, cancel := context.WithCancel(context.Background()) - errCh := make(chan error, 1) - go func() { - _, err := svc.Status(ctx) - errCh <- err - }() - - <-startedFour - cancel() - - select { - case err := <-errCh: - require.ErrorIs(t, err, context.Canceled) - case <-time.After(250 * time.Millisecond): - t.Fatal("Status did not return after context cancellation while waiting for semaphore") - } -} + return "/backups/shop/orders/manual/file.bak", nil + }).Once() -func TestService_RunForSchedule_StoresTierAndAppliesRetention(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - storage := outmocks.NewMockBackupStorage(t) - containerSvc := inmocks.NewMockContainerService(t) - - containerSvc.EXPECT().List(mock.Anything).Return(map[string]*domain.Container{ - "app.example.com": {}, - }) - containerSvc.EXPECT().ListAttachments(mock.Anything, "app.example.com").Return([]domain.Attachment{ - {Name: "postgres", Image: "postgres:17", ContainerID: "db123", Status: "running"}, - }) - - runtime.EXPECT().ExecInContainer(mock.Anything, "db123", mock.MatchedBy(func(cmd []string) bool { - if len(cmd) != 3 || cmd[0] != "sh" || cmd[1] != "-c" { - return false - } - return bytes.Contains([]byte(cmd[2]), []byte("pg_dump -Fc")) - })).Return(&outiface.ExecResult{ExitCode: 0, Stdout: []byte("backup-data")}, nil) + result, err := svc.RunBackup(ctx, "shop", "api", "orders") + require.NoError(t, err) + assert.Equal(t, "shop", result.Job.App) + assert.Equal(t, "api", result.Job.Service) + assert.Equal(t, "orders", result.Job.DBName) + assert.Equal(t, int64(len("dump-bytes")), result.Job.SizeBytes) + assert.Equal(t, domain.BackupStatusCompleted, result.Job.Status) +} - runtime.EXPECT().CopyFromContainer(mock.Anything, "db123", mock.MatchedBy(func(path string) bool { - return path != "" - })).Return(io.NopCloser(bytes.NewReader([]byte("backup-data"))), nil) +// TestService_RunBackup_SelectorRules proves the explicit-selector rules: +// a missing selector with several candidates lists them, and an unknown +// target is not found. +func TestService_RunBackup_SelectorRules(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppStateReader(t) + active := activeWith("shop", map[string]domain.AppService{ + "api": appSpecWithDatabase("api", "orders", "daily"), + "web": appSpecWithDatabase("web", "users", "daily"), + }, "ctr-x") + state.EXPECT().LoadIntent(mock.Anything, "shop").Return(domain.AppStopIntent{App: "shop"}, nil) + state.EXPECT().LoadActive(mock.Anything, "shop").Return(active, true, nil) + + _, _, svc := backupTestService(t, state) + + _, err := svc.RunBackup(ctx, "shop", "", "") + require.Error(t, err) + assert.Contains(t, err.Error(), "ambiguous") + assert.Contains(t, err.Error(), "api/orders") + assert.Contains(t, err.Error(), "web/users") - runtime.EXPECT().ExecInContainer(mock.Anything, "db123", mock.MatchedBy(func(cmd []string) bool { - return len(cmd) == 3 && cmd[0] == "sh" && cmd[1] == "-c" && bytes.Contains([]byte(cmd[2]), []byte("rm -f")) - })).Return(&outiface.ExecResult{ExitCode: 0}, nil) + _, err = svc.RunBackup(ctx, "shop", "api", "missing") + require.Error(t, err) + assert.Contains(t, err.Error(), "not found") - storage.EXPECT().Store( - mock.Anything, - "app.example.com", - "postgres", - domain.ScheduleDaily, - mock.Anything, - mock.Anything, - ).Return("/tmp/daily-backup.bak", nil) + _, err = svc.RunBackup(ctx, "shop", "missing", "orders") + require.Error(t, err) + assert.Contains(t, err.Error(), "not found") +} - storage.EXPECT().ApplyRetention(mock.Anything, "app.example.com", domain.RetentionPolicy{Daily: 7}).Return(0, nil) +// TestService_RunForSchedule_RunsOnlyTheDeclaredSchedule proves the +// scheduler filters by each database's own declaration. +func TestService_RunForSchedule_RunsOnlyTheDeclaredSchedule(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppStateReader(t) + state.EXPECT().ListApps(mock.Anything).Return([]string{"shop"}, nil).Once() + spec := appSpecWithDatabase("api", "orders", "daily") + spec.Databases = append(spec.Databases, domain.AppDatabase{Name: "users", Type: domain.AppDBPostgres, Schedule: "hourly"}) + spec.Backup.Postgres = append(spec.Backup.Postgres, "users") + state.EXPECT().LoadIntent(mock.Anything, "shop").Return(domain.AppStopIntent{App: "shop"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "shop").Return(activeWith("shop", map[string]domain.AppService{"api": spec}, "ctr-api"), true, nil).Once() + + runtime, storage, svc := backupTestService(t, state) + runtime.EXPECT().ExecInContainer(mock.Anything, "ctr-api", mock.MatchedBy(func(cmd []string) bool { + return strings.Contains(cmd[2], "'users'") + })).Return(&out.ExecResult{ExitCode: 0}, nil).Once() + runtime.EXPECT().CopyFromContainer(mock.Anything, "ctr-api", mock.Anything). + Return(io.NopCloser(strings.NewReader("x")), nil).Once() + runtime.EXPECT().ExecInContainer(mock.Anything, "ctr-api", mock.MatchedBy(func(cmd []string) bool { + return strings.HasPrefix(cmd[2], "rm -f ") + })).Return(&out.ExecResult{ExitCode: 0}, nil).Once() + storage.EXPECT().Store(mock.Anything, "shop", "api", "users", domain.ScheduleHourly, mock.Anything, mock.Anything). + RunAndReturn(func(_ context.Context, _, _, _ string, _ domain.BackupSchedule, _ time.Time, data io.Reader) (string, error) { + if _, err := io.Copy(io.Discard, data); err != nil { + return "", err + } + return "/backups/shop/users/hourly/file.bak", nil + }).Once() + storage.EXPECT().ApplyRetention(mock.Anything, "shop", mock.Anything).Return(0, nil).Once() - svc := NewService(runtime, storage, containerSvc, domain.BackupConfig{Retention: domain.RetentionPolicy{Daily: 7}}, zerowrap.Default()) + require.NoError(t, svc.RunForSchedule(ctx, domain.ScheduleHourly)) +} - err := svc.RunForSchedule(context.Background(), domain.ScheduleDaily) - require.NoError(t, err) +// TestService_RunForSchedule_RejectsUnknownSchedule proves a caller cannot +// invent a schedule tier. +func TestService_RunForSchedule_RejectsUnknownSchedule(t *testing.T) { + state := outmocks.NewMockAppStateReader(t) + _, _, svc := backupTestService(t, state) + require.Error(t, svc.RunForSchedule(context.Background(), domain.BackupSchedule("sometimes"))) } -func TestService_RunForSchedule_RejectsInvalidSchedule(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - storage := outmocks.NewMockBackupStorage(t) - containerSvc := inmocks.NewMockContainerService(t) +// TestService_ListBackupsUsesAppIdentity proves listing is app-keyed and +// never domain-keyed. +func TestService_ListBackupsUsesAppIdentity(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppStateReader(t) + state.EXPECT().ListApps(mock.Anything).Return([]string{"shop"}, nil).Once() - svc := NewService(runtime, storage, containerSvc, domain.BackupConfig{}, zerowrap.Default()) + _, storage, svc := backupTestService(t, state) + storage.EXPECT().List(mock.Anything, "shop", mock.Anything).Return([]domain.BackupJob{{ + ID: "job-1", App: "shop", Service: "api", DBName: "orders", + Status: domain.BackupStatusCompleted, StartedAt: time.Now().UTC(), + }}, nil).Once() - err := svc.RunForSchedule(context.Background(), domain.BackupSchedule("every-minute")) - require.Error(t, err) - assert.Contains(t, err.Error(), "invalid backup schedule") + jobs, err := svc.ListBackups(ctx, "") + require.NoError(t, err) + require.Len(t, jobs, 1) + assert.Equal(t, "shop", jobs[0].App) } diff --git a/internal/usecase/backup/targets.go b/internal/usecase/backup/targets.go new file mode 100644 index 000000000..14e17dc36 --- /dev/null +++ b/internal/usecase/backup/targets.go @@ -0,0 +1,172 @@ +package backup + +import ( + "context" + "fmt" + "sort" + "strings" + + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/domain" +) + +// This file resolves backup targets from the declarative app state. There +// is no image, port, container-label, or attachment heuristic anywhere: +// a target exists because the app's ACTIVE record declares it, and its +// identity is always (app, service, resource). + +// AppStateReader is the ACTIVE/intent subset target resolution needs. +// It is narrower than out.AppState so backup paths never gain write +// access. +type AppStateReader = out.AppStateReader + +// declaredTargets walks the ACTIVE record of one app and returns its +// declared database and volume targets. STOPPED-intent apps have no +// target: their workloads are intentionally down. +func declaredTargets(ctx context.Context, state AppStateReader, app string) (databaseTargets []domain.DatabaseTarget, volumeTargets []domain.VolumeBackupTarget, err error) { + intent, err := state.LoadIntent(ctx, app) + if err != nil { + return nil, nil, fmt.Errorf("backup: load intent for %q: %w", app, err) + } + if intent.Stopped { + return nil, nil, nil + } + active, ok, err := state.LoadActive(ctx, app) + if err != nil { + return nil, nil, fmt.Errorf("backup: load active for %q: %w", app, err) + } + if !ok { + return nil, nil, nil + } + names := make([]string, 0, len(active.Services)) + for name := range active.Services { + names = append(names, name) + } + sort.Strings(names) + for _, name := range names { + eff := active.Services[name] + databaseTargets = append(databaseTargets, databaseTargetsForService(app, name, eff)...) + volumeTargets = append(volumeTargets, volumeTargetsForService(app, name, eff)...) + } + return databaseTargets, volumeTargets, nil +} + +// databaseTargetsForService returns the declared PostgreSQL databases of +// one service that its backup declaration references. Only PostgreSQL is +// supported, and the spec validation already rejects anything else. +func databaseTargetsForService(app, service string, eff domain.AppEffectiveService) []domain.DatabaseTarget { + referenced := make(map[string]struct{}, len(eff.Spec.Backup.Postgres)) + for _, ref := range eff.Spec.Backup.Postgres { + referenced[ref] = struct{}{} + } + var targets []domain.DatabaseTarget + for _, db := range eff.Spec.Databases { + if _, ok := referenced[db.Name]; !ok { + continue + } + targets = append(targets, domain.DatabaseTarget{ + App: app, + Service: service, + Database: db.Name, + Schedule: domain.BackupSchedule(db.Schedule), + ContainerID: eff.Container, + }) + } + return targets +} + +// volumeTargetsForService returns the declared volumes of one service that +// its backup declaration references, with the runtime volume name and +// mount path the archive export needs. +func volumeTargetsForService(app, service string, eff domain.AppEffectiveService) []domain.VolumeBackupTarget { + referenced := make(map[string]struct{}, len(eff.Spec.Backup.Volume)) + for _, ref := range eff.Spec.Backup.Volume { + referenced[ref] = struct{}{} + } + var targets []domain.VolumeBackupTarget + for _, volume := range eff.Spec.Volumes { + if _, ok := referenced[volume.Name]; !ok { + continue + } + targets = append(targets, domain.VolumeBackupTarget{ + App: app, + Service: service, + VolumeName: volume.Name, + RuntimeVolumeName: domain.RuntimeVolumeName(app, service, volume.Name), + MountPath: volume.Path, + }) + } + return targets +} + +// selectTarget applies the explicit-selector rules to one candidate list: +// an explicit service and resource must match exactly, while an omitted +// selector succeeds only when exactly one candidate remains. Nothing is +// ever chosen by guessing, and an ambiguity never lists anything but the +// safe target identity (service and resource names). +func selectTarget[T any](candidates []T, kind, service, resource string, key func(T) (string, string)) (T, error) { + var zero T + matched := candidates + if service != "" { + matched = filterTargets(matched, func(candidate T) bool { + candidateService, _ := key(candidate) + return candidateService == service + }) + } + if resource != "" { + matched = filterTargets(matched, func(candidate T) bool { + _, candidateResource := key(candidate) + return strings.EqualFold(candidateResource, resource) + }) + } + switch len(matched) { + case 1: + return matched[0], nil + case 0: + return zero, fmt.Errorf("backup: %s %s not found", kind, selectorLabel(service, resource)) + default: + names := make([]string, 0, len(matched)) + for _, candidate := range matched { + candidateService, candidateResource := key(candidate) + names = append(names, candidateService+"/"+candidateResource) + } + sort.Strings(names) + return zero, fmt.Errorf( + "backup: %s selector %s is ambiguous: candidates %s", + kind, selectorLabel(service, resource), strings.Join(names, ", "), + ) + } +} + +func filterTargets[T any](candidates []T, keep func(T) bool) []T { + matched := make([]T, 0, len(candidates)) + for _, candidate := range candidates { + if keep(candidate) { + matched = append(matched, candidate) + } + } + return matched +} + +// selectorLabel renders the selectors a caller supplied, for errors. +func selectorLabel(service, resource string) string { + switch { + case service != "" && resource != "": + return service + "/" + resource + case resource != "": + return resource + case service != "": + return "of service " + service + default: + return "without selectors" + } +} + +func appNames(ctx context.Context, state AppStateReader) ([]string, error) { + apps, err := state.ListApps(ctx) + if err != nil { + return nil, fmt.Errorf("backup: list apps: %w", err) + } + sort.Strings(apps) + return apps, nil +} diff --git a/internal/usecase/backup/volume_service.go b/internal/usecase/backup/volume_service.go index ccb82421d..e391c43d6 100644 --- a/internal/usecase/backup/volume_service.go +++ b/internal/usecase/backup/volume_service.go @@ -14,22 +14,24 @@ import ( "github.com/bnema/gordon/internal/domain" ) -// VolumeService orchestrates volume archive backups. +// VolumeService orchestrates declarative app volume archive backups. +// Targets come from the app's ACTIVE record: a service declares its +// volumes and which of them are backed up, and the app name is the +// backup identity. No container label or attachment is ever consulted. type VolumeService struct { - runtime out.ContainerRuntime exporter out.VolumeArchiveExporter storage out.VolumeBackupStorage config domain.VolumeBackupConfig log zerowrap.Logger + state out.AppStateReader mu sync.Mutex recent map[string]domain.VolumeBackupJob } // NewVolumeService creates a volume backup service. -func NewVolumeService(runtime out.ContainerRuntime, exporter out.VolumeArchiveExporter, storage out.VolumeBackupStorage, config domain.VolumeBackupConfig, log zerowrap.Logger) *VolumeService { +func NewVolumeService(exporter out.VolumeArchiveExporter, storage out.VolumeBackupStorage, config domain.VolumeBackupConfig, log zerowrap.Logger) *VolumeService { return &VolumeService{ - runtime: runtime, exporter: exporter, storage: storage, config: config, @@ -38,21 +40,61 @@ func NewVolumeService(runtime out.ContainerRuntime, exporter out.VolumeArchiveEx } } -// ListVolumeBackups lists completed volume backups for a domain, or all domains when empty. -func (s *VolumeService) ListVolumeBackups(ctx context.Context, domainName string) ([]domain.VolumeBackupJob, error) { +// WithAppState wires the ACTIVE app state target resolution reads. +func (s *VolumeService) WithAppState(state out.AppStateReader) *VolumeService { + s.state = state + return s +} + +// VolumeTargets returns every declared volume target of one app. +func (s *VolumeService) VolumeTargets(ctx context.Context, app string) ([]domain.VolumeBackupTarget, error) { + if s.state == nil { + return nil, fmt.Errorf("backup: app state is not wired") + } + _, targets, err := declaredTargets(ctx, s.state, app) + if err != nil { + return nil, err + } + return targets, nil +} + +// allVolumeTargets returns the declared volume targets of every app. +func (s *VolumeService) allVolumeTargets(ctx context.Context) ([]domain.VolumeBackupTarget, error) { + if s.state == nil { + return nil, fmt.Errorf("backup: app state is not wired") + } + apps, err := appNames(ctx, s.state) + if err != nil { + return nil, err + } + var targets []domain.VolumeBackupTarget + for _, app := range apps { + appTargets, err := s.VolumeTargets(ctx, app) + if err != nil { + return nil, err + } + targets = append(targets, appTargets...) + } + return targets, nil +} + +// ListVolumeBackups lists completed volume backups of one app, or every +// app when app is empty. +func (s *VolumeService) ListVolumeBackups(ctx context.Context, app string) ([]domain.VolumeBackupJob, error) { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ zerowrap.FieldLayer: "usecase", zerowrap.FieldUseCase: "ListVolumeBackups", - "domain": domainName, + "app": app, }) - jobs, err := s.storage.ListVolumeArchives(ctx, domainName) + jobs, err := s.storage.ListVolumeArchives(ctx, app) if err != nil { return nil, fmt.Errorf("list volume archives: %w", err) } return jobs, nil } -// VolumeBackupStatus returns completed backup artifacts plus current/recent in-memory job state. +// VolumeBackupStatus returns completed backup artifacts plus current or +// recent in-memory jobs. func (s *VolumeService) VolumeBackupStatus(ctx context.Context) ([]domain.VolumeBackupJob, error) { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ zerowrap.FieldLayer: "usecase", @@ -71,148 +113,100 @@ func (s *VolumeService) VolumeBackupStatus(ctx context.Context) ([]domain.Volume } s.mu.Unlock() - sort.Slice(jobs, func(i, j int) bool { - if !jobs[i].StartedAt.Equal(jobs[j].StartedAt) { - return jobs[i].StartedAt.After(jobs[j].StartedAt) - } - return jobs[i].VolumeName < jobs[j].VolumeName - }) + sortVolumeJobs(jobs) return jobs, nil } -// RunVolumeBackups runs volume backups for all eligible targets, optionally scoped to a domain and volume. -func (s *VolumeService) RunVolumeBackups(ctx context.Context, domainName, volumeName string) ([]domain.VolumeBackupJob, error) { +// RunVolumeBackups runs the declared volume backups of one app. service +// and volume are explicit selectors: an omitted selector succeeds only +// when exactly one compatible target exists. +func (s *VolumeService) RunVolumeBackups(ctx context.Context, app, service, volume string) ([]domain.VolumeBackupJob, error) { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ zerowrap.FieldLayer: "usecase", zerowrap.FieldUseCase: "RunVolumeBackups", - "domain": domainName, - "volume": volumeName, + "app": app, + "volume": volume, }) - log := zerowrap.FromCtx(ctx) if !s.config.Enabled { return []domain.VolumeBackupJob{}, nil } - - containers, err := s.runtime.ListContainers(ctx, true) + targets, err := s.VolumeTargets(ctx, app) if err != nil { - return nil, fmt.Errorf("failed to list containers for volume backups: %w", err) + return nil, err } - targets := SelectVolumeBackupTargetsForScope(containers, s.config.VolumePrefix, domainName, volumeName) - log.Info(). - Str("domain", domainName). - Str("volume", volumeName). - Int("targets", len(targets)). - Msg("selected volume backup targets") - if len(targets) == 0 { - return []domain.VolumeBackupJob{}, nil - } - - results, firstErr := s.runVolumeBackupTargets(ctx, targets) - if err := s.applyRetentionForSuccessfulVolumeBackups(ctx, results); err != nil && firstErr == nil { - firstErr = err + target, err := selectTarget(targets, "volume", service, volume, volumeTargetKey) + if err != nil { + return nil, err } - return results, firstErr -} - -func (s *VolumeService) runVolumeBackupTargets(ctx context.Context, targets []domain.VolumeBackupTarget) ([]domain.VolumeBackupJob, error) { - runCtx, cancel := context.WithCancel(ctx) - defer cancel() - - sem := make(chan struct{}, max(s.config.MaxConcurrency, 1)) - results := make([]domain.VolumeBackupJob, 0, len(targets)) - var wg sync.WaitGroup - var mu sync.Mutex - var firstErr error - -launchLoop: - for _, target := range targets { - select { - case <-runCtx.Done(): - firstErr = runCtx.Err() - break launchLoop - case sem <- struct{}{}: + job := s.runVolumeBackup(ctx, target) + if job.Status == domain.BackupStatusCompleted { + if _, err := s.storage.ApplyVolumeRetention(ctx, job.App, s.config.Retention); err != nil { + return []domain.VolumeBackupJob{job}, fmt.Errorf("apply volume backup retention for %s: %w", job.App, err) } - wg.Add(1) - go func(target domain.VolumeBackupTarget) { - defer wg.Done() - defer func() { <-sem }() - job := s.runVolumeBackup(runCtx, target) - mu.Lock() - results = append(results, job) - mu.Unlock() - }(target) } - if firstErr != nil { - cancel() + if job.Status == domain.BackupStatusFailed { + return []domain.VolumeBackupJob{job}, fmt.Errorf("volume backup failed for %s/%s: %s", job.App, job.VolumeName, job.Error) } - wg.Wait() - return results, firstVolumeBackupError(results, firstErr) + return []domain.VolumeBackupJob{job}, nil } -func (s *VolumeService) applyRetentionForSuccessfulVolumeBackups(ctx context.Context, results []domain.VolumeBackupJob) error { - successDomains, failedDomains := volumeBackupResultDomains(results) - for domainName := range successDomains { - if _, failed := failedDomains[domainName]; failed { - continue - } - deleted, err := s.storage.ApplyVolumeRetention(ctx, domainName, s.config.Retention) - if err != nil { - return fmt.Errorf("apply volume backup retention for %s: %w", domainName, err) - } - log := zerowrap.FromCtx(ctx) - log.Info(). - Str("domain", domainName). - Int("deleted", deleted). - Msg("volume backup retention applied") +// RunVolumeBackupsForSchedule runs every declared volume backup of every +// app under the installation's schedule, then applies retention. Volume +// declarations carry no schedule of their own: the installation preset is +// the schedule, and the label is only used for the caller's own filter. +func (s *VolumeService) RunVolumeBackupsForSchedule(ctx context.Context, schedule domain.BackupSchedule) error { + if !s.config.Enabled { + return nil } - return nil -} - -func firstVolumeBackupError(results []domain.VolumeBackupJob, firstErr error) error { - if firstErr != nil { - return firstErr + if schedule != "" && !isValidBackupSchedule(schedule) { + return fmt.Errorf("invalid backup schedule: %q", schedule) } - for _, job := range results { - if job.Status == domain.BackupStatusFailed { - return fmt.Errorf("volume backup failed for %s/%s: %s", job.Domain, job.VolumeName, job.Error) + targets, err := s.allVolumeTargets(ctx) + if err != nil { + return err + } + var firstErr error + apps := map[string]struct{}{} + for _, target := range targets { + job := s.runVolumeBackup(ctx, target) + if job.Status != domain.BackupStatusCompleted { + if firstErr == nil { + firstErr = fmt.Errorf("volume backup %s/%s: %s", job.App, job.VolumeName, job.Error) + } + continue } + apps[job.App] = struct{}{} } - return nil -} - -func volumeBackupResultDomains(results []domain.VolumeBackupJob) (map[string]struct{}, map[string]struct{}) { - successDomains := make(map[string]struct{}) - failedDomains := make(map[string]struct{}) - for _, job := range results { - switch job.Status { - case domain.BackupStatusCompleted: - successDomains[job.Domain] = struct{}{} - case domain.BackupStatusFailed: - failedDomains[job.Domain] = struct{}{} + for _, app := range sortedSet(apps) { + if _, err := s.storage.ApplyVolumeRetention(ctx, app, s.config.Retention); err != nil { + if firstErr == nil { + firstErr = fmt.Errorf("apply volume backup retention for %s: %w", app, err) + } } } - return successDomains, failedDomains + return firstErr } func (s *VolumeService) runVolumeBackup(ctx context.Context, target domain.VolumeBackupTarget) domain.VolumeBackupJob { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - "domain": target.Domain, + "app": target.App, + "service": target.Service, "volume": target.VolumeName, - "container_name": target.ContainerName, + "runtime_volume": target.RuntimeVolumeName, "mount_path": target.MountPath, }) log := zerowrap.FromCtx(ctx) started := time.Now().UTC() job := domain.VolumeBackupJob{ - ID: newBackupJobID(started), - Domain: target.Domain, - ContainerName: target.ContainerName, - ContainerID: target.ContainerID, - VolumeName: target.VolumeName, - MountPath: target.MountPath, - Type: domain.BackupTypeVolumeArchive, - Status: domain.BackupStatusRunning, - StartedAt: started, + ID: newBackupJobID(started), + App: target.App, + Service: target.Service, + VolumeName: target.VolumeName, + RuntimeVolumeName: target.RuntimeVolumeName, + MountPath: target.MountPath, + Type: domain.BackupTypeVolumeArchive, + Status: domain.BackupStatusRunning, + StartedAt: started, Metadata: map[string]string{ "compression": string(s.config.Compression), }, @@ -227,7 +221,7 @@ func (s *VolumeService) runVolumeBackup(ctx context.Context, target domain.Volum defer cancel() archive, err := s.exporter.ExportVolumeArchive(exportCtx, domain.VolumeArchiveRequest{ - VolumeName: target.VolumeName, + VolumeName: target.RuntimeVolumeName, MountPath: target.MountPath, Compression: s.config.Compression, HelperImage: s.config.HelperImage, @@ -257,6 +251,11 @@ func (s *VolumeService) runVolumeBackup(ctx context.Context, target domain.Volum return job } +// volumeTargetKey reports the (service, volume) identity of one target. +func volumeTargetKey(target domain.VolumeBackupTarget) (string, string) { + return target.Service, target.VolumeName +} + func (s *VolumeService) failJob(ctx context.Context, job domain.VolumeBackupJob, err error) domain.VolumeBackupJob { job.Status = domain.BackupStatusFailed job.CompletedAt = time.Now().UTC() @@ -270,7 +269,7 @@ func (s *VolumeService) failJob(ctx context.Context, job domain.VolumeBackupJob, func (s *VolumeService) remember(job domain.VolumeBackupJob) { s.mu.Lock() defer s.mu.Unlock() - s.recent[job.Domain+"/"+job.VolumeName] = job + s.recent[job.App+"/"+job.VolumeName] = job if len(s.recent) > 100 { for k := range s.recent { delete(s.recent, k) @@ -282,5 +281,14 @@ func (s *VolumeService) remember(job domain.VolumeBackupJob) { func (s *VolumeService) forget(job domain.VolumeBackupJob) { s.mu.Lock() defer s.mu.Unlock() - delete(s.recent, job.Domain+"/"+job.VolumeName) + delete(s.recent, job.App+"/"+job.VolumeName) +} + +func sortVolumeJobs(jobs []domain.VolumeBackupJob) { + sort.Slice(jobs, func(i, j int) bool { + if !jobs[i].StartedAt.Equal(jobs[j].StartedAt) { + return jobs[i].StartedAt.After(jobs[j].StartedAt) + } + return jobs[i].VolumeName < jobs[j].VolumeName + }) } diff --git a/internal/usecase/backup/volume_service_test.go b/internal/usecase/backup/volume_service_test.go index 73c84b0a6..0f4339c32 100644 --- a/internal/usecase/backup/volume_service_test.go +++ b/internal/usecase/backup/volume_service_test.go @@ -3,7 +3,7 @@ package backup import ( "bytes" "context" - "fmt" + "errors" "io" "testing" "time" @@ -22,10 +22,12 @@ func testLogger() zerowrap.Logger { } type fakeVolumeArchiveExporter struct { - err error + err error + requested []domain.VolumeArchiveRequest } -func (f fakeVolumeArchiveExporter) ExportVolumeArchive(context.Context, domain.VolumeArchiveRequest) (*domain.VolumeArchiveResult, error) { +func (f *fakeVolumeArchiveExporter) ExportVolumeArchive(_ context.Context, request domain.VolumeArchiveRequest) (*domain.VolumeArchiveResult, error) { + f.requested = append(f.requested, request) if f.err != nil { return nil, f.err } @@ -35,15 +37,20 @@ func (f fakeVolumeArchiveExporter) ExportVolumeArchive(context.Context, domain.V type fakeVolumeBackupStorage struct { stored []domain.VolumeBackupJob retention []string + jobs []domain.VolumeBackupJob listErr error + storeErr error } func (f *fakeVolumeBackupStorage) StoreVolumeArchive(_ context.Context, job domain.VolumeBackupJob, data io.Reader) (string, error) { if _, err := io.Copy(io.Discard, data); err != nil { return "", err } + if f.storeErr != nil { + return "", f.storeErr + } f.stored = append(f.stored, job) - return "s3://bucket/" + job.VolumeName, nil + return "s3://bucket/" + job.App + "/" + job.VolumeName, nil } func (f *fakeVolumeBackupStorage) GetVolumeArchive(context.Context, string) (io.ReadCloser, error) { @@ -54,112 +61,180 @@ func (f *fakeVolumeBackupStorage) ListVolumeArchives(context.Context, string) ([ if f.listErr != nil { return nil, f.listErr } - return nil, nil + return f.jobs, nil } func (f *fakeVolumeBackupStorage) DeleteVolumeArchive(context.Context, string) error { return nil } -func (f *fakeVolumeBackupStorage) ApplyVolumeRetention(_ context.Context, domainName string, _ domain.VolumeBackupRetentionPolicy) (int, error) { - f.retention = append(f.retention, domainName) +func (f *fakeVolumeBackupStorage) ApplyVolumeRetention(_ context.Context, app string, _ domain.VolumeBackupRetentionPolicy) (int, error) { + f.retention = append(f.retention, app) return 0, nil } -func TestVolumeServiceRunVolumeBackups(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) +// volumeSpec declares one volume and references it from the service's +// backup declaration. +func volumeSpec(service, volume, path string) domain.AppService { + return domain.AppService{ + Name: service, + Image: "app:1", + Volumes: []domain.AppVolume{{Name: volume, Path: path}}, + Backup: domain.AppBackup{Volume: []string{volume}}, + } +} + +func volumeTestService(t *testing.T, state *outmocks.MockAppStateReader, exporter *fakeVolumeArchiveExporter, storage *fakeVolumeBackupStorage, enabled bool) *VolumeService { + t.Helper() + return NewVolumeService(exporter, storage, domain.VolumeBackupConfig{ + Enabled: enabled, + Compression: domain.VolumeBackupCompressionGzip, + Retention: domain.VolumeBackupRetentionPolicy{Keep: 2}, + Timeout: time.Minute, + HelperImage: "helper:latest", + }, testLogger()).WithAppState(state) +} + +// TestVolumeService_TargetsResolveFromDeclarations proves a volume target +// comes from the ACTIVE declaration, with the runtime volume name and +// mount path the exporter needs. No container label is consulted. +func TestVolumeService_TargetsResolveFromDeclarations(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppStateReader(t) + state.EXPECT().LoadIntent(mock.Anything, "shop").Return(domain.AppStopIntent{App: "shop"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "shop").Return(activeWith("shop", map[string]domain.AppService{ + "api": volumeSpec("api", "data", "/var/lib/data"), + }, "ctr-api"), true, nil).Once() + + svc := volumeTestService(t, state, &fakeVolumeArchiveExporter{}, &fakeVolumeBackupStorage{}, true) + targets, err := svc.VolumeTargets(ctx, "shop") + require.NoError(t, err) + require.Len(t, targets, 1) + assert.Equal(t, "shop", targets[0].App) + assert.Equal(t, "api", targets[0].Service) + assert.Equal(t, "data", targets[0].VolumeName) + assert.Equal(t, domain.RuntimeVolumeName("shop", "api", "data"), targets[0].RuntimeVolumeName) + assert.Equal(t, "/var/lib/data", targets[0].MountPath) +} + +// TestVolumeService_RunVolumeBackups_ExportsTheDeclaredVolume proves the +// archive export targets the runtime volume and the artifact is stored +// under the app identity. +func TestVolumeService_RunVolumeBackups_ExportsTheDeclaredVolume(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppStateReader(t) + state.EXPECT().LoadIntent(mock.Anything, "shop").Return(domain.AppStopIntent{App: "shop"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "shop").Return(activeWith("shop", map[string]domain.AppService{ + "api": volumeSpec("api", "data", "/var/lib/data"), + }, "ctr-api"), true, nil).Once() + + exporter := &fakeVolumeArchiveExporter{} storage := &fakeVolumeBackupStorage{} - svc := NewVolumeService(runtime, fakeVolumeArchiveExporter{}, storage, domain.VolumeBackupConfig{ - Enabled: true, - Compression: domain.VolumeBackupCompressionGzip, - Retention: domain.VolumeBackupRetentionPolicy{Keep: 2}, - Timeout: time.Minute, - MaxConcurrency: 1, - HelperImage: "helper:latest", - VolumePrefix: "gordon", - }, testLogger()) - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - { - ID: "app", - Name: "app", - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelDomain: "app.example.com", - }, - VolumeMounts: []domain.ContainerVolumeMount{{Name: "gordon-app-data", Type: "volume", Destination: "/data"}}, - }, - }, nil) - - jobs, err := svc.RunVolumeBackups(context.Background(), "", "") + svc := volumeTestService(t, state, exporter, storage, true) + + jobs, err := svc.RunVolumeBackups(ctx, "shop", "api", "data") require.NoError(t, err) require.Len(t, jobs, 1) assert.Equal(t, domain.BackupStatusCompleted, jobs[0].Status) - assert.Equal(t, "s3://bucket/gordon-app-data", jobs[0].ArtifactRef) + assert.Equal(t, "shop", jobs[0].App) + assert.Equal(t, "api", jobs[0].Service) + assert.Equal(t, "data", jobs[0].VolumeName) assert.Equal(t, int64(len("archive")), jobs[0].SizeBytes) - assert.Len(t, storage.stored, 1) - assert.Equal(t, []string{"app.example.com"}, storage.retention) + + require.Len(t, exporter.requested, 1) + assert.Equal(t, domain.RuntimeVolumeName("shop", "api", "data"), exporter.requested[0].VolumeName) + assert.Equal(t, "/var/lib/data", exporter.requested[0].MountPath) + require.Len(t, storage.stored, 1) + assert.Equal(t, "shop", storage.stored[0].App) + assert.Equal(t, []string{"shop"}, storage.retention) + assert.Contains(t, jobs[0].ArtifactRef, "shop") } -func TestVolumeServiceRunVolumeBackupsReturnsPartialFailure(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - storage := &fakeVolumeBackupStorage{} - svc := NewVolumeService(runtime, fakeVolumeArchiveExporter{err: fmt.Errorf("boom")}, storage, domain.VolumeBackupConfig{ - Enabled: true, - Compression: domain.VolumeBackupCompressionGzip, - Retention: domain.VolumeBackupRetentionPolicy{Keep: 2}, - Timeout: time.Minute, - MaxConcurrency: 1, - HelperImage: "helper:latest", - VolumePrefix: "gordon", - }, testLogger()) - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - { - ID: "app", - Name: "app", - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelDomain: "app.example.com", - }, - VolumeMounts: []domain.ContainerVolumeMount{{Name: "gordon-app-data", Type: "volume", Destination: "/data"}}, - }, - }, nil) - - jobs, err := svc.RunVolumeBackups(context.Background(), "", "") +// TestVolumeService_RunVolumeBackups_SelectorRules proves the explicit +// selector rules for volume backups too. +func TestVolumeService_RunVolumeBackups_SelectorRules(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppStateReader(t) + state.EXPECT().LoadIntent(mock.Anything, "shop").Return(domain.AppStopIntent{App: "shop"}, nil) + state.EXPECT().LoadActive(mock.Anything, "shop").Return(activeWith("shop", map[string]domain.AppService{ + "api": volumeSpec("api", "data", "/data"), + "web": volumeSpec("web", "uploads", "/uploads"), + }, "ctr-x"), true, nil) + + svc := volumeTestService(t, state, &fakeVolumeArchiveExporter{}, &fakeVolumeBackupStorage{}, true) + + _, err := svc.RunVolumeBackups(ctx, "shop", "", "") require.Error(t, err) - require.Len(t, jobs, 1) - assert.Equal(t, domain.BackupStatusFailed, jobs[0].Status) - assert.Contains(t, jobs[0].Error, "boom") - assert.Empty(t, storage.stored) -} + assert.Contains(t, err.Error(), "ambiguous") + assert.Contains(t, err.Error(), "api/data") + assert.Contains(t, err.Error(), "web/uploads") -func TestVolumeServiceListVolumeBackupsWrapsStorageError(t *testing.T) { - storageErr := fmt.Errorf("s3 unavailable") - svc := NewVolumeService(outmocks.NewMockContainerRuntime(t), fakeVolumeArchiveExporter{}, &fakeVolumeBackupStorage{listErr: storageErr}, domain.VolumeBackupConfig{}, testLogger()) + _, err = svc.RunVolumeBackups(ctx, "shop", "api", "missing") + require.Error(t, err) + assert.Contains(t, err.Error(), "not found") +} - jobs, err := svc.ListVolumeBackups(context.Background(), "app.example.com") +// TestVolumeService_RunVolumeBackupsForSchedule_RunsEveryDeclaredTarget +// proves the scheduled pass enumerates declared targets and applies +// retention per app. +func TestVolumeService_RunVolumeBackupsForSchedule_RunsEveryDeclaredTarget(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppStateReader(t) + state.EXPECT().ListApps(mock.Anything).Return([]string{"shop"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "shop").Return(domain.AppStopIntent{App: "shop"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "shop").Return(activeWith("shop", map[string]domain.AppService{ + "api": volumeSpec("api", "data", "/data"), + }, "ctr-api"), true, nil).Once() + + exporter := &fakeVolumeArchiveExporter{} + storage := &fakeVolumeBackupStorage{} + svc := volumeTestService(t, state, exporter, storage, true) - require.ErrorIs(t, err, storageErr) - assert.Contains(t, err.Error(), "list volume archives") - assert.Nil(t, jobs) + require.NoError(t, svc.RunVolumeBackupsForSchedule(ctx, domain.ScheduleDaily)) + require.Len(t, storage.stored, 1) + assert.Equal(t, "data", storage.stored[0].VolumeName) + assert.Equal(t, []string{"shop"}, storage.retention) } -func TestVolumeServiceStatusWrapsStorageError(t *testing.T) { - storageErr := fmt.Errorf("s3 unavailable") - svc := NewVolumeService(outmocks.NewMockContainerRuntime(t), fakeVolumeArchiveExporter{}, &fakeVolumeBackupStorage{listErr: storageErr}, domain.VolumeBackupConfig{}, testLogger()) +// TestVolumeService_DisabledDoesNothing proves a disabled backup +// configuration runs nothing at all. +func TestVolumeService_DisabledDoesNothing(t *testing.T) { + state := outmocks.NewMockAppStateReader(t) + svc := volumeTestService(t, state, &fakeVolumeArchiveExporter{}, &fakeVolumeBackupStorage{}, false) - jobs, err := svc.VolumeBackupStatus(context.Background()) + jobs, err := svc.RunVolumeBackups(context.Background(), "shop", "", "") + require.NoError(t, err) + assert.Empty(t, jobs) + require.NoError(t, svc.RunVolumeBackupsForSchedule(context.Background(), domain.ScheduleDaily)) +} - require.ErrorIs(t, err, storageErr) - assert.Contains(t, err.Error(), "list volume backup status") - assert.Nil(t, jobs) +// TestVolumeService_ListVolumeBackupsWrapsStorageError proves storage +// failures are reported with context. +func TestVolumeService_ListVolumeBackupsWrapsStorageError(t *testing.T) { + state := outmocks.NewMockAppStateReader(t) + storage := &fakeVolumeBackupStorage{listErr: errors.New("boom")} + svc := volumeTestService(t, state, &fakeVolumeArchiveExporter{}, storage, true) + + _, err := svc.ListVolumeBackups(context.Background(), "shop") + require.Error(t, err) + assert.Contains(t, err.Error(), "list volume archives") } -func TestVolumeServiceDisabledDoesNothing(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - svc := NewVolumeService(runtime, fakeVolumeArchiveExporter{}, &fakeVolumeBackupStorage{}, domain.VolumeBackupConfig{}, testLogger()) +// TestVolumeService_FailedExportReportsTheJob proves a failed export keeps +// a failed job with its error and stores nothing. +func TestVolumeService_FailedExportReportsTheJob(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppStateReader(t) + state.EXPECT().LoadIntent(mock.Anything, "shop").Return(domain.AppStopIntent{App: "shop"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "shop").Return(activeWith("shop", map[string]domain.AppService{ + "api": volumeSpec("api", "data", "/data"), + }, "ctr-api"), true, nil).Once() - jobs, err := svc.RunVolumeBackups(context.Background(), "", "") + storage := &fakeVolumeBackupStorage{} + svc := volumeTestService(t, state, &fakeVolumeArchiveExporter{err: errors.New("helper failed")}, storage, true) - require.NoError(t, err) - assert.Empty(t, jobs) + jobs, err := svc.RunVolumeBackups(ctx, "shop", "api", "data") + require.Error(t, err) + require.Len(t, jobs, 1) + assert.Equal(t, domain.BackupStatusFailed, jobs[0].Status) + assert.Contains(t, jobs[0].Error, "helper failed") + assert.Empty(t, storage.stored) } diff --git a/internal/usecase/backup/volume_targets.go b/internal/usecase/backup/volume_targets.go deleted file mode 100644 index 4eddef7c2..000000000 --- a/internal/usecase/backup/volume_targets.go +++ /dev/null @@ -1,115 +0,0 @@ -package backup - -import ( - "sort" - "strings" - - "github.com/bnema/gordon/internal/domain" -) - -// SelectVolumeBackupTargets returns deterministic volume backup targets from Gordon-managed containers. -func SelectVolumeBackupTargets(containers []*domain.Container, volumePrefix string) []domain.VolumeBackupTarget { - return SelectVolumeBackupTargetsForScope(containers, volumePrefix, "", "") -} - -// SelectVolumeBackupTargetsForScope returns deterministic volume backup targets after applying domain/volume filters. -func SelectVolumeBackupTargetsForScope(containers []*domain.Container, volumePrefix, domainFilter, volumeFilter string) []domain.VolumeBackupTarget { - targets := make([]domain.VolumeBackupTarget, 0) - seenVolumes := make(map[string]struct{}) - - for _, c := range sortedVolumeBackupContainers(containers) { - domainName, ok := selectedVolumeBackupContainerDomain(c, domainFilter) - if !ok { - continue - } - for _, mount := range selectedVolumeBackupMounts(c.VolumeMounts, volumePrefix, volumeFilter) { - if _, ok := seenVolumes[mount.Name]; ok { - continue - } - seenVolumes[mount.Name] = struct{}{} - targets = append(targets, domain.VolumeBackupTarget{ - Domain: domainName, - ContainerName: c.Name, - ContainerID: c.ID, - VolumeName: mount.Name, - MountPath: mount.Destination, - }) - } - } - - return targets -} - -func sortedVolumeBackupContainers(containers []*domain.Container) []*domain.Container { - sorted := make([]*domain.Container, 0, len(containers)) - for _, c := range containers { - if c != nil { - sorted = append(sorted, c) - } - } - sort.Slice(sorted, func(i, j int) bool { - leftDomain := volumeBackupDomain(sorted[i]) - rightDomain := volumeBackupDomain(sorted[j]) - if leftDomain != rightDomain { - return leftDomain < rightDomain - } - return sorted[i].Name < sorted[j].Name - }) - return sorted -} - -func selectedVolumeBackupContainerDomain(c *domain.Container, domainFilter string) (string, bool) { - if c.Labels == nil || c.Labels[domain.LabelManaged] != "true" { - return "", false - } - domainName := volumeBackupDomain(c) - if domainName == "" || (domainFilter != "" && domainName != domainFilter) { - return "", false - } - return domainName, true -} - -func selectedVolumeBackupMounts(mounts []domain.ContainerVolumeMount, volumePrefix, volumeFilter string) []domain.ContainerVolumeMount { - selected := append([]domain.ContainerVolumeMount(nil), mounts...) - sort.Slice(selected, func(i, j int) bool { - if selected[i].Name != selected[j].Name { - return selected[i].Name < selected[j].Name - } - return selected[i].Destination < selected[j].Destination - }) - out := selected[:0] - for _, mount := range selected { - if volumeFilter != "" && mount.Name != volumeFilter { - continue - } - if !isEligibleVolumeMount(mount, volumePrefix) { - continue - } - out = append(out, mount) - } - return out -} - -func volumeBackupDomain(c *domain.Container) string { - if c == nil || c.Labels == nil { - return "" - } - if c.Labels[domain.LabelAttachment] == "true" { - return c.Labels[domain.LabelAttachedTo] - } - if domainName := c.Labels[domain.LabelDomain]; domainName != "" { - return domainName - } - return c.Labels[domain.LabelRoute] -} - -func isEligibleVolumeMount(mount domain.ContainerVolumeMount, volumePrefix string) bool { - if mount.Type != "volume" || mount.Name == "" || mount.Destination == "" { - return false - } - prefix := strings.TrimSpace(volumePrefix) - if prefix == "" { - return true - } - return strings.HasPrefix(mount.Name, prefix+"-") -} diff --git a/internal/usecase/backup/volume_targets_test.go b/internal/usecase/backup/volume_targets_test.go deleted file mode 100644 index 5a0c7ed70..000000000 --- a/internal/usecase/backup/volume_targets_test.go +++ /dev/null @@ -1,115 +0,0 @@ -package backup - -import ( - "testing" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" - - "github.com/bnema/gordon/internal/domain" -) - -func TestSelectVolumeBackupTargets(t *testing.T) { - containers := []*domain.Container{ - { - ID: "unmanaged", - Name: "unmanaged", - Labels: map[string]string{ - domain.LabelDomain: "app.example.com", - }, - VolumeMounts: []domain.ContainerVolumeMount{{Name: "gordon-app-data", Type: "volume", Destination: "/data"}}, - }, - { - ID: "app", - Name: "app", - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelDomain: "app.example.com", - }, - VolumeMounts: []domain.ContainerVolumeMount{ - {Name: "gordon-app-example-com-data", Type: "volume", Destination: "/data"}, - {Name: "gordon-app-example-com-cache", Type: "bind", Destination: "/cache"}, - {Name: "random-anonymous", Type: "volume", Destination: "/anon"}, - {Name: "", Type: "volume", Destination: "/empty"}, - }, - }, - { - ID: "postgres", - Name: "postgres", - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: "app.example.com", - }, - VolumeMounts: []domain.ContainerVolumeMount{ - {Name: "gordon-postgres-data", Type: "volume", Destination: "/var/lib/postgresql/data"}, - {Name: "gordon-app-example-com-data", Type: "volume", Destination: "/shared"}, - }, - }, - } - - targets := SelectVolumeBackupTargets(containers, "gordon") - - assert.Equal(t, []domain.VolumeBackupTarget{ - { - Domain: "app.example.com", - ContainerName: "app", - ContainerID: "app", - VolumeName: "gordon-app-example-com-data", - MountPath: "/data", - }, - { - Domain: "app.example.com", - ContainerName: "postgres", - ContainerID: "postgres", - VolumeName: "gordon-postgres-data", - MountPath: "/var/lib/postgresql/data", - }, - }, targets) -} - -func TestSelectVolumeBackupTargetsForScopeDedupesAfterFiltering(t *testing.T) { - containers := []*domain.Container{ - { - ID: "alpha", - Name: "alpha", - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelDomain: "alpha.example.com", - }, - VolumeMounts: []domain.ContainerVolumeMount{{Name: "gordon-shared", Type: "volume", Destination: "/data"}}, - }, - { - ID: "beta", - Name: "beta", - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelDomain: "beta.example.com", - }, - VolumeMounts: []domain.ContainerVolumeMount{{Name: "gordon-shared", Type: "volume", Destination: "/data"}}, - }, - } - - targets := SelectVolumeBackupTargetsForScope(containers, "gordon", "beta.example.com", "") - - require.Len(t, targets, 1) - assert.Equal(t, "beta.example.com", targets[0].Domain) - assert.Equal(t, "gordon-shared", targets[0].VolumeName) -} - -func TestSelectVolumeBackupTargetsRequiresDomain(t *testing.T) { - containers := []*domain.Container{ - { - ID: "missing-domain", - Name: "missing-domain", - Labels: map[string]string{ - domain.LabelManaged: "true", - }, - VolumeMounts: []domain.ContainerVolumeMount{{Name: "gordon-data", Type: "volume", Destination: "/data"}}, - }, - } - - targets := SelectVolumeBackupTargets(containers, "gordon") - - assert.Empty(t, targets) -} diff --git a/internal/usecase/config/service.go b/internal/usecase/config/service.go index 800903776..72c31d268 100644 --- a/internal/usecase/config/service.go +++ b/internal/usecase/config/service.go @@ -2,15 +2,8 @@ package config import ( - "cmp" "context" "fmt" - "os" - "path/filepath" - "reflect" - "regexp" - "slices" - "strconv" "strings" "sync" "sync/atomic" @@ -18,7 +11,6 @@ import ( "github.com/bnema/zerowrap" "github.com/fsnotify/fsnotify" - "github.com/pelletier/go-toml/v2" "github.com/spf13/viper" "github.com/bnema/gordon/internal/boundaries/out" @@ -27,36 +19,20 @@ import ( // Config holds the loaded configuration. type Config struct { - ServerPort int - RegistryPort int - RegistryDomain string - LegacyRegistryDomains []string - DataDir string - AutoRouteEnabled bool - AutoRouteAllowedDomains []string `mapstructure:"auto_route_allowed_domains" json:"auto_route_allowed_domains,omitempty"` - PreviewEnabled bool - PreviewTTL time.Duration - PreviewSeparator string - PreviewTagPatterns []string - PreviewDataCopy bool - PreviewEnvCopy bool - NetworkIsolation bool - NetworkPrefix string - Routes map[string]routeConfig - ExternalRoutes map[string]string // domain -> "host:port" - RegistryAuthEnabled bool - RegistryAuthUsername string - RegistryAuthPassword string - VolumeAutoCreate bool - VolumePrefix string - VolumePreserve bool - NetworkGroups map[string][]string - Attachments map[string][]string -} - -type routeConfig struct { - Image string `toml:"image"` - HTTPS bool `toml:"https"` + ServerPort int + RegistryPort int + RegistryDomain string + LegacyRegistryDomains []string + DataDir string + NetworkIsolation bool + NetworkPrefix string + ExternalRoutes map[string]string // domain -> "host:port" + RegistryAuthEnabled bool + RegistryAuthUsername string + RegistryAuthPassword string + VolumeAutoCreate bool + VolumePrefix string + VolumePreserve bool } // Service implements the ConfigService interface. @@ -91,30 +67,17 @@ func (s *Service) Load(ctx context.Context) error { newConfig := s.loadConfigValues() - routes, err := loadCanonicalRoutes(s.viper.Get("routes")) - if err != nil { - return log.WrapErr(err, "failed to load routes") - } - newConfig.Routes = routes externalRoutes, err := loadExternalRoutes(s.viper.Get("external_routes")) if err != nil { return log.WrapErr(err, "failed to load external routes") } - for domainName := range externalRoutes { - if _, exists := routes[domainName]; exists { - return fmt.Errorf("external route %q conflicts with configured route", domainName) - } - } newConfig.ExternalRoutes = externalRoutes - newConfig.NetworkGroups = loadStringArrayMap(s.viper.Get("network_groups")) - newConfig.Attachments = loadStringArrayMap(s.viper.Get("attachments")) s.config = newConfig log.Info(). Int("server_port", s.config.ServerPort). Int("registry_port", s.config.RegistryPort). - Int(zerowrap.FieldCount, len(s.config.Routes)). Msg("configuration loaded") return nil @@ -139,67 +102,33 @@ func (s *Service) Reload(ctx context.Context) error { return s.Load(ctx) } -// loadConfigValues loads simple config values from viper. +// loadConfigValues loads installation config values from viper. +// Retired application keys (routes, attachments, network_groups, auto, +// preview) are NOT loaded here: presence fails startup/reload closed +// with a config-retired diagnostic before any mutation. func (s *Service) loadConfigValues() Config { - // Prefer gordon_domain over registry_domain - registryDomain := s.viper.GetString("server.gordon_domain") + // RegistryDomain identifies the registry in image references. It may be a + // DNS name or an IP address (with an optional port). Fall back to the + // Gordon domain only for installations that share one public endpoint. + registryDomain := s.viper.GetString("server.registry_domain") if registryDomain == "" { - registryDomain = s.viper.GetString("server.registry_domain") - } - - // Backward compat: try new keys first, fall back to old - autoEnabled := s.viper.GetBool("auto.enabled") - if !autoEnabled { - autoEnabled = s.viper.GetBool("auto_route.enabled") - } - allowedDomains := s.viper.GetStringSlice("auto.allowed_domains") - if len(allowedDomains) == 0 { - allowedDomains = s.viper.GetStringSlice("auto_route_allowed_domains") - } - - // Preview config - previewTTLStr := s.viper.GetString("auto.preview.ttl") - previewTTL, err := time.ParseDuration(previewTTLStr) - if err != nil && previewTTLStr != "" { - log := zerowrap.FromCtx(context.Background()) - log.Warn().Err(err).Str("ttl", previewTTLStr).Msg("invalid preview TTL format, using default 48h (use Go duration strings, e.g. \"168h\" for 7 days)") - } - if previewTTL == 0 { - previewTTL = 48 * time.Hour - } - previewSep := s.viper.GetString("auto.preview.separator") - if previewSep == "" { - previewSep = "--" + registryDomain = s.viper.GetString("server.gordon_domain") } legacyRegistryDomains := append([]string{}, s.viper.GetStringSlice("server.legacy_registry_domains")...) return Config{ - ServerPort: s.viper.GetInt("server.port"), - RegistryPort: s.viper.GetInt("server.registry_port"), - RegistryDomain: registryDomain, - LegacyRegistryDomains: legacyRegistryDomains, - DataDir: s.viper.GetString("server.data_dir"), - AutoRouteEnabled: autoEnabled, - AutoRouteAllowedDomains: append([]string{}, allowedDomains...), - PreviewEnabled: s.viper.GetBool("auto.preview.enabled"), - PreviewTTL: previewTTL, - PreviewSeparator: previewSep, - PreviewTagPatterns: append([]string{}, s.viper.GetStringSlice("auto.preview.tag_patterns")...), - PreviewDataCopy: s.viper.GetBool("auto.preview.data_copy"), - PreviewEnvCopy: !s.viper.IsSet("auto.preview.env_copy") || s.viper.GetBool("auto.preview.env_copy"), - NetworkIsolation: s.viper.GetBool("network_isolation.enabled"), - NetworkPrefix: s.viper.GetString("network_isolation.network_prefix"), - RegistryAuthEnabled: s.viper.GetBool("auth.enabled"), - RegistryAuthUsername: s.viper.GetString("auth.username"), - RegistryAuthPassword: s.viper.GetString("auth.password"), - VolumeAutoCreate: s.viper.GetBool("volumes.auto_create"), - VolumePrefix: s.viper.GetString("volumes.prefix"), - VolumePreserve: s.viper.GetBool("volumes.preserve"), - Routes: make(map[string]routeConfig), - ExternalRoutes: make(map[string]string), - NetworkGroups: make(map[string][]string), - Attachments: make(map[string][]string), + ServerPort: s.viper.GetInt("server.port"), + RegistryPort: s.viper.GetInt("server.registry_port"), + RegistryDomain: registryDomain, + LegacyRegistryDomains: legacyRegistryDomains, + DataDir: s.viper.GetString("server.data_dir"), + NetworkIsolation: s.viper.GetBool("network_isolation.enabled"), + NetworkPrefix: s.viper.GetString("network_isolation.network_prefix"), + VolumeAutoCreate: s.viper.GetBool("volumes.auto_create"), + VolumePrefix: s.viper.GetString("volumes.prefix"), + VolumePreserve: s.viper.GetBool("volumes.preserve"), + ExternalRoutes: make(map[string]string), } } @@ -249,946 +178,94 @@ func loadExternalRoutes(raw any) (map[string]string, error) { } // loadStringArrayMap loads a map[string][]string from a viper value. -func loadStringArrayMap(raw any) map[string][]string { - result := make(map[string][]string) - if raw == nil { - return result - } - m, ok := raw.(map[string]any) - if !ok { - return result - } - for k, v := range m { - if arr, ok := v.([]any); ok { - var strs []string - for _, item := range arr { - if s, ok := item.(string); ok { - strs = append(strs, s) - } - } - result[k] = strs - } - } - return result -} - -func loadCanonicalRoutes(raw any) (map[string]routeConfig, error) { - result := make(map[string]routeConfig) - if raw == nil { - return result, nil - } - - routes, ok := raw.(map[string]any) - if !ok { - return nil, fmt.Errorf("routes section must be a table, got %T", raw) - } - - for key, value := range routes { - if strings.HasPrefix(key, "http://") { - continue - } +func (s *Service) Watch(ctx context.Context, onChange func()) error { + log := zerowrap.FromCtx(ctx) - route, err := parseCanonicalRouteEntry(key, value) - if err != nil { - return nil, err - } - canonicalKey, _ := domain.CanonicalRouteDomain(key) - if _, exists := result[canonicalKey]; exists { - return nil, fmt.Errorf("duplicate route key %q canonicalizes to %q", key, canonicalKey) + s.viper.OnConfigChange(func(e fsnotify.Event) { + // Check if this event is within the debounce window of our own Save + lastSave := atomic.LoadInt64(&s.lastSaveTime) + if lastSave > 0 && time.Now().UnixNano()-lastSave < s.debounceDelay { + log.Debug().Str("file", e.Name).Msg("skipping config reload (triggered by save)") + return } - result[canonicalKey] = route - } - for key, value := range routes { - if !strings.HasPrefix(key, "http://") { - continue - } + log.Info().Str("file", e.Name).Msg("config file changed") - domainName := strings.TrimPrefix(key, "http://") - canonicalDomain, ok := domain.CanonicalRouteDomain(domainName) - if !ok { - return nil, fmt.Errorf("invalid route key %q: %w", legacyRouteStorageKey(domainName), domain.ErrRouteDomainInvalid) - } - if _, exists := result[canonicalDomain]; exists { - continue + if err := s.viper.ReadInConfig(); err != nil { + log.WrapErr(err, "failed to reload config") + return } - route, err := parseLegacyRouteEntry(domainName, value) - if err != nil { - return nil, err + if err := s.Load(ctx); err != nil { + log.WrapErr(err, "failed to load updated config") + return } - result[canonicalDomain] = route - } - - return result, nil -} - -func parseCanonicalRouteEntry(domainName string, raw any) (routeConfig, error) { - canonicalDomain, ok := domain.CanonicalRouteDomain(domainName) - if !ok { - return routeConfig{}, fmt.Errorf("invalid route key %q: %w", domainName, domain.ErrRouteDomainInvalid) - } - domainName = canonicalDomain - switch value := raw.(type) { - case string: - if value == "" { - return routeConfig{}, fmt.Errorf("route %q has empty image", domainName) + if onChange != nil { + onChange() } - return routeConfig{Image: value, HTTPS: true}, nil - case map[string]any: - return parseRouteTable(domainName, value) - default: - return routeConfig{}, fmt.Errorf("route %q has unsupported value type %T", domainName, raw) - } -} - -func parseLegacyRouteEntry(domainName string, raw any) (routeConfig, error) { - canonicalDomain, ok := domain.CanonicalRouteDomain(domainName) - if !ok { - return routeConfig{}, fmt.Errorf("invalid route key %q: %w", legacyRouteStorageKey(domainName), domain.ErrRouteDomainInvalid) - } - domainName = canonicalDomain - - value, ok := raw.(string) - if !ok { - return routeConfig{}, fmt.Errorf("legacy route %q must be a string", legacyRouteStorageKey(domainName)) - } - if value == "" { - return routeConfig{}, fmt.Errorf("legacy route %q has empty image", legacyRouteStorageKey(domainName)) - } - - return routeConfig{Image: value, HTTPS: false}, nil -} - -func parseRouteTable(domainName string, raw map[string]any) (routeConfig, error) { - imageValue, ok := raw["image"].(string) - if !ok || imageValue == "" { - return routeConfig{}, fmt.Errorf("route %q has invalid image field", domainName) - } + }) - httpsValue, ok := raw["https"] - if !ok { - return routeConfig{Image: imageValue, HTTPS: true}, nil - } - https, ok := httpsValue.(bool) - if !ok { - return routeConfig{}, fmt.Errorf("route %q has invalid https field", domainName) - } + s.viper.WatchConfig() + log.Info().Msg("watching for configuration changes") - return routeConfig{Image: imageValue, HTTPS: https}, nil + return nil } -// GetRoutes returns all configured routes. -func (s *Service) GetRoutes(_ context.Context) []domain.Route { +// GetServerPort returns the configured server port. +func (s *Service) GetServerPort() int { s.mu.RLock() defer s.mu.RUnlock() - - var routes []domain.Route - for domainName, route := range s.config.Routes { - if strings.HasPrefix(domainName, "http://") { - continue - } - routes = append(routes, domain.Route{Domain: domainName, Image: route.Image, HTTPS: route.HTTPS}) - } - - return routes + return s.config.ServerPort } -// GetRoute returns a single route by domain. -func (s *Service) GetRoute(_ context.Context, domainName string) (*domain.Route, error) { +// GetRegistryPort returns the configured registry port. +func (s *Service) GetRegistryPort() int { s.mu.RLock() defer s.mu.RUnlock() - - domainName, ok := domain.CanonicalRouteDomain(domainName) - if !ok { - return nil, domain.ErrRouteNotFound - } - routeCfg, ok := s.config.Routes[domainName] - if !ok { - return nil, domain.ErrRouteNotFound - } - - routeValue := domain.Route{Domain: domainName, Image: routeCfg.Image, HTTPS: routeCfg.HTTPS} - route := &routeValue - - return route, nil + return s.config.RegistryPort } -// FindRoutesByImage returns all routes whose image matches the given image name. -// Image names are normalized by stripping the registry domain prefix before comparison. -// When the input has no tag, only the name portion is compared (i.e. "myapp" matches "myapp:latest"). -func (s *Service) FindRoutesByImage(_ context.Context, imageName string) []domain.Route { +// GetRegistryDomain returns the configured registry domain. +func (s *Service) GetRegistryDomain() string { s.mu.RLock() defer s.mu.RUnlock() - - var routes []domain.Route - for domainName, route := range s.config.Routes { - if strings.HasPrefix(domainName, "http://") { - continue - } - if matchesImageName(imageName, route.Image, s.config.RegistryDomain, s.config.LegacyRegistryDomains) { - routes = append(routes, domain.Route{Domain: domainName, Image: route.Image, HTTPS: route.HTTPS}) - } - } - - return routes + return s.config.RegistryDomain } -// FindAttachmentTargetsByImage returns all attachment targets whose image matches the given image name. -func (s *Service) FindAttachmentTargetsByImage(_ context.Context, imageName string) []string { +// GetLegacyRegistryDomains returns the configured legacy registry domains. +func (s *Service) GetLegacyRegistryDomains() []string { s.mu.RLock() defer s.mu.RUnlock() - - var targets []string - for target, images := range s.config.Attachments { - for _, image := range images { - if matchesImageName(imageName, image, s.config.RegistryDomain, s.config.LegacyRegistryDomains) { - targets = append(targets, target) - break - } - } - } - - return targets -} - -func matchesImageName(inputImage, candidateImage, registryDomain string, legacyRegistryDomains []string) bool { - normalizedInput := NormalizeRegistryImage(inputImage, registryDomain, legacyRegistryDomains) - inputName, inputHasTag := splitImageNameTag(normalizedInput) - normalizedCandidate := NormalizeRegistryImage(candidateImage, registryDomain, legacyRegistryDomains) - - if inputHasTag { - return strings.EqualFold(normalizedCandidate, normalizedInput) - } - - candidateName, _ := splitImageNameTag(normalizedCandidate) - return strings.EqualFold(candidateName, inputName) -} - -// splitImageNameTag splits "name:tag" into ("name", true) or ("name", false) when no tag is present. -func splitImageNameTag(image string) (name string, hasTag bool) { - if idx := strings.LastIndex(image, ":"); idx != -1 { - return image[:idx], true - } - return image, false -} - -// NormalizeRegistryImage strips the current or legacy Gordon registry domain -// prefix from an image name for comparison. -func NormalizeRegistryImage(imageName, registryDomain string, legacyRegistryDomains []string) string { - return domain.StripKnownGordonRegistry(imageName, registryDomain, legacyRegistryDomains) -} - -// NormalizeBootstrapImage converts a user-supplied image argument into -// the canonical "registry/name" format expected by push. -// Bare names get the registry domain prepended; tags are stripped. -func NormalizeBootstrapImage(image, registryDomain string) (string, error) { - if image == "" { - return "", fmt.Errorf("image is required") - } - if registryDomain == "" { - return "", fmt.Errorf("registry domain is not configured") - } - - registryDomain = strings.TrimSuffix(registryDomain, "/") - - if slashIdx := strings.LastIndex(image, "/"); slashIdx != -1 { - nameTag := image[slashIdx+1:] - if colonIdx := strings.LastIndex(nameTag, ":"); colonIdx != -1 { - image = image[:slashIdx+1] + nameTag[:colonIdx] - } - } else { - if colonIdx := strings.LastIndex(image, ":"); colonIdx != -1 { - image = image[:colonIdx] - } - } - - if !strings.Contains(image, "/") { - image = registryDomain + "/" + image - } - - return image, nil -} - -func legacyRouteStorageKey(domainName string) string { - return "http://" + domainName -} - -func cloneRouteConfigs(routes map[string]routeConfig) map[string]routeConfig { - if routes == nil { - return make(map[string]routeConfig) - } - - cloned := make(map[string]routeConfig, len(routes)) - for key, route := range routes { - cloned[key] = route - } - - return cloned + return append([]string{}, s.config.LegacyRegistryDomains...) } -// AddRoute adds a new route to the configuration and persists it. -func (s *Service) AddRoute(ctx context.Context, route domain.Route) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "AddRoute", - "domain": route.Domain, - }) - log := zerowrap.FromCtx(ctx) - - // Validate route - if route.Domain == "" { - return domain.ErrRouteDomainEmpty - } - canonicalDomain, ok := domain.CanonicalRouteDomain(route.Domain) - if !ok { - return domain.ErrRouteDomainInvalid - } - route.Domain = canonicalDomain - if route.Image == "" { - return domain.ErrRouteImageEmpty - } - - // Store previous value for rollback - s.mu.Lock() - currentConfig := s.config - if currentConfig.Routes == nil { - currentConfig.Routes = make(map[string]routeConfig) - } - if _, exists := currentConfig.ExternalRoutes[route.Domain]; exists { - s.mu.Unlock() - return fmt.Errorf("%w: route %q conflicts with external route", domain.ErrRouteConflict, route.Domain) - } - previousCanonicalRoute, canonicalExisted := currentConfig.Routes[route.Domain] - legacyKey := legacyRouteStorageKey(route.Domain) - _, legacyExisted := currentConfig.Routes[legacyKey] - newRoute := routeConfig{Image: route.Image, HTTPS: route.HTTPS} - if canonicalExisted && previousCanonicalRoute == newRoute && !legacyExisted { - s.mu.Unlock() - return nil - } - nextRoutes := cloneRouteConfigs(currentConfig.Routes) - nextRoutes[route.Domain] = newRoute - if legacyExisted { - delete(nextRoutes, legacyKey) - } - - nextConfig := currentConfig - nextConfig.Routes = nextRoutes - - if err := s.persistConfig(ctx, nextConfig); err != nil { - log.Warn().Err(err).Msg("failed to persist route to disk") - s.mu.Unlock() - return err - } - - s.config.Routes = nextRoutes - s.mu.Unlock() - - log.Info().Str("image", route.Image).Msg("route added to configuration") - return nil +// GetDataDir returns the configured data directory. +func (s *Service) GetDataDir() string { + s.mu.RLock() + defer s.mu.RUnlock() + return s.config.DataDir } -// UpdateRoute updates an existing route and persists it. -func (s *Service) UpdateRoute(ctx context.Context, route domain.Route) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "UpdateRoute", - "domain": route.Domain, - }) - log := zerowrap.FromCtx(ctx) - - // Validate route - if route.Domain == "" { - return domain.ErrRouteDomainEmpty - } - canonicalDomain, ok := domain.CanonicalRouteDomain(route.Domain) - if !ok { - return domain.ErrRouteDomainInvalid - } - route.Domain = canonicalDomain - if route.Image == "" { - return domain.ErrRouteImageEmpty - } - - // Store previous value for rollback - s.mu.Lock() - currentConfig := s.config - previousRoute, ok := currentConfig.Routes[route.Domain] - if !ok { - s.mu.Unlock() - return domain.ErrRouteNotFound - } - updatedRoute := previousRoute - legacyKey := legacyRouteStorageKey(route.Domain) - nextRoutes := cloneRouteConfigs(currentConfig.Routes) - delete(nextRoutes, legacyKey) - updatedRoute.Image = route.Image - updatedRoute.HTTPS = route.HTTPS - nextRoutes[route.Domain] = updatedRoute - - nextConfig := currentConfig - nextConfig.Routes = nextRoutes - - if err := s.persistConfig(ctx, nextConfig); err != nil { - log.Warn().Err(err).Msg("failed to persist route update to disk") - s.mu.Unlock() - return err - } - - s.config.Routes = nextRoutes - s.mu.Unlock() - - log.Info().Str("image", route.Image).Msg("route updated") - return nil +// IsNetworkIsolationEnabled returns whether network isolation is enabled. +func (s *Service) IsNetworkIsolationEnabled() bool { + s.mu.RLock() + defer s.mu.RUnlock() + return s.config.NetworkIsolation } -// RemoveRoute removes a route from the configuration and persists it. -func (s *Service) RemoveRoute(ctx context.Context, domainName string) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "RemoveRoute", - "domain": domainName, - }) - log := zerowrap.FromCtx(ctx) - - if domainName == "" { - return domain.ErrRouteDomainEmpty - } - canonicalDomain, ok := domain.CanonicalRouteDomain(domainName) - if !ok { - return domain.ErrRouteDomainInvalid - } - domainName = canonicalDomain - - // Store previous value for rollback - s.mu.Lock() - currentConfig := s.config - _, ok = currentConfig.Routes[domainName] - if !ok { - s.mu.Unlock() - return domain.ErrRouteNotFound - } - legacyKey := legacyRouteStorageKey(domainName) - nextRoutes := cloneRouteConfigs(currentConfig.Routes) - delete(nextRoutes, legacyKey) - delete(nextRoutes, domainName) - - nextConfig := currentConfig - nextConfig.Routes = nextRoutes - - if err := s.persistConfig(ctx, nextConfig); err != nil { - log.Warn().Err(err).Msg("failed to persist route removal to disk") - s.mu.Unlock() - return err - } - - s.config.Routes = nextRoutes - s.mu.Unlock() - - log.Info().Msg("route removed") - return nil +// GetNetworkPrefix returns the prefix for created networks. +func (s *Service) GetNetworkPrefix() string { + s.mu.RLock() + defer s.mu.RUnlock() + return s.config.NetworkPrefix } -// Save persists the current configuration to disk. -func (s *Service) Save(ctx context.Context) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "SaveConfig", - }) - - s.mu.Lock() - defer s.mu.Unlock() - - return s.persistConfig(ctx, s.config) -} - -func (s *Service) persistConfig(ctx context.Context, config Config) error { - log := zerowrap.FromCtx(ctx) - - configFile := s.viper.ConfigFileUsed() - if configFile == "" { - return log.WrapErr(fmt.Errorf("no config file path"), "cannot save config") - } - - if err := backupConfigFile(configFile); err != nil { - log.Warn().Err(err).Msg("failed to create config backup") - } - - snapshot := s.snapshotCriticalFields() - - if err := s.writeConfigSurgical(configFile, config); err != nil { - return log.WrapErr(err, "failed to write config") - } - - atomic.StoreInt64(&s.lastSaveTime, time.Now().UnixNano()) - - if err := s.viper.ReadInConfig(); err != nil { - return log.WrapErr(err, "failed to re-read config after save") - } - - if err := s.verifyCriticalFields(snapshot); err != nil { - log.Error().Err(err).Msg("config corruption detected after save, restoring backup") - if restoreErr := restoreConfigBackup(configFile); restoreErr != nil { - log.Error().Err(restoreErr).Msg("CRITICAL: failed to restore config backup") - } else if reloadErr := s.viper.ReadInConfig(); reloadErr != nil { - log.Error().Err(reloadErr).Msg("failed to reload config after restore") - } - return log.WrapErr(err, "config verification failed after save") - } - - log.Info().Msg("configuration saved to disk") - return nil -} - -type configSnapshot struct { - values map[string]any -} - -// mutableSections lists config sections that are intentionally modified by service operations. -var mutableSections = []string{"routes.", "external_routes.", "attachments.", "network_groups.", "auto_route_allowed_domains", "auto."} - -func isMutableKey(key string) bool { - for _, prefix := range mutableSections { - if strings.HasPrefix(key, prefix) { - return true - } - } - - switch key { - case "routes", "external_routes", "attachments", "network_groups", "auto_route_allowed_domains", "auto": - return true - } - - return false -} - -func (s *Service) snapshotCriticalFields() configSnapshot { - snap := configSnapshot{values: make(map[string]any)} - for _, key := range s.viper.AllKeys() { - if isMutableKey(key) { - continue - } - snap.values[key] = s.viper.Get(key) - } - return snap -} - -func (s *Service) verifyCriticalFields(snap configSnapshot) error { - for key, oldVal := range snap.values { - if oldVal == nil { - continue - } - newVal := s.viper.Get(key) - if !reflect.DeepEqual(oldVal, newVal) { - return fmt.Errorf("config key %q changed from %v to %v after save", key, oldVal, newVal) - } - } - - return nil -} - -func (s *Service) writeConfigSurgical(configFile string, snapshotConfig Config) error { - data, err := os.ReadFile(configFile) - var configMap map[string]any - if err != nil { - if os.IsNotExist(err) { - configMap = s.viper.AllSettings() - } else { - return fmt.Errorf("failed to read config file: %w", err) - } - } - - if configMap == nil { - configMap = make(map[string]any) - } - if len(data) > 0 { - if err := toml.Unmarshal(data, &configMap); err != nil { - return fmt.Errorf("failed to parse config file: %w", err) - } - } - - delete(configMap, "routes") - configMap["external_routes"] = snapshotConfig.ExternalRoutes - configMap["attachments"] = canonicalAttachmentsForSave(snapshotConfig.Attachments, snapshotConfig.RegistryDomain, snapshotConfig.LegacyRegistryDomains) - configMap["network_groups"] = snapshotConfig.NetworkGroups - applyAutoAllowedDomains(configMap, snapshotConfig.AutoRouteAllowedDomains) - - out, err := toml.Marshal(configMap) - if err != nil { - return fmt.Errorf("failed to marshal config: %w", err) - } - - routes, err := canonicalRoutesForSave(snapshotConfig.Routes, snapshotConfig.RegistryDomain, snapshotConfig.LegacyRegistryDomains) - if err != nil { - return err - } - routesOut := renderCanonicalRoutesSection(routes) - if routesOut != "" { - if len(out) > 0 { - out = append(out, '\n') - } - out = append(out, []byte(routesOut)...) - } - - configDir := filepath.Dir(configFile) - tmpFile, err := os.CreateTemp(configDir, ".gordon-config-*.tmp") - if err != nil { - return fmt.Errorf("failed to create temp config file: %w", err) - } - tmpPath := tmpFile.Name() - - if _, writeErr := tmpFile.Write(out); writeErr != nil { - _ = tmpFile.Close() - _ = os.Remove(tmpPath) - return fmt.Errorf("failed to write temp config: %w", writeErr) - } - if chmodErr := tmpFile.Chmod(0600); chmodErr != nil { - _ = tmpFile.Close() - _ = os.Remove(tmpPath) - return fmt.Errorf("failed to set temp config permissions: %w", chmodErr) - } - if closeErr := tmpFile.Close(); closeErr != nil { - _ = os.Remove(tmpPath) - return fmt.Errorf("failed to close temp config: %w", closeErr) - } - if renameErr := os.Rename(tmpPath, configFile); renameErr != nil { - _ = os.Remove(tmpPath) - return fmt.Errorf("failed to rename temp config: %w", renameErr) - } - - return nil -} - -func applyAutoAllowedDomains(config map[string]any, allowedDomains []string) { - // Write to both old and new keys for backward compat - config["auto_route_allowed_domains"] = allowedDomains - if autoSection, ok := config["auto"].(map[string]any); ok { - autoSection["allowed_domains"] = allowedDomains - return - } - - config["auto"] = map[string]any{ - "allowed_domains": allowedDomains, - } -} - -func canonicalAttachmentsForSave(attachments map[string][]string, registryDomain string, legacyRegistryDomains []string) map[string][]string { - if attachments == nil { - return nil - } - - result := make(map[string][]string, len(attachments)) - for target, images := range attachments { - if images == nil { - result[target] = nil - continue - } - - canonicalImages := make([]string, 0, len(images)) - seen := make(map[string]struct{}, len(images)) - for _, image := range images { - canonicalImage := domain.CanonicalizeGordonImageRef(image, registryDomain, legacyRegistryDomains) - if _, ok := seen[canonicalImage]; ok { - continue - } - seen[canonicalImage] = struct{}{} - canonicalImages = append(canonicalImages, canonicalImage) - } - result[target] = canonicalImages - } - - return result -} - -func canonicalRoutesForSave(routes map[string]routeConfig, registryDomain string, legacyRegistryDomains []string) (map[string]routeConfig, error) { - result := make(map[string]routeConfig) - if routes == nil { - return result, nil - } - - canonicalizeRoute := func(route routeConfig) routeConfig { - route.Image = domain.CanonicalizeGordonImageRef(route.Image, registryDomain, legacyRegistryDomains) - return route - } - - for key, route := range routes { - if strings.HasPrefix(key, "http://") { - continue - } - if !domain.IsValidRouteDomain(key) { - return nil, fmt.Errorf("invalid route key %q: %w", key, domain.ErrRouteDomainInvalid) - } - result[key] = canonicalizeRoute(route) - } - - for key, route := range routes { - if !strings.HasPrefix(key, "http://") { - continue - } - - domainName := strings.TrimPrefix(key, "http://") - if !domain.IsValidRouteDomain(domainName) { - return nil, fmt.Errorf("invalid route key %q: %w", key, domain.ErrRouteDomainInvalid) - } - if _, exists := result[domainName]; exists { - continue - } - route.HTTPS = false - result[domainName] = canonicalizeRoute(route) - } - - return result, nil -} - -func renderCanonicalRoutesSection(routes map[string]routeConfig) string { - if len(routes) == 0 { - return "" - } - - keys := make([]string, 0, len(routes)) - for key := range routes { - keys = append(keys, key) - } - slices.Sort(keys) - - var b strings.Builder - b.WriteString("[routes]\n") - for _, key := range keys { - route := routes[key] - b.WriteString(strconv.Quote(key)) - b.WriteString(" = { image = ") - b.WriteString(strconv.Quote(route.Image)) - b.WriteString(", https = ") - if route.HTTPS { - b.WriteString("true") - } else { - b.WriteString("false") - } - b.WriteString(" }\n") - } - - return b.String() -} - -func backupConfigFile(configFile string) error { - src, err := os.ReadFile(configFile) - if err != nil { - if os.IsNotExist(err) { - return nil - } - return fmt.Errorf("failed to read config for backup: %w", err) - } - - backupPath := fmt.Sprintf("%s.bak.%d", configFile, time.Now().UnixNano()) - if err := os.WriteFile(backupPath, src, 0600); err != nil { - return fmt.Errorf("failed to write backup config: %w", err) - } - if err := cleanupOldBackups(configFile, 5); err != nil { - return fmt.Errorf("failed to clean up old backups: %w", err) - } - - return nil -} - -// cleanupOldBackups removes old backup files, keeping only the most recent `keep` backups. -// It only removes files matching the pattern `.bak.`. -func cleanupOldBackups(configFile string, keep int) error { - dir := filepath.Dir(configFile) - base := filepath.Base(configFile) - entries, err := os.ReadDir(dir) - if err != nil { - return fmt.Errorf("failed to read backup directory: %w", err) - } - - prefix := base + ".bak." - type tsBackup struct { - ts int64 - name string - } - var backups []tsBackup - for _, entry := range entries { - if entry.IsDir() { - continue - } - name := entry.Name() - if !strings.HasPrefix(name, prefix) { - continue - } - suffix := strings.TrimPrefix(name, prefix) - ts, err := strconv.ParseInt(suffix, 10, 64) - if err != nil { - continue - } - backups = append(backups, tsBackup{ts: ts, name: name}) - } - - slices.SortFunc(backups, func(a, b tsBackup) int { - return cmp.Compare(b.ts, a.ts) - }) - if len(backups) <= keep { - return nil - } - - var firstErr error - for _, backup := range backups[keep:] { - backupPath := filepath.Join(dir, backup.name) - if err := os.Remove(backupPath); err != nil && firstErr == nil { - firstErr = fmt.Errorf("failed to remove old backup %s: %w", backupPath, err) - } - } - - return firstErr -} - -func restoreConfigBackup(configFile string) error { - dir := filepath.Dir(configFile) - base := filepath.Base(configFile) - entries, err := os.ReadDir(dir) - if err != nil { - return fmt.Errorf("failed to read backup directory: %w", err) - } - - prefix := base + ".bak." - type tsBackup struct { - ts int64 - name string - } - var backups []tsBackup - for _, entry := range entries { - if entry.IsDir() { - continue - } - name := entry.Name() - if !strings.HasPrefix(name, prefix) { - continue - } - suffix := strings.TrimPrefix(name, prefix) - ts, err := strconv.ParseInt(suffix, 10, 64) - if err != nil { - continue - } - backups = append(backups, tsBackup{ts: ts, name: name}) - } - - slices.SortFunc(backups, func(a, b tsBackup) int { - return cmp.Compare(b.ts, a.ts) - }) - - backupPath := configFile + ".bak" - if len(backups) > 0 { - backupPath = filepath.Join(dir, backups[0].name) - } - - src, err := os.ReadFile(backupPath) - if err != nil { - return fmt.Errorf("failed to read backup: %w", err) - } - - if err := os.WriteFile(configFile, src, 0600); err != nil { - return fmt.Errorf("failed to restore backup: %w", err) - } - - return nil -} - -// Watch starts watching for configuration changes. -func (s *Service) Watch(ctx context.Context, onChange func()) error { - log := zerowrap.FromCtx(ctx) - - s.viper.OnConfigChange(func(e fsnotify.Event) { - // Check if this event is within the debounce window of our own Save - lastSave := atomic.LoadInt64(&s.lastSaveTime) - if lastSave > 0 && time.Now().UnixNano()-lastSave < s.debounceDelay { - log.Debug().Str("file", e.Name).Msg("skipping config reload (triggered by save)") - return - } - - log.Info().Str("file", e.Name).Msg("config file changed") - - if err := s.viper.ReadInConfig(); err != nil { - log.WrapErr(err, "failed to reload config") - return - } - - if err := s.Load(ctx); err != nil { - log.WrapErr(err, "failed to load updated config") - return - } - - if onChange != nil { - onChange() - } - }) - - s.viper.WatchConfig() - log.Info().Msg("watching for configuration changes") - - return nil -} - -// GetServerPort returns the configured server port. -func (s *Service) GetServerPort() int { - s.mu.RLock() - defer s.mu.RUnlock() - return s.config.ServerPort -} - -// GetRegistryPort returns the configured registry port. -func (s *Service) GetRegistryPort() int { - s.mu.RLock() - defer s.mu.RUnlock() - return s.config.RegistryPort -} - -// GetRegistryDomain returns the configured registry domain. -func (s *Service) GetRegistryDomain() string { - s.mu.RLock() - defer s.mu.RUnlock() - return s.config.RegistryDomain -} - -// GetLegacyRegistryDomains returns the configured legacy registry domains. -func (s *Service) GetLegacyRegistryDomains() []string { - s.mu.RLock() - defer s.mu.RUnlock() - return append([]string{}, s.config.LegacyRegistryDomains...) -} - -// GetDataDir returns the configured data directory. -func (s *Service) GetDataDir() string { - s.mu.RLock() - defer s.mu.RUnlock() - return s.config.DataDir -} - -// IsAutoRouteEnabled returns whether auto-route is enabled. -func (s *Service) IsAutoRouteEnabled() bool { - s.mu.RLock() - defer s.mu.RUnlock() - return s.config.AutoRouteEnabled -} - -// IsNetworkIsolationEnabled returns whether network isolation is enabled. -func (s *Service) IsNetworkIsolationEnabled() bool { - s.mu.RLock() - defer s.mu.RUnlock() - return s.config.NetworkIsolation -} - -// GetNetworkPrefix returns the prefix for created networks. -func (s *Service) GetNetworkPrefix() string { - s.mu.RLock() - defer s.mu.RUnlock() - return s.config.NetworkPrefix -} - -// GetConfig returns a copy of the current configuration. -func (s *Service) GetConfig() Config { - s.mu.RLock() - defer s.mu.RUnlock() - return s.config +// GetConfig returns a copy of the current configuration. +func (s *Service) GetConfig() Config { + s.mu.RLock() + defer s.mu.RUnlock() + return s.config } // GetVolumeConfig returns volume configuration. @@ -1198,356 +275,6 @@ func (s *Service) GetVolumeConfig() (autoCreate bool, prefix string, preserve bo return s.config.VolumeAutoCreate, s.config.VolumePrefix, s.config.VolumePreserve } -// GetNetworkGroups returns network group configuration. -func (s *Service) GetNetworkGroups() map[string][]string { - s.mu.RLock() - defer s.mu.RUnlock() - - result := make(map[string][]string) - for k, v := range s.config.NetworkGroups { - result[k] = append([]string{}, v...) - } - return result -} - -// GetAttachmentConfig returns a consistent snapshot of attachments and network -// groups under a single lock, preventing cross-field races during config reloads. -func (s *Service) GetAttachmentConfig() out.AttachmentConfigSnapshot { - s.mu.RLock() - defer s.mu.RUnlock() - - attachments := make(map[string][]string, len(s.config.Attachments)) - for k, v := range s.config.Attachments { - attachments[k] = append([]string{}, v...) - } - - networkGroups := make(map[string][]string, len(s.config.NetworkGroups)) - for k, v := range s.config.NetworkGroups { - networkGroups[k] = append([]string{}, v...) - } - - return out.AttachmentConfigSnapshot{ - Attachments: attachments, - NetworkGroups: networkGroups, - } -} - -// GetAllAttachments returns all configured attachments. -func (s *Service) GetAllAttachments(_ context.Context) map[string][]string { - s.mu.RLock() - defer s.mu.RUnlock() - - result := make(map[string][]string) - for k, v := range s.config.Attachments { - result[k] = append([]string{}, v...) - } - return result -} - -// GetAttachmentsFor returns attachments for a specific domain or network group. -func (s *Service) GetAttachmentsFor(_ context.Context, domainOrGroup string) ([]string, error) { - s.mu.RLock() - defer s.mu.RUnlock() - - images, exists := s.config.Attachments[domainOrGroup] - if !exists { - return nil, domain.ErrAttachmentNotFound - } - - return append([]string{}, images...), nil -} - -// AddAttachment adds an image to a domain/group's attachments. -func (s *Service) AddAttachment(ctx context.Context, domainOrGroup, image string) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "AddAttachment", - "target": domainOrGroup, - "image": image, - }) - log := zerowrap.FromCtx(ctx) - - // Validate input - if domainOrGroup == "" { - return domain.ErrAttachmentTargetEmpty - } - if image == "" { - return domain.ErrAttachmentImageEmpty - } - - s.mu.Lock() - - // Initialize map if needed - if s.config.Attachments == nil { - s.config.Attachments = make(map[string][]string) - } - - // Check if already exists - existing := s.config.Attachments[domainOrGroup] - canonicalImage := domain.CanonicalizeGordonImageRef(image, s.config.RegistryDomain, s.config.LegacyRegistryDomains) - for _, img := range existing { - if img == image || domain.CanonicalizeGordonImageRef(img, s.config.RegistryDomain, s.config.LegacyRegistryDomains) == canonicalImage { - s.mu.Unlock() - return domain.ErrAttachmentExists - } - } - - // Store previous value for rollback - previousImages := append([]string{}, existing...) - hadKey := len(existing) > 0 - - // Add the image - s.config.Attachments[domainOrGroup] = append(existing, image) - s.mu.Unlock() - - // Persist to disk - rollback on failure - if err := s.Save(ctx); err != nil { - log.Warn().Err(err).Msg("failed to persist attachment to disk, rolling back") - s.mu.Lock() - if hadKey { - s.config.Attachments[domainOrGroup] = previousImages - } else { - delete(s.config.Attachments, domainOrGroup) - } - s.mu.Unlock() - return err - } - - log.Info().Msg("attachment added to configuration") - return nil -} - -// RemoveAttachment removes an image from a domain/group's attachments. -func (s *Service) RemoveAttachment(ctx context.Context, domainOrGroup, image string) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "RemoveAttachment", - "target": domainOrGroup, - "image": image, - }) - log := zerowrap.FromCtx(ctx) - - // Validate input - if domainOrGroup == "" { - return domain.ErrAttachmentTargetEmpty - } - if image == "" { - return domain.ErrAttachmentImageEmpty - } - - s.mu.Lock() - - existing, exists := s.config.Attachments[domainOrGroup] - if !exists { - s.mu.Unlock() - return domain.ErrAttachmentNotFound - } - - // Find and remove the image - found := false - newImages := make([]string, 0, len(existing)) - for _, img := range existing { - if img == image { - found = true - } else { - newImages = append(newImages, img) - } - } - - if !found { - s.mu.Unlock() - return domain.ErrAttachmentNotFound - } - - // Store previous value for rollback - previousImages := existing - - // Update or remove the key - if len(newImages) == 0 { - delete(s.config.Attachments, domainOrGroup) - } else { - s.config.Attachments[domainOrGroup] = newImages - } - s.mu.Unlock() - - // Persist to disk - rollback on failure - if err := s.Save(ctx); err != nil { - log.Warn().Err(err).Msg("failed to persist attachment removal to disk, rolling back") - s.mu.Lock() - s.config.Attachments[domainOrGroup] = previousImages - s.mu.Unlock() - return err - } - - log.Info().Msg("attachment removed from configuration") - return nil -} - -// GetPreviewConfig returns the preview environment configuration. -func (s *Service) GetPreviewConfig() domain.PreviewConfig { - s.mu.RLock() - defer s.mu.RUnlock() - return domain.PreviewConfig{ - Enabled: s.config.PreviewEnabled, - TTL: s.config.PreviewTTL, - Separator: s.config.PreviewSeparator, - TagPatterns: s.config.PreviewTagPatterns, - DataCopy: s.config.PreviewDataCopy, - EnvCopy: s.config.PreviewEnvCopy, - } -} - -// IsPreviewEnabled returns whether preview environments are enabled. -func (s *Service) IsPreviewEnabled() bool { - s.mu.RLock() - defer s.mu.RUnlock() - return s.config.PreviewEnabled -} - -// IsAutoEnabled returns whether the auto feature is enabled (alias for IsAutoRouteEnabled). -func (s *Service) IsAutoEnabled() bool { - return s.IsAutoRouteEnabled() -} - -// GetPreviewTagPatterns returns the configured tag patterns for preview environments. -func (s *Service) GetPreviewTagPatterns() []string { - s.mu.RLock() - defer s.mu.RUnlock() - return append([]string{}, s.config.PreviewTagPatterns...) -} - -// GetAllowedDomains returns the auto-route allowed domain patterns. -func (s *Service) GetAllowedDomains() []string { - s.mu.RLock() - defer s.mu.RUnlock() - return append([]string{}, s.config.AutoRouteAllowedDomains...) -} - -// GetAutoRouteAllowedDomains returns the configured auto-route domain allowlist. -func (s *Service) GetAutoRouteAllowedDomains(_ context.Context) ([]string, error) { - s.mu.RLock() - defer s.mu.RUnlock() - - return append([]string{}, s.config.AutoRouteAllowedDomains...), nil -} - -// AddAutoRouteAllowedDomain adds a domain pattern to the auto-route allowlist. -func (s *Service) AddAutoRouteAllowedDomain(ctx context.Context, pattern string) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "AddAutoRouteAllowedDomain", - "pattern": pattern, - }) - log := zerowrap.FromCtx(ctx) - - if err := validateDomainPattern(pattern); err != nil { - return err - } - - s.mu.Lock() - for _, existing := range s.config.AutoRouteAllowedDomains { - if existing == pattern { - s.mu.Unlock() - return nil - } - } - snapshot := make([]string, len(s.config.AutoRouteAllowedDomains)) - copy(snapshot, s.config.AutoRouteAllowedDomains) - s.config.AutoRouteAllowedDomains = append(s.config.AutoRouteAllowedDomains, pattern) - s.mu.Unlock() - - if err := s.Save(ctx); err != nil { - log.Warn().Err(err).Msg("failed to persist allowed domain, rolling back") - s.mu.Lock() - s.config.AutoRouteAllowedDomains = snapshot - s.mu.Unlock() - return err - } - - log.Info().Msg("auto-route allowed domain added") - return nil -} - -// RemoveAutoRouteAllowedDomain removes a domain pattern from the auto-route allowlist. -func (s *Service) RemoveAutoRouteAllowedDomain(ctx context.Context, pattern string) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "RemoveAutoRouteAllowedDomain", - "pattern": pattern, - }) - log := zerowrap.FromCtx(ctx) - - s.mu.Lock() - previous := append([]string{}, s.config.AutoRouteAllowedDomains...) - filtered := make([]string, 0, len(s.config.AutoRouteAllowedDomains)) - removed := false - for _, existing := range s.config.AutoRouteAllowedDomains { - if existing == pattern { - removed = true - continue - } - filtered = append(filtered, existing) - } - s.config.AutoRouteAllowedDomains = filtered - s.mu.Unlock() - - if !removed { - log.Debug().Msg("pattern not found in allowlist") - return nil - } - - if err := s.Save(ctx); err != nil { - log.Warn().Err(err).Msg("failed to persist allowed domain removal, rolling back") - s.mu.Lock() - s.config.AutoRouteAllowedDomains = previous - s.mu.Unlock() - return err - } - - log.Info().Msg("auto-route allowed domain removed") - return nil -} - -func validateDomainPattern(pattern string) error { - if pattern == "" { - return fmt.Errorf("%w: pattern is required", domain.ErrInvalidDomainPattern) - } - if pattern == "*" { - return nil - } - if pattern != strings.ToLower(pattern) { - return fmt.Errorf("%w: must be lowercase", domain.ErrInvalidDomainPattern) - } - if strings.HasSuffix(pattern, ".") { - return fmt.Errorf("%w: must not have trailing dots", domain.ErrInvalidDomainPattern) - } - if strings.Contains(pattern, "**") { - return fmt.Errorf("%w: must not contain double wildcards", domain.ErrInvalidDomainPattern) - } - if strings.Count(pattern, "*") > 1 { - return fmt.Errorf("%w: must not contain multiple wildcards", domain.ErrInvalidDomainPattern) - } - if strings.Contains(pattern, "*") { - if !strings.HasPrefix(pattern, "*.") { - return fmt.Errorf("%w: wildcard must be in *.domain form", domain.ErrInvalidDomainPattern) - } - if !isValidExactDomain(strings.TrimPrefix(pattern, "*.")) { - return fmt.Errorf("%w: invalid domain after wildcard", domain.ErrInvalidDomainPattern) - } - return nil - } - if !isValidExactDomain(pattern) { - return fmt.Errorf("%w: invalid domain format", domain.ErrInvalidDomainPattern) - } - return nil -} - -var exactDomainPattern = regexp.MustCompile(`^[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?(?:\.[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?)+$`) - -func isValidExactDomain(pattern string) bool { - return exactDomainPattern.MatchString(pattern) -} - // GetExternalRoutes returns all configured external routes. func (s *Service) GetExternalRoutes() map[string]string { s.mu.RLock() diff --git a/internal/usecase/config/service_test.go b/internal/usecase/config/service_test.go index d6fff0362..78adcc679 100644 --- a/internal/usecase/config/service_test.go +++ b/internal/usecase/config/service_test.go @@ -2,10 +2,8 @@ package config import ( "context" - "fmt" "os" "path/filepath" - "strings" "testing" "github.com/bnema/zerowrap" @@ -14,51 +12,12 @@ import ( "github.com/stretchr/testify/require" "github.com/bnema/gordon/internal/boundaries/out/mocks" - "github.com/bnema/gordon/internal/domain" ) func testContext() context.Context { return zerowrap.WithCtx(context.Background(), zerowrap.Default()) } -func requireRoute(t *testing.T, routes map[string]routeConfig, domainName string) routeConfig { - t.Helper() - route, ok := routes[domainName] - require.True(t, ok, "route %q not found", domainName) - return route -} - -func setupServiceWithCoexistingLegacyRoute(t *testing.T, canonicalImage, legacyImage string) (*Service, context.Context) { - t.Helper() - - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configFile, []byte(fmt.Sprintf("[routes]\n\"app.example.com\" = %q\n\"http://app.example.com\" = %q\n", canonicalImage, legacyImage)), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - v.Set("routes", map[string]any{ - "app.example.com": canonicalImage, - "http://app.example.com": legacyImage, - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - require.NoError(t, svc.Load(ctx)) - - svc.mu.Lock() - if svc.config.Routes == nil { - svc.config.Routes = make(map[string]routeConfig) - } - svc.config.Routes["app.example.com"] = routeConfig{Image: canonicalImage, HTTPS: true} - svc.config.Routes["http://app.example.com"] = routeConfig{Image: legacyImage, HTTPS: false} - svc.mu.Unlock() - - return svc, ctx -} - func TestService_Load(t *testing.T) { v := viper.New() v.Set("server.port", 8080) @@ -73,2331 +32,51 @@ func TestService_Load(t *testing.T) { v.Set("auth.password", "secret") v.Set("volumes.auto_create", true) v.Set("volumes.prefix", "gordon") - v.Set("volumes.preserve", false) - v.Set("routes", map[string]interface{}{ - "app.example.com": "myapp:latest", - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - err := svc.Load(ctx) - - assert.NoError(t, err) - assert.Equal(t, 8080, svc.GetServerPort()) - assert.Equal(t, 5000, svc.GetRegistryPort()) - assert.Equal(t, "registry.example.com", svc.GetRegistryDomain()) - assert.Equal(t, "/var/gordon", svc.GetDataDir()) - assert.True(t, svc.IsAutoRouteEnabled()) - assert.True(t, svc.IsNetworkIsolationEnabled()) - assert.Equal(t, "gordon", svc.GetNetworkPrefix()) - - autoCreate, prefix, preserve := svc.GetVolumeConfig() - assert.True(t, autoCreate) - assert.Equal(t, "gordon", prefix) - assert.False(t, preserve) -} - -func TestService_Load_CanonicalInlineRoute(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configFile, []byte(`[routes] -"insecure.example.com" = { image = "myapp:latest", https = false } -`), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - require.NoError(t, v.ReadInConfig()) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - err = svc.Load(ctx) - require.NoError(t, err) - - route, err := svc.GetRoute(ctx, "insecure.example.com") - require.NoError(t, err) - assert.Equal(t, "myapp:latest", route.Image) - assert.False(t, route.HTTPS) -} - -func TestService_Load_CanonicalInlineRouteDefaultsHTTPS(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configFile, []byte(`[routes] -"secure.example.com" = { image = "myapp:latest" } -`), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - require.NoError(t, v.ReadInConfig()) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - err = svc.Load(ctx) - require.NoError(t, err) - - route, err := svc.GetRoute(ctx, "secure.example.com") - require.NoError(t, err) - assert.Equal(t, "myapp:latest", route.Image) - assert.True(t, route.HTTPS) -} - -func TestService_Reload_InvalidRouteKeyPreservesPreviousState(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - initialConfig := `[routes] -"app.example.com" = "myapp:v1" -` - err := os.WriteFile(configFile, []byte(initialConfig), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - require.NoError(t, v.ReadInConfig()) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - err = svc.Load(ctx) - require.NoError(t, err) - - updatedConfig := `[routes] -"localhost" = "myapp:v2" -` - err = os.WriteFile(configFile, []byte(updatedConfig), 0600) - require.NoError(t, err) - - err = svc.Reload(ctx) - require.Error(t, err) - - route, getErr := svc.GetRoute(ctx, "app.example.com") - require.NoError(t, getErr) - assert.Equal(t, "myapp:v1", route.Image) - assert.True(t, route.HTTPS) -} - -func TestService_GetRoutes_DeduplicatesCanonicalAndLegacyEntries(t *testing.T) { - svc, ctx := setupServiceWithCoexistingLegacyRoute(t, "canonical:v2", "legacy:v1") - - routes := svc.GetRoutes(ctx) - - require.Len(t, routes, 1) - assert.Equal(t, "app.example.com", routes[0].Domain) - assert.Equal(t, "canonical:v2", routes[0].Image) - assert.True(t, routes[0].HTTPS) -} - -func TestService_FindRoutesByImage_DeduplicatesCanonicalAndLegacyEntries(t *testing.T) { - v := viper.New() - v.Set("server.registry_domain", "registry.example.com") - v.Set("routes", map[string]any{}) - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - svc.mu.Lock() - svc.config.Routes = make(map[string]routeConfig) - svc.config.Routes["app.example.com"] = routeConfig{Image: "registry.example.com/myapp:v2", HTTPS: true} - svc.config.Routes["http://app.example.com"] = routeConfig{Image: "myapp:v1", HTTPS: false} - svc.mu.Unlock() - - routes := svc.FindRoutesByImage(ctx, "myapp") - - require.Len(t, routes, 1) - assert.Equal(t, "app.example.com", routes[0].Domain) - assert.Equal(t, "registry.example.com/myapp:v2", routes[0].Image) - assert.True(t, routes[0].HTTPS) -} - -func TestCanonicalRoutesForSave(t *testing.T) { - routes := map[string]routeConfig{ - "legacy.example.com": {Image: "old-registry.example.com/app:latest", HTTPS: true}, - "current.example.com": {Image: "registry.example.com/current:v1", HTTPS: true}, - "bare.example.com": {Image: "worker:v2", HTTPS: true}, - "external.example.com": {Image: "docker.io/library/nginx:latest", HTTPS: true}, - "http://legacy-http.example.com": {Image: "old-registry.example.com:5000/http:v3", HTTPS: true}, - } - - savedRoutes, err := canonicalRoutesForSave(routes, "registry.example.com", []string{"old-registry.example.com", "old-registry.example.com:5000"}) - require.NoError(t, err) - - assert.Equal(t, routeConfig{Image: "registry.example.com/app:latest", HTTPS: true}, requireRoute(t, savedRoutes, "legacy.example.com")) - assert.Equal(t, routeConfig{Image: "registry.example.com/current:v1", HTTPS: true}, requireRoute(t, savedRoutes, "current.example.com")) - assert.Equal(t, routeConfig{Image: "worker:v2", HTTPS: true}, requireRoute(t, savedRoutes, "bare.example.com")) - assert.Equal(t, routeConfig{Image: "docker.io/library/nginx:latest", HTTPS: true}, requireRoute(t, savedRoutes, "external.example.com")) - assert.Equal(t, routeConfig{Image: "registry.example.com/http:v3", HTTPS: false}, requireRoute(t, savedRoutes, "legacy-http.example.com")) - - assert.Equal(t, "old-registry.example.com/app:latest", routes["legacy.example.com"].Image) - assert.Equal(t, "old-registry.example.com:5000/http:v3", routes["http://legacy-http.example.com"].Image) -} - -func TestService_AddRoute_StoresHTTPSFalseInCanonicalForm(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configFile, []byte("[routes]\n"), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - err = svc.AddRoute(ctx, domain.Route{Domain: "new.example.com", Image: "newapp:latest", HTTPS: false}) - require.NoError(t, err) - - routeCfg := requireRoute(t, svc.GetConfig().Routes, "new.example.com") - assert.Equal(t, "newapp:latest", routeCfg.Image) - assert.False(t, routeCfg.HTTPS) -} - -func TestService_AddRoute_RewritesRoutesToCanonicalInlineTables(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - initialConfig := `[server] -gordon_domain = "registry.example.com" -legacy_registry_domains = ["old-registry.example.com"] - -[routes] -"secure.example.com" = "old-registry.example.com/secure:v1" -"http://insecure.example.com" = "old-registry.example.com/insecure:v1" -` - err := os.WriteFile(configFile, []byte(initialConfig), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - require.NoError(t, v.ReadInConfig()) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - err = svc.Load(ctx) - require.NoError(t, err) - - err = svc.AddRoute(ctx, domain.Route{Domain: "new.example.com", Image: "new:v1"}) - require.NoError(t, err) - - content, err := os.ReadFile(configFile) - require.NoError(t, err) - text := string(content) - assert.Contains(t, text, `"secure.example.com" = { image = "registry.example.com/secure:v1", https = true }`) - assert.Contains(t, text, `"insecure.example.com" = { image = "registry.example.com/insecure:v1", https = false }`) - assert.Contains(t, text, `"new.example.com" = { image = "new:v1", https = false }`) - assert.NotContains(t, text, "http://insecure.example.com") - assert.Contains(t, text, "[routes]") -} - -func TestService_SaveCanonicalizesRouteImages(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - initialConfig := `[server] -gordon_domain = "registry.example.com" -legacy_registry_domains = ["old-registry.example.com", "old-registry.example.com:5000"] - -[routes] -"legacy.example.com" = "old-registry.example.com/app:latest" -"legacy-port.example.com" = "old-registry.example.com:5000/api:v2" -"current.example.com" = "registry.example.com/web:v3" -"bare.example.com" = "worker:v4" -"external.example.com" = "docker.io/library/nginx:latest" -` - err := os.WriteFile(configFile, []byte(initialConfig), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - require.NoError(t, v.ReadInConfig()) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - require.NoError(t, svc.Load(ctx)) - assert.Equal(t, "old-registry.example.com/app:latest", requireRoute(t, svc.GetConfig().Routes, "legacy.example.com").Image) - - require.NoError(t, svc.Save(ctx)) - - content, err := os.ReadFile(configFile) - require.NoError(t, err) - text := string(content) - assert.Contains(t, text, `"legacy.example.com" = { image = "registry.example.com/app:latest", https = true }`) - assert.Contains(t, text, `"legacy-port.example.com" = { image = "registry.example.com/api:v2", https = true }`) - assert.Contains(t, text, `"current.example.com" = { image = "registry.example.com/web:v3", https = true }`) - assert.Contains(t, text, `"bare.example.com" = { image = "worker:v4", https = true }`) - assert.Contains(t, text, `"external.example.com" = { image = "docker.io/library/nginx:latest", https = true }`) -} - -func TestCanonicalAttachmentsForSave(t *testing.T) { - attachments := map[string][]string{ - "app.example.com": { - "old-registry.example.com/postgres:18", - "registry.example.com/postgres:18", - "rabbitmq:3", - "docker.io/library/nginx:latest", - }, - } - - saved := canonicalAttachmentsForSave(attachments, "registry.example.com", []string{"old-registry.example.com"}) - - assert.Equal(t, []string{ - "registry.example.com/postgres:18", - "rabbitmq:3", - "docker.io/library/nginx:latest", - }, saved["app.example.com"]) - assert.Equal(t, []string{ - "old-registry.example.com/postgres:18", - "registry.example.com/postgres:18", - "rabbitmq:3", - "docker.io/library/nginx:latest", - }, attachments["app.example.com"]) -} - -func TestService_SaveCanonicalizesAttachmentImages(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - initialConfig := `[server] - gordon_domain = "registry.example.com" - legacy_registry_domains = ["old-registry.example.com"] - - [attachments] - "app.example.com" = [ - "old-registry.example.com/postgres:18", - "registry.example.com/postgres:18", - "registry.example.com/redis:7", - "rabbitmq:3", - "docker.io/library/nginx:latest", - ] - ` - err := os.WriteFile(configFile, []byte(initialConfig), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - require.NoError(t, v.ReadInConfig()) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - require.NoError(t, svc.Load(ctx)) - require.Equal(t, []string{ - "old-registry.example.com/postgres:18", - "registry.example.com/postgres:18", - "registry.example.com/redis:7", - "rabbitmq:3", - "docker.io/library/nginx:latest", - }, svc.GetConfig().Attachments["app.example.com"], "load path should preserve legacy attachment refs") - - require.NoError(t, svc.Save(ctx)) - require.Equal(t, []string{ - "old-registry.example.com/postgres:18", - "registry.example.com/postgres:18", - "registry.example.com/redis:7", - "rabbitmq:3", - "docker.io/library/nginx:latest", - }, svc.GetConfig().Attachments["app.example.com"], "save path should not rewrite in-memory attachment refs") - - saved := viper.New() - saved.SetConfigFile(configFile) - require.NoError(t, saved.ReadInConfig()) - - var attachments map[string][]string - require.NoError(t, saved.UnmarshalKey("attachments", &attachments)) - assert.Equal(t, []string{ - "registry.example.com/postgres:18", - "registry.example.com/redis:7", - "rabbitmq:3", - "docker.io/library/nginx:latest", - }, attachments["app.example.com"]) - - content, err := os.ReadFile(configFile) - require.NoError(t, err) - text := string(content) - assert.Contains(t, text, `registry.example.com/postgres:18`) - assert.Equal(t, 1, strings.Count(text, `registry.example.com/postgres:18`)) - assert.Contains(t, text, `registry.example.com/redis:7`) - assert.Contains(t, text, `rabbitmq:3`) - assert.Contains(t, text, `docker.io/library/nginx:latest`) - assert.NotContains(t, text, `old-registry.example.com/postgres:18`) -} - -func TestService_Reload(t *testing.T) { - t.Run("success - picks up config file changes", func(t *testing.T) { - // Create temp config file - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - initialConfig := `[server] -port = 8080 - -[routes] -"app.example.com" = "myapp:v1" -` - err := os.WriteFile(configFile, []byte(initialConfig), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - err = v.ReadInConfig() - require.NoError(t, err) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - // Initial load - err = svc.Load(ctx) - require.NoError(t, err) - assert.Equal(t, 8080, svc.GetServerPort()) - route := requireRoute(t, svc.GetConfig().Routes, "app.example.com") - assert.Equal(t, "myapp:v1", route.Image) - assert.True(t, route.HTTPS) - - // Modify config file on disk - updatedConfig := `[server] -port = 9090 - -[routes] -"app.example.com" = "myapp:v2" -"new.example.com" = "newapp:latest" -` - err = os.WriteFile(configFile, []byte(updatedConfig), 0600) - require.NoError(t, err) - - // Reload should pick up new values - err = svc.Reload(ctx) - require.NoError(t, err) - - assert.Equal(t, 9090, svc.GetServerPort()) - route = requireRoute(t, svc.GetConfig().Routes, "app.example.com") - assert.Equal(t, "myapp:v2", route.Image) - assert.True(t, route.HTTPS) - route = requireRoute(t, svc.GetConfig().Routes, "new.example.com") - assert.Equal(t, "newapp:latest", route.Image) - assert.True(t, route.HTTPS) - }) - - t.Run("error - config file not found", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - - // Create config, load it, then delete it - err := os.WriteFile(configFile, []byte("[server]\nport = 8080\n"), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - err = v.ReadInConfig() - require.NoError(t, err) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - err = svc.Load(ctx) - require.NoError(t, err) - - // Delete the config file - err = os.Remove(configFile) - require.NoError(t, err) - - // Reload should fail - err = svc.Reload(ctx) - assert.Error(t, err) - assert.Contains(t, err.Error(), "failed to read config file") - }) - - t.Run("error - invalid config syntax", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - - // Create valid config initially - err := os.WriteFile(configFile, []byte("[server]\nport = 8080\n"), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - err = v.ReadInConfig() - require.NoError(t, err) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - err = svc.Load(ctx) - require.NoError(t, err) - - // Write invalid TOML - err = os.WriteFile(configFile, []byte("[server\nport = invalid syntax"), 0600) - require.NoError(t, err) - - // Reload should fail - err = svc.Reload(ctx) - assert.Error(t, err) - }) -} - -func TestService_GetRoutes(t *testing.T) { - v := viper.New() - v.Set("routes", map[string]interface{}{ - "app1.example.com": "myapp:latest", - "app2.example.com": "otherapp:v2", - "http://insecure.example": "insecureapp:latest", - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - routes := svc.GetRoutes(ctx) - - assert.Len(t, routes, 3) - - // Check routes (order is not guaranteed) - routeMap := make(map[string]domain.Route) - for _, r := range routes { - routeMap[r.Domain] = r - } - - assert.Equal(t, "myapp:latest", routeMap["app1.example.com"].Image) - assert.True(t, routeMap["app1.example.com"].HTTPS) - - assert.Equal(t, "otherapp:v2", routeMap["app2.example.com"].Image) - assert.True(t, routeMap["app2.example.com"].HTTPS) - - assert.Equal(t, "insecureapp:latest", routeMap["insecure.example"].Image) - assert.False(t, routeMap["insecure.example"].HTTPS) -} - -func TestService_GetRoute(t *testing.T) { - v := viper.New() - v.Set("routes", map[string]interface{}{ - "app.example.com": "myapp:latest", - "http://insecure.example": "insecureapp:latest", - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - t.Run("existing route", func(t *testing.T) { - route, err := svc.GetRoute(ctx, "app.example.com") - require.NoError(t, err) - assert.Equal(t, "app.example.com", route.Domain) - assert.Equal(t, "myapp:latest", route.Image) - assert.True(t, route.HTTPS) - }) - - t.Run("legacy http route found via fallback", func(t *testing.T) { - route, err := svc.GetRoute(ctx, "insecure.example") - require.NoError(t, err) - assert.Equal(t, "insecure.example", route.Domain) - assert.Equal(t, "insecureapp:latest", route.Image) - assert.False(t, route.HTTPS) - }) - - t.Run("non-existent route", func(t *testing.T) { - route, err := svc.GetRoute(ctx, "notfound.example.com") - assert.ErrorIs(t, err, domain.ErrRouteNotFound) - assert.Nil(t, route) - }) -} - -func TestLoadCanonicalRoutes_RejectsDuplicateCanonicalRouteKeys(t *testing.T) { - _, err := loadCanonicalRoutes(map[string]any{ - "App.Example.com": "app:v1", - "app.example.com": "app:v2", - }) - - require.Error(t, err) - assert.Contains(t, err.Error(), "duplicate route key") -} - -func TestService_Load_RejectsExternalRouteConflictingWithRoute(t *testing.T) { - v := viper.New() - v.Set("routes", map[string]any{ - "App.Example.com": "app:latest", - }) - v.Set("external_routes", map[string]any{ - "app.example.com": "203.0.113.10:5000", - }) - - svc := NewService(v, mocks.NewMockEventPublisher(t)) - err := svc.Load(testContext()) - - require.Error(t, err) - assert.Contains(t, err.Error(), "conflicts with configured route") -} - -func TestLoadExternalRoutes_RejectsInvalidAndDuplicateKeys(t *testing.T) { - t.Run("invalid", func(t *testing.T) { - _, err := loadExternalRoutes(map[string]any{ - "bad.example.com:8080": "203.0.113.10:5000", - }) - - require.Error(t, err) - assert.Contains(t, err.Error(), "invalid external route key") - }) - - t.Run("duplicate canonical", func(t *testing.T) { - _, err := loadExternalRoutes(map[string]any{ - "Reg.Example.com": "203.0.113.10:5000", - "reg.example.com": "203.0.113.11:5000", - }) - - require.Error(t, err) - assert.Contains(t, err.Error(), "duplicate external route key") - }) -} - -func TestService_AddRoute_RejectsExternalRouteConflict(t *testing.T) { - v := viper.New() - v.Set("external_routes", map[string]any{ - "app.example.com": "203.0.113.10:5000", - }) - - svc := NewService(v, mocks.NewMockEventPublisher(t)) - ctx := testContext() - require.NoError(t, svc.Load(ctx)) - - err := svc.AddRoute(ctx, domain.Route{Domain: "App.Example.com", Image: "app:latest"}) - - require.Error(t, err) - assert.ErrorIs(t, err, domain.ErrRouteConflict) - assert.Contains(t, err.Error(), "conflicts with external route") -} - -func TestService_AddRoute_CanonicalizesDomainCaseAndRejectsInvalidAuthority(t *testing.T) { - configFile := filepath.Join(t.TempDir(), "gordon.toml") - require.NoError(t, os.WriteFile(configFile, []byte("[routes]\n"), 0600)) - v := viper.New() - v.SetConfigFile(configFile) - svc := NewService(v, mocks.NewMockEventPublisher(t)) - ctx := testContext() - require.NoError(t, svc.Load(ctx)) - - require.NoError(t, svc.AddRoute(ctx, domain.Route{Domain: "App.Example.com", Image: "app:latest", HTTPS: true})) - route, err := svc.GetRoute(ctx, "app.example.com") - require.NoError(t, err) - assert.Equal(t, "app.example.com", route.Domain) - _, err = svc.GetRoute(ctx, "APP.EXAMPLE.COM") - require.NoError(t, err) - - err = svc.AddRoute(ctx, domain.Route{Domain: "bad.example.com:8080", Image: "app:latest"}) - assert.ErrorIs(t, err, domain.ErrRouteDomainInvalid) - err = svc.AddRoute(ctx, domain.Route{Domain: "bad.example.com.", Image: "app:latest"}) - assert.ErrorIs(t, err, domain.ErrRouteDomainInvalid) -} - -func TestService_AddRoute(t *testing.T) { - t.Run("success", func(t *testing.T) { - // Create temp config file for Save() to work - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configFile, []byte("[routes]\n"), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - route := domain.Route{ - Domain: "new.example.com", - Image: "newapp:latest", - HTTPS: true, - } - - err = svc.AddRoute(ctx, route) - - assert.NoError(t, err) - - // Verify route was added - config := svc.GetConfig() - routeCfg := requireRoute(t, config.Routes, "new.example.com") - assert.Equal(t, "newapp:latest", routeCfg.Image) - assert.True(t, routeCfg.HTTPS) - }) - - t.Run("zero-value route stores HTTPS false in canonical form", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configFile, []byte("[routes]\n"), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - _ = svc.Load(ctx) - - // Zero-value HTTPS field (false) should be preserved - route := domain.Route{ - Domain: "app.example.com", - Image: "myapp:latest", - } - err = svc.AddRoute(ctx, route) - require.NoError(t, err) - - config := svc.GetConfig() - routeCfg := requireRoute(t, config.Routes, "app.example.com") - assert.Equal(t, "myapp:latest", routeCfg.Image, - "zero-value route should store under plain domain key") - assert.False(t, routeCfg.HTTPS) - _, hasHTTPKey := config.Routes["http://app.example.com"] - assert.False(t, hasHTTPKey, "zero-value route should NOT get http:// prefix") - - // Verify it round-trips as HTTPS=false via GetRoutes - routes := svc.GetRoutes(ctx) - require.Len(t, routes, 1) - assert.False(t, routes[0].HTTPS, "zero-value route should be treated as HTTP") - }) - - t.Run("HTTPS explicitly true stores under plain domain key", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configFile, []byte("[routes]\n"), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - _ = svc.Load(ctx) - - // Explicitly set HTTPS: true — should store under plain domain key - route := domain.Route{ - Domain: "secure.example.com", - Image: "myapp:latest", - HTTPS: true, - } - err = svc.AddRoute(ctx, route) - require.NoError(t, err) - - config := svc.GetConfig() - routeCfg := requireRoute(t, config.Routes, "secure.example.com") - assert.Equal(t, "myapp:latest", routeCfg.Image, - "HTTPS=true route should store under plain domain key") - assert.True(t, routeCfg.HTTPS) - }) - - t.Run("idempotent same domain and image saves once", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configFile, []byte("[routes]\n"), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - route := domain.Route{ - Domain: "new.example.com", - Image: "newapp:latest", - HTTPS: true, - } - - err = svc.AddRoute(ctx, route) - require.NoError(t, err) - - before, err := os.Stat(configFile) - require.NoError(t, err) - - err = svc.AddRoute(ctx, route) - require.NoError(t, err) - - after, err := os.Stat(configFile) - require.NoError(t, err) - - assert.Equal(t, before.ModTime(), after.ModTime()) - routeCfg := requireRoute(t, svc.GetConfig().Routes, "new.example.com") - assert.Equal(t, "newapp:latest", routeCfg.Image) - assert.True(t, routeCfg.HTTPS) - }) - - t.Run("empty domain", func(t *testing.T) { - v := viper.New() - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - route := domain.Route{ - Domain: "", - Image: "myapp:latest", - } - - err := svc.AddRoute(ctx, route) - assert.ErrorIs(t, err, domain.ErrRouteDomainEmpty) - }) - - t.Run("empty image", func(t *testing.T) { - v := viper.New() - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - route := domain.Route{ - Domain: "example.com", - Image: "", - } - - err := svc.AddRoute(ctx, route) - assert.ErrorIs(t, err, domain.ErrRouteImageEmpty) - }) - - t.Run("invalid domain - IP address", func(t *testing.T) { - v := viper.New() - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - route := domain.Route{ - Domain: "192.168.1.1", - Image: "myapp:latest", - } - - err := svc.AddRoute(ctx, route) - assert.ErrorIs(t, err, domain.ErrRouteDomainInvalid) - }) - - t.Run("invalid domain - localhost", func(t *testing.T) { - v := viper.New() - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - route := domain.Route{ - Domain: "localhost", - Image: "myapp:latest", - } - - err := svc.AddRoute(ctx, route) - assert.ErrorIs(t, err, domain.ErrRouteDomainInvalid) - }) - - t.Run("invalid domain - internal TLD", func(t *testing.T) { - v := viper.New() - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - route := domain.Route{ - Domain: "myapp.local", - Image: "myapp:latest", - } - - err := svc.AddRoute(ctx, route) - assert.ErrorIs(t, err, domain.ErrRouteDomainInvalid) - }) - - t.Run("invalid domain - with port", func(t *testing.T) { - v := viper.New() - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - route := domain.Route{ - Domain: "example.com:8080", - Image: "myapp:latest", - } - - err := svc.AddRoute(ctx, route) - assert.ErrorIs(t, err, domain.ErrRouteDomainInvalid) - }) - - t.Run("reconciles coexisting legacy http key", func(t *testing.T) { - svc, ctx := setupServiceWithCoexistingLegacyRoute(t, "myapp:v1", "myapp:old") - preConfig := svc.GetConfig() - requireRoute(t, preConfig.Routes, "app.example.com") - requireRoute(t, preConfig.Routes, "http://app.example.com") - - route := domain.Route{ - Domain: "app.example.com", - Image: "myapp:v2", - HTTPS: true, - } - - err := svc.AddRoute(ctx, route) - assert.NoError(t, err) - - config := svc.GetConfig() - routeCfg := requireRoute(t, config.Routes, "app.example.com") - assert.Equal(t, "myapp:v2", routeCfg.Image, - "canonical route should be updated") - assert.True(t, routeCfg.HTTPS) - _, exists := config.Routes["http://app.example.com"] - assert.False(t, exists, "legacy http route key should be removed during reconciliation") - }) -} - -func TestService_UpdateRoute(t *testing.T) { - t.Run("success", func(t *testing.T) { - // Create temp config file for Save() to work - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configFile, []byte("[routes]\n\"app.example.com\" = \"myapp:v1\"\n"), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - v.Set("routes", map[string]any{ - "app.example.com": "myapp:v1", - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - route := domain.Route{ - Domain: "app.example.com", - Image: "myapp:v2", - HTTPS: true, - } - - err = svc.UpdateRoute(ctx, route) - - assert.NoError(t, err) - - config := svc.GetConfig() - routeCfg := requireRoute(t, config.Routes, "app.example.com") - assert.Equal(t, "myapp:v2", routeCfg.Image) - assert.True(t, routeCfg.HTTPS) - }) - - t.Run("updates https setting", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configFile, []byte("[routes]\n\"app.example.com\" = \"myapp:v1\"\n"), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - v.Set("routes", map[string]any{ - "app.example.com": "myapp:v1", - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - route := domain.Route{ - Domain: "app.example.com", - Image: "myapp:v2", - HTTPS: false, - } - - err = svc.UpdateRoute(ctx, route) - require.NoError(t, err) - - routeCfg := requireRoute(t, svc.GetConfig().Routes, "app.example.com") - assert.Equal(t, "myapp:v2", routeCfg.Image) - assert.False(t, routeCfg.HTTPS) - - content, err := os.ReadFile(configFile) - require.NoError(t, err) - assert.Contains(t, string(content), `"app.example.com" = { image = "myapp:v2", https = false }`) - }) - - t.Run("not found", func(t *testing.T) { - v := viper.New() - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - route := domain.Route{ - Domain: "nonexistent.example.com", - Image: "myapp:latest", - } - - err := svc.UpdateRoute(ctx, route) - assert.ErrorIs(t, err, domain.ErrRouteNotFound) - }) - - t.Run("empty domain", func(t *testing.T) { - v := viper.New() - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - route := domain.Route{ - Domain: "", - Image: "myapp:latest", - } - - err := svc.UpdateRoute(ctx, route) - assert.ErrorIs(t, err, domain.ErrRouteDomainEmpty) - }) - - t.Run("empty image", func(t *testing.T) { - v := viper.New() - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - route := domain.Route{ - Domain: "example.com", - Image: "", - } - - err := svc.UpdateRoute(ctx, route) - assert.ErrorIs(t, err, domain.ErrRouteImageEmpty) - }) - - t.Run("invalid domain - IP address", func(t *testing.T) { - v := viper.New() - v.Set("routes", map[string]any{ - "192.168.1.1": "myapp:v1", - }) - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - route := domain.Route{ - Domain: "192.168.1.1", - Image: "myapp:v2", - } - - err := svc.UpdateRoute(ctx, route) - assert.ErrorIs(t, err, domain.ErrRouteDomainInvalid) - }) - - t.Run("invalid domain - localhost", func(t *testing.T) { - v := viper.New() - v.Set("routes", map[string]any{ - "localhost": "myapp:v1", - }) - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - route := domain.Route{ - Domain: "localhost", - Image: "myapp:v2", - } - - err := svc.UpdateRoute(ctx, route) - assert.ErrorIs(t, err, domain.ErrRouteDomainInvalid) - }) - - t.Run("insecure route found via http prefix fallback", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configFile, []byte("[routes]\n\"http://insecure.example.com\" = \"myapp:v1\"\n"), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - v.Set("routes", map[string]any{ - "http://insecure.example.com": "myapp:v1", - }) - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - _ = svc.Load(ctx) - - // Caller uses clean domain for update - route := domain.Route{ - Domain: "insecure.example.com", - Image: "myapp:v2", - HTTPS: false, - } - - err = svc.UpdateRoute(ctx, route) - assert.NoError(t, err, "UpdateRoute should find insecure route via http:// fallback") - - config := svc.GetConfig() - routeCfg := requireRoute(t, config.Routes, "insecure.example.com") - assert.Equal(t, "myapp:v2", routeCfg.Image, - "insecure route update should preserve the plain hostname key") - assert.False(t, routeCfg.HTTPS) - }) - - t.Run("reconciles coexisting legacy http key", func(t *testing.T) { - svc, ctx := setupServiceWithCoexistingLegacyRoute(t, "myapp:v1", "myapp:old") - preConfig := svc.GetConfig() - requireRoute(t, preConfig.Routes, "app.example.com") - requireRoute(t, preConfig.Routes, "http://app.example.com") - - route := domain.Route{ - Domain: "app.example.com", - Image: "myapp:v2", - HTTPS: true, - } - - err := svc.UpdateRoute(ctx, route) - assert.NoError(t, err) - - config := svc.GetConfig() - routeCfg := requireRoute(t, config.Routes, "app.example.com") - assert.Equal(t, "myapp:v2", routeCfg.Image, - "canonical route should be updated") - assert.True(t, routeCfg.HTTPS) - _, exists := config.Routes["http://app.example.com"] - assert.False(t, exists, "legacy http route key should be removed during reconciliation") - }) -} - -func TestService_RemoveRoute(t *testing.T) { - // Create temp config file for Save() to work - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configFile, []byte("[routes]\n\"app.example.com\" = \"myapp:latest\"\n"), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - v.Set("routes", map[string]interface{}{ - "app.example.com": "myapp:latest", - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - err = svc.RemoveRoute(ctx, "app.example.com") - - assert.NoError(t, err) - - config := svc.GetConfig() - _, exists := config.Routes["app.example.com"] - assert.False(t, exists) -} - -func TestService_RemoveRoute_InvalidDomain(t *testing.T) { - v := viper.New() - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - err := svc.RemoveRoute(ctx, "localhost") - assert.ErrorIs(t, err, domain.ErrRouteDomainInvalid) -} - -func TestService_RemoveRoute_NotFound(t *testing.T) { - v := viper.New() - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - err := svc.RemoveRoute(ctx, "nonexistent.example.com") - - assert.ErrorIs(t, err, domain.ErrRouteNotFound) -} - -func TestService_RemoveRoute_InsecureFallback(t *testing.T) { - // Setup: insecure route stored with http:// prefix - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configFile, []byte("[routes]\n\"http://insecure.example.com\" = \"myapp:latest\"\n"), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - v.Set("routes", map[string]interface{}{ - "http://insecure.example.com": "myapp:latest", - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - // Caller passes plain domain (not the storage key) - err = svc.RemoveRoute(ctx, "insecure.example.com") - assert.NoError(t, err, "RemoveRoute should find insecure route via http:// fallback") - - config := svc.GetConfig() - _, exists := config.Routes["insecure.example.com"] - assert.False(t, exists, "insecure route should be removed") -} - -func TestService_RemoveRoute_ReconcilesCoexistingLegacyKey(t *testing.T) { - svc, ctx := setupServiceWithCoexistingLegacyRoute(t, "myapp:latest", "myapp:old") - - err := svc.RemoveRoute(ctx, "app.example.com") - assert.NoError(t, err) - - config := svc.GetConfig() - _, exists := config.Routes["app.example.com"] - assert.False(t, exists, "canonical route should be removed") - _, exists = config.Routes["http://app.example.com"] - assert.False(t, exists, "legacy http route key should be removed during reconciliation") -} - -func TestService_Save_WrapsInvalidRouteDomainError(t *testing.T) { - t.Run("canonical key", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configFile, []byte("[routes]\n\"app.example.com\" = \"myapp:latest\"\n"), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - require.NoError(t, v.ReadInConfig()) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - require.NoError(t, svc.Load(ctx)) - - svc.mu.Lock() - svc.config.Routes["localhost"] = routeConfig{Image: "myapp:latest", HTTPS: true} - svc.mu.Unlock() - - err = svc.Save(ctx) - assert.ErrorIs(t, err, domain.ErrRouteDomainInvalid) - }) - - t.Run("legacy key", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configFile, []byte("[routes]\n\"app.example.com\" = \"myapp:latest\"\n"), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - require.NoError(t, v.ReadInConfig()) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - require.NoError(t, svc.Load(ctx)) - - svc.mu.Lock() - svc.config.Routes["http://localhost"] = routeConfig{Image: "myapp:latest", HTTPS: false} - svc.mu.Unlock() - - err = svc.Save(ctx) - assert.ErrorIs(t, err, domain.ErrRouteDomainInvalid) - }) -} - -func TestService_GetNetworkGroups(t *testing.T) { - v := viper.New() - v.Set("network_groups", map[string]interface{}{ - "frontend": []interface{}{"app1.example.com", "app2.example.com"}, - "backend": []interface{}{"api.example.com", "db.example.com"}, - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - groups := svc.GetNetworkGroups() - - assert.Len(t, groups, 2) - assert.ElementsMatch(t, []string{"app1.example.com", "app2.example.com"}, groups["frontend"]) - assert.ElementsMatch(t, []string{"api.example.com", "db.example.com"}, groups["backend"]) -} - -func TestService_GetAttachmentConfig(t *testing.T) { - v := viper.New() - v.Set("attachments", map[string]interface{}{ - "app.example.com": []interface{}{"redis:latest", "postgres:18"}, - }) - v.Set("network_groups", map[string]interface{}{ - "shared": []interface{}{"app.example.com", "api.example.com"}, - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - snapshot := svc.GetAttachmentConfig() - - assert.Len(t, snapshot.Attachments, 1) - assert.ElementsMatch(t, []string{"redis:latest", "postgres:18"}, snapshot.Attachments["app.example.com"]) - assert.Len(t, snapshot.NetworkGroups, 1) - assert.ElementsMatch(t, []string{"app.example.com", "api.example.com"}, snapshot.NetworkGroups["shared"]) - - snapshot.Attachments["app.example.com"][0] = "mutated" - snapshot.NetworkGroups["shared"][0] = "mutated.example.com" - - freshSnapshot := svc.GetAttachmentConfig() - assert.ElementsMatch(t, []string{"redis:latest", "postgres:18"}, freshSnapshot.Attachments["app.example.com"]) - assert.ElementsMatch(t, []string{"app.example.com", "api.example.com"}, freshSnapshot.NetworkGroups["shared"]) -} - -func TestService_GetAllAttachments(t *testing.T) { - t.Run("returns all attachments", func(t *testing.T) { - v := viper.New() - v.Set("attachments", map[string]interface{}{ - "app.example.com": []interface{}{"redis:latest", "postgres:18"}, - "api.example.com": []interface{}{"rabbitmq:3"}, - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - attachments := svc.GetAllAttachments(ctx) - - assert.Len(t, attachments, 2) - assert.ElementsMatch(t, []string{"redis:latest", "postgres:18"}, attachments["app.example.com"]) - assert.ElementsMatch(t, []string{"rabbitmq:3"}, attachments["api.example.com"]) - }) - - t.Run("returns empty map when no attachments", func(t *testing.T) { - v := viper.New() - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - attachments := svc.GetAllAttachments(ctx) - - assert.Empty(t, attachments) - }) - - t.Run("returns copy not reference", func(t *testing.T) { - v := viper.New() - v.Set("attachments", map[string]interface{}{ - "app.example.com": []interface{}{"redis:latest"}, - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - attachments := svc.GetAllAttachments(ctx) - // Modify the returned map - attachments["app.example.com"] = append(attachments["app.example.com"], "postgres:18") - - // Original should be unchanged - original := svc.GetAllAttachments(ctx) - assert.Len(t, original["app.example.com"], 1) - }) -} - -func TestService_GetAttachmentsFor(t *testing.T) { - t.Run("existing domain", func(t *testing.T) { - v := viper.New() - v.Set("attachments", map[string]interface{}{ - "app.example.com": []interface{}{"redis:latest", "postgres:18"}, - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - images, err := svc.GetAttachmentsFor(ctx, "app.example.com") - - require.NoError(t, err) - assert.ElementsMatch(t, []string{"redis:latest", "postgres:18"}, images) - }) - - t.Run("existing network group", func(t *testing.T) { - v := viper.New() - v.Set("attachments", map[string]interface{}{ - "backend": []interface{}{"rabbitmq:3", "redis:latest"}, - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - images, err := svc.GetAttachmentsFor(ctx, "backend") - - require.NoError(t, err) - assert.ElementsMatch(t, []string{"rabbitmq:3", "redis:latest"}, images) - }) - - t.Run("non-existent target", func(t *testing.T) { - v := viper.New() - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - images, err := svc.GetAttachmentsFor(ctx, "notfound.example.com") - - assert.ErrorIs(t, err, domain.ErrAttachmentNotFound) - assert.Nil(t, images) - }) - - t.Run("returns copy not reference", func(t *testing.T) { - v := viper.New() - v.Set("attachments", map[string]interface{}{ - "app.example.com": []interface{}{"redis:latest"}, - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - images, err := svc.GetAttachmentsFor(ctx, "app.example.com") - require.NoError(t, err) - - // Modify the returned slice (use _ to satisfy linter) - _ = append(images, "postgres:18") - - // Original should be unchanged - original, _ := svc.GetAttachmentsFor(ctx, "app.example.com") - assert.Len(t, original, 1) - }) -} - -func TestService_AddAttachment(t *testing.T) { - t.Run("success - new target", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configFile, []byte("[attachments]\n"), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - err = svc.AddAttachment(ctx, "app.example.com", "postgres:18") - - assert.NoError(t, err) - - // Verify attachment was added - config := svc.GetConfig() - assert.Contains(t, config.Attachments["app.example.com"], "postgres:18") - }) - - t.Run("success - existing target", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configFile, []byte("[attachments]\n\"app.example.com\" = [\"redis:latest\"]\n"), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - v.Set("attachments", map[string]interface{}{ - "app.example.com": []interface{}{"redis:latest"}, - }) - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - err = svc.AddAttachment(ctx, "app.example.com", "postgres:18") - - assert.NoError(t, err) - - // Verify attachment was added alongside existing - config := svc.GetConfig() - assert.ElementsMatch(t, []string{"redis:latest", "postgres:18"}, config.Attachments["app.example.com"]) - }) - - t.Run("legacy Gordon ref is canonicalized on disk only", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - initialConfig := `[server] - gordon_domain = "registry.example.com" - legacy_registry_domains = ["old-registry.example.com"] - - [attachments] - ` - err := os.WriteFile(configFile, []byte(initialConfig), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - require.NoError(t, v.ReadInConfig()) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - require.NoError(t, svc.Load(ctx)) - - err = svc.AddAttachment(ctx, "app.example.com", "old-registry.example.com/postgres:18") - require.NoError(t, err) - - require.Equal(t, []string{"old-registry.example.com/postgres:18"}, svc.GetConfig().Attachments["app.example.com"]) - - saved := viper.New() - saved.SetConfigFile(configFile) - require.NoError(t, saved.ReadInConfig()) - - var attachments map[string][]string - require.NoError(t, saved.UnmarshalKey("attachments", &attachments)) - assert.Equal(t, []string{"registry.example.com/postgres:18"}, attachments["app.example.com"]) - - content, err := os.ReadFile(configFile) - require.NoError(t, err) - text := string(content) - assert.Contains(t, text, `registry.example.com/postgres:18`) - assert.NotContains(t, text, `old-registry.example.com/postgres:18`) - }) - - t.Run("rejects canonical duplicate when existing legacy ref matches current ref", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - initialConfig := `[server] - gordon_domain = "registry.example.com" - legacy_registry_domains = ["old-registry.example.com"] - - [attachments] - "app.example.com" = ["old-registry.example.com/postgres:18"] - ` - err := os.WriteFile(configFile, []byte(initialConfig), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - require.NoError(t, v.ReadInConfig()) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - require.NoError(t, svc.Load(ctx)) - - err = svc.AddAttachment(ctx, "app.example.com", "registry.example.com/postgres:18") - - assert.ErrorIs(t, err, domain.ErrAttachmentExists) - assert.Equal(t, []string{"old-registry.example.com/postgres:18"}, svc.GetConfig().Attachments["app.example.com"]) - }) - - t.Run("rejects canonical duplicate when existing current ref matches legacy ref", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - initialConfig := `[server] - gordon_domain = "registry.example.com" - legacy_registry_domains = ["old-registry.example.com"] - - [attachments] - "app.example.com" = ["registry.example.com/postgres:18"] - ` - err := os.WriteFile(configFile, []byte(initialConfig), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - require.NoError(t, v.ReadInConfig()) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - require.NoError(t, svc.Load(ctx)) - - err = svc.AddAttachment(ctx, "app.example.com", "old-registry.example.com/postgres:18") - - assert.ErrorIs(t, err, domain.ErrAttachmentExists) - assert.Equal(t, []string{"registry.example.com/postgres:18"}, svc.GetConfig().Attachments["app.example.com"]) - }) - - t.Run("duplicate attachment", func(t *testing.T) { - v := viper.New() - v.Set("attachments", map[string]interface{}{ - "app.example.com": []interface{}{"redis:latest"}, - }) - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - err := svc.AddAttachment(ctx, "app.example.com", "redis:latest") - - assert.ErrorIs(t, err, domain.ErrAttachmentExists) - }) - - t.Run("empty target", func(t *testing.T) { - v := viper.New() - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - err := svc.AddAttachment(ctx, "", "postgres:18") - - assert.ErrorIs(t, err, domain.ErrAttachmentTargetEmpty) - }) - - t.Run("empty image", func(t *testing.T) { - v := viper.New() - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - err := svc.AddAttachment(ctx, "app.example.com", "") - - assert.ErrorIs(t, err, domain.ErrAttachmentImageEmpty) - }) - - t.Run("network group target", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configFile, []byte("[attachments]\n"), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - err = svc.AddAttachment(ctx, "backend", "rabbitmq:3") - - assert.NoError(t, err) - - config := svc.GetConfig() - assert.Contains(t, config.Attachments["backend"], "rabbitmq:3") - }) -} - -func TestService_RemoveAttachment(t *testing.T) { - t.Run("success - single attachment", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configFile, []byte("[attachments]\n\"app.example.com\" = [\"redis:latest\"]\n"), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - v.Set("attachments", map[string]interface{}{ - "app.example.com": []interface{}{"redis:latest"}, - }) - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - err = svc.RemoveAttachment(ctx, "app.example.com", "redis:latest") - - assert.NoError(t, err) - - // Verify target was removed (no attachments left) - config := svc.GetConfig() - _, exists := config.Attachments["app.example.com"] - assert.False(t, exists) - }) - - t.Run("success - multiple attachments", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configFile, []byte("[attachments]\n\"app.example.com\" = [\"redis:latest\", \"postgres:18\"]\n"), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - v.Set("attachments", map[string]interface{}{ - "app.example.com": []interface{}{"redis:latest", "postgres:18"}, - }) - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - err = svc.RemoveAttachment(ctx, "app.example.com", "redis:latest") - - assert.NoError(t, err) - - // Verify only redis was removed - config := svc.GetConfig() - assert.Equal(t, []string{"postgres:18"}, config.Attachments["app.example.com"]) - }) - - t.Run("target not found", func(t *testing.T) { - v := viper.New() - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - err := svc.RemoveAttachment(ctx, "notfound.example.com", "redis:latest") - - assert.ErrorIs(t, err, domain.ErrAttachmentNotFound) - }) - - t.Run("image not found in target", func(t *testing.T) { - v := viper.New() - v.Set("attachments", map[string]interface{}{ - "app.example.com": []interface{}{"redis:latest"}, - }) - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - err := svc.RemoveAttachment(ctx, "app.example.com", "postgres:18") - - assert.ErrorIs(t, err, domain.ErrAttachmentNotFound) - }) - - t.Run("empty target", func(t *testing.T) { - v := viper.New() - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - err := svc.RemoveAttachment(ctx, "", "redis:latest") - - assert.ErrorIs(t, err, domain.ErrAttachmentTargetEmpty) - }) - - t.Run("empty image", func(t *testing.T) { - v := viper.New() - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - err := svc.RemoveAttachment(ctx, "app.example.com", "") - - assert.ErrorIs(t, err, domain.ErrAttachmentImageEmpty) - }) - - t.Run("network group target", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - err := os.WriteFile(configFile, []byte("[attachments]\n\"backend\" = [\"rabbitmq:3\"]\n"), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - v.Set("attachments", map[string]interface{}{ - "backend": []interface{}{"rabbitmq:3"}, - }) - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - err = svc.RemoveAttachment(ctx, "backend", "rabbitmq:3") - - assert.NoError(t, err) - - config := svc.GetConfig() - _, exists := config.Attachments["backend"] - assert.False(t, exists) - }) -} - -func TestService_AutoRouteAllowedDomains(t *testing.T) { - t.Run("get returns copy", func(t *testing.T) { - v := viper.New() - v.Set("auto_route_allowed_domains", []string{"example.com", "*.example.com"}) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - require.NoError(t, svc.Load(ctx)) - - domains, err := svc.GetAutoRouteAllowedDomains(ctx) - require.NoError(t, err) - domains[0] = "changed.com" - - again, err := svc.GetAutoRouteAllowedDomains(ctx) - require.NoError(t, err) - assert.Equal(t, []string{"example.com", "*.example.com"}, again) - }) - - t.Run("add and remove persist idempotently", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - require.NoError(t, os.WriteFile(configFile, []byte("auto_route_allowed_domains = [\"example.com\"]\n"), 0600)) - - v := viper.New() - v.SetConfigFile(configFile) - v.Set("auto_route_allowed_domains", []interface{}{"example.com"}) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - require.NoError(t, svc.Load(ctx)) - require.NoError(t, svc.AddAutoRouteAllowedDomain(ctx, "*.example.com")) - require.NoError(t, svc.AddAutoRouteAllowedDomain(ctx, "*.example.com")) - - domains, err := svc.GetAutoRouteAllowedDomains(ctx) - require.NoError(t, err) - assert.Equal(t, []string{"example.com", "*.example.com"}, domains) - - require.NoError(t, svc.RemoveAutoRouteAllowedDomain(ctx, "example.com")) - require.NoError(t, svc.RemoveAutoRouteAllowedDomain(ctx, "missing.com")) - - domains, err = svc.GetAutoRouteAllowedDomains(ctx) - require.NoError(t, err) - assert.Equal(t, []string{"*.example.com"}, domains) - }) - - t.Run("add validates patterns", func(t *testing.T) { - v := viper.New() - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - require.NoError(t, svc.Load(ctx)) - - invalid := []string{"", "EXAMPLE.com", "example.com.", "**.example.com", "foo.*.example.com", "*.example.com."} - for _, pattern := range invalid { - err := svc.AddAutoRouteAllowedDomain(ctx, pattern) - assert.ErrorIs(t, err, domain.ErrInvalidDomainPattern, "pattern: %s", pattern) - } - }) -} - -func TestService_GetExternalRoutes(t *testing.T) { - v := viper.New() - v.Set("external_routes", map[string]interface{}{ - "Reg.Example.com": "localhost:5000", - "cache.example.com": "127.0.0.1:6379", - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - routes := svc.GetExternalRoutes() - - assert.Len(t, routes, 2) - assert.Equal(t, "localhost:5000", routes["reg.example.com"]) - assert.Equal(t, "127.0.0.1:6379", routes["cache.example.com"]) -} - -func TestService_GetExternalRoutes_Empty(t *testing.T) { - v := viper.New() - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - _ = svc.Load(ctx) - - routes := svc.GetExternalRoutes() - - assert.Empty(t, routes) -} - -func TestSplitImageNameTag(t *testing.T) { - tests := []struct { - name string - image string - wantName string - wantHasTag bool - }{ - {"with tag", "myapp:latest", "myapp", true}, - {"with version tag", "myapp:v1.2.0", "myapp", true}, - {"no tag", "myapp", "myapp", false}, - {"empty string", "", "", false}, - {"tag only colon", "myapp:", "myapp", true}, - {"registry prefixed with tag", "reg.example.com/myapp:latest", "reg.example.com/myapp", true}, - {"registry prefixed no tag", "reg.example.com/myapp", "reg.example.com/myapp", false}, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - name, hasTag := splitImageNameTag(tt.image) - assert.Equal(t, tt.wantName, name) - assert.Equal(t, tt.wantHasTag, hasTag) - }) - } -} - -func TestNormalizeRegistryImage(t *testing.T) { - tests := []struct { - name string - imageName string - registryDomain string - legacyRegistryDomains []string - expected string - }{ - {"strips current registry prefix", "reg.example.com/myapp:latest", "reg.example.com", nil, "myapp:latest"}, - {"strips legacy registry prefix", "old-reg.example.com/myapp:latest", "new-reg.example.com", []string{"old-reg.example.com"}, "myapp:latest"}, - {"strips legacy registry prefix with explicit port and digest", "old-reg.example.com:5000/myapp@sha256:deadbeef", "new-reg.example.com", []string{"old-reg.example.com:5000"}, "myapp@sha256:deadbeef"}, - {"no prefix to strip", "myapp:latest", "reg.example.com", nil, "myapp:latest"}, - {"empty registry domain", "myapp:latest", "", nil, "myapp:latest"}, - {"registry with trailing slash", "reg.example.com/myapp:latest", "reg.example.com/", nil, "myapp:latest"}, - {"hostile lookalike is not stripped", "old-reg.example.com.evil/myapp:latest", "new-reg.example.com", []string{"old-reg.example.com"}, "old-reg.example.com.evil/myapp:latest"}, - {"bare name no registry", "myapp", "reg.example.com", nil, "myapp"}, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - result := NormalizeRegistryImage(tt.imageName, tt.registryDomain, tt.legacyRegistryDomains) - assert.Equal(t, tt.expected, result) - }) - } -} - -func TestNormalizeBootstrapImage(t *testing.T) { - tests := []struct { - name string - image string - registryDomain string - expected string - wantErr string - }{ - {name: "bare name prepends registry", image: "myapp", registryDomain: "reg.example.com", expected: "reg.example.com/myapp"}, - {name: "bare name strips tag", image: "myapp:v1", registryDomain: "reg.example.com", expected: "reg.example.com/myapp"}, - {name: "qualified image unchanged", image: "reg.example.com/myapp", registryDomain: "reg.example.com", expected: "reg.example.com/myapp"}, - {name: "qualified image strips tag", image: "reg.example.com/myapp:abc123", registryDomain: "reg.example.com", expected: "reg.example.com/myapp"}, - {name: "other registry strips tag", image: "other.registry.com/myapp:v2", registryDomain: "reg.example.com", expected: "other.registry.com/myapp"}, - {name: "empty image", image: "", registryDomain: "reg.example.com", wantErr: "image is required"}, - {name: "empty registry domain", image: "pitlane", registryDomain: "", wantErr: "registry domain is not configured"}, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - result, err := NormalizeBootstrapImage(tt.image, tt.registryDomain) - if tt.wantErr != "" { - require.Error(t, err) - assert.Equal(t, tt.wantErr, err.Error()) - assert.Empty(t, result) - return - } - - require.NoError(t, err) - assert.Equal(t, tt.expected, result) - }) - } -} - -func TestService_FindRoutesByImage(t *testing.T) { - t.Run("exact match with tag", func(t *testing.T) { - v := viper.New() - v.Set("routes", map[string]interface{}{ - "app.example.com": "myapp:latest", - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - _ = svc.Load(ctx) - - routes := svc.FindRoutesByImage(ctx, "myapp:latest") - assert.Len(t, routes, 1) - assert.Equal(t, "app.example.com", routes[0].Domain) - assert.Equal(t, "myapp:latest", routes[0].Image) - }) - - t.Run("bare name matches route with tag", func(t *testing.T) { - v := viper.New() - v.Set("routes", map[string]interface{}{ - "app.example.com": "myapp:latest", - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - _ = svc.Load(ctx) - - routes := svc.FindRoutesByImage(ctx, "myapp") - assert.Len(t, routes, 1) - assert.Equal(t, "app.example.com", routes[0].Domain) - }) - - t.Run("bare name matches route with version tag", func(t *testing.T) { - v := viper.New() - v.Set("routes", map[string]interface{}{ - "app.example.com": "myapp:v2.0.0", - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - _ = svc.Load(ctx) - - routes := svc.FindRoutesByImage(ctx, "myapp") - assert.Len(t, routes, 1) - assert.Equal(t, "app.example.com", routes[0].Domain) - }) - - t.Run("tag mismatch does not match", func(t *testing.T) { - v := viper.New() - v.Set("routes", map[string]interface{}{ - "app.example.com": "myapp:v1", - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - _ = svc.Load(ctx) - - routes := svc.FindRoutesByImage(ctx, "myapp:v2") - assert.Empty(t, routes) - }) - - t.Run("no match", func(t *testing.T) { - v := viper.New() - v.Set("routes", map[string]interface{}{ - "app.example.com": "myapp:latest", - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - _ = svc.Load(ctx) - - routes := svc.FindRoutesByImage(ctx, "otherapp") - assert.Empty(t, routes) - }) - - t.Run("strips registry prefix before matching", func(t *testing.T) { - v := viper.New() - v.Set("server.registry_domain", "reg.example.com") - v.Set("routes", map[string]interface{}{ - "app.example.com": "reg.example.com/myapp:latest", - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - _ = svc.Load(ctx) - - routes := svc.FindRoutesByImage(ctx, "myapp") - assert.Len(t, routes, 1) - assert.Equal(t, "app.example.com", routes[0].Domain) - }) - - t.Run("strips registry prefix from input too", func(t *testing.T) { - v := viper.New() - v.Set("server.registry_domain", "reg.example.com") - v.Set("routes", map[string]interface{}{ - "app.example.com": "reg.example.com/myapp:latest", - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - _ = svc.Load(ctx) - - routes := svc.FindRoutesByImage(ctx, "reg.example.com/myapp:latest") - assert.Len(t, routes, 1) - assert.Equal(t, "app.example.com", routes[0].Domain) - }) - - t.Run("matches current image when route stores legacy registry host", func(t *testing.T) { - v := viper.New() - v.Set("server.gordon_domain", "new-registry.example.com") - v.Set("server.legacy_registry_domains", []string{"old-registry.example.com"}) - v.Set("routes", map[string]interface{}{ - "app.example.com": "old-registry.example.com/myapp:latest", - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - require.NoError(t, svc.Load(ctx)) - - routes := svc.FindRoutesByImage(ctx, "new-registry.example.com/myapp:latest") - assert.Len(t, routes, 1) - assert.Equal(t, "app.example.com", routes[0].Domain) - }) - - t.Run("multiple routes for same image", func(t *testing.T) { - v := viper.New() - v.Set("routes", map[string]interface{}{ - "app1.example.com": "myapp:latest", - "app2.example.com": "myapp:latest", - "other.example.com": "otherapp:latest", - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - _ = svc.Load(ctx) - - routes := svc.FindRoutesByImage(ctx, "myapp") - assert.Len(t, routes, 2) - - domains := []string{routes[0].Domain, routes[1].Domain} - assert.ElementsMatch(t, []string{"app1.example.com", "app2.example.com"}, domains) - }) - - t.Run("case insensitive match", func(t *testing.T) { - v := viper.New() - v.Set("routes", map[string]interface{}{ - "app.example.com": "MyApp:latest", - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - _ = svc.Load(ctx) - - routes := svc.FindRoutesByImage(ctx, "myapp") - assert.Len(t, routes, 1) - }) - - t.Run("http prefix route sets HTTPS false", func(t *testing.T) { - v := viper.New() - v.Set("routes", map[string]interface{}{ - "http://insecure.example.com": "myapp:latest", - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - _ = svc.Load(ctx) - - routes := svc.FindRoutesByImage(ctx, "myapp") - assert.Len(t, routes, 1) - assert.Equal(t, "insecure.example.com", routes[0].Domain) - assert.False(t, routes[0].HTTPS) - }) - - t.Run("empty routes", func(t *testing.T) { - v := viper.New() - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - _ = svc.Load(ctx) - - routes := svc.FindRoutesByImage(ctx, "myapp") - assert.Empty(t, routes) - }) -} - -func TestService_FindAttachmentTargetsByImage(t *testing.T) { - t.Run("exact tagged matches", func(t *testing.T) { - v := viper.New() - v.Set("attachments", map[string]interface{}{ - "app.example.com": []interface{}{"redis:latest", "postgres:18"}, - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - _ = svc.Load(ctx) - - targets := svc.FindAttachmentTargetsByImage(ctx, "postgres:18") - assert.Equal(t, []string{"app.example.com"}, targets) - }) - - t.Run("bare name matches tagged attachments", func(t *testing.T) { - v := viper.New() - v.Set("attachments", map[string]interface{}{ - "app.example.com": []interface{}{"postgres:18"}, - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - _ = svc.Load(ctx) - - targets := svc.FindAttachmentTargetsByImage(ctx, "postgres") - assert.Equal(t, []string{"app.example.com"}, targets) - }) - - t.Run("registry qualified input is normalized", func(t *testing.T) { - v := viper.New() - v.Set("server.registry_domain", "reg.example.com") - v.Set("attachments", map[string]interface{}{ - "backend": []interface{}{"reg.example.com/postgres:18"}, - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - _ = svc.Load(ctx) - - targets := svc.FindAttachmentTargetsByImage(ctx, "reg.example.com/postgres:18") - assert.Equal(t, []string{"backend"}, targets) - }) - - t.Run("matches current image when attachment stores legacy registry host", func(t *testing.T) { - v := viper.New() - v.Set("server.gordon_domain", "new-registry.example.com") - v.Set("server.legacy_registry_domains", []string{"old-registry.example.com"}) - v.Set("attachments", map[string]interface{}{ - "backend": []interface{}{"old-registry.example.com/postgres:18"}, - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - require.NoError(t, svc.Load(ctx)) - - targets := svc.FindAttachmentTargetsByImage(ctx, "new-registry.example.com/postgres:18") - assert.Equal(t, []string{"backend"}, targets) - }) - - t.Run("shared attachment used by multiple targets", func(t *testing.T) { - v := viper.New() - v.Set("attachments", map[string]interface{}{ - "app.example.com": []interface{}{"redis:latest"}, - "backend": []interface{}{"redis:latest", "postgres:18"}, - "worker": []interface{}{"rabbitmq:3"}, - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - _ = svc.Load(ctx) - - targets := svc.FindAttachmentTargetsByImage(ctx, "redis") - assert.ElementsMatch(t, []string{"app.example.com", "backend"}, targets) - }) - - t.Run("no matches", func(t *testing.T) { - v := viper.New() - v.Set("attachments", map[string]interface{}{ - "app.example.com": []interface{}{"redis:latest"}, - }) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - _ = svc.Load(ctx) - - targets := svc.FindAttachmentTargetsByImage(ctx, "postgres") - assert.Empty(t, targets) - }) -} - -func TestSavePreservesAllConfigFields(t *testing.T) { - t.Run("does not write predictable tmp path", func(t *testing.T) { - dir := t.TempDir() - configFile := filepath.Join(dir, "gordon.toml") - err := os.WriteFile(configFile, []byte("[server]\nport = 8080\n\n[routes]\n\"app.example.com\" = \"myapp:latest\"\n"), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - err = v.ReadInConfig() - require.NoError(t, err) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - err = svc.Load(ctx) - require.NoError(t, err) - - maliciousTarget := filepath.Join(dir, "pwned.txt") - predictableTmp := configFile + ".tmp" - err = os.Symlink(maliciousTarget, predictableTmp) - require.NoError(t, err) - - err = svc.Save(ctx) - require.NoError(t, err) - - _, statErr := os.Stat(maliciousTarget) - assert.True(t, os.IsNotExist(statErr), "symlink target should not exist - temp file should use unpredictable name") - - info, err := os.Lstat(predictableTmp) - require.NoError(t, err) - assert.NotZero(t, info.Mode()&os.ModeSymlink, "predictable .tmp path should still be the original symlink") - }) - - t.Run("with defaults does not corrupt config", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - initialConfig := ` -[server] -port = 8088 -registry_port = 5000 -registry_domain = "reg.example.com" -tls_enabled = true -tls_port = 8443 -data_dir = "/tmp/test-gordon" -runtime = "auto" + v.Set("volumes.preserve", false) -[auth] -enabled = true -secrets_backend = "pass" -username = "testadmin" -password_hash = "test/auth/password_hash" -token_secret = "test/auth/token_secret" -token_expiry = "720h" -type = "password" + eventBus := mocks.NewMockEventPublisher(t) + svc := NewService(v, eventBus) + ctx := testContext() -[auto_route] -enabled = true + err := svc.Load(ctx) -[network_isolation] -enabled = true -network_prefix = "gordon" -dns_suffix = ".internal" + assert.NoError(t, err) + assert.Equal(t, 8080, svc.GetServerPort()) + assert.Equal(t, 5000, svc.GetRegistryPort()) + assert.Equal(t, "registry.example.com", svc.GetRegistryDomain()) + assert.Equal(t, "/var/gordon", svc.GetDataDir()) + assert.True(t, svc.IsNetworkIsolationEnabled()) + assert.Equal(t, "gordon", svc.GetNetworkPrefix()) -[backups] -enabled = true -schedule = "daily" -storage_dir = "~/.gordon/backups" + autoCreate, prefix, preserve := svc.GetVolumeConfig() + assert.True(t, autoCreate) + assert.Equal(t, "gordon", prefix) + assert.False(t, preserve) +} -[routes] -"app.example.com" = "reg.example.com/myapp:latest" +func TestService_LoadPrefersExplicitRegistryAddress(t *testing.T) { + v := viper.New() + v.Set("server.gordon_domain", "gordon.example.com") + v.Set("server.registry_domain", "100.64.0.10:15000") -[attachments] -"app.example.com" = ["postgres:15"] + svc := NewService(v, mocks.NewMockEventPublisher(t)) + require.NoError(t, svc.Load(testContext())) + assert.Equal(t, "100.64.0.10:15000", svc.GetRegistryDomain()) +} -[network_groups] -backend = ["app.example.com"] +func TestService_Reload(t *testing.T) { + t.Run("success - picks up config file changes", func(t *testing.T) { + // Create temp config file + tmpDir := t.TempDir() + configFile := filepath.Join(tmpDir, "gordon.toml") + initialConfig := `[server] +port = 8080 ` err := os.WriteFile(configFile, []byte(initialConfig), 0600) require.NoError(t, err) v := viper.New() v.SetConfigFile(configFile) - v.SetDefault("server.port", 8088) - v.SetDefault("server.tls_port", 8443) - v.SetDefault("auth.secrets_backend", "") - v.SetDefault("auto_route.enabled", false) - v.SetDefault("network_isolation.enabled", false) - v.SetDefault("backups.enabled", false) - err = v.ReadInConfig() require.NoError(t, err) @@ -2405,57 +84,31 @@ backend = ["app.example.com"] svc := NewService(v, eventBus) ctx := testContext() + // Initial load err = svc.Load(ctx) require.NoError(t, err) + assert.Equal(t, 8080, svc.GetServerPort()) - err = svc.AddRoute(ctx, domain.Route{Domain: "new.example.com", Image: "newapp:latest"}) - require.NoError(t, err) - - v2 := viper.New() - v2.SetConfigFile(configFile) - err = v2.ReadInConfig() + // Modify config file on disk + updatedConfig := `[server] +port = 9090 +` + err = os.WriteFile(configFile, []byte(updatedConfig), 0600) require.NoError(t, err) - assert.Equal(t, 8088, v2.GetInt("server.port")) - assert.Equal(t, 8443, v2.GetInt("server.tls_port")) - assert.Equal(t, "reg.example.com", v2.GetString("server.registry_domain")) - assert.Equal(t, "auto", v2.GetString("server.runtime")) - assert.Equal(t, "pass", v2.GetString("auth.secrets_backend")) - assert.Equal(t, "testadmin", v2.GetString("auth.username")) - assert.Equal(t, "test/auth/password_hash", v2.GetString("auth.password_hash")) - assert.Equal(t, "test/auth/token_secret", v2.GetString("auth.token_secret")) - assert.Equal(t, "password", v2.GetString("auth.type")) - assert.True(t, v2.GetBool("auto_route.enabled")) - assert.True(t, v2.GetBool("network_isolation.enabled")) - assert.Equal(t, ".internal", v2.GetString("network_isolation.dns_suffix")) - assert.True(t, v2.GetBool("backups.enabled")) - assert.Equal(t, "~/.gordon/backups", v2.GetString("backups.storage_dir")) - - savedRoutes, err := loadCanonicalRoutes(v2.Get("routes")) + // Reload should pick up new values + err = svc.Reload(ctx) require.NoError(t, err) - routeCfg := requireRoute(t, savedRoutes, "new.example.com") - assert.Equal(t, "newapp:latest", routeCfg.Image) - assert.False(t, routeCfg.HTTPS) - routeCfg = requireRoute(t, savedRoutes, "app.example.com") - assert.Equal(t, "reg.example.com/myapp:latest", routeCfg.Image) - assert.True(t, routeCfg.HTTPS) - matches, err := filepath.Glob(configFile + ".bak.*") - require.NoError(t, err) - assert.NotEmpty(t, matches) + assert.Equal(t, 9090, svc.GetServerPort()) }) - t.Run("backup created before write", func(t *testing.T) { + t.Run("error - config file not found", func(t *testing.T) { tmpDir := t.TempDir() configFile := filepath.Join(tmpDir, "gordon.toml") - initialConfig := ` -[server] -port = 9999 -[routes] -"app.example.com" = "myapp:latest" -` - err := os.WriteFile(configFile, []byte(initialConfig), 0600) + // Create config, load it, then delete it + err := os.WriteFile(configFile, []byte("[server]\nport = 8080\n"), 0600) require.NoError(t, err) v := viper.New() @@ -2466,41 +119,28 @@ port = 9999 eventBus := mocks.NewMockEventPublisher(t) svc := NewService(v, eventBus) ctx := testContext() - err = svc.Load(ctx) - require.NoError(t, err) - err = svc.AddRoute(ctx, domain.Route{Domain: "new.example.com", Image: "newapp:latest"}) + err = svc.Load(ctx) require.NoError(t, err) - matches, err := filepath.Glob(configFile + ".bak.*") + // Delete the config file + err = os.Remove(configFile) require.NoError(t, err) - require.NotEmpty(t, matches) - backupData, err := os.ReadFile(matches[0]) - require.NoError(t, err) - assert.Contains(t, string(backupData), "port = 9999") - assert.NotContains(t, string(backupData), "new.example.com") + // Reload should fail + err = svc.Reload(ctx) + assert.Error(t, err) + assert.Contains(t, err.Error(), "failed to read config file") }) - t.Run("old backups are cleaned up keeping 5", func(t *testing.T) { + t.Run("error - invalid config syntax", func(t *testing.T) { tmpDir := t.TempDir() configFile := filepath.Join(tmpDir, "gordon.toml") - initialConfig := ` -[server] -port = 9999 -[routes] -"app.example.com" = "myapp:latest" -` - err := os.WriteFile(configFile, []byte(initialConfig), 0600) + // Create valid config initially + err := os.WriteFile(configFile, []byte("[server]\nport = 8080\n"), 0600) require.NoError(t, err) - for i := 1000000000000; i < 1000000000007; i++ { - backupPath := fmt.Sprintf("%s.bak.%d", configFile, i) - err := os.WriteFile(backupPath, []byte("old backup"), 0600) - require.NoError(t, err) - } - v := viper.New() v.SetConfigFile(configFile) err = v.ReadInConfig() @@ -2509,275 +149,73 @@ port = 9999 eventBus := mocks.NewMockEventPublisher(t) svc := NewService(v, eventBus) ctx := testContext() + err = svc.Load(ctx) require.NoError(t, err) - err = svc.AddRoute(ctx, domain.Route{Domain: "new.example.com", Image: "newapp:latest"}) + // Write invalid TOML + err = os.WriteFile(configFile, []byte("[server\nport = invalid syntax"), 0600) require.NoError(t, err) - matches, err := filepath.Glob(configFile + ".bak.*") - require.NoError(t, err) - assert.Equal(t, 5, len(matches)) + // Reload should fail + err = svc.Reload(ctx) + assert.Error(t, err) }) +} - t.Run("AddRoute updates existing dotted domain route", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - initialConfig := ` -[server] -port = 8088 - -[auth] -username = "testadmin" -token_secret = "testsecret123" - -[routes] -"pitlane.example.com" = "pitlane:latest" -` - err := os.WriteFile(configFile, []byte(initialConfig), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - err = v.ReadInConfig() - require.NoError(t, err) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - err = svc.Load(ctx) - require.NoError(t, err) - - err = svc.AddRoute(ctx, domain.Route{ - Domain: "pitlane.example.com", - Image: "reg.example.com/pitlane:latest", +func TestLoadExternalRoutes_RejectsInvalidAndDuplicateKeys(t *testing.T) { + t.Run("invalid", func(t *testing.T) { + _, err := loadExternalRoutes(map[string]any{ + "bad.example.com:8080": "203.0.113.10:5000", }) - require.NoError(t, err, "AddRoute should succeed when updating an existing dotted-domain route") - - routes := svc.GetRoutes(ctx) - require.Len(t, routes, 1) - assert.Equal(t, "pitlane.example.com", routes[0].Domain) - assert.Equal(t, "reg.example.com/pitlane:latest", routes[0].Image) - - v2 := viper.New() - v2.SetConfigFile(configFile) - err = v2.ReadInConfig() - require.NoError(t, err) - assert.Equal(t, 8088, v2.GetInt("server.port")) - assert.Equal(t, "testadmin", v2.GetString("auth.username")) - assert.Equal(t, "testsecret123", v2.GetString("auth.token_secret")) - }) - - t.Run("RemoveRoute removes dotted domain route", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - initialConfig := ` -[server] -port = 8088 - -[auth] -username = "testadmin" -token_secret = "testsecret123" - -[routes] -"pitlane.example.com" = "reg.example.com/pitlane:latest" -"other.example.com" = "reg.example.com/other:latest" -` - err := os.WriteFile(configFile, []byte(initialConfig), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - err = v.ReadInConfig() - require.NoError(t, err) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - err = svc.Load(ctx) - require.NoError(t, err) - - err = svc.RemoveRoute(ctx, "pitlane.example.com") - require.NoError(t, err, "RemoveRoute should succeed for a dotted-domain route") - routes := svc.GetRoutes(ctx) - require.Len(t, routes, 1) - assert.Equal(t, "other.example.com", routes[0].Domain, "other routes should be preserved") - assert.Equal(t, "reg.example.com/other:latest", routes[0].Image, "other routes should be preserved") - - v2 := viper.New() - v2.SetConfigFile(configFile) - err = v2.ReadInConfig() - require.NoError(t, err) - assert.Equal(t, 8088, v2.GetInt("server.port")) - assert.Equal(t, "testadmin", v2.GetString("auth.username")) - assert.Equal(t, "testsecret123", v2.GetString("auth.token_secret")) + require.Error(t, err) + assert.Contains(t, err.Error(), "invalid external route key") }) - t.Run("network_groups are persisted to disk", func(t *testing.T) { - // Setup: config file with auth, server, and network_groups - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - initialConfig := ` -[server] -port = 9999 - -[auth] -username = "admin" -token_secret = "supersecretvalue" - -[routes] -"app.example.com" = "myapp:latest" - -[network_groups] -frontend = ["app1.example.com", "app2.example.com"] -` - err := os.WriteFile(configFile, []byte(initialConfig), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - err = v.ReadInConfig() - require.NoError(t, err) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - err = svc.Load(ctx) - require.NoError(t, err) - - // Verify network_groups loaded correctly in memory - groups := svc.GetNetworkGroups() - require.Len(t, groups, 1) - require.ElementsMatch(t, []string{"app1.example.com", "app2.example.com"}, groups["frontend"]) - - // Trigger a Save via AddRoute (which calls Save internally) - err = svc.AddRoute(ctx, domain.Route{Domain: "new.example.com", Image: "newapp:latest"}) - require.NoError(t, err) - - // Re-read the file to verify network_groups was persisted - v2 := viper.New() - v2.SetConfigFile(configFile) - err = v2.ReadInConfig() - require.NoError(t, err) - - // Verify network_groups is still intact and correct after save - savedGroups := loadStringArrayMap(v2.Get("network_groups")) - assert.Len(t, savedGroups, 1, "network_groups should still be present in saved config") - assert.ElementsMatch(t, []string{"app1.example.com", "app2.example.com"}, savedGroups["frontend"]) - // Verify auth fields are preserved - assert.Equal(t, "admin", v2.GetString("auth.username")) - assert.Equal(t, "supersecretvalue", v2.GetString("auth.token_secret")) + t.Run("duplicate canonical", func(t *testing.T) { + _, err := loadExternalRoutes(map[string]any{ + "Reg.Example.com": "203.0.113.10:5000", + "reg.example.com": "203.0.113.11:5000", + }) - // Verify server fields are preserved - assert.Equal(t, 9999, v2.GetInt("server.port")) + require.Error(t, err) + assert.Contains(t, err.Error(), "duplicate external route key") }) +} - // Regression test: external_routes keys contain dots (e.g. "reg.example.com"). - // Without explicitly calling viper.Set("external_routes", ...) before WriteConfig, - // viper splits dotted keys into nested subtrees, corrupting the data on re-read. - t.Run("external_routes with dotted domain keys are not corrupted on Save", func(t *testing.T) { - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - initialConfig := ` -[server] -port = 8080 - -[routes] -"app.example.com" = "myapp:latest" - -[external_routes] -"reg.example.com" = "localhost:5000" -` - err := os.WriteFile(configFile, []byte(initialConfig), 0600) - require.NoError(t, err) - - v := viper.New() - v.SetConfigFile(configFile) - err = v.ReadInConfig() - require.NoError(t, err) - - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() - - err = svc.Load(ctx) - require.NoError(t, err) - - // Verify external_routes loaded correctly in memory - extRoutes := svc.GetExternalRoutes() - require.Equal(t, "localhost:5000", extRoutes["reg.example.com"]) - - // Trigger Save - err = svc.AddRoute(ctx, domain.Route{Domain: "new.example.com", Image: "newapp:latest"}) - require.NoError(t, err) - - // Re-read the file with a fresh viper instance - v2 := viper.New() - v2.SetConfigFile(configFile) - err = v2.ReadInConfig() - require.NoError(t, err) - - // external_routes must survive Save with correct flat map structure. - // Without the fix, viper splits "reg.example.com" into nested TOML subtrees - // and GetString("external_routes.reg\\.example\\.com") returns empty string. - savedExtRoutes := loadStringMap(v2.Get("external_routes")) - assert.Equal(t, "localhost:5000", savedExtRoutes["reg.example.com"], - "external_routes must not be corrupted by viper's dot-path splitting on WriteConfig") +func TestService_GetExternalRoutes(t *testing.T) { + v := viper.New() + v.Set("external_routes", map[string]interface{}{ + "Reg.Example.com": "localhost:5000", + "cache.example.com": "127.0.0.1:6379", }) - t.Run("network_groups added in memory are written to disk on Save", func(t *testing.T) { - // Setup: config file WITHOUT network_groups - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - initialConfig := ` -[server] -port = 8080 - -[routes] -"app.example.com" = "myapp:latest" -` - err := os.WriteFile(configFile, []byte(initialConfig), 0600) - require.NoError(t, err) + eventBus := mocks.NewMockEventPublisher(t) + svc := NewService(v, eventBus) + ctx := testContext() - v := viper.New() - v.SetConfigFile(configFile) - err = v.ReadInConfig() - require.NoError(t, err) + _ = svc.Load(ctx) - // Inject network_groups into viper as if they were loaded from some other source - // This simulates a case where network_groups exist in memory but not on disk - v.Set("network_groups", map[string]interface{}{ - "backend": []interface{}{"api.example.com"}, - }) + routes := svc.GetExternalRoutes() - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(v, eventBus) - ctx := testContext() + assert.Len(t, routes, 2) + assert.Equal(t, "localhost:5000", routes["reg.example.com"]) + assert.Equal(t, "127.0.0.1:6379", routes["cache.example.com"]) +} - err = svc.Load(ctx) - require.NoError(t, err) +func TestService_GetExternalRoutes_Empty(t *testing.T) { + v := viper.New() - // Verify network_groups is loaded in memory - groups := svc.GetNetworkGroups() - require.Len(t, groups, 1) + eventBus := mocks.NewMockEventPublisher(t) + svc := NewService(v, eventBus) + ctx := testContext() - // Trigger Save - err = svc.AddRoute(ctx, domain.Route{Domain: "new.example.com", Image: "newapp:latest"}) - require.NoError(t, err) + _ = svc.Load(ctx) - // Re-read the file - v2 := viper.New() - v2.SetConfigFile(configFile) - err = v2.ReadInConfig() - require.NoError(t, err) + routes := svc.GetExternalRoutes() - // network_groups must be present in the saved file - savedGroups := loadStringArrayMap(v2.Get("network_groups")) - assert.Len(t, savedGroups, 1, "network_groups added in memory must be written on Save") - assert.ElementsMatch(t, []string{"api.example.com"}, savedGroups["backend"]) - }) + assert.Empty(t, routes) } func TestExtractDomainFromImageName(t *testing.T) { @@ -2880,50 +318,3 @@ func TestLoadStringMap(t *testing.T) { }) } } - -func TestLoadStringArrayMap(t *testing.T) { - tests := []struct { - name string - input any - expected map[string][]string - }{ - { - name: "nil input", - input: nil, - expected: map[string][]string{}, - }, - { - name: "valid map", - input: map[string]any{ - "group1": []any{"a", "b"}, - "group2": []any{"c", "d", "e"}, - }, - expected: map[string][]string{ - "group1": {"a", "b"}, - "group2": {"c", "d", "e"}, - }, - }, - { - name: "invalid type", - input: "not a map", - expected: map[string][]string{}, - }, - { - name: "non-array value", - input: map[string]any{ - "array": []any{"a", "b"}, - "string": "not an array", - }, - expected: map[string][]string{ - "array": {"a", "b"}, - }, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - result := loadStringArrayMap(tt.input) - assert.Equal(t, tt.expected, result) - }) - } -} diff --git a/internal/usecase/container/autoroute.go b/internal/usecase/container/autoroute.go deleted file mode 100644 index 4c74b3894..000000000 --- a/internal/usecase/container/autoroute.go +++ /dev/null @@ -1,473 +0,0 @@ -package container - -import ( - "bytes" - "context" - "fmt" - "os" - "path/filepath" - "sort" - "strings" - - "github.com/bnema/zerowrap" - - "github.com/bnema/gordon/internal/boundaries/in" - "github.com/bnema/gordon/internal/boundaries/out" - "github.com/bnema/gordon/internal/domain" - "github.com/bnema/gordon/internal/usecase/auto" -) - -// EnvFileExtractor defines the interface for extracting env files from images. -type EnvFileExtractor interface { - ExtractEnvFileFromImage(ctx context.Context, imageRef, envFilePath string) ([]byte, error) -} - -type registryDomainsProvider interface { - GetRegistryDomain() string - GetLegacyRegistryDomains() []string -} - -// AutoRouteHandler handles image.pushed events for auto-route from labels. -type AutoRouteHandler struct { - configSvc in.ConfigService - containerSvc in.ContainerService - blobStorage out.BlobStorage - extractor EnvFileExtractor - registryDomain string - legacyRegistryDomains []string - envDir string - ctx context.Context -} - -// NewAutoRouteHandler creates a new AutoRouteHandler. -func NewAutoRouteHandler( - ctx context.Context, - configSvc in.ConfigService, - containerSvc in.ContainerService, - blobStorage out.BlobStorage, - registryDomain string, - legacyRegistryDomains ...string, -) *AutoRouteHandler { - return &AutoRouteHandler{ - configSvc: configSvc, - containerSvc: containerSvc, - blobStorage: blobStorage, - registryDomain: registryDomain, - legacyRegistryDomains: append([]string(nil), legacyRegistryDomains...), - ctx: ctx, - } -} - -// WithEnvExtractor sets the env file extractor for auto env-file copy feature. -func (h *AutoRouteHandler) WithEnvExtractor(extractor EnvFileExtractor, envDir string) *AutoRouteHandler { - h.extractor = extractor - h.envDir = envDir - return h -} - -// Handle handles an image.pushed event and creates routes from labels. -func (h *AutoRouteHandler) Handle(ctx context.Context, event domain.Event) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldHandler: "AutoRouteHandler", - zerowrap.FieldEvent: string(event.Type), - "event_id": event.ID, - }) - log := zerowrap.FromCtx(ctx) - - if !h.configSvc.IsAutoRouteEnabled() { - log.Debug().Msg("auto-route disabled, skipping label extraction") - return nil - } - - payload, ok := event.Data.(domain.ImagePushedPayload) - if !ok { - log.Warn().Msg("invalid event payload type") - return nil - } - - if len(payload.Manifest) == 0 { - log.Debug().Msg("no manifest data in event, skipping") - return nil - } - - log.Info(). - Str("image", payload.Name). - Str("reference", payload.Reference). - Msg("processing image for auto-route labels") - - labels, err := h.extractLabels(ctx, payload.Manifest) - if err != nil { - log.Debug().Err(err).Msg("failed to extract labels, skipping auto-route") - return nil - } - - domains := h.collectDomains(labels) - if len(domains) == 0 { - log.Debug().Str("image", payload.Name).Msg("no gordon.domain label found, skipping") - return nil - } - - imageName := h.buildImageName(payload.Name, payload.Reference) - h.processRoutes(ctx, domains, imageName, labels) - - return nil -} - -// collectDomains collects all domains from labels. -func (h *AutoRouteHandler) collectDomains(labels *domain.ImageLabels) []string { - var domains []string - if labels.Domain != "" { - domains = append(domains, labels.Domain) - } - domains = append(domains, labels.Domains...) - return domains -} - -// buildImageName constructs the full image name with tag or digest. -// For digest references (sha256:...), uses @ separator per Docker spec. -// For tags, uses : separator. -func (h *AutoRouteHandler) buildImageName(name, reference string) string { - if reference == "" { - return name - } - // Digest references use @ separator, tags use : - if strings.HasPrefix(reference, "sha256:") { - return fmt.Sprintf("%s@%s", name, reference) - } - return fmt.Sprintf("%s:%s", name, reference) -} - -func (h *AutoRouteHandler) registryDomains() (string, []string) { - provider, ok := h.configSvc.(registryDomainsProvider) - if !ok { - return h.registryDomain, append([]string(nil), h.legacyRegistryDomains...) - } - return provider.GetRegistryDomain(), provider.GetLegacyRegistryDomains() -} - -// buildFullImageRef constructs a fully qualified image reference with registry domain. -// This is needed for Docker operations since images are stored with the registry prefix. -func (h *AutoRouteHandler) buildFullImageRef(imageName string) string { - registryDomain, _ := h.registryDomains() - if registryDomain == "" { - return imageName - } - - // Don't prefix if already has registry domain - if strings.HasPrefix(imageName, registryDomain+"/") { - return imageName - } - - // Don't prefix if it's an external registry reference (contains dots and slashes) - repoPart := strings.Split(imageName, ":")[0] - if strings.Contains(repoPart, "@") { - repoPart = strings.Split(imageName, "@")[0] - } - if strings.Contains(repoPart, ".") && strings.Contains(repoPart, "/") { - return imageName - } - - return fmt.Sprintf("%s/%s", registryDomain, imageName) -} - -// processRoutes processes each domain and creates/updates routes. -func (h *AutoRouteHandler) processRoutes(ctx context.Context, domains []string, imageName string, labels *domain.ImageLabels) { - log := zerowrap.FromCtx(ctx) - - // Load allowed domains once for all routes in this event. - allowedDomains, err := h.configSvc.GetAutoRouteAllowedDomains(ctx) - if err != nil { - log.Warn().Err(err).Msg("failed to load auto-route allowed domains, skipping route processing") - return - } - - // Build full image reference for env extraction (Docker needs fully qualified name) - fullImageRef := h.buildFullImageRef(imageName) - - for _, routeDomain := range domains { - routeDomain = strings.TrimSpace(routeDomain) - if routeDomain == "" { - continue - } - - canonicalDomain, ok := domain.CanonicalRouteDomain(routeDomain) - if !ok { - log.Warn().Str("domain", routeDomain).Msg("auto-route rejected invalid domain") - continue - } - routeDomain = canonicalDomain - - created := h.createOrUpdateRoute(ctx, routeDomain, imageName, allowedDomains) - - if created && labels.EnvFile != "" && h.extractor != nil && h.envDir != "" { - // Use full image ref for extraction (Docker knows image by this name) - if err := h.extractAndMergeEnvFile(ctx, fullImageRef, routeDomain, labels.EnvFile); err != nil { - log.Warn(). - Err(err). - Str("domain", routeDomain). - Str("env_file", labels.EnvFile). - Msg("failed to extract env file from image") - } - } - } -} - -// createOrUpdateRoute creates a new route or updates an existing one. -// It also triggers deployment for new routes since the ImagePushedHandler -// may have already finished processing (handlers run concurrently). -func (h *AutoRouteHandler) createOrUpdateRoute(ctx context.Context, routeDomain, imageName string, allowedDomains []string) bool { - log := zerowrap.FromCtx(ctx) - - route := domain.Route{ - Domain: routeDomain, - Image: imageName, - HTTPS: true, - } - - registryDomain, legacyRegistryDomains := h.registryDomains() - existingRoutes := h.configSvc.GetRoutes(ctx) - for _, existing := range existingRoutes { - if existing.Domain == routeDomain { - route.HTTPS = existing.HTTPS - if auto.ExtractRepoName(existing.Image, registryDomain, legacyRegistryDomains...) != auto.ExtractRepoName(imageName, registryDomain, legacyRegistryDomains...) { - log.Warn().Str("domain", routeDomain).Str("existing_image", existing.Image).Str("image", imageName).Msg("auto-route update rejected due to repository ownership mismatch") - return false - } - if existing.Image != imageName { - if err := h.configSvc.UpdateRoute(ctx, route); err != nil { - log.Warn().Err(err).Str("domain", routeDomain).Msg("failed to update route") - } else { - log.Info().Str("domain", routeDomain).Str("image", imageName).Msg("auto-route updated from image labels") - // Trigger deploy for updated route - h.triggerDeploy(ctx, route) - } - } - return false - } - } - - if !auto.MatchesDomainAllowlist(routeDomain, allowedDomains) { - log.Warn().Str("domain", routeDomain).Strs("allowed_domains", allowedDomains).Msg("auto-route create rejected due to domain allowlist") - return false - } - - if err := h.configSvc.AddRoute(ctx, route); err != nil { - log.Warn().Err(err).Str("domain", routeDomain).Msg("failed to add route") - } else { - log.Info().Str("domain", routeDomain).Str("image", imageName).Msg("auto-route added from image labels") - // Trigger deploy for new route since ImagePushedHandler may have already finished - h.triggerDeploy(ctx, route) - return true - } - - return false -} - -// triggerDeploy initiates deployment for a route. -func (h *AutoRouteHandler) triggerDeploy(ctx context.Context, route domain.Route) { - log := zerowrap.FromCtx(ctx) - - if h.containerSvc == nil { - log.Debug().Str("domain", route.Domain).Msg("no container service, skipping deploy trigger") - return - } - - // Mark context as internal deploy - the event originated from our own registry, - // so we can use internal registry (localhost) for image pulls. - internalCtx := domain.WithInternalDeploy(ctx) - - if _, err := h.containerSvc.Deploy(internalCtx, route); err != nil { - log.Warn().Err(err).Str("domain", route.Domain).Msg("failed to trigger deploy for auto-route") - } else { - log.Info().Str("domain", route.Domain).Msg("deploy triggered for auto-route") - } -} - -func allowedAbsoluteEnvFileRoots() []string { - return []string{"/app", "/workspace", "/usr/src/app"} -} - -func validateAutoRouteEnvFilePath(envFilePath string) (string, error) { - envFilePath = strings.TrimSpace(envFilePath) - if envFilePath == "" { - return "", fmt.Errorf("env file path cannot be empty") - } - if strings.Contains(envFilePath, "\x00") { - return "", fmt.Errorf("env file path contains NUL byte") - } - - cleaned := filepath.Clean(envFilePath) - if cleaned == "." || cleaned == string(filepath.Separator) || cleaned == ".." || strings.HasPrefix(cleaned, ".."+string(filepath.Separator)) { - return "", fmt.Errorf("unsafe env file path: %s", envFilePath) - } - if filepath.IsAbs(cleaned) && !pathHasAllowedRoot(cleaned, allowedAbsoluteEnvFileRoots()) { - return "", fmt.Errorf("unsafe env file path: %s", envFilePath) - } - return cleaned, nil -} - -func pathHasAllowedRoot(path string, roots []string) bool { - for _, root := range roots { - if path == root || strings.HasPrefix(path, root+string(filepath.Separator)) { - return true - } - } - return false -} - -// extractAndMergeEnvFile extracts an env file from the image and merges it with existing secrets. -func (h *AutoRouteHandler) extractAndMergeEnvFile(ctx context.Context, imageRef, routeDomain, envFilePath string) error { - log := zerowrap.FromCtx(ctx) - - log.Info(). - Str("domain", routeDomain). - Str("image", imageRef). - Str("env_file", envFilePath). - Msg("extracting env file from image") - - safeEnvFilePath, err := validateAutoRouteEnvFilePath(envFilePath) - if err != nil { - return err - } - - // Extract env file from image - envData, err := h.extractor.ExtractEnvFileFromImage(ctx, imageRef, safeEnvFilePath) - if err != nil { - return fmt.Errorf("failed to extract env file: %w", err) - } - - if len(envData) == 0 { - log.Debug().Msg("env file is empty, skipping") - return nil - } - - // Parse the extracted env file - imageEnv, err := domain.ParseEnvData(envData) - if err != nil { - return fmt.Errorf("failed to parse env file: %w", err) - } - - if len(imageEnv) == 0 { - log.Debug().Msg("no environment variables found in env file") - return nil - } - - if err := rejectSecretReferences(imageEnv); err != nil { - return err - } - - // Load existing env file for this domain - envFileName, err := domainToEnvFileName(routeDomain) - if err != nil { - return fmt.Errorf("invalid env storage domain: %w", err) - } - envFileDst := filepath.Join(h.envDir, envFileName) - - existingEnv := make(map[string]string) - if data, err := os.ReadFile(envFileDst); err == nil { - var parseErr error - existingEnv, parseErr = domain.ParseEnvData(data) - if parseErr != nil { - return fmt.Errorf("failed to parse existing env file %q: %w", envFileDst, parseErr) - } - } else if !os.IsNotExist(err) { - return fmt.Errorf("failed to read existing env file %q: %w", envFileDst, err) - } - - // Merge: image values are defaults, existing values win - merged := make(map[string]string) - for k, v := range imageEnv { - merged[k] = v - } - for k, v := range existingEnv { - merged[k] = v // Existing values override image values - } - - // Write merged env file - if err := os.MkdirAll(h.envDir, 0700); err != nil { - return fmt.Errorf("failed to create env directory: %w", err) - } - - if err := writeEnvFile(envFileDst, merged); err != nil { - return fmt.Errorf("failed to write env file: %w", err) - } - - log.Info(). - Str("domain", routeDomain). - Int("image_vars", len(imageEnv)). - Int("existing_vars", len(existingEnv)). - Int("merged_vars", len(merged)). - Msg("env file extracted and merged from image") - - return nil -} - -// CanHandle returns whether this handler can handle the given event type. -func (h *AutoRouteHandler) CanHandle(eventType domain.EventType) bool { - return eventType == domain.EventImagePushed -} - -// extractLabels extracts Gordon labels from an image manifest. -func (h *AutoRouteHandler) extractLabels(ctx context.Context, manifestData []byte) (*domain.ImageLabels, error) { - log := zerowrap.FromCtx(ctx) - - labels, err := auto.ExtractLabels(ctx, manifestData, h.blobStorage) - if err != nil { - return nil, err - } - - // Log the config digest for observability (best-effort parse for logging only) - if digest, parseErr := auto.ParseConfigDigest(manifestData); parseErr == nil && digest != "" { - log.Debug().Str("config_digest", digest).Msg("found config digest") - } - - return labels, nil -} - -// domainToEnvFileName converts a domain to an env file name. -// Must match the naming convention in envloader.FileLoader.getEnvFilePath. -func domainToEnvFileName(domainName string) (string, error) { - storageKey, err := domain.NewEnvStorageKey(domainName) - if err != nil { - return "", err - } - - return storageKey.FileName(), nil -} - -func rejectSecretReferences(env map[string]string) error { - // Security check: reject imported env files containing secret references - // to prevent attacker-controlled images from persisting ${provider:path} - // syntax that would later resolve against host secret providers. - for key, value := range env { - if domain.ContainsSecretReference(value) { - return fmt.Errorf("env key %q contains secret reference: %w", key, domain.ErrEnvContainsSecretRef) - } - } - return nil -} - -// writeEnvFile writes environment variables to a file. -func writeEnvFile(path string, env map[string]string) error { - var buf bytes.Buffer - - // Sort keys for consistent output - keys := make([]string, 0, len(env)) - for k := range env { - keys = append(keys, k) - } - sort.Strings(keys) - - // Write each key-value pair - for _, k := range keys { - v := env[k] - // Quote values that contain special characters - if strings.ContainsAny(v, " \t\n\"'$\\") { - v = fmt.Sprintf("\"%s\"", strings.ReplaceAll(v, "\"", "\\\"")) - } - fmt.Fprintf(&buf, "%s=%s\n", k, v) - } - - return os.WriteFile(path, buf.Bytes(), 0600) -} diff --git a/internal/usecase/container/autoroute_test.go b/internal/usecase/container/autoroute_test.go deleted file mode 100644 index 0836318cb..000000000 --- a/internal/usecase/container/autoroute_test.go +++ /dev/null @@ -1,1322 +0,0 @@ -package container - -import ( - "context" - "encoding/json" - "io" - "os" - "path/filepath" - "strings" - "testing" - - "github.com/bnema/zerowrap" - "github.com/spf13/viper" - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/mock" - "github.com/stretchr/testify/require" - - inmocks "github.com/bnema/gordon/internal/boundaries/in/mocks" - "github.com/bnema/gordon/internal/boundaries/out/mocks" - "github.com/bnema/gordon/internal/domain" - "github.com/bnema/gordon/internal/usecase/auto" - configusecase "github.com/bnema/gordon/internal/usecase/config" -) - -// domainToEnvFileName tests - -func TestDomainToEnvFileName(t *testing.T) { - tests := []struct { - name string - domain string - expected string - }{ - { - name: "simple domain", - domain: "app.example.com", - expected: "YXBwLmV4YW1wbGUuY29t.env", - }, - { - name: "domain with port", - domain: "app.example.com:8080", - expected: "YXBwLmV4YW1wbGUuY29tOjgwODA.env", - }, - { - name: "domain with path", - domain: "app.example.com/api", - expected: "YXBwLmV4YW1wbGUuY29tL2FwaQ.env", - }, - { - name: "complex domain", - domain: "sub.app.example.com:443/v1", - expected: "c3ViLmFwcC5leGFtcGxlLmNvbTo0NDMvdjE.env", - }, - { - name: "single word", - domain: "localhost", - expected: "bG9jYWxob3N0.env", - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - result, err := domainToEnvFileName(tt.domain) - require.NoError(t, err) - assert.Equal(t, tt.expected, result) - }) - } -} - -// ParseEnvData tests - -func TestParseEnvFile(t *testing.T) { - tests := []struct { - name string - input string - expected map[string]string - wantErr bool - }{ - { - name: "simple key-value pairs", - input: "FOO=bar\nBAZ=qux", - expected: map[string]string{ - "FOO": "bar", - "BAZ": "qux", - }, - }, - { - name: "with comments and empty lines", - input: "# This is a comment\nFOO=bar\n\n# Another comment\nBAZ=qux\n", - expected: map[string]string{ - "FOO": "bar", - "BAZ": "qux", - }, - }, - { - name: "with quoted values - double quotes", - input: `FOO="bar baz"`, - expected: map[string]string{ - "FOO": "bar baz", - }, - }, - { - name: "with quoted values - single quotes", - input: `FOO='bar baz'`, - expected: map[string]string{ - "FOO": "bar baz", - }, - }, - { - name: "value with equals sign", - input: "DATABASE_URL=postgres://user:pass@host/db?option=value", - expected: map[string]string{ - "DATABASE_URL": "postgres://user:pass@host/db?option=value", - }, - }, - { - name: "empty value", - input: "EMPTY=", - expected: map[string]string{ - "EMPTY": "", - }, - }, - { - name: "empty file", - input: "", - expected: map[string]string{}, - }, - { - name: "only comments", - input: "# comment 1\n# comment 2", - expected: map[string]string{}, - }, - { - name: "malformed line without equals", - input: "MALFORMED\nGOOD=value", - expected: map[string]string{"GOOD": "value"}, - }, - { - name: "whitespace trimming", - input: " FOO = bar \n BAZ=qux", - expected: map[string]string{ - "FOO": "bar", - "BAZ": "qux", - }, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - result, err := domain.ParseEnvData([]byte(tt.input)) - - if tt.wantErr { - assert.Error(t, err) - return - } - - require.NoError(t, err) - assert.Equal(t, tt.expected, result) - }) - } -} - -// parseConfigDigest tests — now delegated to auto package; tested there directly. -// These tests call auto.ParseConfigDigest to maintain coverage for this package. - -func TestParseConfigDigest(t *testing.T) { - tests := []struct { - name string - manifest map[string]any - expected string - wantErr bool - }{ - { - name: "valid manifest v2", - manifest: map[string]any{ - "schemaVersion": 2, - "mediaType": "application/vnd.docker.distribution.manifest.v2+json", - "config": map[string]any{ - "mediaType": "application/vnd.docker.container.image.v1+json", - "digest": "sha256:abc123", - "size": 1234, - }, - }, - expected: "sha256:abc123", - }, - { - name: "OCI manifest", - manifest: map[string]any{ - "schemaVersion": 2, - "mediaType": "application/vnd.oci.image.manifest.v1+json", - "config": map[string]any{ - "mediaType": "application/vnd.oci.image.config.v1+json", - "digest": "sha256:def456", - "size": 5678, - }, - }, - expected: "sha256:def456", - }, - { - name: "empty config digest", - manifest: map[string]any{ - "schemaVersion": 2, - "config": map[string]any{ - "digest": "", - }, - }, - expected: "", - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - data, err := json.Marshal(tt.manifest) - require.NoError(t, err) - - result, err := auto.ParseConfigDigest(data) - - if tt.wantErr { - assert.Error(t, err) - return - } - - require.NoError(t, err) - assert.Equal(t, tt.expected, result) - }) - } -} - -func TestParseConfigDigest_InvalidJSON(t *testing.T) { - _, err := auto.ParseConfigDigest([]byte("invalid json")) - assert.Error(t, err) -} - -// parseImageLabels tests — now delegated to auto package; tested there directly. - -func TestParseImageLabels(t *testing.T) { - tests := []struct { - name string - config map[string]any - expected *domain.ImageLabels - }{ - { - name: "single domain label", - config: map[string]any{ - "config": map[string]any{ - "Labels": map[string]any{ - "gordon.domain": "app.example.com", - }, - }, - }, - expected: &domain.ImageLabels{ - Domain: "app.example.com", - }, - }, - { - name: "multiple domains label", - config: map[string]any{ - "config": map[string]any{ - "Labels": map[string]any{ - "gordon.domains": "app1.example.com, app2.example.com", - }, - }, - }, - expected: &domain.ImageLabels{ - Domains: []string{"app1.example.com", "app2.example.com"}, - }, - }, - { - name: "all gordon labels", - config: map[string]any{ - "config": map[string]any{ - "Labels": map[string]any{ - "gordon.domain": "app.example.com", - "gordon.health": "/healthz", - "gordon.proxy.port": "8080", - "gordon.env-file": ".env.prod", - }, - }, - }, - expected: &domain.ImageLabels{ - Domain: "app.example.com", - Health: "/healthz", - Port: "8080", - EnvFile: ".env.prod", - }, - }, - { - name: "no labels", - config: map[string]any{ - "config": map[string]any{ - "Labels": nil, - }, - }, - expected: &domain.ImageLabels{}, - }, - { - name: "no gordon labels", - config: map[string]any{ - "config": map[string]any{ - "Labels": map[string]any{ - "other.label": "value", - }, - }, - }, - expected: &domain.ImageLabels{}, - }, - { - name: "whitespace in labels", - config: map[string]any{ - "config": map[string]any{ - "Labels": map[string]any{ - "gordon.domain": " app.example.com ", - }, - }, - }, - expected: &domain.ImageLabels{ - Domain: "app.example.com", - }, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - data, err := json.Marshal(tt.config) - require.NoError(t, err) - - result, err := auto.ParseImageLabels(data) - require.NoError(t, err) - - assert.Equal(t, tt.expected.Domain, result.Domain) - assert.Equal(t, tt.expected.Domains, result.Domains) - assert.Equal(t, tt.expected.Health, result.Health) - assert.Equal(t, tt.expected.Port, result.Port) - assert.Equal(t, tt.expected.EnvFile, result.EnvFile) - }) - } -} - -func TestParseImageLabels_InvalidJSON(t *testing.T) { - _, err := auto.ParseImageLabels([]byte("invalid json")) - assert.Error(t, err) -} - -func TestMatchesDomainAllowlist(t *testing.T) { - tests := []struct { - name string - domain string - patterns []string - want bool - }{ - {name: "exact match", domain: "app.example.com", patterns: []string{"app.example.com"}, want: true}, - {name: "no match", domain: "app.example.com", patterns: []string{"api.example.com"}, want: false}, - {name: "wildcard match", domain: "app.example.com", patterns: []string{"*.example.com"}, want: true}, - {name: "wildcard no match on root", domain: "example.com", patterns: []string{"*.example.com"}, want: false}, - {name: "wildcard matches one level only", domain: "api.app.example.com", patterns: []string{"*.example.com"}, want: false}, - {name: "empty allowlist", domain: "app.example.com", patterns: nil, want: false}, - {name: "case insensitive", domain: "App.Example.Com", patterns: []string{"APP.EXAMPLE.COM"}, want: true}, - {name: "multiple patterns", domain: "api.example.com", patterns: []string{"foo.com", "*.example.com"}, want: true}, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - assert.Equal(t, tt.want, auto.MatchesDomainAllowlist(tt.domain, tt.patterns)) - }) - } -} - -func TestExtractRepoName(t *testing.T) { - tests := []struct { - name string - imageRef string - registryDomain string - want string - }{ - {name: "simple tag", imageRef: "myapp:latest", want: "myapp"}, - {name: "version tag", imageRef: "myapp:v2.0.0", want: "myapp"}, - {name: "digest", imageRef: "myapp@sha256:abc123", want: "myapp"}, - {name: "strip registry", imageRef: "registry.example.com/myapp:latest", registryDomain: "registry.example.com", want: "myapp"}, - {name: "org image", imageRef: "org/myapp:latest", want: "org/myapp"}, - {name: "registry org image", imageRef: "registry.example.com/org/myapp:latest", registryDomain: "registry.example.com", want: "org/myapp"}, - {name: "lowercase", imageRef: "MyApp:Latest", want: "myapp"}, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - assert.Equal(t, tt.want, auto.ExtractRepoName(tt.imageRef, tt.registryDomain)) - }) - } -} - -// AutoRouteHandler tests - -func TestAutoRouteHandler_CanHandle(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "registry.example.com") - - assert.True(t, handler.CanHandle(domain.EventImagePushed)) - assert.False(t, handler.CanHandle(domain.EventConfigReload)) -} - -func TestAutoRouteHandler_Handle_DisabledAutoRoute(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "registry.example.com") - - configSvc.EXPECT().IsAutoRouteEnabled().Return(false) - - event := domain.Event{ - ID: "event-123", - Type: domain.EventImagePushed, - Data: domain.ImagePushedPayload{ - Name: "myapp", - Reference: "latest", - Manifest: []byte(`{"schemaVersion": 2}`), - }, - } - - err := handler.Handle(context.Background(), event) - - assert.NoError(t, err) -} - -func TestAutoRouteHandler_Handle_InvalidPayload(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "registry.example.com") - - configSvc.EXPECT().IsAutoRouteEnabled().Return(true) - - event := domain.Event{ - ID: "event-123", - Type: domain.EventImagePushed, - Data: "invalid payload type", - } - - err := handler.Handle(context.Background(), event) - - // Handler skips invalid payload, doesn't error - assert.NoError(t, err) -} - -func TestAutoRouteHandler_Handle_EmptyManifest(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "registry.example.com") - - configSvc.EXPECT().IsAutoRouteEnabled().Return(true) - - event := domain.Event{ - ID: "event-123", - Type: domain.EventImagePushed, - Data: domain.ImagePushedPayload{ - Name: "myapp", - Reference: "latest", - Manifest: []byte{}, - }, - } - - err := handler.Handle(context.Background(), event) - - assert.NoError(t, err) -} - -func TestAutoRouteHandler_Handle_CreatesNewRoute(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "registry.example.com") - - // Build manifest with config digest - manifest := map[string]any{ - "schemaVersion": 2, - "config": map[string]any{ - "digest": "sha256:configdigest123", - }, - } - manifestData, _ := json.Marshal(manifest) - - // Build image config with gordon labels - imageConfig := map[string]any{ - "config": map[string]any{ - "Labels": map[string]any{ - "gordon.domain": "app.example.com", - }, - }, - } - configData, _ := json.Marshal(imageConfig) - - configSvc.EXPECT().IsAutoRouteEnabled().Return(true) - blobStorage.EXPECT().GetBlob("sha256:configdigest123").Return(io.NopCloser(strings.NewReader(string(configData))), nil) - configSvc.EXPECT().GetAutoRouteAllowedDomains(mock.Anything).Return([]string{"app.example.com"}, nil) - - // No existing routes - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{}) - - // Expect route to be added - configSvc.EXPECT().AddRoute(mock.Anything, domain.Route{ - Domain: "app.example.com", - Image: "myapp:latest", - HTTPS: true, - }).Return(nil) - - // Expect deploy to be triggered - containerSvc.EXPECT().Deploy(mock.Anything, domain.Route{ - Domain: "app.example.com", - Image: "myapp:latest", - HTTPS: true, - }).Return(&domain.Container{ID: "container-1"}, nil) - - event := domain.Event{ - ID: "event-123", - Type: domain.EventImagePushed, - Data: domain.ImagePushedPayload{ - Name: "myapp", - Reference: "latest", - Manifest: manifestData, - }, - } - - err := handler.Handle(context.Background(), event) - - assert.NoError(t, err) -} - -func TestAutoRouteHandler_Handle_UpdatesExistingRoute(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "registry.example.com") - - // Build manifest with config digest - manifest := map[string]any{ - "schemaVersion": 2, - "config": map[string]any{ - "digest": "sha256:configdigest123", - }, - } - manifestData, _ := json.Marshal(manifest) - - // Build image config with gordon labels - imageConfig := map[string]any{ - "config": map[string]any{ - "Labels": map[string]any{ - "gordon.domain": "app.example.com", - }, - }, - } - configData, _ := json.Marshal(imageConfig) - - configSvc.EXPECT().IsAutoRouteEnabled().Return(true) - blobStorage.EXPECT().GetBlob("sha256:configdigest123").Return(io.NopCloser(strings.NewReader(string(configData))), nil) - configSvc.EXPECT().GetAutoRouteAllowedDomains(mock.Anything).Return([]string{"*"}, nil) - - // Existing route with different image - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{ - {Domain: "app.example.com", Image: "myapp:v1", HTTPS: true}, - }) - - // Expect route to be updated - configSvc.EXPECT().UpdateRoute(mock.Anything, domain.Route{ - Domain: "app.example.com", - Image: "myapp:v2", - HTTPS: true, - }).Return(nil) - - // Expect deploy to be triggered - containerSvc.EXPECT().Deploy(mock.Anything, domain.Route{ - Domain: "app.example.com", - Image: "myapp:v2", - HTTPS: true, - }).Return(&domain.Container{ID: "container-1"}, nil) - - event := domain.Event{ - ID: "event-123", - Type: domain.EventImagePushed, - Data: domain.ImagePushedPayload{ - Name: "myapp", - Reference: "v2", - Manifest: manifestData, - }, - } - - err := handler.Handle(context.Background(), event) - - assert.NoError(t, err) -} - -func TestAutoRouteHandler_Handle_NoUpdateIfSameImage(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "registry.example.com") - - // Build manifest with config digest - manifest := map[string]any{ - "schemaVersion": 2, - "config": map[string]any{ - "digest": "sha256:configdigest123", - }, - } - manifestData, _ := json.Marshal(manifest) - - // Build image config with gordon labels - imageConfig := map[string]any{ - "config": map[string]any{ - "Labels": map[string]any{ - "gordon.domain": "app.example.com", - }, - }, - } - configData, _ := json.Marshal(imageConfig) - - configSvc.EXPECT().IsAutoRouteEnabled().Return(true) - blobStorage.EXPECT().GetBlob("sha256:configdigest123").Return(io.NopCloser(strings.NewReader(string(configData))), nil) - configSvc.EXPECT().GetAutoRouteAllowedDomains(mock.Anything).Return([]string{"*"}, nil) - - // Existing route with same image - no update needed - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{ - {Domain: "app.example.com", Image: "myapp:latest"}, - }) - - // No AddRoute, UpdateRoute, or Deploy calls expected - - event := domain.Event{ - ID: "event-123", - Type: domain.EventImagePushed, - Data: domain.ImagePushedPayload{ - Name: "myapp", - Reference: "latest", - Manifest: manifestData, - }, - } - - err := handler.Handle(context.Background(), event) - - assert.NoError(t, err) -} - -func TestAutoRouteHandler_Handle_NoDomainLabel(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "registry.example.com") - - // Build manifest with config digest - manifest := map[string]any{ - "schemaVersion": 2, - "config": map[string]any{ - "digest": "sha256:configdigest123", - }, - } - manifestData, _ := json.Marshal(manifest) - - // Build image config with no gordon labels - imageConfig := map[string]any{ - "config": map[string]any{ - "Labels": map[string]any{ - "other.label": "value", - }, - }, - } - configData, _ := json.Marshal(imageConfig) - - configSvc.EXPECT().IsAutoRouteEnabled().Return(true) - blobStorage.EXPECT().GetBlob("sha256:configdigest123").Return(io.NopCloser(strings.NewReader(string(configData))), nil) - - // No route operations expected - - event := domain.Event{ - ID: "event-123", - Type: domain.EventImagePushed, - Data: domain.ImagePushedPayload{ - Name: "myapp", - Reference: "latest", - Manifest: manifestData, - }, - } - - err := handler.Handle(context.Background(), event) - - assert.NoError(t, err) -} - -func TestAutoRouteHandler_Handle_MultipleDomains(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "registry.example.com") - - // Build manifest with config digest - manifest := map[string]any{ - "schemaVersion": 2, - "config": map[string]any{ - "digest": "sha256:configdigest123", - }, - } - manifestData, _ := json.Marshal(manifest) - - // Build image config with multiple domains - imageConfig := map[string]any{ - "config": map[string]any{ - "Labels": map[string]any{ - "gordon.domain": "app.example.com", - "gordon.domains": "api.example.com, www.example.com", - }, - }, - } - configData, _ := json.Marshal(imageConfig) - - configSvc.EXPECT().IsAutoRouteEnabled().Return(true) - blobStorage.EXPECT().GetBlob("sha256:configdigest123").Return(io.NopCloser(strings.NewReader(string(configData))), nil) - configSvc.EXPECT().GetAutoRouteAllowedDomains(mock.Anything).Return([]string{"app.example.com", "api.example.com", "www.example.com"}, nil).Once() - - // No existing routes - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{}).Times(3) - - // Expect routes to be added for all three domains - configSvc.EXPECT().AddRoute(mock.Anything, domain.Route{ - Domain: "app.example.com", - Image: "myapp:latest", - HTTPS: true, - }).Return(nil) - configSvc.EXPECT().AddRoute(mock.Anything, domain.Route{ - Domain: "api.example.com", - Image: "myapp:latest", - HTTPS: true, - }).Return(nil) - configSvc.EXPECT().AddRoute(mock.Anything, domain.Route{ - Domain: "www.example.com", - Image: "myapp:latest", - HTTPS: true, - }).Return(nil) - - // Expect deploy for all three - containerSvc.EXPECT().Deploy(mock.Anything, mock.AnythingOfType("domain.Route")).Return(&domain.Container{ID: "container-1"}, nil).Times(3) - - event := domain.Event{ - ID: "event-123", - Type: domain.EventImagePushed, - Data: domain.ImagePushedPayload{ - Name: "myapp", - Reference: "latest", - Manifest: manifestData, - }, - } - - err := handler.Handle(context.Background(), event) - - assert.NoError(t, err) -} - -func TestAutoRouteHandler_TriggerDeploy_NilContainerService(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - blobStorage := mocks.NewMockBlobStorage(t) - - // Create handler without container service - handler := NewAutoRouteHandler(ctx, configSvc, nil, blobStorage, "registry.example.com") - - route := domain.Route{ - Domain: "app.example.com", - Image: "myapp:latest", - } - - // Should not panic, just log and skip - handler.triggerDeploy(ctx, route) -} - -func TestAutoRouteHandler_NewDomain_Allowed(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "registry.example.com") - - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{}) - configSvc.EXPECT().AddRoute(mock.Anything, domain.Route{Domain: "app.example.com", Image: "myapp:latest", HTTPS: true}).Return(nil) - containerSvc.EXPECT().Deploy(mock.Anything, domain.Route{Domain: "app.example.com", Image: "myapp:latest", HTTPS: true}).Return(&domain.Container{ID: "c1"}, nil) - - created := handler.createOrUpdateRoute(context.Background(), "app.example.com", "myapp:latest", []string{"*.example.com"}) - assert.True(t, created) -} - -func TestAutoRouteHandler_NewDomain_NotAllowed(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "registry.example.com") - - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{}) - - created := handler.createOrUpdateRoute(context.Background(), "app.example.com", "myapp:latest", []string{"*.allowed.com"}) - assert.False(t, created) -} - -func TestAutoRouteHandler_ExistingDomain_SameRepo(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "registry.example.com") - - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{{Domain: "app.example.com", Image: "myapp:v1", HTTPS: true}}) - configSvc.EXPECT().UpdateRoute(mock.Anything, domain.Route{Domain: "app.example.com", Image: "myapp:v2", HTTPS: true}).Return(nil) - containerSvc.EXPECT().Deploy(mock.Anything, domain.Route{Domain: "app.example.com", Image: "myapp:v2", HTTPS: true}).Return(&domain.Container{ID: "c1"}, nil) - - created := handler.createOrUpdateRoute(context.Background(), "app.example.com", "myapp:v2", []string{"*.example.com"}) - assert.False(t, created) -} - -func TestAutoRouteHandler_ExistingDomain_SameRepoAcrossLegacyAndCurrentRegistryHosts_UpdatesRoute(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "new-registry.example.com", "old-registry.example.com") - - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{{ - Domain: "app.example.com", - Image: "old-registry.example.com/app:v1", - HTTPS: true, - }}) - configSvc.EXPECT().UpdateRoute(mock.Anything, domain.Route{ - Domain: "app.example.com", - Image: "new-registry.example.com/app:v2", - HTTPS: true, - }).Return(nil) - containerSvc.EXPECT().Deploy(mock.Anything, domain.Route{ - Domain: "app.example.com", - Image: "new-registry.example.com/app:v2", - HTTPS: true, - }).Return(&domain.Container{ID: "c1"}, nil) - - created := handler.createOrUpdateRoute(context.Background(), "app.example.com", "new-registry.example.com/app:v2", []string{"*.example.com"}) - assert.False(t, created) -} - -func TestAutoRouteHandler_UsesRefreshedRegistryDomainsAfterReload(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - tmpDir := t.TempDir() - configFile := filepath.Join(tmpDir, "gordon.toml") - require.NoError(t, os.WriteFile(configFile, []byte(`[server] - gordon_domain = "old-registry.example.com" - - [routes] - "app.example.com" = "old-registry.example.com/app:v1" - `), 0600)) - - v := viper.New() - v.SetConfigFile(configFile) - require.NoError(t, v.ReadInConfig()) - - configSvc := configusecase.NewService(v, mocks.NewMockEventPublisher(t)) - require.NoError(t, configSvc.Load(ctx)) - - containerSvc := inmocks.NewMockContainerService(t) - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, mocks.NewMockBlobStorage(t), "old-registry.example.com") - - require.NoError(t, os.WriteFile(configFile, []byte(`[server] - gordon_domain = "new-registry.example.com" - legacy_registry_domains = ["old-registry.example.com"] - - [routes] - "app.example.com" = "old-registry.example.com/app:v1" - `), 0600)) - require.NoError(t, configSvc.Reload(ctx)) - - containerSvc.EXPECT().Deploy(mock.Anything, domain.Route{ - Domain: "app.example.com", - Image: "new-registry.example.com/app:v2", - HTTPS: true, - }).Return(&domain.Container{ID: "c1"}, nil) - - created := handler.createOrUpdateRoute(context.Background(), "app.example.com", "new-registry.example.com/app:v2", []string{"*.example.com"}) - assert.False(t, created) - - route, err := configSvc.GetRoute(ctx, "app.example.com") - require.NoError(t, err) - assert.Equal(t, "new-registry.example.com/app:v2", route.Image) -} - -func TestAutoRouteHandler_ExistingDomain_DifferentRepo(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "registry.example.com") - - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{{Domain: "app.example.com", Image: "oldapp:v1"}}) - - created := handler.createOrUpdateRoute(context.Background(), "app.example.com", "newapp:v2", []string{"*.example.com"}) - assert.False(t, created) -} - -func TestAutoRouteHandler_ExistingDomain_DifferentRepoAcrossLegacyAndCurrentRegistryHosts_RejectsRoute(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "new-registry.example.com", "old-registry.example.com") - - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{{ - Domain: "app.example.com", - Image: "old-registry.example.com/oldapp:v1", - HTTPS: true, - }}) - - created := handler.createOrUpdateRoute(context.Background(), "app.example.com", "new-registry.example.com/newapp:v2", []string{"*.example.com"}) - assert.False(t, created) -} - -func TestAutoRouteHandler_EnvExtractionOnlyOnCreate(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - extractor := &testEnvExtractor{} - - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "registry.example.com").WithEnvExtractor(extractor, t.TempDir()) - - configSvc.EXPECT().GetAutoRouteAllowedDomains(mock.Anything).Return([]string{"*.example.com"}, nil).Twice() - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{}).Once() - configSvc.EXPECT().AddRoute(mock.Anything, domain.Route{Domain: "app.example.com", Image: "myapp:latest", HTTPS: true}).Return(nil) - containerSvc.EXPECT().Deploy(mock.Anything, domain.Route{Domain: "app.example.com", Image: "myapp:latest", HTTPS: true}).Return(&domain.Container{ID: "c1"}, nil) - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{{Domain: "app.example.com", Image: "myapp:latest"}}).Once() - - handler.processRoutes(context.Background(), []string{"app.example.com"}, "myapp:latest", &domain.ImageLabels{EnvFile: ".env"}) - handler.processRoutes(context.Background(), []string{"app.example.com"}, "myapp:latest", &domain.ImageLabels{EnvFile: ".env"}) - - assert.Equal(t, 1, extractor.calls) - assert.Equal(t, "registry.example.com/myapp:latest", extractor.lastImage) - assert.Equal(t, ".env", extractor.lastPath) -} - -func TestAutoRouteHandler_ExtractAndMergeEnvFile_ReturnsExistingParseError(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - extractor := &testEnvExtractor{data: []byte("FOO=bar\n")} - - envDir := t.TempDir() - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "registry.example.com").WithEnvExtractor(extractor, envDir) - - envFileName, err := domainToEnvFileName("app.example.com") - require.NoError(t, err) - envFileDst := filepath.Join(envDir, envFileName) - err = os.WriteFile(envFileDst, []byte("BIG="+strings.Repeat("a", (1<<20)+1)), 0600) - require.NoError(t, err) - - err = handler.extractAndMergeEnvFile(context.Background(), "myapp:latest", "app.example.com", ".env") - require.Error(t, err) - assert.Contains(t, err.Error(), envFileDst) - assert.Contains(t, err.Error(), "failed to parse existing env file") -} - -type testEnvExtractor struct { - calls int - lastImage string - lastPath string - data []byte - err error -} - -func (e *testEnvExtractor) ExtractEnvFileFromImage(_ context.Context, imageRef, envFilePath string) ([]byte, error) { - e.calls++ - e.lastImage = imageRef - e.lastPath = envFilePath - if e.err != nil { - return nil, e.err - } - if e.data != nil { - return e.data, nil - } - return []byte("FOO=bar\n"), nil -} - -// collectDomains tests - -func TestValidateAutoRouteEnvFilePath(t *testing.T) { - tests := []struct { - name string - path string - want string - wantErr bool - }{ - {name: "relative path", path: ".env", want: ".env"}, - {name: "absolute app path", path: "/app/.env", want: "/app/.env"}, - {name: "absolute workspace path", path: "/workspace/.env", want: "/workspace/.env"}, - {name: "cleans safe path", path: "/app/config/../.env", want: "/app/.env"}, - {name: "absolute traversal escapes app root", path: "/app/../../etc/passwd", wantErr: true}, - {name: "absolute non app path", path: "/etc/passwd", wantErr: true}, - {name: "empty", path: "", wantErr: true}, - {name: "root", path: "/", wantErr: true}, - {name: "traversal", path: "../.env", wantErr: true}, - {name: "proc", path: "/proc/self/environ", wantErr: true}, - {name: "sys", path: "/sys/kernel", wantErr: true}, - {name: "dev", path: "/dev/null", wantErr: true}, - {name: "run", path: "/run/secrets", wantErr: true}, - {name: "var run", path: "/var/run/docker.sock", wantErr: true}, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - got, err := validateAutoRouteEnvFilePath(tt.path) - if tt.wantErr { - assert.Error(t, err) - return - } - require.NoError(t, err) - assert.Equal(t, tt.want, got) - }) - } -} - -func TestAutoRouteHandler_ExtractAndMergeEnvFile_UsesCleanSafeEnvFilePath(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - extractor := &testEnvExtractor{data: []byte("FOO=bar\n")} - handler := NewAutoRouteHandler(ctx, inmocks.NewMockConfigService(t), inmocks.NewMockContainerService(t), mocks.NewMockBlobStorage(t), "registry.example.com"). - WithEnvExtractor(extractor, t.TempDir()) - - err := handler.extractAndMergeEnvFile(ctx, "myapp:latest", "app.example.com", "/app/config/../.env") - require.NoError(t, err) - assert.Equal(t, "/app/.env", extractor.lastPath) -} - -func TestAutoRouteHandler_ExtractAndMergeEnvFile_RejectsSecretReferences(t *testing.T) { - tests := []struct { - name string - envData string - wantErr bool - errContains string - }{ - { - name: "pass reference in value", - envData: "DB_PASSWORD=${pass:myapp/db}\n", - wantErr: true, - errContains: "secret reference", - }, - { - name: "sops reference in value", - envData: "API_KEY=${sops:secrets.yaml#/key}\n", - wantErr: true, - errContains: "secret reference", - }, - { - name: "pass reference embedded in text", - envData: "CONNECTION=prefix-${pass:secret}-suffix\n", - wantErr: true, - errContains: "secret reference", - }, - { - name: "multiple secret references", - envData: "A=${pass:a}\nB=${sops:b}\n", - wantErr: true, - errContains: "secret reference", - }, - { - name: "normal env values are allowed", - envData: "FOO=bar\nBAZ=qux\n", - wantErr: false, - }, - { - name: "dollar sign without reference is allowed", - envData: "PRICE=$100\n", - wantErr: false, - }, - { - name: "empty braces are allowed", - envData: "EMPTY=${}\n", - wantErr: false, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - extractor := &testEnvExtractor{data: []byte(tt.envData)} - - envDir := t.TempDir() - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "registry.example.com"). - WithEnvExtractor(extractor, envDir) - - err := handler.extractAndMergeEnvFile(context.Background(), "myapp:latest", "app.example.com", ".env") - - if tt.wantErr { - assert.Error(t, err) - assert.Contains(t, err.Error(), tt.errContains) - return - } - - assert.NoError(t, err) - }) - } -} - -func TestAutoRouteHandler_ExtractAndMergeEnvFile_DoesNotPersistSecretReferences(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - extractor := &testEnvExtractor{data: []byte("DB_PASSWORD=${pass:myapp/db}\n")} - - envDir := t.TempDir() - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "registry.example.com").WithEnvExtractor(extractor, envDir) - - envFileName, err := domainToEnvFileName("app.example.com") - require.NoError(t, err) - envFileDst := filepath.Join(envDir, envFileName) - - err = handler.extractAndMergeEnvFile(context.Background(), "myapp:latest", "app.example.com", ".env") - require.ErrorIs(t, err, domain.ErrEnvContainsSecretRef) - assert.NoFileExists(t, envFileDst) -} - -func TestAutoRouteHandler_ExtractAndMergeEnvFile_ReturnsReadErrorForExistingFileIssues(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - extractor := &testEnvExtractor{data: []byte("FOO=bar\n")} - - envDir := t.TempDir() - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "registry.example.com").WithEnvExtractor(extractor, envDir) - - envFileName, err := domainToEnvFileName("app.example.com") - require.NoError(t, err) - envFileDst := filepath.Join(envDir, envFileName) - require.NoError(t, os.Mkdir(envFileDst, 0700)) - - err = handler.extractAndMergeEnvFile(context.Background(), "myapp:latest", "app.example.com", ".env") - require.Error(t, err) - assert.Contains(t, err.Error(), envFileDst) - assert.Contains(t, err.Error(), "failed to read existing env file") -} - -func TestAutoRouteHandler_CollectDomains(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "registry.example.com") - - tests := []struct { - name string - labels *domain.ImageLabels - expected []string - }{ - { - name: "single domain", - labels: &domain.ImageLabels{ - Domain: "app.example.com", - }, - expected: []string{"app.example.com"}, - }, - { - name: "domains list only", - labels: &domain.ImageLabels{ - Domains: []string{"api.example.com", "www.example.com"}, - }, - expected: []string{"api.example.com", "www.example.com"}, - }, - { - name: "both domain and domains", - labels: &domain.ImageLabels{ - Domain: "app.example.com", - Domains: []string{"api.example.com"}, - }, - expected: []string{"app.example.com", "api.example.com"}, - }, - { - name: "empty labels", - labels: &domain.ImageLabels{}, - expected: nil, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - result := handler.collectDomains(tt.labels) - assert.Equal(t, tt.expected, result) - }) - } -} - -// buildImageName tests - -func TestAutoRouteHandler_BuildImageName(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, "registry.example.com") - - tests := []struct { - name string - imageName string - reference string - expected string - }{ - { - name: "with tag", - imageName: "myapp", - reference: "latest", - expected: "myapp:latest", - }, - { - name: "with version tag", - imageName: "myapp", - reference: "v1.2.3", - expected: "myapp:v1.2.3", - }, - { - name: "empty reference", - imageName: "myapp", - reference: "", - expected: "myapp", - }, - { - name: "with digest", - imageName: "myapp", - reference: "sha256:abc123", - expected: "myapp@sha256:abc123", - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - result := handler.buildImageName(tt.imageName, tt.reference) - assert.Equal(t, tt.expected, result) - }) - } -} - -// buildFullImageRef tests - -func TestAutoRouteHandler_BuildFullImageRef(t *testing.T) { - ctx := zerowrap.WithCtx(context.Background(), zerowrap.Default()) - configSvc := inmocks.NewMockConfigService(t) - containerSvc := inmocks.NewMockContainerService(t) - blobStorage := mocks.NewMockBlobStorage(t) - - tests := []struct { - name string - registryDomain string - imageName string - expected string - }{ - { - name: "simple image with registry", - registryDomain: "registry.example.com", - imageName: "myapp:latest", - expected: "registry.example.com/myapp:latest", - }, - { - name: "already has registry prefix", - registryDomain: "registry.example.com", - imageName: "registry.example.com/myapp:latest", - expected: "registry.example.com/myapp:latest", - }, - { - name: "external registry reference", - registryDomain: "registry.example.com", - imageName: "docker.io/library/nginx:latest", - expected: "docker.io/library/nginx:latest", - }, - { - name: "digest reference", - registryDomain: "registry.example.com", - imageName: "myapp@sha256:abc123def456", - expected: "registry.example.com/myapp@sha256:abc123def456", - }, - { - name: "empty registry domain", - registryDomain: "", - imageName: "myapp:latest", - expected: "myapp:latest", - }, - { - name: "image without tag", - registryDomain: "registry.example.com", - imageName: "myapp", - expected: "registry.example.com/myapp", - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - handler := NewAutoRouteHandler(ctx, configSvc, containerSvc, blobStorage, tt.registryDomain) - result := handler.buildFullImageRef(tt.imageName) - assert.Equal(t, tt.expected, result) - }) - } -} diff --git a/internal/usecase/container/config_provider.go b/internal/usecase/container/config_provider.go deleted file mode 100644 index 56467c01d..000000000 --- a/internal/usecase/container/config_provider.go +++ /dev/null @@ -1,7 +0,0 @@ -package container - -import "github.com/bnema/gordon/internal/boundaries/out" - -// AttachmentConfigProvider is an alias for the boundary interface, kept for -// internal readability. -type AttachmentConfigProvider = out.AttachmentConfigProvider diff --git a/internal/usecase/container/env_hash.go b/internal/usecase/container/env_hash.go deleted file mode 100644 index b6a1ac644..000000000 --- a/internal/usecase/container/env_hash.go +++ /dev/null @@ -1,23 +0,0 @@ -package container - -import ( - "crypto/sha256" - "fmt" - "slices" -) - -// hashEnvironment computes a stable SHA-256 hash of a set of KEY=VALUE pairs. -// The input is sorted by key before hashing so map iteration order does not -// affect the result. The returned string is the hex-encoded digest. -func hashEnvironment(env []string) string { - sorted := make([]string, len(env)) - copy(sorted, env) - slices.Sort(sorted) - - h := sha256.New() - for _, kv := range sorted { - // Length-prefix each entry so embedded newlines cannot collide. - _, _ = fmt.Fprintf(h, "%d:%s\n", len(kv), kv) - } - return fmt.Sprintf("%x", h.Sum(nil)) -} diff --git a/internal/usecase/container/env_hash_test.go b/internal/usecase/container/env_hash_test.go deleted file mode 100644 index 7fc676415..000000000 --- a/internal/usecase/container/env_hash_test.go +++ /dev/null @@ -1,59 +0,0 @@ -package container - -import ( - "testing" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" -) - -func TestHashEnvironment_Deterministic(t *testing.T) { - env1 := []string{"B=2", "A=1", "C=3"} - env2 := []string{"A=1", "C=3", "B=2"} - assert.Equal(t, hashEnvironment(env1), hashEnvironment(env2)) -} - -func TestHashEnvironment_DifferentValues(t *testing.T) { - env1 := []string{"A=1", "B=2"} - env2 := []string{"A=1", "B=3"} - assert.NotEqual(t, hashEnvironment(env1), hashEnvironment(env2)) -} - -func TestHashEnvironment_DifferentKeys(t *testing.T) { - env1 := []string{"A=1", "B=2"} - env2 := []string{"A=1", "C=2"} - assert.NotEqual(t, hashEnvironment(env1), hashEnvironment(env2)) -} - -func TestHashEnvironment_Empty(t *testing.T) { - hash := hashEnvironment(nil) - require.NotEmpty(t, hash) - assert.Equal(t, hash, hashEnvironment([]string{})) -} - -func TestHashEnvironment_ExtraKey(t *testing.T) { - env1 := []string{"A=1"} - env2 := []string{"A=1", "B=2"} - assert.NotEqual(t, hashEnvironment(env1), hashEnvironment(env2)) -} - -func TestHashEnvironment_NewlineInValue(t *testing.T) { - env1 := []string{"A=1\nB=2"} - env2 := []string{"A=1", "B=2"} - - assert.NotEqual(t, hashEnvironment(env1), hashEnvironment(env2)) -} - -func TestMergeEnvironmentVariables_Sorted(t *testing.T) { - dockerfile := []string{"Z=26", "A=1"} - user := []string{"M=13", "B=2"} - result := mergeEnvironmentVariables(dockerfile, user) - assert.Equal(t, []string{"A=1", "B=2", "M=13", "Z=26"}, result) -} - -func TestMergeEnvironmentVariables_UserOverridesDockerfile(t *testing.T) { - dockerfile := []string{"A=old"} - user := []string{"A=new"} - result := mergeEnvironmentVariables(dockerfile, user) - assert.Equal(t, []string{"A=new"}, result) -} diff --git a/internal/usecase/container/events.go b/internal/usecase/container/events.go deleted file mode 100644 index d89562f6f..000000000 --- a/internal/usecase/container/events.go +++ /dev/null @@ -1,248 +0,0 @@ -package container - -import ( - "context" - "fmt" - - "github.com/bnema/zerowrap" - - "github.com/bnema/gordon/internal/boundaries/in" - "github.com/bnema/gordon/internal/domain" -) - -// ImagePushedHandler handles image.pushed events. -type ImagePushedHandler struct { - containerSvc in.ContainerService - configSvc in.ConfigService - ctx context.Context -} - -// NewImagePushedHandler creates a new ImagePushedHandler. -func NewImagePushedHandler(ctx context.Context, containerSvc in.ContainerService, configSvc in.ConfigService) *ImagePushedHandler { - return &ImagePushedHandler{ - containerSvc: containerSvc, - configSvc: configSvc, - ctx: ctx, - } -} - -// Handle handles an event. -func (h *ImagePushedHandler) Handle(ctx context.Context, event domain.Event) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldHandler: "ImagePushedHandler", - zerowrap.FieldEvent: string(event.Type), - "event_id": event.ID, - }) - log := zerowrap.FromCtx(ctx) - - if event.ImageName == "" { - return domain.ErrInvalidImageFormat - } - - tag := event.Tag - if tag == "" { - tag = "latest" - } - - fullImageName := fmt.Sprintf("%s:%s", event.ImageName, tag) - - log.Info().Str("image", fullImageName).Msg("processing image push event") - - routes := h.configSvc.FindRoutesByImage(ctx, fullImageName) - if len(routes) == 0 { - log.Debug().Str("image", fullImageName).Msg("no routes configured for pushed image") - return nil - } - - // Mark context as internal deploy - the event originated from our own registry, - // so we can use internal registry auth when pulling images. - internalCtx := domain.WithInternalDeploy(ctx) - - for _, route := range routes { - if _, err := h.containerSvc.Deploy(internalCtx, route); err != nil { - log.WrapErrWithFields(err, "failed to deploy container for route", map[string]any{"domain": route.Domain}) - } - } - - return nil -} - -// CanHandle returns whether this handler can handle the given event type. -func (h *ImagePushedHandler) CanHandle(eventType domain.EventType) bool { - return eventType == domain.EventImagePushed -} - -// ConfigReloadHandler handles config.reload events. -type ConfigReloadHandler struct { - containerSvc in.ContainerService - configSvc in.ConfigService - ctx context.Context -} - -// NewConfigReloadHandler creates a new ConfigReloadHandler. -func NewConfigReloadHandler(ctx context.Context, containerSvc in.ContainerService, configSvc in.ConfigService) *ConfigReloadHandler { - return &ConfigReloadHandler{ - containerSvc: containerSvc, - configSvc: configSvc, - ctx: ctx, - } -} - -// Handle handles an event. -func (h *ConfigReloadHandler) Handle(ctx context.Context, event domain.Event) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldHandler: "ConfigReloadHandler", - zerowrap.FieldEvent: string(event.Type), - "event_id": event.ID, - }) - log := zerowrap.FromCtx(ctx) - - log.Info().Msg("processing configuration reload event") - - // Sync containers first to ensure we have accurate state - // This removes tracking for containers that were stopped externally - if err := h.containerSvc.SyncContainers(ctx); err != nil { - log.Warn().Err(err).Msg("failed to sync containers before reload, proceeding with current state") - } - - // Propagate updated attachment configuration so new attachments take effect - // without requiring a Gordon restart (fixes issue #87 part 2). - attachments := h.configSvc.GetAllAttachments(ctx) - log.Debug().Int("attachment_groups", len(attachments)).Msg("propagating updated attachment configuration") - h.containerSvc.UpdateAttachments(attachments) - log.Debug().Int("attachment_groups", len(attachments)).Msg("attachment configuration propagated") - - currentContainers := h.containerSvc.List(ctx) - - activeRoutes := make(map[string]*domain.Container) - for domainName, container := range currentContainers { - if route := routeDomainForContainer(domainName, container); route != "" { - activeRoutes[route] = container - } - } - - routes := h.configSvc.GetRoutes(ctx) - for _, route := range routes { - if container, exists := activeRoutes[route.Domain]; exists { - currentImage := container.Labels[domain.LabelImage] - if currentImage != route.Image { - log.Info(). - Str("domain", route.Domain). - Str("old_image", currentImage). - Str("new_image", route.Image). - Msg("image changed for route, redeploying") - - if _, err := h.containerSvc.Deploy(domain.WithInternalDeploy(ctx), route); err != nil { - log.WrapErrWithFields(err, "failed to redeploy container", map[string]any{"domain": route.Domain}) - } - } - delete(activeRoutes, route.Domain) - } else { - // Route exists in config but no running container (new route or missing container) - log.Info(). - Str("domain", route.Domain). - Str("image", route.Image). - Msg("route missing container, deploying") - - if _, err := h.containerSvc.Deploy(domain.WithInternalDeploy(ctx), route); err != nil { - log.WrapErrWithFields(err, "failed to deploy container for route", map[string]any{"domain": route.Domain}) - } - } - } - - for route, container := range activeRoutes { - log.Info(). - Str("domain", route). - Str(zerowrap.FieldEntityID, container.ID). - Msg("route no longer configured, reconciling removed route runtime state") - - if _, err := h.containerSvc.ReconcileRemovedRoute(ctx, route); err != nil { - log.WrapErrWithFields(err, "failed to reconcile removed route runtime state", map[string]any{"domain": route, zerowrap.FieldEntityID: container.ID}) - } - } - - return nil -} - -// CanHandle returns whether this handler can handle the given event type. -func (h *ConfigReloadHandler) CanHandle(eventType domain.EventType) bool { - return eventType == domain.EventConfigReload -} - -func routeDomainForContainer(trackedDomain string, container *domain.Container) string { - if container == nil { - return "" - } - if container.Labels != nil { - if route := container.Labels[domain.LabelRoute]; route != "" { - return route - } - if route := container.Labels[domain.LabelDomain]; route != "" { - return route - } - } - return trackedDomain -} - -// ManualDeployHandler handles manual.deploy events for specific routes. -type ManualDeployHandler struct { - containerSvc in.ContainerService - configSvc in.ConfigService - ctx context.Context -} - -// NewManualDeployHandler creates a new ManualDeployHandler. -func NewManualDeployHandler(ctx context.Context, containerSvc in.ContainerService, configSvc in.ConfigService) *ManualDeployHandler { - return &ManualDeployHandler{ - containerSvc: containerSvc, - configSvc: configSvc, - ctx: ctx, - } -} - -// Handle handles a manual deploy event for a specific route. -func (h *ManualDeployHandler) Handle(ctx context.Context, event domain.Event) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldHandler: "ManualDeployHandler", - zerowrap.FieldEvent: string(event.Type), - "event_id": event.ID, - }) - log := zerowrap.FromCtx(ctx) - - payload, ok := event.Data.(*domain.ManualDeployPayload) - if !ok || payload == nil || payload.Domain == "" { - return fmt.Errorf("invalid manual deploy payload") - } - - log.Info().Str("domain", payload.Domain).Msg("processing manual deploy event") - - // Find the route in configuration - routes := h.configSvc.GetRoutes(ctx) - var targetRoute *domain.Route - for _, r := range routes { - if r.Domain == payload.Domain { - targetRoute = &r - break - } - } - - if targetRoute == nil { - return fmt.Errorf("route not found for domain: %s", payload.Domain) - } - - // Manual deploy is an internal trigger, so use internal deploy context. - if _, err := h.containerSvc.Deploy(domain.WithInternalDeploy(ctx), *targetRoute); err != nil { - return log.WrapErr(err, "failed to deploy container") - } - - log.Info().Str("domain", payload.Domain).Msg("manual deploy completed successfully") - return nil -} - -// CanHandle returns whether this handler can handle the given event type. -func (h *ManualDeployHandler) CanHandle(eventType domain.EventType) bool { - return eventType == domain.EventManualDeploy -} diff --git a/internal/usecase/container/events_test.go b/internal/usecase/container/events_test.go deleted file mode 100644 index 813d783e3..000000000 --- a/internal/usecase/container/events_test.go +++ /dev/null @@ -1,546 +0,0 @@ -package container - -import ( - "context" - "errors" - "testing" - - "github.com/bnema/zerowrap" - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/mock" - - inmocks "github.com/bnema/gordon/internal/boundaries/in/mocks" - "github.com/bnema/gordon/internal/domain" -) - -func testCtx() context.Context { - return zerowrap.WithCtx(context.Background(), zerowrap.Default()) -} - -// ImagePushedHandler tests - -func TestImagePushedHandler_CanHandle(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - handler := NewImagePushedHandler(testCtx(), containerSvc, configSvc) - - assert.True(t, handler.CanHandle(domain.EventImagePushed)) - assert.False(t, handler.CanHandle(domain.EventConfigReload)) -} - -func TestImagePushedHandler_Handle_DeploysMatchingRoutes(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - handler := NewImagePushedHandler(testCtx(), containerSvc, configSvc) - - // FindRoutesByImage returns matching routes directly - configSvc.EXPECT().FindRoutesByImage(mock.Anything, "myapp:latest").Return([]domain.Route{ - {Domain: "app1.example.com", Image: "myapp:latest"}, - {Domain: "app2.example.com", Image: "myapp:latest"}, - }) - - // Expect Deploy to be called for both matching routes - containerSvc.EXPECT().Deploy(mock.Anything, domain.Route{ - Domain: "app1.example.com", - Image: "myapp:latest", - }).Return(&domain.Container{ID: "container-1"}, nil) - - containerSvc.EXPECT().Deploy(mock.Anything, domain.Route{ - Domain: "app2.example.com", - Image: "myapp:latest", - }).Return(&domain.Container{ID: "container-2"}, nil) - - event := domain.Event{ - ID: "event-123", - Type: domain.EventImagePushed, - ImageName: "myapp", - Tag: "latest", - } - - err := handler.Handle(context.Background(), event) - - assert.NoError(t, err) -} - -func TestImagePushedHandler_Handle_NoMatchingRoutes(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - handler := NewImagePushedHandler(testCtx(), containerSvc, configSvc) - - // FindRoutesByImage returns no matching routes - configSvc.EXPECT().FindRoutesByImage(mock.Anything, "myapp:latest").Return([]domain.Route{}) - - // No Deploy calls expected - - event := domain.Event{ - ID: "event-123", - Type: domain.EventImagePushed, - ImageName: "myapp", - Tag: "latest", - } - - err := handler.Handle(context.Background(), event) - - assert.NoError(t, err) -} - -func TestImagePushedHandler_Handle_EmptyImageName(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - handler := NewImagePushedHandler(testCtx(), containerSvc, configSvc) - - event := domain.Event{ - ID: "event-123", - Type: domain.EventImagePushed, - ImageName: "", - Tag: "latest", - } - - err := handler.Handle(context.Background(), event) - - assert.ErrorIs(t, err, domain.ErrInvalidImageFormat) -} - -func TestImagePushedHandler_Handle_DefaultTag(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - handler := NewImagePushedHandler(testCtx(), containerSvc, configSvc) - - // Image name "myapp" with empty tag should become "myapp:latest" - configSvc.EXPECT().FindRoutesByImage(mock.Anything, "myapp:latest").Return([]domain.Route{ - {Domain: "app.example.com", Image: "myapp:latest"}, - }) - - containerSvc.EXPECT().Deploy(mock.Anything, domain.Route{ - Domain: "app.example.com", - Image: "myapp:latest", - }).Return(&domain.Container{ID: "container-1"}, nil) - - event := domain.Event{ - ID: "event-123", - Type: domain.EventImagePushed, - ImageName: "myapp", - Tag: "", // Empty tag should default to "latest" - } - - err := handler.Handle(context.Background(), event) - - assert.NoError(t, err) -} - -func TestImagePushedHandler_Handle_DeployError(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - handler := NewImagePushedHandler(testCtx(), containerSvc, configSvc) - - configSvc.EXPECT().FindRoutesByImage(mock.Anything, "myapp:latest").Return([]domain.Route{ - {Domain: "app.example.com", Image: "myapp:latest"}, - }) - - containerSvc.EXPECT().Deploy(mock.Anything, mock.Anything).Return(nil, errors.New("deploy failed")) - - event := domain.Event{ - ID: "event-123", - Type: domain.EventImagePushed, - ImageName: "myapp", - Tag: "latest", - } - - // Handler logs error but doesn't fail - err := handler.Handle(context.Background(), event) - - assert.NoError(t, err) -} - -func TestImagePushedHandler_Handle_StripsRegistryDomain(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - handler := NewImagePushedHandler(testCtx(), containerSvc, configSvc) - - // FindRoutesByImage handles registry domain stripping internally - configSvc.EXPECT().FindRoutesByImage(mock.Anything, "docker.io/library/nginx:latest").Return([]domain.Route{ - {Domain: "app.example.com", Image: "registry.example.com/docker.io/library/nginx:latest"}, - }) - - containerSvc.EXPECT().Deploy(mock.Anything, domain.Route{ - Domain: "app.example.com", - Image: "registry.example.com/docker.io/library/nginx:latest", - }).Return(&domain.Container{ID: "container-1"}, nil) - - event := domain.Event{ - ID: "event-123", - Type: domain.EventImagePushed, - ImageName: "docker.io/library/nginx", - Tag: "latest", - } - - err := handler.Handle(context.Background(), event) - - assert.NoError(t, err) -} - -// ConfigReloadHandler tests - -func TestConfigReloadHandler_CanHandle(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - handler := NewConfigReloadHandler(testCtx(), containerSvc, configSvc) - - assert.True(t, handler.CanHandle(domain.EventConfigReload)) - assert.False(t, handler.CanHandle(domain.EventImagePushed)) -} - -func TestConfigReloadHandler_Handle_DeploysNewRoutes(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - handler := NewConfigReloadHandler(testCtx(), containerSvc, configSvc) - - // SyncContainers is called first - containerSvc.EXPECT().SyncContainers(mock.Anything).Return(nil) - - // Attachment config is propagated after sync - configSvc.EXPECT().GetAllAttachments(mock.Anything).Return(map[string][]string{}) - containerSvc.EXPECT().UpdateAttachments(map[string][]string{}).Return() - - // No existing containers - containerSvc.EXPECT().List(mock.Anything).Return(map[string]*domain.Container{}) - - // New route in config - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{ - {Domain: "newapp.example.com", Image: "newapp:latest"}, - }) - - containerSvc.EXPECT().Deploy(mock.Anything, domain.Route{ - Domain: "newapp.example.com", - Image: "newapp:latest", - }).Return(&domain.Container{ID: "container-1"}, nil) - - event := domain.Event{ - ID: "event-123", - Type: domain.EventConfigReload, - } - - err := handler.Handle(context.Background(), event) - - assert.NoError(t, err) -} - -func TestConfigReloadHandler_Handle_StopsRemovedRoutes(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - handler := NewConfigReloadHandler(testCtx(), containerSvc, configSvc) - - // SyncContainers is called first - containerSvc.EXPECT().SyncContainers(mock.Anything).Return(nil) - - // Attachment config is propagated after sync - configSvc.EXPECT().GetAllAttachments(mock.Anything).Return(map[string][]string{}) - containerSvc.EXPECT().UpdateAttachments(map[string][]string{}).Return() - - // Existing container for a route that's no longer configured - containerSvc.EXPECT().List(mock.Anything).Return(map[string]*domain.Container{ - "removed.example.com": { - ID: "container-old", - Labels: map[string]string{ - "gordon.route": "removed.example.com", - "gordon.image": "oldapp:latest", - }, - }, - }) - - // Empty routes - the route was removed - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{}) - - containerSvc.EXPECT().ReconcileRemovedRoute(mock.Anything, "removed.example.com").Return(&domain.CleanupReport{Domain: "removed.example.com"}, nil) - - event := domain.Event{ - ID: "event-123", - Type: domain.EventConfigReload, - } - - err := handler.Handle(context.Background(), event) - - assert.NoError(t, err) -} - -func TestConfigReloadHandler_Handle_ReconcilesRemovedRouteWithLegacyDomainLabel(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - handler := NewConfigReloadHandler(testCtx(), containerSvc, configSvc) - - containerSvc.EXPECT().SyncContainers(mock.Anything).Return(nil) - configSvc.EXPECT().GetAllAttachments(mock.Anything).Return(map[string][]string{}) - containerSvc.EXPECT().UpdateAttachments(map[string][]string{}).Return() - containerSvc.EXPECT().List(mock.Anything).Return(map[string]*domain.Container{ - "legacy.example.com": { - ID: "container-legacy", - Labels: map[string]string{ - domain.LabelDomain: "legacy.example.com", - domain.LabelImage: "legacy:latest", - }, - }, - }) - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{}) - containerSvc.EXPECT().ReconcileRemovedRoute(mock.Anything, "legacy.example.com").Return(&domain.CleanupReport{Domain: "legacy.example.com"}, nil) - - event := domain.Event{ID: "event-legacy", Type: domain.EventConfigReload} - - err := handler.Handle(context.Background(), event) - - assert.NoError(t, err) -} - -func TestConfigReloadHandler_Handle_RedeploysChangedImage(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - handler := NewConfigReloadHandler(testCtx(), containerSvc, configSvc) - - // SyncContainers is called first - containerSvc.EXPECT().SyncContainers(mock.Anything).Return(nil) - - // Attachment config is propagated after sync - configSvc.EXPECT().GetAllAttachments(mock.Anything).Return(map[string][]string{}) - containerSvc.EXPECT().UpdateAttachments(map[string][]string{}).Return() - - // Existing container with old image - containerSvc.EXPECT().List(mock.Anything).Return(map[string]*domain.Container{ - "app.example.com": { - ID: "container-1", - Labels: map[string]string{ - "gordon.route": "app.example.com", - "gordon.image": "myapp:v1", - }, - }, - }) - - // Config now has different image - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{ - {Domain: "app.example.com", Image: "myapp:v2"}, - }) - - containerSvc.EXPECT().Deploy(mock.Anything, domain.Route{ - Domain: "app.example.com", - Image: "myapp:v2", - }).Return(&domain.Container{ID: "container-2"}, nil) - - event := domain.Event{ - ID: "event-123", - Type: domain.EventConfigReload, - } - - err := handler.Handle(context.Background(), event) - - assert.NoError(t, err) -} - -func TestConfigReloadHandler_Handle_NoChanges(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - handler := NewConfigReloadHandler(testCtx(), containerSvc, configSvc) - - // SyncContainers is called first - containerSvc.EXPECT().SyncContainers(mock.Anything).Return(nil) - - // Attachment config is propagated after sync - configSvc.EXPECT().GetAllAttachments(mock.Anything).Return(map[string][]string{}) - containerSvc.EXPECT().UpdateAttachments(map[string][]string{}).Return() - - // Existing container matches config - containerSvc.EXPECT().List(mock.Anything).Return(map[string]*domain.Container{ - "app.example.com": { - ID: "container-1", - Labels: map[string]string{ - "gordon.route": "app.example.com", - "gordon.image": "myapp:latest", - }, - }, - }) - - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{ - {Domain: "app.example.com", Image: "myapp:latest"}, - }) - - // No Deploy, Stop, or Remove calls expected - - event := domain.Event{ - ID: "event-123", - Type: domain.EventConfigReload, - } - - err := handler.Handle(context.Background(), event) - - assert.NoError(t, err) -} - -// ManualDeployHandler tests - -func TestConfigReloadHandlerUpdatesContainerConfig(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - handler := NewConfigReloadHandler(testCtx(), containerSvc, configSvc) - - // SyncContainers is called first - containerSvc.EXPECT().SyncContainers(mock.Anything).Return(nil) - - // No existing containers - containerSvc.EXPECT().List(mock.Anything).Return(map[string]*domain.Container{}) - - // New attachments from config - newAttachments := map[string][]string{ - "app.example.com": {"postgres:15", "redis:7"}, - } - configSvc.EXPECT().GetAllAttachments(mock.Anything).Return(newAttachments) - - // Expect UpdateAttachments to be called with new attachment config - containerSvc.EXPECT().UpdateAttachments(newAttachments).Return() - - // No routes configured - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{}) - - event := domain.Event{ - ID: "event-123", - Type: domain.EventConfigReload, - } - - err := handler.Handle(context.Background(), event) - - assert.NoError(t, err) -} - -func TestManualDeployHandler_CanHandle(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - handler := NewManualDeployHandler(testCtx(), containerSvc, configSvc) - - assert.True(t, handler.CanHandle(domain.EventManualDeploy)) - assert.False(t, handler.CanHandle(domain.EventImagePushed)) -} - -func TestManualDeployHandler_Handle_DeploysSpecificRoute(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - handler := NewManualDeployHandler(testCtx(), containerSvc, configSvc) - - // Configure routes - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{ - {Domain: "app1.example.com", Image: "app1:latest"}, - {Domain: "app2.example.com", Image: "app2:latest"}, - }) - - // Only the requested route should be deployed - containerSvc.EXPECT().Deploy(mock.Anything, domain.Route{ - Domain: "app1.example.com", - Image: "app1:latest", - }).Return(&domain.Container{ID: "container-1"}, nil) - - event := domain.Event{ - ID: "event-123", - Type: domain.EventManualDeploy, - Data: &domain.ManualDeployPayload{Domain: "app1.example.com"}, - } - - err := handler.Handle(context.Background(), event) - - assert.NoError(t, err) -} - -func TestManualDeployHandler_Handle_RouteNotFound(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - handler := NewManualDeployHandler(testCtx(), containerSvc, configSvc) - - // No matching routes - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{ - {Domain: "other.example.com", Image: "other:latest"}, - }) - - event := domain.Event{ - ID: "event-123", - Type: domain.EventManualDeploy, - Data: &domain.ManualDeployPayload{Domain: "unknown.example.com"}, - } - - err := handler.Handle(context.Background(), event) - - assert.Error(t, err) - assert.Contains(t, err.Error(), "route not found") -} - -func TestManualDeployHandler_Handle_InvalidPayload(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - handler := NewManualDeployHandler(testCtx(), containerSvc, configSvc) - - // Event with nil payload - event := domain.Event{ - ID: "event-123", - Type: domain.EventManualDeploy, - Data: nil, - } - - err := handler.Handle(context.Background(), event) - - assert.Error(t, err) - assert.Contains(t, err.Error(), "invalid manual deploy payload") -} - -func TestManualDeployHandler_Handle_EmptyDomain(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - handler := NewManualDeployHandler(testCtx(), containerSvc, configSvc) - - // Event with empty domain - event := domain.Event{ - ID: "event-123", - Type: domain.EventManualDeploy, - Data: &domain.ManualDeployPayload{Domain: ""}, - } - - err := handler.Handle(context.Background(), event) - - assert.Error(t, err) - assert.Contains(t, err.Error(), "invalid manual deploy payload") -} - -func TestManualDeployHandler_Handle_DeployError(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - handler := NewManualDeployHandler(testCtx(), containerSvc, configSvc) - - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{ - {Domain: "app.example.com", Image: "app:latest"}, - }) - - containerSvc.EXPECT().Deploy(mock.Anything, mock.Anything).Return(nil, errors.New("deploy failed")) - - event := domain.Event{ - ID: "event-123", - Type: domain.EventManualDeploy, - Data: &domain.ManualDeployPayload{Domain: "app.example.com"}, - } - - err := handler.Handle(context.Background(), event) - - assert.Error(t, err) - assert.Contains(t, err.Error(), "failed to deploy") -} diff --git a/internal/usecase/container/monitor.go b/internal/usecase/container/monitor.go deleted file mode 100644 index f29d49185..000000000 --- a/internal/usecase/container/monitor.go +++ /dev/null @@ -1,265 +0,0 @@ -package container - -import ( - "context" - "sync" - "time" - - "github.com/bnema/zerowrap" - "go.opentelemetry.io/otel/attribute" - "go.opentelemetry.io/otel/metric" - - "github.com/bnema/gordon/internal/domain" -) - -const ( - monitorDefaultInterval = 15 * time.Second - crashLoopThreshold = 3 - crashLoopWindow = 5 * time.Minute - backoffCap = 15 * time.Minute - stableRunningDuration = 5 * time.Minute -) - -// restartRecord tracks restart attempts for crash loop detection. -type restartRecord struct { - attempts []time.Time - backoffEnd time.Time - consecutive int - lastSeen time.Time // last time the container was seen running -} - -// Monitor watches tracked containers and restarts crashed ones. -type Monitor struct { - service *Service - stopCh chan struct{} - stopped chan struct{} - interval time.Duration - mu sync.Mutex - history map[string]*restartRecord // keyed by domain -} - -// newMonitor creates a new container monitor. -func newMonitor(service *Service) *Monitor { - return &Monitor{ - service: service, - stopCh: make(chan struct{}), - stopped: make(chan struct{}), - interval: monitorDefaultInterval, - history: make(map[string]*restartRecord), - } -} - -// Start begins the background monitoring loop. -func (m *Monitor) Start(ctx context.Context) { - log := zerowrap.FromCtx(ctx) - log.Info().Dur("interval", m.interval).Msg("container monitor started") - - go m.run(ctx) -} - -// Stop signals the monitor to stop and waits for it to finish. -func (m *Monitor) Stop() { - close(m.stopCh) - <-m.stopped -} - -func (m *Monitor) run(ctx context.Context) { - defer close(m.stopped) - - ticker := time.NewTicker(m.interval) - defer ticker.Stop() - - for { - select { - case <-m.stopCh: - return - case <-ctx.Done(): - return - case <-ticker.C: - m.check(ctx) - } - } -} - -func (m *Monitor) check(ctx context.Context) { - log := zerowrap.FromCtx(ctx) - - // Snapshot tracked containers under lock. - m.service.mu.RLock() - snapshot := make(map[string]*domain.Container, len(m.service.containers)) - for d, c := range m.service.containers { - snapshot[d] = c - } - m.service.mu.RUnlock() - - now := time.Now() - - for domainName, tracked := range snapshot { - m.checkContainer(ctx, log, domainName, tracked, now) - } -} - -func (m *Monitor) checkContainer(ctx context.Context, log zerowrap.Logger, domainName string, tracked *domain.Container, now time.Time) { - if tracked == nil { - return - } - - inspected, err := m.service.runtime.InspectContainer(ctx, tracked.ID) - if err != nil { - log.Debug().Err(err).Str("domain", domainName).Str("container_id", tracked.ID). - Msg("monitor: failed to inspect container, may have been replaced by deploy") - return - } - - switch { - case inspected.Status == string(domain.ContainerStatusRunning): - m.handleRunning(ctx, log, domainName, tracked.ID, now) - - case inspected.Status == string(domain.ContainerStatusExited): - if inspected.ExitCode == 0 { - // Graceful stop (exit 0): respect user intent, don't restart. - log.Debug().Str("domain", domainName).Msg("monitor: container exited gracefully (code 0), skipping") - return - } - // Crashed (non-zero exit): restart if not in backoff. - m.handleCrash(ctx, log, domainName, tracked.ID, inspected.ExitCode, now) - } -} - -func (m *Monitor) handleRunning(ctx context.Context, log zerowrap.Logger, domainName, containerID string, now time.Time) { - // Check Docker health status for running containers. - healthStatus, hasHealthcheck, err := m.service.runtime.GetContainerHealthStatus(ctx, containerID) - if err != nil { - return - } - - if hasHealthcheck && healthStatus == "unhealthy" { - if !m.isContainerStillTracked(domainName, containerID) { - log.Debug().Str("domain", domainName).Msg("monitor: container no longer tracked, skipping unhealthy restart") - return - } - log.Warn().Str("domain", domainName).Msg("monitor: container is unhealthy, restarting") - if err := m.service.runtime.RestartContainer(ctx, containerID); err != nil { - log.Warn().Err(err).Str("domain", domainName).Msg("monitor: failed to restart unhealthy container") - } - return - } - - // Container is running and healthy — clear backoff if stable long enough. - m.mu.Lock() - rec, exists := m.history[domainName] - if exists { - if rec.lastSeen.IsZero() { - rec.lastSeen = now - } - if now.Sub(rec.lastSeen) >= stableRunningDuration { - // Container has been running for 5+ min — clear crash history. - delete(m.history, domainName) - m.mu.Unlock() - log.Info().Str("domain", domainName).Msg("monitor: container stable, cleared crash history") - return - } - } - m.mu.Unlock() -} - -func (m *Monitor) isContainerStillTracked(domainName, containerID string) bool { - m.service.mu.RLock() - current := m.service.containers[domainName] - m.service.mu.RUnlock() - return current != nil && current.ID == containerID -} - -func (m *Monitor) handleCrash(ctx context.Context, log zerowrap.Logger, domainName, containerID string, exitCode int, now time.Time) { - m.mu.Lock() - rec := m.history[domainName] - if rec == nil { - rec = &restartRecord{} - m.history[domainName] = rec - } - - // Check backoff. - if now.Before(rec.backoffEnd) { - m.mu.Unlock() - log.Debug().Str("domain", domainName). - Time("backoff_until", rec.backoffEnd). - Msg("monitor: in crash loop backoff, skipping restart") - return - } - - // Record this crash attempt. - rec.attempts = append(rec.attempts, now) - rec.consecutive++ - rec.lastSeen = time.Time{} // reset stable timer - - // Prune old attempts outside the crash loop window. - cutoff := now.Add(-crashLoopWindow) - pruned := rec.attempts[:0] - for _, t := range rec.attempts { - if t.After(cutoff) { - pruned = append(pruned, t) - } - } - rec.attempts = pruned - - // Check for crash loop. - if len(rec.attempts) >= crashLoopThreshold { - backoff := m.computeBackoff(rec.consecutive) - rec.backoffEnd = now.Add(backoff) - m.mu.Unlock() - - log.Error().Str("domain", domainName). - Int("consecutive", rec.consecutive). - Dur("backoff", backoff). - Msg("monitor: crash loop detected, backing off") - - // Record crash loop metric - if m.service.metrics != nil { - attrs := metric.WithAttributes(attribute.String("domain", domainName)) - m.service.metrics.ContainerCrashLoops.Add(ctx, 1, attrs) - } - return - } - m.mu.Unlock() - - // Verify container ID still matches tracked state (deploy may have replaced it). - if !m.isContainerStillTracked(domainName, containerID) { - log.Debug().Str("domain", domainName).Msg("monitor: container replaced by deploy, skipping restart") - return - } - - log.Warn().Str("domain", domainName).Str("container_id", containerID). - Int("exit_code", exitCode). - Msg("monitor: restarting crashed container") - - if err := m.service.runtime.StartContainer(ctx, containerID); err != nil { - log.Warn().Err(err).Str("domain", domainName).Msg("monitor: failed to restart container") - return - } - - // Record restart metric only on successful restart - if m.service.metrics != nil { - attrs := metric.WithAttributes( - attribute.String("domain", domainName), - attribute.String("source", "monitor"), - ) - m.service.metrics.ContainerRestarts.Add(ctx, 1, attrs) - } -} - -func (m *Monitor) computeBackoff(consecutive int) time.Duration { - // Exponential backoff: 1min, 2min, 4min, 8min, cap 15min. - base := time.Minute - shift := consecutive - crashLoopThreshold - if shift < 0 { - shift = 0 - } - if shift > 4 { - shift = 4 // cap shift to avoid overflow (max 16min, then capped to backoffCap) - } - backoff := base << uint(shift) // #nosec G115 -- shift is bounded [0,4] - if backoff > backoffCap { - backoff = backoffCap - } - return backoff -} diff --git a/internal/usecase/container/monitor_test.go b/internal/usecase/container/monitor_test.go deleted file mode 100644 index 470d382e1..000000000 --- a/internal/usecase/container/monitor_test.go +++ /dev/null @@ -1,347 +0,0 @@ -package container - -import ( - "context" - "testing" - "time" - - "github.com/bnema/zerowrap" - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/mock" - - "github.com/bnema/gordon/internal/boundaries/out/mocks" - "github.com/bnema/gordon/internal/domain" -) - -func newTestService(runtime *mocks.MockContainerRuntime) *Service { - return &Service{ - runtime: runtime, - config: Config{}, - containers: make(map[string]*domain.Container), - attachments: make(map[string][]string), - } -} - -func monitorTestContext() context.Context { - return zerowrap.WithCtx(context.Background(), zerowrap.Default()) -} - -func TestMonitor_RestartsCrashedContainer(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := newTestService(runtime) - - // Track a container. - svc.containers["app.example.com"] = &domain.Container{ - ID: "ctr-1", - Image: "myapp:latest", - } - - // Inspect returns exited with non-zero exit code. - runtime.EXPECT().InspectContainer(mock.Anything, "ctr-1").Return(&domain.Container{ - ID: "ctr-1", - Status: string(domain.ContainerStatusExited), - ExitCode: 1, - }, nil) - - // Monitor should restart it. - runtime.EXPECT().StartContainer(mock.Anything, "ctr-1").Return(nil) - - m := newMonitor(svc) - m.check(monitorTestContext()) - - runtime.AssertExpectations(t) -} - -func TestMonitor_SkipsGracefulExit(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := newTestService(runtime) - - svc.containers["app.example.com"] = &domain.Container{ - ID: "ctr-1", - Image: "myapp:latest", - } - - // Exited with code 0 — graceful stop. - runtime.EXPECT().InspectContainer(mock.Anything, "ctr-1").Return(&domain.Container{ - ID: "ctr-1", - Status: string(domain.ContainerStatusExited), - ExitCode: 0, - }, nil) - - // Should NOT call StartContainer. - - m := newMonitor(svc) - m.check(monitorTestContext()) - - runtime.AssertExpectations(t) - runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) -} - -func TestMonitor_SkipsRunningContainer(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := newTestService(runtime) - - svc.containers["app.example.com"] = &domain.Container{ - ID: "ctr-1", - Image: "myapp:latest", - } - - runtime.EXPECT().InspectContainer(mock.Anything, "ctr-1").Return(&domain.Container{ - ID: "ctr-1", - Status: string(domain.ContainerStatusRunning), - }, nil) - - // Running container — check health status. - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "ctr-1").Return("healthy", true, nil) - - m := newMonitor(svc) - m.check(monitorTestContext()) - - runtime.AssertExpectations(t) - runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) -} - -func TestMonitor_RestartsUnhealthyContainer(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := newTestService(runtime) - - svc.containers["app.example.com"] = &domain.Container{ - ID: "ctr-1", - Image: "myapp:latest", - } - - runtime.EXPECT().InspectContainer(mock.Anything, "ctr-1").Return(&domain.Container{ - ID: "ctr-1", - Status: string(domain.ContainerStatusRunning), - }, nil) - - // Running but Docker health says unhealthy. - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "ctr-1").Return("unhealthy", true, nil) - - // Should restart the unhealthy container. - runtime.EXPECT().RestartContainer(mock.Anything, "ctr-1").Return(nil) - - m := newMonitor(svc) - m.check(monitorTestContext()) - - runtime.AssertExpectations(t) -} - -func TestMonitor_SkipsUnhealthyRestartWhenContainerWasUntracked(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := newTestService(runtime) - - tracked := &domain.Container{ID: "ctr-1", Image: "myapp:latest"} - svc.containers["app.example.com"] = tracked - - runtime.EXPECT().InspectContainer(mock.Anything, "ctr-1").Run(func(context.Context, string) { - svc.mu.Lock() - delete(svc.containers, "app.example.com") - svc.mu.Unlock() - }).Return(&domain.Container{ - ID: "ctr-1", - Status: string(domain.ContainerStatusRunning), - }, nil) - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "ctr-1").Return("unhealthy", true, nil) - - m := newMonitor(svc) - m.check(monitorTestContext()) - - runtime.AssertNotCalled(t, "RestartContainer", mock.Anything, "ctr-1") -} - -func TestMonitor_CrashLoopBackoff(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := newTestService(runtime) - - svc.containers["app.example.com"] = &domain.Container{ - ID: "ctr-1", - Image: "myapp:latest", - } - - m := newMonitor(svc) - ctx := monitorTestContext() - now := time.Now() - - // Simulate 3 crashes within the crash loop window. - m.mu.Lock() - m.history["app.example.com"] = &restartRecord{ - attempts: []time.Time{now.Add(-2 * time.Minute), now.Add(-1 * time.Minute)}, - consecutive: 2, - } - m.mu.Unlock() - - // Third crash — this should trigger backoff. - runtime.EXPECT().InspectContainer(mock.Anything, "ctr-1").Return(&domain.Container{ - ID: "ctr-1", - Status: string(domain.ContainerStatusExited), - ExitCode: 137, - }, nil) - - // Should NOT call StartContainer because crash loop is detected. - m.check(ctx) - - runtime.AssertExpectations(t) - runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) - - // Verify backoff was set. - m.mu.Lock() - rec := m.history["app.example.com"] - assert.False(t, rec.backoffEnd.IsZero(), "backoff should be set after crash loop detection") - assert.True(t, rec.backoffEnd.After(time.Now()), "backoff end should be in the future") - m.mu.Unlock() -} - -func TestMonitor_BackoffPreventsRestart(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := newTestService(runtime) - - svc.containers["app.example.com"] = &domain.Container{ - ID: "ctr-1", - Image: "myapp:latest", - } - - m := newMonitor(svc) - - // Set backoff into the future. - m.mu.Lock() - m.history["app.example.com"] = &restartRecord{ - backoffEnd: time.Now().Add(5 * time.Minute), - consecutive: 3, - } - m.mu.Unlock() - - // Container is crashed. - runtime.EXPECT().InspectContainer(mock.Anything, "ctr-1").Return(&domain.Container{ - ID: "ctr-1", - Status: string(domain.ContainerStatusExited), - ExitCode: 1, - }, nil) - - // Should NOT restart — in backoff. - m.check(monitorTestContext()) - - runtime.AssertExpectations(t) - runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) -} - -func TestMonitor_ClearsCrashHistoryAfterStableRunning(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := newTestService(runtime) - - svc.containers["app.example.com"] = &domain.Container{ - ID: "ctr-1", - Image: "myapp:latest", - } - - m := newMonitor(svc) - - // Container has been running for longer than stableRunningDuration. - m.mu.Lock() - m.history["app.example.com"] = &restartRecord{ - attempts: []time.Time{time.Now().Add(-10 * time.Minute)}, - consecutive: 2, - lastSeen: time.Now().Add(-6 * time.Minute), // 6 min ago > stableRunningDuration (5 min) - } - m.mu.Unlock() - - runtime.EXPECT().InspectContainer(mock.Anything, "ctr-1").Return(&domain.Container{ - ID: "ctr-1", - Status: string(domain.ContainerStatusRunning), - }, nil) - - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "ctr-1").Return("healthy", true, nil) - - m.check(monitorTestContext()) - - // History should be cleared. - m.mu.Lock() - _, exists := m.history["app.example.com"] - m.mu.Unlock() - assert.False(t, exists, "crash history should be cleared after stable running") - - runtime.AssertExpectations(t) -} - -func TestMonitor_SkipsReplacedContainer(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := newTestService(runtime) - - // Track the OLD container ID. - svc.containers["app.example.com"] = &domain.Container{ - ID: "ctr-NEW", - Image: "myapp:latest", - } - - m := newMonitor(svc) - - // Monitor snapshot captured "ctr-OLD" but a deploy replaced it with "ctr-NEW". - // We simulate this by directly calling checkContainer with the old ID. - ctx := monitorTestContext() - log := zerowrap.FromCtx(ctx) - - // Inspect the old container — it's exited. - runtime.EXPECT().InspectContainer(mock.Anything, "ctr-OLD").Return(&domain.Container{ - ID: "ctr-OLD", - Status: string(domain.ContainerStatusExited), - ExitCode: 1, - }, nil) - - // Should NOT restart — tracked container has been replaced. - oldTracked := &domain.Container{ID: "ctr-OLD", Image: "myapp:latest"} - m.checkContainer(ctx, log, "app.example.com", oldTracked, time.Now()) - - runtime.AssertExpectations(t) - runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) -} - -func TestMonitor_ComputeBackoff(t *testing.T) { - m := &Monitor{} - - tests := []struct { - consecutive int - expected time.Duration - }{ - {3, 1 * time.Minute}, // first backoff - {4, 2 * time.Minute}, // second - {5, 4 * time.Minute}, // third - {6, 8 * time.Minute}, // fourth - {7, 15 * time.Minute}, // capped - {10, 15 * time.Minute}, // still capped - } - - for _, tt := range tests { - got := m.computeBackoff(tt.consecutive) - assert.Equal(t, tt.expected, got, "consecutive=%d", tt.consecutive) - } -} - -func TestMonitor_StartStop(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := newTestService(runtime) - - m := newMonitor(svc) - m.interval = 50 * time.Millisecond // fast ticking for test - - ctx, cancel := context.WithCancel(monitorTestContext()) - defer cancel() - - m.Start(ctx) - - // Let it tick a couple times. - time.Sleep(150 * time.Millisecond) - - // Stop should return promptly. - done := make(chan struct{}) - go func() { - m.Stop() - close(done) - }() - - select { - case <-done: - // OK - case <-time.After(2 * time.Second): - t.Fatal("monitor.Stop() did not return within 2 seconds") - } -} diff --git a/internal/usecase/container/readiness.go b/internal/usecase/container/readiness.go deleted file mode 100644 index d7ac078f0..000000000 --- a/internal/usecase/container/readiness.go +++ /dev/null @@ -1,170 +0,0 @@ -package container - -import ( - "context" - "errors" - "fmt" - "net" - "net/http" - "time" - - "github.com/bnema/zerowrap" -) - -var errProbeTimeout = errors.New("probe timeout") - -func probeLoop( - ctx context.Context, - deadline time.Time, - baseSleep time.Duration, - defaultAttemptTimeout time.Duration, - work func(attemptCtx context.Context) (success bool, err error), -) error { - for { - if err := ctx.Err(); err != nil { - return err - } - if time.Now().After(deadline) { - return errProbeTimeout - } - - remaining := time.Until(deadline) - attemptTimeout := defaultAttemptTimeout - if remaining < attemptTimeout { - attemptTimeout = remaining - } - - attemptCtx, cancel := context.WithTimeout(ctx, attemptTimeout) - success, err := work(attemptCtx) - cancel() - if err != nil { - return err - } - if success { - return nil - } - - sleepInterval := baseSleep - if remProbe := time.Until(deadline); remProbe < sleepInterval { - sleepInterval = remProbe - } - if ctxDeadline, ok := ctx.Deadline(); ok { - if remCtx := time.Until(ctxDeadline); remCtx < sleepInterval { - sleepInterval = remCtx - } - } - if sleepInterval <= 0 { - if ctx.Err() != nil { - return ctx.Err() - } - return errProbeTimeout - } - - t := time.NewTimer(sleepInterval) - select { - case <-t.C: - case <-ctx.Done(): - t.Stop() - return ctx.Err() - } - } -} - -// tcpProbe attempts a TCP connection to addr, retrying every 500ms until -// success or timeout. This is the universal fallback readiness check — -// it verifies the process is at least accepting connections. -func tcpProbe(ctx context.Context, addr string, timeout time.Duration) error { - log := zerowrap.FromCtx(ctx) - deadline := time.Now().Add(timeout) - dialer := &net.Dialer{} - attempts := 0 - var lastErr error - err := probeLoop(ctx, deadline, 500*time.Millisecond, time.Second, func(attemptCtx context.Context) (bool, error) { - conn, err := dialer.DialContext(attemptCtx, "tcp", addr) - attempts++ - if err == nil { - conn.Close() - log.Debug().Str("addr", addr).Int("attempts", attempts).Msg("TCP probe connected") - return true, nil - } - - lastErr = err - if attempts <= 3 || attempts%10 == 0 { - log.Debug().Err(err).Str("addr", addr).Int("attempt", attempts).Msg("TCP probe attempt failed") - } - return false, nil - }) - if errors.Is(err, errProbeTimeout) { - return fmt.Errorf("TCP probe timeout after %s: %s not reachable (attempts=%d, last_error=%v)", timeout, addr, attempts, lastErr) - } - return err -} - -// httpProbe performs HTTP GET requests to url, retrying every 1s until a -// 2xx/3xx response or timeout. Used when gordon.health label is set. -func httpProbe(ctx context.Context, url string, timeout time.Duration) error { - client := &http.Client{ - CheckRedirect: func(req *http.Request, via []*http.Request) error { - // Don't follow redirects — treat 3xx as a successful response. - return http.ErrUseLastResponse - }, - } - deadline := time.Now().Add(timeout) - var lastStatus int - err := probeLoop(ctx, deadline, time.Second, 2*time.Second, func(attemptCtx context.Context) (bool, error) { - req, reqErr := http.NewRequestWithContext(attemptCtx, http.MethodGet, url, nil) - if reqErr != nil { - return false, reqErr - } - - resp, err := client.Do(req) - if err != nil { - return false, nil - } - - lastStatus = resp.StatusCode - resp.Body.Close() - if lastStatus >= 200 && lastStatus < 400 { - return true, nil - } - - return false, nil - }) - if errors.Is(err, errProbeTimeout) { - return fmt.Errorf("HTTP probe timeout after %s: last status %d from %s", timeout, lastStatus, url) - } - return err -} - -// httpAliveProbe performs HTTP GET requests to url, retrying every 1s until -// any HTTP response is received or timeout. Unlike httpProbe, ANY status code -// counts as success -- the goal is to verify the HTTP server is accepting and -// processing requests, not that a specific endpoint is healthy. -// Used as the default readiness check for web containers. -func httpAliveProbe(ctx context.Context, url string, timeout time.Duration) error { - client := &http.Client{ - CheckRedirect: func(req *http.Request, via []*http.Request) error { - return http.ErrUseLastResponse - }, - } - deadline := time.Now().Add(timeout) - err := probeLoop(ctx, deadline, time.Second, 2*time.Second, func(attemptCtx context.Context) (bool, error) { - req, reqErr := http.NewRequestWithContext(attemptCtx, http.MethodGet, url, nil) - if reqErr != nil { - return false, reqErr - } - - resp, err := client.Do(req) - if err != nil { - // Connection refused, timeout, etc. -- not ready yet. - return false, nil - } - resp.Body.Close() - // Any HTTP response means the server is alive and processing. - return true, nil - }) - if errors.Is(err, errProbeTimeout) { - return fmt.Errorf("HTTP alive probe timeout after %s: no HTTP response from %s", timeout, url) - } - return err -} diff --git a/internal/usecase/container/readiness_test.go b/internal/usecase/container/readiness_test.go deleted file mode 100644 index 811ba9ff7..000000000 --- a/internal/usecase/container/readiness_test.go +++ /dev/null @@ -1,167 +0,0 @@ -package container - -import ( - "context" - "net" - "net/http" - "net/http/httptest" - "sync/atomic" - "syscall" - "testing" - "time" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" -) - -// --- TCP Probe Tests --- - -func TestTCPProbe_Success(t *testing.T) { - ln, err := net.Listen("tcp", "127.0.0.1:0") - require.NoError(t, err) - defer ln.Close() - - // Accept connections in background so the probe succeeds - go func() { - for { - conn, err := ln.Accept() - if err != nil { - return - } - conn.Close() - } - }() - - ctx := testContext() - err = tcpProbe(ctx, ln.Addr().String(), 5*time.Second) - assert.NoError(t, err) -} - -func TestTCPProbe_Timeout(t *testing.T) { - // Reserve an ephemeral port and close it immediately to get a deterministic - // unreachable target, avoiding assumptions about privileged ports. - ln, err := net.Listen("tcp", "127.0.0.1:0") - require.NoError(t, err) - addr := ln.Addr().String() - ln.Close() - - ctx := testContext() - err = tcpProbe(ctx, addr, 200*time.Millisecond) - assert.Error(t, err) - assert.Contains(t, err.Error(), "TCP probe timeout") -} - -func TestTCPProbe_ContextCancelled(t *testing.T) { - ctx, cancel := context.WithCancel(testContext()) - cancel() - err := tcpProbe(ctx, "127.0.0.1:1", 5*time.Second) - assert.Error(t, err) -} - -func TestTCPProbe_DelayedListener(t *testing.T) { - ln, err := net.Listen("tcp", "127.0.0.1:0") - require.NoError(t, err) - // Close initially so probe fails at first - addr := ln.Addr().String() - ln.Close() - - // Re-open after 1 second - go func() { - time.Sleep(time.Second) - lc := net.ListenConfig{ - Control: func(network, address string, c syscall.RawConn) error { - return c.Control(func(fd uintptr) { - _ = syscall.SetsockoptInt(int(fd), syscall.SOL_SOCKET, syscall.SO_REUSEADDR, 1) - }) - }, - } - newLn, err := lc.Listen(context.Background(), "tcp", addr) - if err != nil { - t.Logf("delayed listener failed to bind: %v", err) - return - } - defer newLn.Close() - for { - conn, err := newLn.Accept() - if err != nil { - return - } - conn.Close() - } - }() - - ctx := testContext() - err = tcpProbe(ctx, addr, 5*time.Second) - assert.NoError(t, err) -} - -// --- HTTP Probe Tests --- - -func TestHTTPProbe_Success(t *testing.T) { - srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - if r.URL.Path == "/healthz" { - w.WriteHeader(200) - return - } - w.WriteHeader(404) - })) - defer srv.Close() - - ctx := testContext() - err := httpProbe(ctx, srv.URL+"/healthz", 5*time.Second) - assert.NoError(t, err) -} - -func TestHTTPProbe_ServerError_ThenSuccess(t *testing.T) { - var calls int32 - srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - n := atomic.AddInt32(&calls, 1) - if n < 3 { - w.WriteHeader(503) - return - } - w.WriteHeader(200) - })) - defer srv.Close() - - ctx := testContext() - err := httpProbe(ctx, srv.URL+"/healthz", 10*time.Second) - assert.NoError(t, err) - assert.GreaterOrEqual(t, atomic.LoadInt32(&calls), int32(3)) -} - -func TestHTTPProbe_Timeout(t *testing.T) { - srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - w.WriteHeader(503) - })) - defer srv.Close() - - ctx := testContext() - err := httpProbe(ctx, srv.URL+"/healthz", 500*time.Millisecond) - assert.Error(t, err) - assert.Contains(t, err.Error(), "HTTP probe timeout") -} - -func TestHTTPProbe_ContextCancelled(t *testing.T) { - srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - w.WriteHeader(503) - })) - defer srv.Close() - - ctx, cancel := context.WithCancel(testContext()) - cancel() - err := httpProbe(ctx, srv.URL+"/healthz", 5*time.Second) - assert.Error(t, err) -} - -func TestHTTPProbe_RedirectIsSuccess(t *testing.T) { - srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - http.Redirect(w, r, "/other", http.StatusFound) - })) - defer srv.Close() - - ctx := testContext() - // The probe should treat 3xx as success without following the redirect - err := httpProbe(ctx, srv.URL+"/healthz", 5*time.Second) - assert.NoError(t, err) -} diff --git a/internal/usecase/container/routes_test.go b/internal/usecase/container/routes_test.go deleted file mode 100644 index fd020a7af..000000000 --- a/internal/usecase/container/routes_test.go +++ /dev/null @@ -1,173 +0,0 @@ -package container - -import ( - "testing" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/mock" - - "github.com/bnema/gordon/internal/boundaries/out/mocks" - "github.com/bnema/gordon/internal/domain" -) - -func TestService_ListRoutesWithDetails(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{NetworkPrefix: "gordon"}, nil) - ctx := testContext() - - svc.containers["app.example.com"] = &domain.Container{ - ID: "container-1", - Image: "registry/app:latest", - Status: "running", - Labels: map[string]string{ - domain.LabelImage: "app:latest", - }, - } - - runtime.EXPECT().GetContainerNetwork(mock.Anything, "container-1").Return("gordon-app-example-com", nil) - // getAllAttachments calls GetContainerNetwork for each attachment - runtime.EXPECT().GetContainerNetwork(mock.Anything, "attach-1").Return("gordon-app-example-com", nil) - runtime.EXPECT().GetContainerNetwork(mock.Anything, "attach-2").Return("gordon-other-example-com", nil) - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - { - ID: "attach-1", - Name: "gordon-app-example-com-postgres", - Image: "postgres:15", - Status: "running", - Labels: map[string]string{ - domain.LabelAttachment: "true", - domain.LabelAttachedTo: "app.example.com", - domain.LabelImage: "postgres:15", - }, - }, - { - ID: "attach-2", - Name: "gordon-other-redis", - Image: "redis:7", - Status: "running", - Labels: map[string]string{ - domain.LabelAttachment: "true", - domain.LabelAttachedTo: "other.example.com", - domain.LabelImage: "redis:7", - }, - }, - { - ID: "other", - Name: "gordon-other", - Image: "busybox", - Status: "running", - Labels: map[string]string{ - domain.LabelAttachment: "false", - }, - }, - }, nil) - - results := svc.ListRoutesWithDetails(ctx) - - assert.Len(t, results, 1) - assert.Equal(t, "app.example.com", results[0].Domain) - assert.Equal(t, "app:latest", results[0].Image) - assert.Equal(t, "container-1", results[0].ContainerID) - assert.Equal(t, "running", results[0].ContainerStatus) - assert.Equal(t, "gordon-app-example-com", results[0].Network) - if assert.Len(t, results[0].Attachments, 1) { - assert.Equal(t, "postgres", results[0].Attachments[0].Name) - assert.Equal(t, "postgres:15", results[0].Attachments[0].Image) - assert.Equal(t, "attach-1", results[0].Attachments[0].ContainerID) - assert.Equal(t, "gordon-app-example-com", results[0].Attachments[0].Network) - } -} - -func TestService_ListRoutesWithDetails_Empty(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{NetworkPrefix: "gordon"}, nil) - ctx := testContext() - - // getAllAttachments is called even with empty containers - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - - results := svc.ListRoutesWithDetails(ctx) - - assert.Empty(t, results) -} - -func TestService_ListRoutesWithDetails_NetworkError(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{NetworkPrefix: "gordon"}, nil) - ctx := testContext() - - svc.containers["app.example.com"] = &domain.Container{ - ID: "container-1", - Image: "app:latest", - Status: "running", - } - - runtime.EXPECT().GetContainerNetwork(mock.Anything, "container-1").Return("", assert.AnError) - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - - results := svc.ListRoutesWithDetails(ctx) - - assert.Len(t, results, 1) - assert.Equal(t, "", results[0].Network) -} - -func TestService_ListAttachments(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{}, nil) - ctx := testContext() - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - { - ID: "attach-1", - Name: "gordon-app-example-com-postgres", - Image: "postgres:15", - Status: "running", - Labels: map[string]string{ - domain.LabelAttachment: "true", - domain.LabelAttachedTo: "app.example.com", - domain.LabelImage: "postgres:15", - }, - }, - { - ID: "attach-2", - Name: "gordon-other-redis", - Image: "redis:7", - Status: "running", - Labels: map[string]string{ - domain.LabelAttachment: "true", - domain.LabelAttachedTo: "other.example.com", - domain.LabelImage: "redis:7", - }, - }, - }, nil) - runtime.EXPECT().GetContainerNetwork(mock.Anything, "attach-1").Return("gordon-app-example-com", nil) - - attachments := svc.ListAttachments(ctx, "app.example.com") - - assert.Len(t, attachments, 1) - assert.Equal(t, "postgres", attachments[0].Name) - assert.Equal(t, "postgres:15", attachments[0].Image) - assert.Equal(t, "attach-1", attachments[0].ContainerID) - assert.Equal(t, "gordon-app-example-com", attachments[0].Network) -} - -func TestService_ListNetworks(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{NetworkPrefix: "gordon"}, nil) - ctx := testContext() - - runtime.EXPECT().ListNetworks(mock.Anything).Return([]*domain.NetworkInfo{ - {Name: "gordon-app", Labels: map[string]string{domain.LabelManaged: "true"}}, - {Name: "bridge"}, - {Name: "gordon-shared", Labels: map[string]string{domain.LabelManaged: "true"}}, - {Name: "gordon-unmanaged"}, - }, nil) - - networks, err := svc.ListNetworks(ctx) - - assert.NoError(t, err) - assert.Len(t, networks, 2) - assert.Equal(t, "gordon-app", networks[0].Name) - assert.Equal(t, "gordon-shared", networks[1].Name) - assert.NotContains(t, []string{networks[0].Name, networks[1].Name}, "gordon-unmanaged") -} diff --git a/internal/usecase/container/secrets_handler.go b/internal/usecase/container/secrets_handler.go deleted file mode 100644 index af05769ca..000000000 --- a/internal/usecase/container/secrets_handler.go +++ /dev/null @@ -1,155 +0,0 @@ -package container - -import ( - "context" - "sync" - "time" - - "github.com/bnema/zerowrap" - - "github.com/bnema/gordon/internal/boundaries/in" - "github.com/bnema/gordon/internal/domain" -) - -const DefaultSecretsDebounce = 60 * time.Second - -// SecretsChangedHandler handles secrets.changed events with per-domain -// debouncing. When secrets are modified rapidly (e.g. multiple -// "gordon secrets set" calls), the handler resets a timer on each -// event and only triggers a deploy once the domain is quiet for -// the configured delay. -type SecretsChangedHandler struct { - containerSvc in.ContainerService - configSvc in.ConfigService - ctx context.Context - cancel context.CancelFunc - debounceDelay time.Duration - - mu sync.Mutex - timers map[string]*time.Timer - generations map[string]uint64 -} - -// NewSecretsChangedHandler creates a new SecretsChangedHandler. -func NewSecretsChangedHandler( - ctx context.Context, - containerSvc in.ContainerService, - configSvc in.ConfigService, - debounceDelay time.Duration, -) *SecretsChangedHandler { - ctx, cancel := context.WithCancel(ctx) - return &SecretsChangedHandler{ - containerSvc: containerSvc, - configSvc: configSvc, - ctx: ctx, - cancel: cancel, - debounceDelay: debounceDelay, - timers: make(map[string]*time.Timer), - generations: make(map[string]uint64), - } -} - -// Handle processes a secrets.changed event by resetting the debounce -// timer for the affected domain. Returns immediately - the deploy -// runs asynchronously when the timer fires. -// -// Uses the handler's long-lived context (h.ctx) rather than the event -// context because the debounced deploy fires after this method returns; -// using the event context would cause premature cancellation. -func (h *SecretsChangedHandler) Handle(_ context.Context, event domain.Event) error { - payload, ok := event.Data.(domain.SecretsChangedPayload) - if !ok { - return nil - } - if payload.Domain == "" { - return nil - } - - ctx := zerowrap.CtxWithFields(h.ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldHandler: "SecretsChangedHandler", - "domain": payload.Domain, - "operation": payload.Operation, - }) - log := zerowrap.FromCtx(ctx) - - log.Debug(). - Strs("keys", payload.Keys). - Dur("debounce", h.debounceDelay). - Msg("secrets changed, resetting debounce timer") - - h.resetTimer(ctx, payload.Domain) - return nil -} - -// CanHandle returns true for secrets.changed events. -func (h *SecretsChangedHandler) CanHandle(eventType domain.EventType) bool { - return eventType == domain.EventSecretsChanged -} - -// Stop cancels the handler context and all pending timers. -// Call before graceful shutdown to prevent deploys during drain. -func (h *SecretsChangedHandler) Stop() { - h.cancel() - - h.mu.Lock() - defer h.mu.Unlock() - - for domainName, timer := range h.timers { - timer.Stop() - delete(h.timers, domainName) - delete(h.generations, domainName) - } -} - -// resetTimer replaces any existing timer for the domain with a new one. -func (h *SecretsChangedHandler) resetTimer(ctx context.Context, domainName string) { - h.mu.Lock() - defer h.mu.Unlock() - - if existing, ok := h.timers[domainName]; ok { - existing.Stop() - } - - h.generations[domainName]++ - gen := h.generations[domainName] - - h.timers[domainName] = time.AfterFunc(h.debounceDelay, func() { - h.fireDeploy(ctx, domainName, gen) - }) -} - -// fireDeploy looks up the route and triggers a deploy. -func (h *SecretsChangedHandler) fireDeploy(ctx context.Context, domainName string, generation uint64) { - // Bail if the handler has been stopped (shutdown in progress). - if h.ctx.Err() != nil { - return - } - - log := zerowrap.FromCtx(ctx) - - h.mu.Lock() - if h.generations[domainName] != generation { - h.mu.Unlock() - return - } - delete(h.timers, domainName) - delete(h.generations, domainName) - h.mu.Unlock() - - route, err := h.configSvc.GetRoute(ctx, domainName) - if err != nil { - log.WrapErr(err, "failed to get route for secrets-changed deploy") - return - } - if route == nil { - log.Debug().Msg("no route configured for domain, skipping secrets-changed deploy") - return - } - - log.Info().Str("image", route.Image).Msg("debounce fired, deploying after secrets change") - - if _, err := h.containerSvc.Deploy(domain.WithInternalDeploy(ctx), *route); err != nil { - log.WrapErr(err, "secrets-changed deploy failed") - } -} diff --git a/internal/usecase/container/secrets_handler_test.go b/internal/usecase/container/secrets_handler_test.go deleted file mode 100644 index 333b74292..000000000 --- a/internal/usecase/container/secrets_handler_test.go +++ /dev/null @@ -1,214 +0,0 @@ -package container - -import ( - "context" - "sync/atomic" - "testing" - "time" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/mock" - "github.com/stretchr/testify/require" - - inmocks "github.com/bnema/gordon/internal/boundaries/in/mocks" - "github.com/bnema/gordon/internal/domain" -) - -func TestSecretsChangedHandler_CanHandle(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - h := NewSecretsChangedHandler(testCtx(), containerSvc, configSvc, 200*time.Millisecond) - defer h.Stop() - - assert.True(t, h.CanHandle(domain.EventSecretsChanged)) - assert.False(t, h.CanHandle(domain.EventConfigReload)) -} - -func TestSecretsChangedHandler_DebouncesBurstEvents(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - debounceDelay := 200 * time.Millisecond - h := NewSecretsChangedHandler(testCtx(), containerSvc, configSvc, debounceDelay) - defer h.Stop() - - route := &domain.Route{Domain: "app.example.com", Image: "app:latest"} - configSvc.EXPECT().GetRoute(mock.Anything, route.Domain).Return(route, nil).Once() - - var deployCalls atomic.Int32 - containerSvc.EXPECT().Deploy(mock.Anything, *route).RunAndReturn(func(_ context.Context, deployedRoute domain.Route) (*domain.Container, error) { - deployCalls.Add(1) - return &domain.Container{ID: "container-1"}, nil - }).Once() - - event := domain.Event{ - Type: domain.EventSecretsChanged, - Data: domain.SecretsChangedPayload{ - Domain: route.Domain, - Operation: "set", - Keys: []string{"A", "B"}, - }, - } - - for range 5 { - assert.NoError(t, h.Handle(context.Background(), event)) - time.Sleep(10 * time.Millisecond) - } - - require.Eventually(t, func() bool { - return deployCalls.Load() == 1 - }, debounceDelay+time.Second, 10*time.Millisecond, "expected exactly 1 deploy call after debounce") -} - -func TestSecretsChangedHandler_IndependentDomains(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - debounceDelay := 100 * time.Millisecond - h := NewSecretsChangedHandler(testCtx(), containerSvc, configSvc, debounceDelay) - defer h.Stop() - - routeOne := &domain.Route{Domain: "app1.example.com", Image: "app1:latest"} - routeTwo := &domain.Route{Domain: "app2.example.com", Image: "app2:latest"} - - configSvc.EXPECT().GetRoute(mock.Anything, routeOne.Domain).Return(routeOne, nil).Once() - configSvc.EXPECT().GetRoute(mock.Anything, routeTwo.Domain).Return(routeTwo, nil).Once() - - var deployCalls atomic.Int32 - containerSvc.EXPECT().Deploy(mock.Anything, *routeOne).RunAndReturn(func(_ context.Context, _ domain.Route) (*domain.Container, error) { - deployCalls.Add(1) - return &domain.Container{ID: "container-1"}, nil - }).Once() - containerSvc.EXPECT().Deploy(mock.Anything, *routeTwo).RunAndReturn(func(_ context.Context, _ domain.Route) (*domain.Container, error) { - deployCalls.Add(1) - return &domain.Container{ID: "container-2"}, nil - }).Once() - - assert.NoError(t, h.Handle(context.Background(), domain.Event{ - Type: domain.EventSecretsChanged, - Data: domain.SecretsChangedPayload{Domain: routeOne.Domain, Operation: "set", Keys: []string{"A"}}, - })) - assert.NoError(t, h.Handle(context.Background(), domain.Event{ - Type: domain.EventSecretsChanged, - Data: domain.SecretsChangedPayload{Domain: routeTwo.Domain, Operation: "delete", Keys: []string{"B"}}, - })) - - require.Eventually(t, func() bool { - return deployCalls.Load() == 2 - }, debounceDelay+time.Second, 10*time.Millisecond, "expected 2 independent deploy calls") -} - -func TestSecretsChangedHandler_NoRouteFound_NoOp(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - debounceDelay := 100 * time.Millisecond - h := NewSecretsChangedHandler(testCtx(), containerSvc, configSvc, debounceDelay) - defer h.Stop() - - configSvc.EXPECT().GetRoute(mock.Anything, "missing.example.com").Return(nil, nil).Once() - - assert.NoError(t, h.Handle(context.Background(), domain.Event{ - Type: domain.EventSecretsChanged, - Data: domain.SecretsChangedPayload{Domain: "missing.example.com", Operation: "set", Keys: []string{"A"}}, - })) - - time.Sleep(debounceDelay + 100*time.Millisecond) -} - -func TestSecretsChangedHandler_InvalidPayload_NoOp(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - h := NewSecretsChangedHandler(testCtx(), containerSvc, configSvc, 100*time.Millisecond) - defer h.Stop() - - err := h.Handle(context.Background(), domain.Event{ - Type: domain.EventSecretsChanged, - Data: "wrong-payload", - }) - - assert.NoError(t, err) -} - -func TestSecretsChangedHandler_FireDeploySkipsStaleGeneration(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - h := NewSecretsChangedHandler(testCtx(), containerSvc, configSvc, time.Second) - defer h.Stop() - - const domainName = "app.example.com" - - // White-box: directly set internal state to simulate a stale generation. - // This exercises the generation-based debounce logic that prevents - // already-fired timers from triggering duplicate deploys. - h.mu.Lock() - h.generations[domainName] = 2 - h.timers[domainName] = time.NewTimer(time.Hour) - h.mu.Unlock() - - staleDone := make(chan struct{}) - go func() { - defer close(staleDone) - h.fireDeploy(context.Background(), domainName, 1) - }() - - select { - case <-staleDone: - case <-time.After(2 * time.Second): - t.Fatal("stale fireDeploy did not return") - } - - h.mu.Lock() - _, timerExists := h.timers[domainName] - gen := h.generations[domainName] - h.mu.Unlock() - - assert.True(t, timerExists) - assert.Equal(t, uint64(2), gen) -} - -func TestSecretsChangedHandler_FireDeployClearsCurrentGeneration(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - h := NewSecretsChangedHandler(testCtx(), containerSvc, configSvc, time.Second) - defer h.Stop() - - route := &domain.Route{Domain: "app.example.com", Image: "app:latest"} - configSvc.EXPECT().GetRoute(mock.Anything, route.Domain).Return(route, nil).Once() - containerSvc.EXPECT().Deploy(mock.Anything, *route).Return(&domain.Container{ID: "container-1"}, nil).Once() - - h.mu.Lock() - h.generations[route.Domain] = 1 - h.timers[route.Domain] = time.NewTimer(time.Hour) - h.mu.Unlock() - - h.fireDeploy(context.Background(), route.Domain, 1) - - h.mu.Lock() - _, timerExists := h.timers[route.Domain] - _, generationExists := h.generations[route.Domain] - h.mu.Unlock() - - assert.False(t, timerExists) - assert.False(t, generationExists) -} - -func TestSecretsChangedHandler_StopCancelsPendingDeploy(t *testing.T) { - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - debounceDelay := 50 * time.Millisecond - h := NewSecretsChangedHandler(testCtx(), containerSvc, configSvc, debounceDelay) - - assert.NoError(t, h.Handle(context.Background(), domain.Event{ - Type: domain.EventSecretsChanged, - Data: domain.SecretsChangedPayload{Domain: "app.example.com", Operation: "set", Keys: []string{"A"}}, - })) - - h.Stop() - time.Sleep(debounceDelay + 50*time.Millisecond) -} diff --git a/internal/usecase/container/service.go b/internal/usecase/container/service.go index 14816bd54..328733834 100644 --- a/internal/usecase/container/service.go +++ b/internal/usecase/container/service.go @@ -1,3827 +1,102 @@ -// Package container implements the container management use case. +// Package container implements the retained container lifecycle use case. +// +// The v3 declarative-apps cutover removed the route-container engine +// (deploy/restart/remove/reconcile/attachments/sync/autostart, image-label +// inference, readiness cascade, monitor). Workload effects belong to the +// deployment engine through out.ContainerRuntime directly. What remains: +// +// - ListNetworks: read-only inventory of Gordon-managed networks. +// - Shutdown: graceful teardown (log writer close; containers are left +// running across Gordon restarts by design). package container import ( - "bufio" "context" - "encoding/binary" - "encoding/hex" - "errors" - "fmt" - "io" - "maps" - "net" - "slices" - "strconv" "strings" "sync" - "time" "github.com/bnema/zerowrap" - "go.opentelemetry.io/otel" - "go.opentelemetry.io/otel/attribute" - "go.opentelemetry.io/otel/codes" - "go.opentelemetry.io/otel/metric" - "go.opentelemetry.io/otel/trace" - "github.com/bnema/gordon/internal/adapters/out/telemetry" "github.com/bnema/gordon/internal/boundaries/out" "github.com/bnema/gordon/internal/domain" ) // Config holds configuration needed by the container service. type Config struct { - RegistryAuthEnabled bool - RegistryDomain string - LegacyRegistryDomains []string - RegistryPort int - ServiceTokenUsername string - ServiceToken string - InternalRegistryUsername string - InternalRegistryPassword string - PullPolicy string - VolumeAutoCreate bool - VolumePrefix string - VolumePreserve bool - NetworkIsolation bool - NetworkPrefix string - NetworkGroups map[string][]string - NetworkInternal bool - Attachments map[string][]string - AllowedRegistries []string - RequireImageDigest bool - SecurityProfile string - ReadinessDelay time.Duration // Delay after container starts before considering it ready - ReadinessMode string // Readiness strategy: auto, docker-health, delay - HealthTimeout time.Duration // Max wait for health-based readiness - DrainDelay time.Duration // Grace period after cache invalidation before stopping old container - DrainDelayConfigured bool // True when deploy.drain_delay was explicitly configured - DrainMode string // Drain strategy: auto, inflight, delay - DrainTimeout time.Duration // Max wait for in-flight requests to drain - StabilizationDelay time.Duration // Post-switch monitoring window (default 2s) - TCPProbeTimeout time.Duration // TCP probe timeout (default 30s) - HTTPProbeTimeout time.Duration // HTTP probe timeout (default 60s) - AttachmentReadinessTimeout time.Duration // Max wait for attachment readiness (default 30s) - DefaultMemoryLimit int64 // Default memory limit in bytes for containers (0 = no limit) - DefaultNanoCPUs int64 // Default CPU quota in nanoseconds for containers (0 = no limit) - DefaultPidsLimit int64 // Default max PIDs for containers (0 = no limit) + NetworkPrefix string } -var tracer = otel.Tracer("gordon.container") - -const ( - PullPolicyAlways = "always" - PullPolicyIfNotPresent = "if-not-present" - PullPolicyIfTagChanged = "if-tag-changed" - - // readinessRecoveryWindow is an additional grace window used when a container - // is briefly not running at the end of readiness delay. This avoids false - // negatives during short startup flaps. - readinessRecoveryWindow = 30 * time.Second - failedContainerCleanupTimeout = 30 * time.Second - internalPullMaxAttempts = 3 - deployFailureLogLimit = 20 -) - // Service implements the ContainerService interface. type Service struct { - runtime out.ContainerRuntime - envLoader out.EnvLoader - eventBus out.EventPublisher - logWriter out.ContainerLogWriter - cacheInvalidator out.ProxyCacheInvalidator - drainWaiter out.ProxyDrainWaiter - config Config - configProvider AttachmentConfigProvider // live config reads for attachments/networks (may be nil) - metrics *telemetry.Metrics - containers map[string]*domain.Container - attachments map[string][]string // ownerDomain → []containerIDs - managedCount int64 // tracks UpDownCounter value for delta computation - mu sync.RWMutex - deployMu sync.Map // per-domain deploy locks (domain → *domainDeployLock) - cleanupWg sync.WaitGroup // tracks background old-container cleanup goroutines - monitor *Monitor -} - -// domainDeployLock is a context-aware mutex using a buffered channel. -// It allows Lock() to be interrupted by context cancellation. -type domainDeployLock struct { - ch chan struct{} -} - -// newDomainDeployLock creates a new lock that is initially unlocked. -func newDomainDeployLock() *domainDeployLock { - l := &domainDeployLock{ch: make(chan struct{}, 1)} - l.ch <- struct{}{} - return l -} - -// Lock acquires the lock or returns an error if the context is cancelled. -func (l *domainDeployLock) Lock(ctx context.Context) error { - // Fast path: honor already-canceled contexts before blocking. - if err := ctx.Err(); err != nil { - return err - } - select { - case <-ctx.Done(): - return ctx.Err() - case <-l.ch: - // Re-check context after acquisition: if canceled while we were - // in the select, release the token so the lock isn't consumed. - if err := ctx.Err(); err != nil { - l.ch <- struct{}{} - return err - } - return nil - } -} - -// Unlock releases the lock. -func (l *domainDeployLock) Unlock() { - select { - case l.ch <- struct{}{}: - default: - // This should never happen if Lock/Unlock are used correctly. - // Panic to detect double-unlocks or other misuse early. - panic("domainDeployLock: double unlock detected") - } + runtime out.ContainerRuntime + eventBus out.EventPublisher + logWriter out.ContainerLogWriter + config Config + mu sync.RWMutex } // NewService creates a new container service. func NewService( runtime out.ContainerRuntime, - envLoader out.EnvLoader, eventBus out.EventPublisher, logWriter out.ContainerLogWriter, config Config, - configProvider AttachmentConfigProvider, ) *Service { return &Service{ - runtime: runtime, - envLoader: envLoader, - eventBus: eventBus, - logWriter: logWriter, - config: config, - configProvider: configProvider, - containers: make(map[string]*domain.Container), - attachments: make(map[string][]string), - } -} - -// SetMetrics sets the telemetry metrics for the container service. -func (s *Service) SetMetrics(m *telemetry.Metrics) { - s.metrics = m -} - -// SetProxyCacheInvalidator sets the proxy cache invalidator for synchronous -// cache invalidation during zero-downtime deployments. This must be called -// after construction because the proxy service is created after the container -// service during application initialization. -func (s *Service) SetProxyCacheInvalidator(inv out.ProxyCacheInvalidator) { - s.mu.Lock() - s.cacheInvalidator = inv - s.mu.Unlock() -} - -// SetProxyDrainWaiter sets the proxy in-flight drain waiter used during -// zero-downtime replacement before stopping the old container. -func (s *Service) SetProxyDrainWaiter(waiter out.ProxyDrainWaiter) { - s.mu.Lock() - s.drainWaiter = waiter - s.mu.Unlock() -} - -// domainDeployMu returns a per-domain lock for serializing deploy operations. -func (s *Service) domainDeployMu(domain string) *domainDeployLock { - v, _ := s.deployMu.LoadOrStore(domain, newDomainDeployLock()) - // Type assertion is safe: LoadOrStore always stores *domainDeployLock values, - // and any concurrent LoadOrStore calls will also store *domainDeployLock. - return v.(*domainDeployLock) -} - -// acquireDomainDeployLock acquires the per-domain deploy lock with context cancellation support. -// Returns an unlock function that must be called to release the lock. -func (s *Service) acquireDomainDeployLock(ctx context.Context, domain string) (func(), error) { - lock := s.domainDeployMu(domain) - if err := lock.Lock(ctx); err != nil { - return nil, err - } - return func() { lock.Unlock() }, nil -} - -// buildContainerConfig constructs the container configuration for deployment. -func (s *Service) deploymentContainerName(containerDomain string, existing *domain.Container) string { - canonicalName := managedContainerName(containerDomain) - if existing == nil { - return canonicalName - } - - // Alternate temporary names if the tracked container was left with a temp suffix. - // This prevents name collisions after interrupted zero-downtime deploys. - newName := canonicalName + "-new" - nextName := canonicalName + "-next" - switch existing.Name { - case newName: - return nextName - case nextName: - return newName - default: - return newName + runtime: runtime, + eventBus: eventBus, + logWriter: logWriter, + config: config, } } -// containerConfigInput groups the parameters for building a container configuration. -type containerConfigInput struct { - Domain string - Image string - ImageRef string - ExposedPorts []int - EnvVars []string - EnvHash string - Volumes map[string]string - NetworkName string - ImageLabels map[string]string - Existing *domain.Container -} - -func (s *Service) buildContainerConfig(in containerConfigInput) *domain.ContainerConfig { +// ListNetworks returns Gordon-managed networks (prefix-filtered, +// manager-labeled). +func (s *Service) ListNetworks(ctx context.Context) ([]*domain.NetworkInfo, error) { s.mu.RLock() cfg := s.config s.mu.RUnlock() - containerName := s.deploymentContainerName(in.Domain, in.Existing) - labels := map[string]string{ - domain.LabelDomain: in.Domain, - domain.LabelEnvHash: in.EnvHash, - domain.LabelImage: in.Image, - domain.LabelManaged: "true", - domain.LabelRoute: in.Domain, - } - - // Propagate proxy/port labels from image so readiness probes can find them - for _, key := range []string{domain.LabelProxyPort, domain.LabelPort, domain.LabelHealth} { - if v, ok := in.ImageLabels[key]; ok && v != "" { - labels[key] = v - } + networks, err := s.runtime.ListNetworks(ctx) + if err != nil { + return nil, err } - // Ensure the label-specified proxy port is included in exposed ports - // so Docker creates a host port binding for it. - ports := in.ExposedPorts - if portStr, ok := labels[domain.LabelProxyPort]; ok { - if port, err := strconv.Atoi(portStr); err == nil && port > 0 && port <= 65535 { - if !slices.Contains(ports, port) { - ports = append(ports, port) - } + var filtered []*domain.NetworkInfo + for _, network := range networks { + if strings.HasPrefix(network.Name, cfg.NetworkPrefix+"-") && network.Labels[domain.LabelManaged] == "true" { + filtered = append(filtered, network) } } - containerConfig := &domain.ContainerConfig{ - Image: in.ImageRef, - Name: containerName, - Ports: ports, - Env: in.EnvVars, - Volumes: in.Volumes, - NetworkMode: in.NetworkName, - Hostname: in.Domain, - Labels: labels, - AutoRemove: false, - RestartPolicy: domain.RestartPolicyAlways, - MemoryLimit: cfg.DefaultMemoryLimit, - NanoCPUs: cfg.DefaultNanoCPUs, - PidsLimit: cfg.DefaultPidsLimit, - } - applySecurityProfile(containerConfig, cfg) - return containerConfig + return filtered, nil } -// Deploy creates and starts a container for the given route. -// Implements zero-downtime deployment: new container starts before old one stops. -func (s *Service) Deploy(ctx context.Context, route domain.Route) (*domain.Container, error) { - ctx, span := tracer.Start(ctx, "container.deploy", - trace.WithAttributes( - attribute.String("domain", route.Domain), - attribute.String("image", route.Image), - )) - defer span.End() - - deployStart := time.Now() - - // Serialize deploys for the same domain to prevent race conditions - // (e.g. multiple image.pushed events + explicit deploy call from CLI). - unlock, err := s.acquireDomainDeployLock(ctx, route.Domain) - if err != nil { - return nil, err - } - defer unlock() - - // Enrich context with use case fields for all downstream logs +// Shutdown gracefully shuts down the container manager. +// Containers are left running across Gordon restarts by design; +// boot reconciliation picks them back up from ACTIVE state. +func (s *Service) Shutdown(ctx context.Context) error { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "Deploy", - "domain": route.Domain, + zerowrap.FieldUseCase: "Shutdown", }) log := zerowrap.FromCtx(ctx) + log.Info().Msg("shutting down container manager...") - // Record deploy metrics and trace status - defer s.recordDeployMetrics(ctx, span, route, deployStart, &err) - - existing, hasExisting := s.resolveExistingContainer(ctx, route.Domain) - - resources, err := s.prepareDeployResources(ctx, route, existing) - if err != nil { - if deployErr := s.wrapPullDeployFailure(route, existing, err); deployErr != nil { - err = deployErr - return nil, err - } - return nil, err - } - - // Skip redundant deploy: if the existing container is already running - // the exact same image (by Docker image ID), return it immediately. - // This prevents the double-deploy caused by the event-based deploy - // (triggered by image.pushed) racing with the explicit CLI deploy call. - if hasExisting { - existingForSkip := existing - if existingForSkip.ImageID == "" && existingForSkip.Image != "" && normalizeImageRef(existingForSkip.Image) == normalizeImageRef(route.Image) { - existingForSkip = s.containerForRedundantCheck(ctx, existingForSkip) - } - if existingForSkip.ImageID != "" { - if skip, container := s.skipRedundantDeploy(ctx, existingForSkip, resources.actualImageRef, resources.envHash); skip { - return container, nil - } - } - } - - newContainer, err := s.createStartedContainer(ctx, route, existing, resources) - if err != nil { - return nil, err - } - - invalidated := s.activateDeployedContainer(ctx, route.Domain, newContainer) - - // Post-switch stabilization: verify new container stays running - if hasExisting { - stable, stabilizeErr := s.stabilizeNewContainer(ctx, route.Domain, newContainer, existing) - if stabilizeErr != nil { - // Both old and new containers are dead; assign to named return so - // the deferred recordDeployMetrics and span see the failure. - err = stabilizeErr - return nil, err - } - if !stable { - // Rollback performed — old container is restored - return existing, nil - } - } - - // Finalize old container in the background — the new container is already - // serving traffic, so there's no reason to block the deploy response while - // waiting for the old container to stop (which can take 20s if the app - // doesn't handle SIGTERM). - s.cleanupWg.Add(1) - go func() { - defer s.cleanupWg.Done() - s.finalizePreviousContainer(context.WithoutCancel(ctx), route.Domain, existing, hasExisting, invalidated, newContainer.ID) - }() - - // Start container log collection (non-blocking, errors don't fail deployment) - s.startLogCollection(ctx, newContainer.ID, route.Domain) - - log.Info(). - Str("image", route.Image). - Str(zerowrap.FieldEntityID, newContainer.ID). - Ints("ports", newContainer.Ports). - Str("network", resources.networkName). - Bool("zero_downtime", hasExisting). - Msg("container deployed successfully") - - return newContainer, nil -} - -// recordDeployMetrics records span error status and deploy metrics at the end of a Deploy call. -// It is called via defer and receives a pointer to the named return error so it can observe -// the final error value after all deferred functions have run. -func (s *Service) recordDeployMetrics(ctx context.Context, span trace.Span, route domain.Route, start time.Time, errPtr *error) { - err := *errPtr - if err != nil { - span.RecordError(err) - span.SetStatus(codes.Error, err.Error()) - } - if s.metrics == nil { - return - } - attrs := metric.WithAttributes( - attribute.String("domain", route.Domain), - attribute.String("image", route.Image), - ) - s.metrics.DeployTotal.Add(ctx, 1, attrs) - s.metrics.DeployDuration.Record(ctx, time.Since(start).Seconds(), attrs) - if err != nil { - s.metrics.DeployErrors.Add(ctx, 1, attrs) - } -} - -// skipRedundantDeploy checks whether the existing container is already running -// the same image (by Docker image ID) as the one we are about to deploy. -// When a push triggers both an event-based deploy and an explicit CLI deploy, -// the second one arrives after the first has already completed; this avoids -// a full redundant create-start-readiness cycle. -func (s *Service) skipRedundantDeploy(ctx context.Context, existing *domain.Container, actualImageRef, envHash string) (bool, *domain.Container) { - log := zerowrap.FromCtx(ctx) - - newImageID, err := s.runtime.GetImageID(ctx, actualImageRef) - if err != nil { - log.Debug().Err(err).Msg("cannot resolve image ID for redundancy check, proceeding with deploy") - return false, nil - } - - if existing.ImageID != newImageID { - return false, nil - } - - // Verify the container is actually still running before skipping. - running, err := s.runtime.IsContainerRunning(ctx, existing.ID) - if err != nil || !running { - return false, nil - } - - if existing.Labels != nil { - existingEnvHash, hasEnvHash := existing.Labels[domain.LabelEnvHash] - if !hasEnvHash || existingEnvHash != envHash { - log.Info(). - Str("container_id", existing.ID). - Bool("has_env_hash", hasEnvHash). - Msg("env changed, proceeding with deploy despite same image") - return false, nil - } - } else { - return false, nil - } - - log.Info(). - Str("container_id", existing.ID). - Str("image_id", newImageID). - Msg("skipping redundant deploy: container already running this image") - - return true, existing -} - -func (s *Service) containerForRedundantCheck(ctx context.Context, existing *domain.Container) *domain.Container { - if existing == nil || existing.ImageID != "" || existing.ID == "" { - return existing - } - - log := zerowrap.FromCtx(ctx) - inspected, err := s.runtime.InspectContainer(ctx, existing.ID) - if err != nil { - log.Debug().Err(err).Str("container_id", existing.ID).Msg("cannot inspect existing container for redundancy check") - return existing - } - if inspected == nil || inspected.ImageID == "" { - return existing - } - - existingCopy := *existing - existingCopy.ImageID = inspected.ImageID - return &existingCopy -} - -type deployResources struct { - networkName string - actualImageRef string - exposedPorts []int - imageLabels map[string]string - envVars []string - envHash string - volumes map[string]string -} - -// resolveExistingContainer returns the currently running container for a domain. -// It first checks the in-memory map (fast path). If not found, it queries the -// runtime for a running container with matching name and managed label. This -// handles cases where Gordon restarted and in-memory state is stale. -// -// The slow path checks canonical, -new, and -next names because a previous -// deploy may have been interrupted (e.g. eventbus timeout, Gordon restart) -// leaving the active container under a temp name. Canonical is preferred when -// multiple matches exist. -func (s *Service) resolveExistingContainer(ctx context.Context, domainName string) (*domain.Container, bool) { - // Fast path: check in-memory state - s.mu.RLock() - container, ok := s.containers[domainName] - s.mu.RUnlock() - if ok { - return container, true - } - - // Slow path: query runtime for running containers - log := zerowrap.FromCtx(ctx) - canonicalName := managedContainerName(domainName) - candidateNames := map[string]bool{ - canonicalName: true, - canonicalName + "-new": true, - canonicalName + "-next": true, - } - - running, err := s.runtime.ListContainers(ctx, false) - if err != nil { - log.Warn().Err(err).Msg("failed to list running containers for existing container resolution") - return nil, false - } - - // Prefer canonical name; fall back to any running managed temp container. - var best *domain.Container - for _, c := range running { - if !candidateNames[c.Name] || c.Labels[domain.LabelManaged] != "true" { - continue - } - if best == nil || c.Name == canonicalName { - best = c - } - } - - if best != nil { - log.Info(). - Str("container_id", best.ID). - Str("container_name", best.Name). - Msg("resolved existing container from runtime (in-memory state was stale)") - - // Update in-memory state so subsequent lookups are fast - s.mu.Lock() - if _, alreadyTracked := s.containers[domainName]; !alreadyTracked { - s.managedCount++ - } - s.containers[domainName] = best - s.mu.Unlock() - - return best, true - } - - return nil, false -} - -func (s *Service) prepareDeployResources(ctx context.Context, route domain.Route, existing *domain.Container) (*deployResources, error) { - log := zerowrap.FromCtx(ctx) - - existingID := "" - if existing != nil { - existingID = existing.ID - } - if err := s.cleanupOrphanedContainers(ctx, route.Domain, existingID); err != nil { - log.WrapErr(err, "failed to cleanup orphaned containers") - } - - imageRef, err := s.buildValidatedImageRef(ctx, route.Image) - if err != nil { - return nil, err - } - actualImageRef, err := s.ensureImage(ctx, imageRef) - if err != nil { - return nil, err - } - - networkName := s.getNetworkForApp(route.Domain) - if err := s.createNetworkIfNeeded(ctx, networkName); err != nil { - return nil, log.WrapErr(err, "failed to create network") - } - if err := s.deployAttachments(ctx, route.Domain, networkName); err != nil { - return nil, log.WrapErr(err, "failed to deploy attachments") - } - - exposedPorts, err := s.runtime.GetImageExposedPorts(ctx, actualImageRef) - if err != nil { - log.WrapErr(err, "failed to get exposed ports, using defaults") - exposedPorts = []int{80, 8080, 3000} - } - - imageLabels, err := s.runtime.GetImageLabels(ctx, actualImageRef) - if err != nil { - log.Warn().Err(err).Msg("failed to get image labels, skipping label propagation") - imageLabels = nil - } - - envVars, err := s.loadEnvironment(ctx, route.Env, route.Domain, actualImageRef) - if err != nil { - return nil, err - } - envHash := hashEnvironment(envVars) - - volumes, err := s.setupVolumes(ctx, route.Domain, actualImageRef, nil) - if err != nil { - return nil, err - } - - return &deployResources{ - networkName: networkName, - actualImageRef: actualImageRef, - exposedPorts: exposedPorts, - imageLabels: imageLabels, - envVars: envVars, - envHash: envHash, - volumes: volumes, - }, nil -} - -func (s *Service) createStartedContainer(ctx context.Context, route domain.Route, existing *domain.Container, resources *deployResources) (*domain.Container, error) { - ctx, span := tracer.Start(ctx, "container.create_and_start") - defer span.End() - - log := zerowrap.FromCtx(ctx) - - containerConfig := s.buildContainerConfig(containerConfigInput{ - Domain: route.Domain, - Image: route.Image, - ImageRef: resources.actualImageRef, - ExposedPorts: resources.exposedPorts, - EnvVars: resources.envVars, - EnvHash: resources.envHash, - Volumes: resources.volumes, - NetworkName: resources.networkName, - ImageLabels: resources.imageLabels, - Existing: existing, - }) - - newContainer, err := s.runtime.CreateContainer(ctx, containerConfig) - if err != nil { - return nil, log.WrapErr(err, "failed to create container") - } - if err := s.runtime.StartContainer(ctx, newContainer.ID); err != nil { - s.runtime.RemoveContainer(ctx, newContainer.ID, true) - return nil, log.WrapErr(err, "failed to start container") - } - if !domain.IsSkipReadiness(ctx) { - if err := s.waitForReady(ctx, newContainer.ID, containerConfig); err != nil { - deployErr := s.newDeployFailure(newContainer, "container failed readiness check", err, s.captureRecentContainerLogs(ctx, newContainer.ID)) - s.cleanupFailedContainer(ctx, newContainer.ID) - return nil, deployErr - } - } - - inspected, err := s.runtime.InspectContainer(ctx, newContainer.ID) - if err != nil { - s.cleanupFailedContainer(ctx, newContainer.ID) - return nil, log.WrapErr(err, "failed to inspect started container") - } - - return inspected, nil -} - -func (s *Service) newDeployFailure(container *domain.Container, cause string, err error, logs []string) error { - deployErr := &domain.DeployFailureError{ - Summary: "failed to deploy", - Cause: cause, - Hint: inferDeployFailureHint(err), - Logs: logs, - Err: err, - } - if container != nil { - deployErr.ContainerName = container.Name - deployErr.ContainerID = container.ID - } - - return deployErr -} - -func (s *Service) wrapPullDeployFailure(route domain.Route, existing *domain.Container, err error) error { - if !isPullFailure(err) { - return nil - } - - return s.newDeployFailure(&domain.Container{Name: s.deploymentContainerName(route.Domain, existing)}, "failed to pull image", err, nil) -} - -func (s *Service) captureRecentContainerLogs(ctx context.Context, containerID string) []string { - if ctx.Err() != nil { - return nil - } - - logStream, err := s.runtime.GetContainerLogs(ctx, containerID, false) - if err != nil { - return nil - } - defer func() { - _ = logStream.Close() - }() - - logs, err := parseDockerLogLines(logStream, deployFailureLogLimit) - if err != nil { - return nil - } - - return logs -} - -func parseDockerLogLines(r io.Reader, maxLines int) ([]string, error) { - reader := bufio.NewReader(r) - collector := newLogLineCollector(maxLines) - - header, err := reader.Peek(8) - if err != nil && !errors.Is(err, io.EOF) && !errors.Is(err, bufio.ErrBufferFull) { - return nil, err - } - - if len(header) >= 8 && looksLikeDockerLogStream(header) { - return parseDockerFramedLogLines(reader, collector) - } - - return parsePlainLogLines(reader, collector) -} - -func parseDockerFramedLogLines(reader *bufio.Reader, collector *logLineCollector) ([]string, error) { - for { - frameHeader := make([]byte, 8) - if _, err := io.ReadFull(reader, frameHeader); err != nil { - if isLogStreamEOF(err) { - collector.flush() - return collector.lines, nil - } - return nil, err - } - - size := int(binary.BigEndian.Uint32(frameHeader[4:8])) - if size <= 0 { - continue - } - - payload, err := readLogPayload(reader, size) - if err != nil { - collector.addChunk(string(payload)) - collector.flush() - if isLogStreamEOF(err) { - return collector.lines, nil - } - return nil, err - } - - collector.addChunk(string(payload)) - } -} - -func parsePlainLogLines(reader *bufio.Reader, collector *logLineCollector) ([]string, error) { - - scanner := bufio.NewScanner(reader) - for scanner.Scan() { - collector.addLine(scanner.Text()) - } - if err := scanner.Err(); err != nil { - return nil, err - } - - collector.flush() - return collector.lines, nil -} - -func readLogPayload(reader *bufio.Reader, size int) ([]byte, error) { - payload := make([]byte, size) - _, err := io.ReadFull(reader, payload) - return payload, err -} - -func isLogStreamEOF(err error) bool { - return errors.Is(err, io.EOF) || errors.Is(err, io.ErrUnexpectedEOF) -} - -func looksLikeDockerLogStream(data []byte) bool { - return len(data) >= 8 && data[1] == 0 && data[2] == 0 && data[3] == 0 -} - -type logLineCollector struct { - lines []string - pending string - maxLines int -} - -func newLogLineCollector(maxLines int) *logLineCollector { - return &logLineCollector{maxLines: maxLines} -} - -func (c *logLineCollector) addChunk(chunk string) { - for _, part := range strings.SplitAfter(chunk, "\n") { - c.pending += part - if strings.HasSuffix(part, "\n") { - c.flush() + // Close log writer to stop all log collection + if s.logWriter != nil { + if err := s.logWriter.Close(); err != nil { + log.Warn().Err(err).Msg("failed to close container log writer") } } -} - -func (c *logLineCollector) addLine(line string) { - line = strings.TrimRight(line, "\r") - if line == "" { - return - } - c.lines = append(c.lines, line) - if c.maxLines > 0 && len(c.lines) > c.maxLines { - c.lines = c.lines[len(c.lines)-c.maxLines:] - } -} -func (c *logLineCollector) flush() { - if c.pending == "" { - return - } - c.addLine(strings.TrimRight(c.pending, "\r\n")) - c.pending = "" + log.Info().Msg("container manager shutdown complete") + return nil } -func inferDeployFailureHint(err error) string { - if err == nil { - return "" - } - - msg := strings.ToLower(err.Error()) - switch { - case strings.Contains(msg, "healthcheck timeout"): - return "check startup logs and confirm the health endpoint becomes ready within the configured timeout" - case strings.Contains(msg, "unhealthy"): - return "inspect recent app logs and verify the healthcheck command reflects actual readiness" - case strings.Contains(msg, "no healthcheck detected"): - return "add a Docker healthcheck or switch readiness mode" - case isPullFailure(err), strings.Contains(msg, "access denied"), strings.Contains(msg, "unauthorized"): - return "verify registry auth and confirm the image exists at the requested tag" - default: - return "" - } -} - -func isPullFailure(err error) bool { - if err == nil { - return false - } - - if errors.Is(err, domain.ErrImagePullFailed) { - return true - } - - msg := strings.ToLower(err.Error()) - return strings.Contains(msg, "failed to pull image") || - strings.Contains(msg, "pull access denied") -} - -func (s *Service) activateDeployedContainer(ctx context.Context, domainName string, container *domain.Container) bool { - s.mu.Lock() - _, wasTracked := s.containers[domainName] - s.containers[domainName] = container - if !wasTracked { - s.managedCount++ - } - s.mu.Unlock() - - // Track managed container count (only increment for new domains, not replacements) - if !wasTracked && s.metrics != nil { - s.metrics.ManagedContainers.Add(ctx, 1) - } - - s.publishContainerDeployed(ctx, domainName, container.ID) - - s.mu.RLock() - inv := s.cacheInvalidator - s.mu.RUnlock() - if inv != nil { - inv.InvalidateTarget(ctx, domainName) - return true - } - - return false -} - -// stabilizeNewContainer monitors the new container briefly after traffic switch. -// If it crashes during this window, rolls back to old container. -// Returns true if stabilization succeeded, false if rollback was performed. -// Returns an error only when both new and old containers are dead. -func (s *Service) stabilizeNewContainer(ctx context.Context, domainName string, newContainer, oldContainer *domain.Container) (bool, error) { - log := zerowrap.FromCtx(ctx) - - if oldContainer == nil { - return true, nil - } - - s.mu.RLock() - delay := s.config.StabilizationDelay - s.mu.RUnlock() - if delay == 0 { - delay = 2 * time.Second - } - - // Brief stabilization: verify new container is still running after delay - select { - case <-time.After(delay): - case <-ctx.Done(): - return true, nil - } - - running, err := s.runtime.IsContainerRunning(ctx, newContainer.ID) - if err != nil || !running { - log.Error(). - Str("new_container", newContainer.ID). - Str("old_container", oldContainer.ID). - Msg("new container crashed during stabilization, rolling back to old") - - // Verify old container is still running before restoring it. - // If both old and new are dead (e.g., OOM), returning a dead - // container as "existing" would leave the domain in a broken state. - oldRunning, oldErr := s.runtime.IsContainerRunning(ctx, oldContainer.ID) - if oldErr != nil || !oldRunning { - log.Error(). - Str("old_container", oldContainer.ID). - Msg("old container is also not running, cannot rollback") - - // Cleanup failed new container - if stopErr := s.runtime.StopContainer(ctx, newContainer.ID); stopErr != nil { - log.WrapErrWithFields(stopErr, "failed to stop failed new container during rollback", map[string]any{zerowrap.FieldEntityID: newContainer.ID}) - } - if removeErr := s.runtime.RemoveContainer(ctx, newContainer.ID, true); removeErr != nil { - log.WrapErrWithFields(removeErr, "failed to remove failed new container during rollback", map[string]any{zerowrap.FieldEntityID: newContainer.ID}) - } - - return false, fmt.Errorf("stabilization failed: new container crashed and old container %s is also not running", oldContainer.ID) - } - - // Rollback: restore old container as tracked - s.mu.Lock() - s.containers[domainName] = oldContainer - s.mu.Unlock() - - // Re-invalidate proxy cache to point back to old - s.mu.RLock() - inv := s.cacheInvalidator - s.mu.RUnlock() - if inv != nil { - inv.InvalidateTarget(ctx, domainName) - } - - // Cleanup failed new container - if stopErr := s.runtime.StopContainer(ctx, newContainer.ID); stopErr != nil { - log.WrapErrWithFields(stopErr, "failed to stop failed new container during rollback", map[string]any{zerowrap.FieldEntityID: newContainer.ID}) - } - if removeErr := s.runtime.RemoveContainer(ctx, newContainer.ID, true); removeErr != nil { - log.WrapErrWithFields(removeErr, "failed to remove failed new container during rollback", map[string]any{zerowrap.FieldEntityID: newContainer.ID}) - } - - return false, nil - } - - return true, nil -} - -func (s *Service) finalizePreviousContainer(ctx context.Context, domainName string, existing *domain.Container, hasExisting, invalidated bool, newContainerID string) { - if !hasExisting { - return - } - - if invalidated { - s.waitForDrain(ctx, existing.ID) - } - - s.cleanupOldContainer(ctx, existing, newContainerID, domainName) -} - -func (s *Service) waitForDrain(ctx context.Context, oldContainerID string) { - log := zerowrap.FromCtx(ctx) - - s.mu.RLock() - cfg := s.config - waiter := s.drainWaiter - s.mu.RUnlock() - - drainMode := cfg.DrainMode - if drainMode == "" { - drainMode = "delay" - } - - shouldUseInFlight := (drainMode == "inflight" || drainMode == "auto") && waiter != nil - if shouldUseInFlight { - timeout := cfg.DrainTimeout - if timeout == 0 { - timeout = 30 * time.Second - } - drained := waiter.WaitForNoInFlight(ctx, oldContainerID, timeout) - if !drained { - log.Warn(). - Str("old_container_id", oldContainerID). - Dur("drain_timeout", timeout). - Msg("drain wait timed out; old container may still have in-flight traffic") - } - return - } - - drainDelay := 2 * time.Second - if cfg.DrainDelayConfigured { - drainDelay = cfg.DrainDelay - } - if drainDelay <= 0 { - return - } - select { - case <-time.After(drainDelay): - case <-ctx.Done(): - } -} - -// restartContainerWithRecovery restarts a container, reconciling stale state if needed. -// It returns the (possibly refreshed) container and updated attachment IDs. -func (s *Service) restartContainerWithRecovery(ctx context.Context, domainName string, container *domain.Container, attachmentIDs []string) (*domain.Container, []string, error) { - log := zerowrap.FromCtx(ctx) - - if err := s.runtime.RestartContainer(ctx, container.ID); err != nil { - if !isContainerNotFoundError(err) { - return nil, nil, log.WrapErr(err, "failed to restart container") - } - - // In-memory container ID can become stale after external runtime changes. - // Re-sync and retry once with the latest tracked container. - log.Warn(). - Err(err). - Str(zerowrap.FieldEntityID, container.ID). - Msg("tracked container missing during restart, attempting state reconciliation") - - if syncErr := s.SyncContainers(ctx); syncErr != nil { - log.Warn().Err(syncErr).Msg("failed to sync container state during restart recovery") - } - - s.mu.RLock() - refreshed, refreshedExists := s.containers[domainName] - attachmentIDs = append([]string{}, s.attachments[domainName]...) - s.mu.RUnlock() - - if !refreshedExists || refreshed == nil { - return nil, nil, domain.ErrContainerNotFound - } - if err := s.runtime.RestartContainer(ctx, refreshed.ID); err != nil { - return nil, nil, log.WrapErr(err, "failed to restart container after state reconciliation") - } - return refreshed, attachmentIDs, nil - } - - return container, attachmentIDs, nil -} - -// Restart restarts a running container for the given domain. -func (s *Service) Restart(ctx context.Context, domainName string, withAttachments bool) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "Restart", - "domain": domainName, - }) - log := zerowrap.FromCtx(ctx) - - s.mu.RLock() - container, exists := s.containers[domainName] - attachments := s.attachments[domainName] - var attachmentIDs []string - if len(attachments) > 0 { - attachmentIDs = make([]string, len(attachments)) - copy(attachmentIDs, attachments) - } - s.mu.RUnlock() - - if !exists || container == nil { - return domain.ErrContainerNotFound - } - - // When attachments are requested, sync runtime state first to get accurate - // attachment tracking, then check if configured attachments are deployed. - // If fewer attachment containers are tracked than configured, fail early with - // guidance rather than restarting the main container into a broken state - // (e.g., missing database hostname). - if withAttachments { - if syncErr := s.SyncContainers(ctx); syncErr != nil { - log.Warn().Err(syncErr).Msg("failed to sync container state before attachment check") - } - - // Re-read attachment IDs after sync - s.mu.RLock() - if freshAttachments := s.attachments[domainName]; len(freshAttachments) > 0 { - attachmentIDs = make([]string, len(freshAttachments)) - copy(attachmentIDs, freshAttachments) - } else { - attachmentIDs = nil - } - s.mu.RUnlock() - - configuredAttachments := s.resolveAttachmentsForDomain(domainName) - if len(configuredAttachments) > 0 && len(attachmentIDs) < len(configuredAttachments) { - missing := len(configuredAttachments) - len(attachmentIDs) - return fmt.Errorf("%w: domain %q has %d configured attachment(s) but only %d deployed (%d missing)", - domain.ErrAttachmentNotDeployed, domainName, len(configuredAttachments), len(attachmentIDs), missing) - } - } - - // Restart the main container (with stale-state recovery) - container, attachmentIDs, err := s.restartContainerWithRecovery(ctx, domainName, container, attachmentIDs) - if err != nil { - return err - } - log.Info().Str(zerowrap.FieldEntityID, container.ID).Msg("container restarted") - - // Record restart metric - if s.metrics != nil { - s.metrics.ContainerRestarts.Add(ctx, 1, metric.WithAttributes( - attribute.String("domain", domainName), - attribute.String("source", "api"), - )) - } - - // Restart attachments if requested - if withAttachments && len(attachmentIDs) > 0 { - for _, attachID := range attachmentIDs { - if err := s.runtime.RestartContainer(ctx, attachID); err != nil { - log.Warn().Err(err).Str("attachment_id", attachID).Msg("failed to restart attachment") - continue // Don't fail the whole operation for one attachment - } - log.Info().Str("attachment_id", attachID).Msg("attachment restarted") - } - } - - return nil -} - -// Stop stops a running container. -func (s *Service) Stop(ctx context.Context, containerID string) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "Stop", - zerowrap.FieldEntityID: containerID, - }) - log := zerowrap.FromCtx(ctx) - - // Stop log collection before stopping container - if s.logWriter != nil { - if err := s.logWriter.StopLogging(containerID); err != nil { - log.Warn().Err(err).Msg("failed to stop container log collection") - } - } - - if err := s.runtime.StopContainer(ctx, containerID); err != nil { - return log.WrapErr(err, "failed to stop container") - } - - log.Info().Msg("container stopped") - return nil -} - -// PreviewRemovedRouteCleanup reports retained runtime state for a route that is -// no longer configured, without mutating containers or in-memory tracking. -func (s *Service) PreviewRemovedRouteCleanup(ctx context.Context, domainName string) (*domain.CleanupReport, error) { - canonicalDomain, ok := domain.CanonicalRouteDomain(domainName) - if !ok { - return &domain.CleanupReport{Domain: domainName}, domain.ErrRouteDomainInvalid - } - domainName = canonicalDomain - - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "PreviewRemovedRouteCleanup", - "domain": domainName, - }) - log := zerowrap.FromCtx(ctx) - - report := &domain.CleanupReport{Domain: domainName} - - unlock, err := s.acquireDomainDeployLock(ctx, domainName) - if err != nil { - return report, err - } - defer unlock() - - allContainers, err := s.runtime.ListContainers(ctx, true) - if err != nil { - return report, log.WrapErr(err, "failed to list containers for removed route cleanup preview") - } - - routeContainers := routeContainersForDomain(allContainers, domainName) - report.PreservedAttachments = preservedAttachmentsForDomain(allContainers, domainName) - if len(report.PreservedAttachments) > 0 { - report.Hints = append(report.Hints, "attachments preserved; remove or purge attachments explicitly if they are no longer needed") - } - for _, c := range routeContainers { - report.OrphanedEntities = append(report.OrphanedEntities, domain.CleanupOrphanedEntity{ - Kind: "route_container", - ID: c.ID, - Name: c.Name, - Status: c.Status, - Reason: "route runtime state still exists after route removal", - }) - } - - return report, nil -} - -// ReconcileRemovedRoute removes active runtime containers for a route that was -// removed from configuration while preserving stateful resources for explicit -// follow-up cleanup. It intentionally removes only main route containers, never -// attachment containers or volumes. -func (s *Service) ReconcileRemovedRoute(ctx context.Context, domainName string) (*domain.CleanupReport, error) { - canonicalDomain, ok := domain.CanonicalRouteDomain(domainName) - if !ok { - return &domain.CleanupReport{Domain: domainName}, domain.ErrRouteDomainInvalid - } - domainName = canonicalDomain - - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "ReconcileRemovedRoute", - "domain": domainName, - }) - log := zerowrap.FromCtx(ctx) - - report := &domain.CleanupReport{Domain: domainName} - - unlock, err := s.acquireDomainDeployLock(ctx, domainName) - if err != nil { - return report, err - } - defer unlock() - - allContainers, err := s.runtime.ListContainers(ctx, true) - if err != nil { - return report, log.WrapErr(err, "failed to list containers for removed route cleanup") - } - - routeContainers := routeContainersForDomain(allContainers, domainName) - - // Clear in-memory state before stopping containers so the monitor cannot - // interpret the deliberate removal as a crash that should be restarted. - s.clearRouteContainerTracking(ctx, domainName) - if inv := s.proxyCacheInvalidator(); inv != nil { - inv.InvalidateTarget(ctx, domainName) - } - - report.PreservedAttachments = preservedAttachmentsForDomain(allContainers, domainName) - if len(report.PreservedAttachments) > 0 { - report.Hints = append(report.Hints, "attachments preserved; remove or purge attachments explicitly if they are no longer needed") - } - - if len(routeContainers) == 0 { - report.Warnings = append(report.Warnings, "no active route container found for removed route") - return report, nil - } - - s.removeRouteContainersForCleanup(ctx, log, routeContainers, report) - - if len(report.PartialFailures) > 0 { - report.Warnings = append(report.Warnings, "route cleanup completed with partial failures; manual follow-up may be required") - } - - return report, nil -} - -func routeContainersForDomain(containers []*domain.Container, domainName string) []*domain.Container { - var routeContainers []*domain.Container - for _, c := range containers { - if isManagedRouteContainerForDomain(c, domainName) { - routeContainers = append(routeContainers, c) - } - } - return routeContainers -} - -func preservedAttachmentsForDomain(containers []*domain.Container, domainName string) []domain.CleanupAttachment { - var attachments []domain.CleanupAttachment - for _, c := range containers { - if isPreservedAttachmentForDomain(c, domainName) { - attachments = append(attachments, domain.CleanupAttachment{ - Name: c.Name, - Image: c.Image, - ContainerID: c.ID, - Status: c.Status, - Owner: c.Labels[domain.LabelAttachedTo], - Reason: "attachments are preserved by default; remove or purge them explicitly", - }) - } - } - return attachments -} - -func (s *Service) removeRouteContainersForCleanup(ctx context.Context, log zerowrap.Logger, containers []*domain.Container, report *domain.CleanupReport) { - for _, c := range containers { - s.stopRouteLogCollectionForCleanup(log, c, report) - s.stopRouteContainerForCleanup(ctx, log, c, report) - if err := s.runtime.RemoveContainer(ctx, c.ID, true); err != nil { - report.PartialFailures = append(report.PartialFailures, domain.CleanupFailure{Action: "remove", Kind: "container", ID: c.ID, Name: c.Name, Error: err.Error()}) - log.Warn().Err(err).Str(zerowrap.FieldEntityID, c.ID).Msg("failed to remove removed route container") - continue - } - report.RemovedContainers = append(report.RemovedContainers, domain.CleanupContainer{ID: c.ID, Name: c.Name, Image: c.Image, Status: c.Status}) - } -} - -func (s *Service) stopRouteLogCollectionForCleanup(log zerowrap.Logger, c *domain.Container, report *domain.CleanupReport) { - if s.logWriter == nil { - return - } - if err := s.logWriter.StopLogging(c.ID); err != nil { - report.PartialFailures = append(report.PartialFailures, domain.CleanupFailure{Action: "stop_logging", Kind: "container", ID: c.ID, Name: c.Name, Error: err.Error()}) - log.Warn().Err(err).Str(zerowrap.FieldEntityID, c.ID).Msg("failed to stop removed route log collection") - } -} - -func (s *Service) stopRouteContainerForCleanup(ctx context.Context, log zerowrap.Logger, c *domain.Container, report *domain.CleanupReport) { - if c.Status != string(domain.ContainerStatusRunning) { - return - } - if err := s.runtime.StopContainer(ctx, c.ID); err != nil { - report.PartialFailures = append(report.PartialFailures, domain.CleanupFailure{Action: "stop", Kind: "container", ID: c.ID, Name: c.Name, Error: err.Error()}) - log.Warn().Err(err).Str(zerowrap.FieldEntityID, c.ID).Msg("failed to stop removed route container") - } -} - -func isManagedRouteContainerForDomain(c *domain.Container, domainName string) bool { - if c == nil || c.Labels == nil || c.Labels[domain.LabelManaged] != "true" { - return false - } - if c.Labels[domain.LabelAttachment] == "true" { - return false - } - return c.Labels[domain.LabelRoute] == domainName || c.Labels[domain.LabelDomain] == domainName || c.Name == managedContainerName(domainName) -} - -func isPreservedAttachmentForDomain(c *domain.Container, domainName string) bool { - if c == nil || c.Labels == nil || c.Labels[domain.LabelManaged] != "true" { - return false - } - return c.Labels[domain.LabelAttachment] == "true" && c.Labels[domain.LabelAttachedTo] == domainName -} - -func (s *Service) proxyCacheInvalidator() out.ProxyCacheInvalidator { - s.mu.RLock() - defer s.mu.RUnlock() - return s.cacheInvalidator -} - -func (s *Service) clearRouteContainerTracking(ctx context.Context, domainName string) { - s.mu.Lock() - removed := false - if _, exists := s.containers[domainName]; exists { - delete(s.containers, domainName) - if s.managedCount > 0 { - s.managedCount-- - } - removed = true - } - s.mu.Unlock() - - if removed && s.metrics != nil { - s.metrics.ManagedContainers.Add(ctx, -1) - } -} - -// Remove removes a container. -func (s *Service) Remove(ctx context.Context, containerID string, force bool) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "Remove", - zerowrap.FieldEntityID: containerID, - }) - log := zerowrap.FromCtx(ctx) - - // Stop log collection before removing container - if s.logWriter != nil { - if err := s.logWriter.StopLogging(containerID); err != nil { - log.Warn().Err(err).Msg("failed to stop container log collection") - } - } - - // Find domain and attachment IDs for cleanup - containerDomain, attachmentIDs, cfg := s.findContainerContext(containerID) - - if err := s.runtime.RemoveContainer(ctx, containerID, force); err != nil { - return log.WrapErr(err, "failed to remove container") - } - - // Cleanup volumes - if containerDomain != "" && !cfg.VolumePreserve { - if err := s.cleanupVolumesForDomain(ctx, containerDomain); err != nil { - log.WrapErrWithFields(err, "failed to cleanup volumes", map[string]any{"domain": containerDomain}) - } - } - - // Clean up attachment containers - s.removeAttachments(ctx, attachmentIDs) - - // Remove from tracking +// UpdateConfig updates the service configuration. +func (s *Service) UpdateConfig(config Config) { s.mu.Lock() - var removedDomain string - for d, c := range s.containers { - if c.ID == containerID { - delete(s.containers, d) - removedDomain = d - s.managedCount-- - log.Info().Str("domain", d).Msg("container removed") - break - } - } - if removedDomain != "" { - delete(s.attachments, removedDomain) - } + s.config = config s.mu.Unlock() - - // Decrement managed container count - if removedDomain != "" && s.metrics != nil { - s.metrics.ManagedContainers.Add(ctx, -1) - } - - // Note: We intentionally do NOT clean up deployMu entries to avoid a race condition: - // If Remove deletes the mutex entry while a concurrent Deploy is acquiring the lock, - // the Deploy would create a fresh mutex, breaking serialization and allowing concurrent - // deploys to the same domain. The memory footprint is acceptable (one small struct per - // domain ever deployed), and this is the safer choice for correctness. - - // Cleanup network - if removedDomain != "" && cfg.NetworkIsolation { - networkName := s.getNetworkForApp(removedDomain) - if err := s.cleanupNetworkIfEmpty(ctx, networkName); err != nil { - log.WrapErrWithFields(err, "failed to cleanup network", map[string]any{"network": networkName}) - } - } - - return nil -} - -// findContainerContext looks up the domain, attachment IDs, and config snapshot for a container. -func (s *Service) findContainerContext(containerID string) (string, []string, Config) { - s.mu.RLock() - defer s.mu.RUnlock() - cfg := s.config - var containerDomain string - for d, c := range s.containers { - if c.ID == containerID { - containerDomain = d - break - } - } - var attachmentIDs []string - if containerDomain != "" { - attachmentIDs = append(attachmentIDs, s.attachments[containerDomain]...) - } - return containerDomain, attachmentIDs, cfg -} - -// removeAttachments stops and removes attachment containers best-effort. -func (s *Service) removeAttachments(ctx context.Context, attachmentIDs []string) { - log := zerowrap.FromCtx(ctx) - for _, attachID := range attachmentIDs { - if s.logWriter != nil { - if err := s.logWriter.StopLogging(attachID); err != nil { - log.Warn().Err(err).Str("attachment_id", attachID).Msg("failed to stop attachment log collection") - } - } - if err := s.runtime.StopContainer(ctx, attachID); err != nil { - log.Warn().Err(err).Str("attachment_id", attachID).Msg("failed to stop attachment container") - } - if err := s.runtime.RemoveContainer(ctx, attachID, true); err != nil { - log.Warn().Err(err).Str("attachment_id", attachID).Msg("failed to remove attachment container") - } - } -} - -// Get retrieves a container by domain. -func (s *Service) Get(_ context.Context, domainName string) (*domain.Container, bool) { - s.mu.RLock() - defer s.mu.RUnlock() - container, exists := s.containers[domainName] - return container, exists -} - -// List returns all managed containers. -func (s *Service) List(_ context.Context) map[string]*domain.Container { - s.mu.RLock() - defer s.mu.RUnlock() - - result := make(map[string]*domain.Container, len(s.containers)) - maps.Copy(result, s.containers) - return result -} - -// ListRoutesWithDetails returns routes with network and attachment info. -// Note: Container map is copied under lock, then external calls are made without lock. -// If containers are removed between copy and runtime calls, errors are handled gracefully -// (network becomes empty string). This trade-off avoids holding locks during I/O. -func (s *Service) ListRoutesWithDetails(ctx context.Context) []domain.RouteInfo { - s.mu.RLock() - containers := make(map[string]*domain.Container, len(s.containers)) - maps.Copy(containers, s.containers) - s.mu.RUnlock() - - // Fetch all attachments once to avoid N+1 queries - attachmentsByDomain := s.getAllAttachments(ctx) - - results := make([]domain.RouteInfo, 0, len(containers)) - for domainName, container := range containers { - network := "" - image := "" - containerID := "" - status := "" - if container != nil { - containerID = container.ID - status = container.Status - image = container.Image - if container.Labels != nil { - if labelImage, ok := container.Labels[domain.LabelImage]; ok && labelImage != "" { - image = labelImage - } - } - // Strip registry domain prefix from image for cleaner display - image = s.stripRegistryPrefix(image) - if networkName, err := s.runtime.GetContainerNetwork(ctx, container.ID); err == nil { - network = networkName - } - } - - results = append(results, domain.RouteInfo{ - Domain: domainName, - Image: image, - ContainerID: containerID, - ContainerStatus: status, - Network: network, - Attachments: attachmentsByDomain[domainName], - }) - } - - return results -} - -// stripRegistryPrefix removes the configured current or legacy Gordon registry -// domain prefix from an image reference. For example, -// "reg.example.com/myapp:latest" becomes "myapp:latest" when -// "reg.example.com" is a current or legacy Gordon registry domain. -func (s *Service) stripRegistryPrefix(image string) string { - s.mu.RLock() - cfg := s.config - s.mu.RUnlock() - - return domain.StripKnownGordonRegistry(image, cfg.RegistryDomain, cfg.LegacyRegistryDomains) -} - -// ListAttachments returns attachments for a domain. -func (s *Service) ListAttachments(ctx context.Context, domainName string) []domain.Attachment { - return s.getAttachmentsForDomain(ctx, domainName) -} - -// ListOrphanedAttachments returns managed attachment containers no longer configured. -func (s *Service) ListOrphanedAttachments(ctx context.Context) ([]domain.CleanupAttachment, error) { - containers, err := s.runtime.ListContainers(ctx, true) - if err != nil { - return nil, fmt.Errorf("failed to list containers with runtime for orphaned attachments: %w", err) - } - return s.orphanedAttachmentsFromContainers(containers), nil -} - -// CleanupOrphanedAttachments optionally stops/removes orphaned attachment containers. -// When owner is non-empty, cleanup is scoped to that attachment owner. -func (s *Service) CleanupOrphanedAttachments(ctx context.Context, owner string, stop bool) (*domain.CleanupReport, error) { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "CleanupOrphanedAttachments", - "owner": owner, - }) - log := zerowrap.FromCtx(ctx) - report := &domain.CleanupReport{Domain: owner} - - containers, err := s.runtime.ListContainers(ctx, true) - if err != nil { - return report, fmt.Errorf("failed to list containers with runtime for orphaned attachment cleanup: %w", err) - } - orphans := filterAttachmentContainersByOwner(s.orphanedAttachmentContainers(containers), owner) - if !stop { - report.PreservedAttachments = cleanupAttachmentsFromContainers(orphans) - if len(report.PreservedAttachments) > 0 { - report.Hints = append(report.Hints, "orphaned attachments preserved; rerun with --stop to stop/remove containers while preserving volumes") - } - return report, nil - } - - for _, c := range orphans { - s.stopRouteLogCollectionForCleanup(log, c, report) - s.stopRouteContainerForCleanup(ctx, log, c, report) - if err := s.runtime.RemoveContainer(ctx, c.ID, true); err != nil { - report.PartialFailures = append(report.PartialFailures, domain.CleanupFailure{Action: "remove", Kind: "attachment", ID: c.ID, Name: c.Name, Error: err.Error()}) - log.Warn().Err(err).Str(zerowrap.FieldEntityID, c.ID).Msg("failed to remove orphaned attachment") - continue - } - report.RemovedContainers = append(report.RemovedContainers, domain.CleanupContainer{ID: c.ID, Name: c.Name, Image: attachmentImage(c), Status: c.Status}) - s.untrackAttachmentContainer(c.Labels[domain.LabelAttachedTo], c.ID) - } - if len(report.RemovedContainers) > 0 { - report.Hints = append(report.Hints, "attachment volumes and data were preserved") - } - if len(report.PartialFailures) > 0 { - report.Warnings = append(report.Warnings, "attachment cleanup completed with partial failures; manual follow-up may be required") - } - return report, nil -} - -func (s *Service) orphanedAttachmentsFromContainers(containers []*domain.Container) []domain.CleanupAttachment { - return cleanupAttachmentsFromContainers(s.orphanedAttachmentContainers(containers)) -} - -func (s *Service) orphanedAttachmentContainers(containers []*domain.Container) []*domain.Container { - var out []*domain.Container - for _, c := range containers { - if !isManagedAttachment(c) { - continue - } - owner := c.Labels[domain.LabelAttachedTo] - image := attachmentImage(c) - if owner == "" || image == "" || !s.isAttachmentConfiguredForOwner(owner, image) { - out = append(out, c) - } - } - return out -} - -func filterAttachmentContainersByOwner(containers []*domain.Container, owner string) []*domain.Container { - if owner == "" { - return containers - } - var out []*domain.Container - for _, c := range containers { - if c != nil && c.Labels != nil && strings.EqualFold(c.Labels[domain.LabelAttachedTo], owner) { - out = append(out, c) - } - } - return out -} - -func cleanupAttachmentsFromContainers(containers []*domain.Container) []domain.CleanupAttachment { - attachments := make([]domain.CleanupAttachment, 0, len(containers)) - for _, c := range containers { - attachments = append(attachments, domain.CleanupAttachment{ - Name: c.Name, - Image: attachmentImage(c), - ContainerID: c.ID, - Status: c.Status, - Owner: c.Labels[domain.LabelAttachedTo], - Reason: "attachment container is no longer configured", - }) - } - return attachments -} - -func isManagedAttachment(c *domain.Container) bool { - return c != nil && c.Labels != nil && c.Labels[domain.LabelManaged] == "true" && c.Labels[domain.LabelAttachment] == "true" -} - -func attachmentImage(c *domain.Container) string { - if c == nil { - return "" - } - if c.Labels != nil && c.Labels[domain.LabelImage] != "" { - return c.Labels[domain.LabelImage] - } - return c.Image -} - -func (s *Service) isAttachmentConfiguredForOwner(owner, image string) bool { - return slices.Contains(s.resolveAttachmentsForDomain(owner), image) -} - -func (s *Service) untrackAttachmentContainer(owner, containerID string) { - s.mu.Lock() - defer s.mu.Unlock() - ids := s.attachments[owner] - for i, id := range ids { - if id == containerID { - s.attachments[owner] = slices.Delete(ids, i, i+1) - break - } - } - if len(s.attachments[owner]) == 0 { - delete(s.attachments, owner) - } -} - -// ListNetworks returns Gordon-managed networks. -func (s *Service) ListNetworks(ctx context.Context) ([]*domain.NetworkInfo, error) { - s.mu.RLock() - cfg := s.config - s.mu.RUnlock() - - networks, err := s.runtime.ListNetworks(ctx) - if err != nil { - return nil, err - } - - var filtered []*domain.NetworkInfo - for _, network := range networks { - if strings.HasPrefix(network.Name, cfg.NetworkPrefix+"-") && network.Labels[domain.LabelManaged] == "true" { - filtered = append(filtered, network) - } - } - - return filtered, nil -} - -// HealthCheck performs health checks on all containers. -func (s *Service) HealthCheck(ctx context.Context) map[string]bool { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "HealthCheck", - }) - log := zerowrap.FromCtx(ctx) - - // Copy container IDs under lock, then release before making Docker API calls - s.mu.RLock() - snapshot := make(map[string]string, len(s.containers)) - for d, c := range s.containers { - snapshot[d] = c.ID - } - s.mu.RUnlock() - - health := make(map[string]bool, len(snapshot)) - for d, id := range snapshot { - running, err := s.runtime.IsContainerRunning(ctx, id) - if err != nil { - log.WrapErrWithFields(err, "health check failed", map[string]any{"domain": d, zerowrap.FieldEntityID: id}) - health[d] = false - } else { - health[d] = running - } - } - return health -} - -// SyncContainers synchronizes containers with runtime state. -func isTrackedManagedContainerStatus(status string) bool { - return status == "" || status == string(domain.ContainerStatusRunning) || status == "restarting" -} - -func (s *Service) SyncContainers(ctx context.Context) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "SyncContainers", - }) - log := zerowrap.FromCtx(ctx) - - // List containers without holding lock to avoid blocking during the runtime call. - // Use all=true so we can rebuild the running-container maps from a single - // snapshot without a second list call. - allContainers, err := s.runtime.ListContainers(ctx, true) - if err != nil { - return log.WrapErr(err, "failed to list containers") - } - - managed := make(map[string]*domain.Container) - attachments := make(map[string][]string) - for _, c := range allContainers { - if !isTrackedManagedContainerStatus(c.Status) { - continue - } - if c.Labels == nil || c.Labels[domain.LabelManaged] != "true" { - continue - } - if c.Labels[domain.LabelAttachment] == "true" { - owner := c.Labels[domain.LabelAttachedTo] - if owner != "" { - attachments[owner] = append(attachments[owner], c.ID) - } - continue - } - if d, ok := c.Labels[domain.LabelDomain]; ok { - managed[d] = c - } - } - - s.mu.Lock() - s.containers = managed - s.attachments = attachments - newCount := int64(len(managed)) - delta := newCount - s.managedCount - s.managedCount = newCount - s.mu.Unlock() - - // Report initial/delta managed container count to OTel UpDownCounter. - if delta != 0 && s.metrics != nil { - s.metrics.ManagedContainers.Add(ctx, delta) - } - - log.Info().Int(zerowrap.FieldCount, len(managed)).Msg("container state synchronized") - return nil -} - -// AutoStart starts containers for the provided routes that aren't running. -// It skips readiness checks to avoid blocking boot; the background monitor -// handles crash recovery. -func (s *Service) AutoStart(ctx context.Context, routes []domain.Route) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "AutoStart", - }) - log := zerowrap.FromCtx(ctx) - - log.Info().Int("route_count", len(routes)).Msg("auto-starting containers for configured routes") - - // Filter out routes that already have a running container. - var pending []domain.Route - skipped := 0 - for _, route := range routes { - if _, exists := s.Get(ctx, route.Domain); exists { - log.Debug().Str("domain", route.Domain).Msg("container already running, skipping") - skipped++ - } else { - pending = append(pending, route) - } - } - - if len(pending) == 0 { - log.Info().Int("skipped", skipped).Msg("auto-start completed, all containers already running") - return nil - } - - // Deploy all pending routes concurrently with readiness checks skipped. - deployCtx := domain.WithSkipReadiness(ctx) - - type result struct { - route domain.Route - err error - } - results := make(chan result, len(pending)) - sem := make(chan struct{}, 4) // limit concurrency - - for _, route := range pending { - sem <- struct{}{} - go func(r domain.Route) { - defer func() { <-sem }() - - log.Info().Str("domain", r.Domain).Str("image", r.Image).Msg("auto-starting container for route") - - if _, err := s.Deploy(deployCtx, r); err != nil { - results <- result{route: r, err: err} - return - } - results <- result{route: r} - }(route) - } - - var started, errCount int - for i := 0; i < len(pending); i++ { - res := <-results - if res.err != nil { - log.Warn().Err(res.err).Str("domain", res.route.Domain).Msg("failed to auto-start container") - errCount++ - } else { - started++ - } - } - - log.Info(). - Int("started", started). - Int("skipped", skipped). - Int("errors", errCount). - Msg("auto-start completed") - - if errCount > 0 { - return fmt.Errorf("auto-start completed with %d errors", errCount) - } - return nil -} - -// Shutdown gracefully shuts down all managed containers. -func (s *Service) Shutdown(ctx context.Context) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "Shutdown", - }) - log := zerowrap.FromCtx(ctx) - log.Info().Msg("shutting down container manager...") - - // Wait for any in-flight background cleanup goroutines to finish - s.cleanupWg.Wait() - - // Containers are left running across Gordon restarts. - // SyncContainers + AutoStart will pick them back up on next boot. - - // Close log writer to stop all log collection - if s.logWriter != nil { - if err := s.logWriter.Close(); err != nil { - log.Warn().Err(err).Msg("failed to close container log writer") - } - } - - log.Info().Msg("container manager shutdown complete") - return nil -} - -// WaitForCleanup blocks until all background cleanup goroutines complete. -// Intended for use in tests to avoid mock assertion races. -func (s *Service) WaitForCleanup() { - s.cleanupWg.Wait() -} - -// StartMonitor begins background monitoring of tracked containers. -// The monitor restarts crashed containers and detects crash loops. -// Safe to call multiple times; subsequent calls are no-ops. -func (s *Service) StartMonitor(ctx context.Context) { - s.mu.Lock() - if s.monitor != nil { - s.mu.Unlock() - return - } - s.monitor = newMonitor(s) - m := s.monitor - m.Start(ctx) - s.mu.Unlock() -} - -// StopMonitor stops the background container monitor. -// Safe to call multiple times; subsequent calls are no-ops. -func (s *Service) StopMonitor() { - s.mu.Lock() - m := s.monitor - s.monitor = nil - s.mu.Unlock() - if m != nil { - m.Stop() - } -} - -// UpdateConfig updates the service configuration. -func (s *Service) UpdateConfig(config Config) { - s.mu.Lock() - s.config = config - s.mu.Unlock() -} - -// UpdateAttachments updates only the attachment configuration in the service. -// This is called after a config reload to propagate attachment changes without restart. -// The incoming map is deep-copied so external callers cannot mutate service state. -func (s *Service) UpdateAttachments(attachments map[string][]string) { - var copied map[string][]string - if attachments != nil { - copied = make(map[string][]string, len(attachments)) - for k, v := range attachments { - sl := make([]string, len(v)) - copy(sl, v) - copied[k] = sl - } - } - s.mu.Lock() - s.config.Attachments = copied - s.mu.Unlock() -} - -// Helper methods - -// cleanupFailedContainer stops and removes a container that failed to start properly. -func (s *Service) cleanupFailedContainer(ctx context.Context, containerID string) { - cleanupCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), failedContainerCleanupTimeout) - defer cancel() - - log := zerowrap.FromCtx(cleanupCtx) - if err := s.runtime.StopContainer(cleanupCtx, containerID); err != nil { - log.Warn().Err(err).Str(zerowrap.FieldEntityID, containerID).Msg("failed to stop container after failure") - } - if err := s.runtime.RemoveContainer(cleanupCtx, containerID, true); err != nil { - log.Warn().Err(err).Str(zerowrap.FieldEntityID, containerID).Msg("failed to remove container after failure") - } -} - -// cleanupOldContainer stops and removes an old container after zero-downtime switch. -// It also renames the new container to the canonical name. -func (s *Service) cleanupOldContainer(ctx context.Context, old *domain.Container, newContainerID, domainName string) { - log := zerowrap.FromCtx(ctx) - log.Info().Str(zerowrap.FieldEntityID, old.ID).Msg("stopping old container after zero-downtime switch") - - if s.logWriter != nil { - if err := s.logWriter.StopLogging(old.ID); err != nil { - log.Warn().Err(err).Str(zerowrap.FieldEntityID, old.ID).Msg("failed to stop logging for old container") - } - } - - if err := s.runtime.StopContainer(ctx, old.ID); err != nil { - log.Warn().Err(err).Str(zerowrap.FieldEntityID, old.ID).Msg("failed to stop old container") - } - - if err := s.runtime.RemoveContainer(ctx, old.ID, true); err != nil { - log.Warn().Err(err).Str(zerowrap.FieldEntityID, old.ID).Msg("failed to remove old container") - } - - // Rename new container to canonical name - canonicalName := managedContainerName(domainName) - if err := s.runtime.RenameContainer(ctx, newContainerID, canonicalName); err != nil { - log.Warn().Err(err).Str("canonical_name", canonicalName).Msg("failed to rename container to canonical name") - } -} - -// startLogCollection starts log collection for a container in the background. -// Errors are logged but don't fail the calling operation. -func (s *Service) startLogCollection(ctx context.Context, containerID, domainName string) { - if s.logWriter == nil { - return - } - - log := zerowrap.FromCtx(ctx) - - // Use context.WithoutCancel so the log stream outlives the HTTP request. - // Without this, the Docker log stream closes when the deploy response completes. - bgCtx := context.WithoutCancel(ctx) - - logStream, err := s.runtime.GetContainerLogs(bgCtx, containerID, true) - if err != nil { - log.Warn().Err(err).Msg("failed to get container logs for collection") - return - } - - if err := s.logWriter.StartLogging(bgCtx, containerID, domainName, logStream); err != nil { - log.Warn().Err(err).Msg("failed to start container log collection") - logStream.Close() - } -} - -func (s *Service) buildValidatedImageRef(ctx context.Context, image string) (string, error) { - imageRef := s.buildImageRef(image) - if err := s.validateImageRef(ctx, imageRef); err != nil { - return "", err - } - return imageRef, nil -} - -func (s *Service) buildImageRef(image string) string { - s.mu.RLock() - cfg := s.config - s.mu.RUnlock() - - if !cfg.RegistryAuthEnabled || cfg.RegistryDomain == "" { - return image - } - - // Normalize registry domain by trimming any trailing slash - reg := strings.TrimSuffix(cfg.RegistryDomain, "/") - - // Check if image already has the registry domain prefix - prefix := reg + "/" - if strings.HasPrefix(image, prefix) { - return image - } - - // Check if image already has an explicit registry (host:port/ or host.domain/). - // We need to detect patterns like: - // - "docker.io/library/nginx:latest" (has dot and slash) - // - "localhost:5001/myapp:latest" (has host:port/ pattern) - // - "myregistry:5000/app:v1" (has host:port/ pattern) - // - "[fd00::1]:5000/app:v1" (IPv6 with port) - if hasExplicitRegistry(image) { - return image - } - - return fmt.Sprintf("%s/%s", reg, image) -} - -// hasExplicitRegistry checks if an image reference already includes an explicit registry. -// Returns true for patterns like "docker.io/image", "localhost:5000/image", "localhost/image", -// "registry:8080/image", "[fd00::1]/image", "[fd00::1]:5000/image". -func hasExplicitRegistry(image string) bool { - // Find the first slash which separates registry from image name - slashIdx := strings.Index(image, "/") - if slashIdx == -1 { - return false // No slash means no registry prefix (e.g., "myapp:latest") - } - - // Extract the part before the first slash (potential registry) - registryPart := image[:slashIdx] - - // Check for bracketed IPv6 address (e.g., "[fd00::1]" or "[fd00::1]:5000") - if strings.HasPrefix(registryPart, "[") { - // Look for closing bracket - if closeBracket := strings.Index(registryPart, "]"); closeBracket != -1 { - // Valid bracketed IPv6: either ends at ] or has ]:port - return true - } - } - - // Check for localhost (with or without port) - if registryPart == "localhost" { - return true - } - - // Check for host:port pattern (e.g., "localhost:5000", "registry:8080") - if colonIdx := strings.LastIndex(registryPart, ":"); colonIdx != -1 { - port := registryPart[colonIdx+1:] - // If everything after colon is digits, it's a port - if len(port) > 0 && isNumeric(port) { - return true - } - } - - // Check for domain pattern (has a dot, e.g., "docker.io", "gcr.io") - if strings.Contains(registryPart, ".") { - return true - } - - return false -} - -func (s *Service) validateImageRef(ctx context.Context, imageRef string) error { - if !hasExplicitRegistry(imageRef) { - return nil - } - return s.validateExternalImageRef(ctx, imageRef, imageRegistry(imageRef)) -} - -func (s *Service) validateImagePullRef(ctx context.Context, imageRef string) error { - return s.validateExternalImageRef(ctx, imageRef, imageRegistryForPolicy(imageRef)) -} - -func (s *Service) validateExternalImageRef(ctx context.Context, imageRef, registry string) error { - s.mu.RLock() - cfg := s.config - s.mu.RUnlock() - - if registry == "" { - return fmt.Errorf("image %q rejected: invalid image reference", imageRef) - } - - if domain.IsGordonRegistryImageRef(imageRef, cfg.RegistryDomain, cfg.LegacyRegistryDomains) || sameRegistry(registry, cfg.RegistryDomain) { - return nil - } - dangerous, err := isDangerousRegistryHost(ctx, imageRegistryHost(registry)) - if err != nil { - return fmt.Errorf("image %q rejected: registry %q could not be verified: %w", imageRef, registry, err) - } - if dangerous { - return fmt.Errorf("image %q rejected: registry %q is not allowed", imageRef, registry) - } - if len(cfg.AllowedRegistries) == 0 { - return fmt.Errorf("image %q rejected: registry %q is not in images.allowed_registries", imageRef, registry) - } - for _, allowed := range cfg.AllowedRegistries { - if sameRegistry(registry, allowed) { - if cfg.RequireImageDigest && !isValidSha256DigestRef(imageRef) { - return fmt.Errorf("image %q rejected: digest reference required", imageRef) - } - return nil - } - } - return fmt.Errorf("image %q rejected: registry %q is not in images.allowed_registries", imageRef, registry) -} - -func imageRegistry(imageRef string) string { - slashIdx := strings.Index(imageRef, "/") - if slashIdx == -1 { - return "" - } - return normalizeRegistry(imageRef[:slashIdx]) -} - -func imageRegistryForPolicy(imageRef string) string { - if hasExplicitRegistry(imageRef) { - return imageRegistry(imageRef) - } - return "docker.io" -} - -// normalizeRegistry normalizes a registry string for comparison purposes only. -// This function is used by sameRegistry to determine if two registry strings -// refer to the same registry. The output intentionally strips IPv6 brackets -// (e.g., "[fd00::1]:5000" -> "fd00::1:5000") which is acceptable for equality -// checks but the result MUST NOT be used to construct or emit valid registry URLs. -func normalizeRegistry(registry string) string { - registry = strings.ToLower(strings.TrimSuffix(strings.TrimSpace(registry), ".")) - registry = strings.TrimSuffix(registry, "/") - if strings.HasPrefix(registry, "[") { - if end := strings.Index(registry, "]"); end != -1 { - host := strings.Trim(registry[:end+1], "[]") - port := registry[end+1:] - return host + port - } - } - return registry -} - -func imageRegistryHost(registry string) string { - if ip := net.ParseIP(registry); ip != nil { - return registry - } - if h, _, err := net.SplitHostPort(registry); err == nil { - return strings.Trim(h, "[]") - } - if strings.HasPrefix(registry, "[") && strings.Contains(registry, "]") { - return strings.TrimPrefix(strings.Split(registry, "]")[0], "[") - } - if colon := strings.LastIndex(registry, ":"); colon != -1 && isNumeric(registry[colon+1:]) { - return registry[:colon] - } - return registry -} - -func sameRegistry(a, b string) bool { - return normalizeRegistry(a) == normalizeRegistry(b) -} - -func isDangerousRegistryHost(ctx context.Context, host string) (bool, error) { - host = strings.ToLower(strings.TrimSuffix(host, ".")) - if host == "localhost" || host == "metadata.google.internal" { - return true, nil - } - if ip := net.ParseIP(host); ip != nil { - return isDangerousRegistryIP(ip), nil - } - addrs, err := net.DefaultResolver.LookupIPAddr(ctx, host) - if err != nil { - return false, err - } - for _, addr := range addrs { - if isDangerousRegistryIP(addr.IP) { - return true, nil - } - } - return false, nil -} - -func isDangerousRegistryIP(ip net.IP) bool { - return ip.IsLoopback() || ip.IsPrivate() || ip.IsLinkLocalUnicast() || ip.IsLinkLocalMulticast() || ip.IsUnspecified() -} - -func applySecurityProfile(config *domain.ContainerConfig, cfg Config) { - if strings.EqualFold(cfg.SecurityProfile, "strict") { - config.ReadOnlyRootFS = true - config.CapDrop = []string{"ALL"} - config.CapAdd = []string{"NET_BIND_SERVICE"} - } -} - -// isNumeric checks if a string contains only digits. -func isNumeric(s string) bool { - for _, c := range s { - if c < '0' || c > '9' { - return false - } - } - return true -} - -func normalizePullPolicy(policy string) string { - switch strings.ToLower(strings.TrimSpace(policy)) { - case PullPolicyAlways: - return PullPolicyAlways - case PullPolicyIfTagChanged: - return PullPolicyIfTagChanged - case PullPolicyIfNotPresent: - return PullPolicyIfNotPresent - default: - return PullPolicyIfNotPresent - } -} - -func isDigestRef(imageRef string) bool { - return isValidSha256DigestRef(imageRef) -} - -func isValidSha256DigestRef(imageRef string) bool { - _, digest, ok := strings.Cut(imageRef, "@sha256:") - if !ok || len(digest) != 64 { - return false - } - _, err := hex.DecodeString(digest) - return err == nil -} - -func (s *Service) pullRefForDeploy(ctx context.Context, imageRef string) (string, bool) { - if !domain.IsInternalDeploy(ctx) { - return imageRef, false - } - s.mu.RLock() - cfg := s.config - s.mu.RUnlock() - return rewriteToLocalRegistry(imageRef, cfg.RegistryDomain, cfg.LegacyRegistryDomains, cfg.RegistryPort), true -} - -// ensureImage ensures the image is available locally, pulling if needed. -// Returns the image reference to use for container operations: -// - For digest references (@sha256:...), returns the pullRef since Docker can't tag digests -// - For tagged images, returns the original imageRef after tagging the pulled image -func (s *Service) ensureImage(ctx context.Context, imageRef string) (string, error) { - ctx, span := tracer.Start(ctx, "container.ensure_image", - trace.WithAttributes(attribute.String("image", imageRef))) - defer span.End() - - ctx = zerowrap.CtxWithField(ctx, "image", imageRef) - log := zerowrap.FromCtx(ctx) - - // Determine if this is an internal deploy and what reference to use for pulls. - pullRef, isInternal := s.pullRefForDeploy(ctx, imageRef) - if err := s.validateImagePullRef(ctx, imageRef); err != nil { - return "", err - } - if pullRef != imageRef { - log.Info(). - Str("original_ref", imageRef). - Str("pull_ref", pullRef). - Msg("internal deploy: using localhost registry for pull") - } - - found, err := s.ensureLocalImage(ctx, imageRef, pullRef) - if err != nil { - return "", err - } - if found { - return imageRef, nil - } - - // Pull image - log.Info().Msg("pulling image from registry") - - if err := s.pullImage(ctx, pullRef, isInternal); err != nil { - return "", err - } - - // For digest references, we can't create a tag (Docker doesn't allow it). - // In this case, use the pullRef directly since the image is available under that reference. - if strings.Contains(imageRef, "@sha256:") { - log.Info(). - Str("pull_ref", pullRef). - Msg("digest reference: using pull reference for container operations") - return pullRef, nil - } - - if err := s.tagImageIfNeeded(ctx, pullRef, imageRef); err != nil { - return "", err - } - - // Clean up the temporary pull reference tag to avoid duplicate entries - if pullRef != imageRef { - if err := s.runtime.UntagImage(ctx, pullRef); err != nil { - // Log but don't fail - the canonical tag is already applied - log.Debug().Err(err).Str("pull_ref", pullRef).Msg("failed to remove temporary pull tag") - } - } - - log.Info().Msg("image pulled successfully") - return imageRef, nil -} - -func (s *Service) ensureLocalImage(ctx context.Context, imageRef, pullRef string) (bool, error) { - log := zerowrap.FromCtx(ctx) - - // For internal deploys (image push events), always pull fresh image - // because the same tag may reference new content (e.g., latest tag updated) - if domain.IsInternalDeploy(ctx) { - log.Info().Msg("internal deploy detected, forcing image pull to ensure latest content") - return false, nil - } - - s.mu.RLock() - cfg := s.config - s.mu.RUnlock() - - pullPolicy := normalizePullPolicy(cfg.PullPolicy) - switch pullPolicy { - case PullPolicyAlways: - log.Info().Str("pull_policy", pullPolicy).Msg("pull policy forces image pull") - return false, nil - case PullPolicyIfTagChanged: - if !isDigestRef(imageRef) { - log.Info().Str("pull_policy", pullPolicy).Msg("tag reference detected, pulling to check for updates") - return false, nil - } - } - - localImages, err := s.runtime.ListImages(ctx) - if err != nil { - log.WrapErr(err, "failed to list local images, will attempt pull") - return false, nil - } - - normalizedRef := normalizeImageRef(imageRef) - normalizedPullRef := normalizeImageRef(pullRef) - for _, img := range localImages { - normalizedImage := normalizeImageRef(img) - if normalizedImage == normalizedRef { - log.Info().Msg("image found locally, skipping pull") - return true, nil - } - if normalizedImage == normalizedPullRef { - if err := s.tagImageIfNeeded(ctx, pullRef, imageRef); err != nil { - return false, err - } - // Clean up the temporary pull reference tag - if pullRef != imageRef { - if err := s.runtime.UntagImage(ctx, pullRef); err != nil { - log.Debug().Err(err).Str("pull_ref", pullRef).Msg("failed to remove temporary pull tag") - } - } - log.Info().Msg("image found locally, skipping pull") - return true, nil - } - } - - return false, nil -} - -func (s *Service) pullImage(ctx context.Context, pullRef string, isInternal bool) error { - s.mu.RLock() - cfg := s.config - s.mu.RUnlock() - - log := zerowrap.FromCtx(ctx) - - switch { - case isInternal && cfg.RegistryAuthEnabled: - if cfg.InternalRegistryUsername == "" || cfg.InternalRegistryPassword == "" { - return log.WrapErr(fmt.Errorf("internal registry auth not configured"), "failed to pull image for internal deploy") - } - if err := s.pullImageWithRetry(ctx, func(ctx context.Context) error { - return s.runtime.PullImageWithAuth(ctx, pullRef, cfg.InternalRegistryUsername, cfg.InternalRegistryPassword) - }, isInternal); err != nil { - return log.WrapErr(fmt.Errorf("%w: %w", domain.ErrImagePullFailed, err), "failed to pull image with internal auth") - } - case isInternal: - if err := s.pullImageWithRetry(ctx, func(ctx context.Context) error { - return s.runtime.PullImage(ctx, pullRef) - }, isInternal); err != nil { - return log.WrapErr(fmt.Errorf("%w: %w", domain.ErrImagePullFailed, err), "failed to pull image") - } - default: - // Never send Gordon's service credentials to an external registry. - // External pulls are anonymous until per-registry credentials have an - // explicit trust-boundary-aware configuration path. - if err := s.runtime.PullImage(ctx, pullRef); err != nil { - return log.WrapErr(fmt.Errorf("%w: %w", domain.ErrImagePullFailed, err), "failed to pull image") - } - } - - return nil -} - -func (s *Service) pullImageWithRetry(ctx context.Context, pullFn func(context.Context) error, isInternal bool) error { - log := zerowrap.FromCtx(ctx) - - for attempt := 1; attempt <= internalPullMaxAttempts; attempt++ { - err := pullFn(ctx) - if err == nil { - return nil - } - if !isInternal || !isConnectionRefusedError(err) || attempt == internalPullMaxAttempts { - return fmt.Errorf("internal image pull failed after %d attempts: %w", attempt, err) - } - - backoff := time.Duration(attempt) * time.Second - log.Warn(). - Err(err). - Int("attempt", attempt). - Dur("backoff", backoff). - Msg("internal image pull failed with connection refused, retrying") - - select { - case <-time.After(backoff): - case <-ctx.Done(): - return ctx.Err() - } - } - return nil -} - -func (s *Service) tagImageIfNeeded(ctx context.Context, sourceRef, targetRef string) error { - if sourceRef == targetRef { - return nil - } - - // Cannot create a tag with a digest reference - Docker/Podman doesn't allow it. - // When using digest references (image@sha256:...), skip tagging as the image - // is already available by its digest. - if strings.Contains(targetRef, "@sha256:") { - return nil - } - - log := zerowrap.FromCtx(ctx) - if err := s.runtime.TagImage(ctx, sourceRef, targetRef); err != nil { - return log.WrapErr(err, "failed to tag image from pull reference") - } - return nil -} - -func (s *Service) loadEnvironment(ctx context.Context, preResolved []string, domainName, imageRef string) ([]string, error) { - log := zerowrap.FromCtx(ctx) - - var userEnvVars []string - if len(preResolved) > 0 { - userEnvVars = preResolved - } else { - var err error - userEnvVars, err = s.envLoader.LoadEnv(ctx, domainName) - if err != nil { - return nil, log.WrapErr(err, "failed to load environment variables") - } - } - - dockerfileEnvVars, err := s.runtime.InspectImageEnv(ctx, imageRef) - if err != nil { - log.WrapErr(err, "failed to inspect image environment") - dockerfileEnvVars = []string{} - } - - return mergeEnvironmentVariables(dockerfileEnvVars, userEnvVars), nil -} - -func (s *Service) setupVolumes(ctx context.Context, domainName, imageRef string, preferredVolumes map[string]namedVolumeMount) (map[string]string, error) { - s.mu.RLock() - cfg := s.config - s.mu.RUnlock() - - log := zerowrap.FromCtx(ctx) - volumes, err := s.validatePreferredVolumes(ctx, preferredVolumes) - if err != nil { - return nil, err - } - - if !cfg.VolumeAutoCreate { - return volumes, nil - } - - volumePaths, err := s.runtime.InspectImageVolumes(ctx, imageRef) - if err != nil { - log.WrapErr(err, "failed to inspect image volumes") - return volumes, nil - } - - for _, path := range volumePaths { - if _, preferred := volumes[path]; preferred { - continue - } - - name := generateVolumeName(cfg.VolumePrefix, domainName, path) - legacyName := legacyVolumeName(cfg.VolumePrefix, domainName, path) - legacyExists, err := s.runtime.VolumeExists(ctx, legacyName) - if err != nil { - log.WrapErrWithFields(err, "failed to check legacy volume", map[string]any{"volume": legacyName}) - continue - } - legacyOwnership := legacyVolumeOwnershipUnverified - if legacyExists { - legacyOwnership, err = s.legacyVolumeOwnership(ctx, legacyName, domainName) - if err != nil { - log.Warn().Err(err).Str("volume", legacyName).Msg("could not verify legacy volume ownership; using stable volume name") - } else if legacyOwnership == legacyVolumeOwnershipOwned { - volumes[path] = legacyName - continue - } - } - - exists, err := s.runtime.VolumeExists(ctx, name) - if err != nil { - log.WrapErrWithFields(err, "failed to check volume", map[string]any{"volume": name}) - continue - } - - if !exists { - if legacyExists && legacyOwnership != legacyVolumeOwnershipForeign { - // No verified mount may mean the owning container was already removed; - // it does not prove that the preserved volume belongs elsewhere. - return nil, fmt.Errorf("legacy volume %q exists but its ownership cannot be verified; refusing to create a replacement volume: %w", legacyName, domain.ErrVolumeOwnershipUnverified) - } - if err := s.runtime.CreateVolume(ctx, name); err != nil { - log.WrapErrWithFields(err, "failed to create volume", map[string]any{"volume": name}) - continue - } - log.Info().Str("volume", name).Str(zerowrap.FieldPath, path).Msg("created volume") - } - - volumes[path] = name - } - - return volumes, nil -} - -func (s *Service) validatePreferredVolumes(ctx context.Context, preferredVolumes map[string]namedVolumeMount) (map[string]string, error) { - volumes := make(map[string]string) - for path, preferred := range preferredVolumes { - if preferred.Name == "" { - continue - } - exists, err := s.runtime.VolumeExists(ctx, preferred.Name) - if err != nil { - return nil, fmt.Errorf("check previously mounted volume %q: %w", preferred.Name, err) - } - if !exists { - return nil, fmt.Errorf("previously mounted volume %q no longer exists: %w", preferred.Name, domain.ErrVolumeNotFound) - } - volumes[path] = preferred.Name - } - return volumes, nil -} - -func (s *Service) getNetworkForApp(domainName string) string { - var networkIsolation bool - var networkGroups map[string][]string - - if s.configProvider != nil { - snap := s.configProvider.GetAttachmentConfig() - networkGroups = snap.NetworkGroups - s.mu.RLock() - networkIsolation = s.config.NetworkIsolation - s.mu.RUnlock() - } else { - s.mu.RLock() - networkIsolation = s.config.NetworkIsolation - networkGroups = s.config.NetworkGroups - s.mu.RUnlock() - } - - if !networkIsolation { - return "bridge" - } - - for groupName, domains := range networkGroups { - if slices.Contains(domains, domainName) { - return s.generateNetworkName(groupName) - } - } - - return s.generateNetworkName(domainName) -} - -func (s *Service) generateNetworkName(identifier string) string { - s.mu.RLock() - cfg := s.config - s.mu.RUnlock() - - return fmt.Sprintf("%s-%s", cfg.NetworkPrefix, domain.StableResourceName(identifier)) -} - -func (s *Service) createNetworkIfNeeded(ctx context.Context, networkName string) error { - if networkName == "bridge" || networkName == "default" { - return nil - } - - ctx = zerowrap.CtxWithField(ctx, "network", networkName) - log := zerowrap.FromCtx(ctx) - - exists, err := s.runtime.NetworkExists(ctx, networkName) - if err != nil { - return log.WrapErr(err, "failed to check network existence") - } - - if !exists { - s.mu.RLock() - internal := s.config.NetworkInternal - s.mu.RUnlock() - if err := s.runtime.CreateNetwork(ctx, networkName, domain.NetworkConfig{ - Driver: "bridge", - Internal: internal, - Labels: map[string]string{domain.LabelManaged: "true"}, - }); err != nil { - return log.WrapErr(err, "failed to create network") - } - log.Info().Msg("created network for app isolation") - } - - return nil -} - -func (s *Service) cleanupOrphanedContainers(ctx context.Context, domainName string, skipContainerID string) error { - log := zerowrap.FromCtx(ctx) - expectedName := managedContainerName(domainName) - expectedNewName := expectedName + "-new" - expectedNextName := expectedName + "-next" - - allContainers, err := s.runtime.ListContainers(ctx, true) - if err != nil { - return err - } - - for _, c := range allContainers { - if (c.Name == expectedName || c.Name == expectedNewName || c.Name == expectedNextName) && c.ID != skipContainerID { - // Without a selected active container, preserve anything that may still - // be serving traffic. When an active container is selected, other active - // temp-name containers are stale candidates left by interrupted deploys. - isActive := c.Status == "running" || c.Status == "restarting" - if isActive && (skipContainerID == "" || c.Name == expectedName) { - log.Debug(). - Str(zerowrap.FieldEntityID, c.ID). - Str("container_name", c.Name). - Str(zerowrap.FieldStatus, c.Status). - Msg("skipping active container during orphan cleanup") - continue - } - - log.Info().Str(zerowrap.FieldEntityID, c.ID).Str("container_name", c.Name).Str(zerowrap.FieldStatus, c.Status).Msg("found orphaned container, removing") - - if err := s.runtime.StopContainer(ctx, c.ID); err != nil { - log.WrapErrWithFields(err, "failed to stop orphaned container", map[string]any{zerowrap.FieldEntityID: c.ID}) - } - - if err := s.runtime.RemoveContainer(ctx, c.ID, true); err != nil { - log.WrapErrWithFields(err, "failed to remove orphaned container", map[string]any{zerowrap.FieldEntityID: c.ID}) - } - } - } - - return nil -} - -func (s *Service) cleanupVolumesForDomain(_ context.Context, _ string) error { - return nil -} - -func (s *Service) cleanupNetworkIfEmpty(ctx context.Context, networkName string) error { - if networkName == "bridge" || networkName == "default" { - return nil - } - - ctx = zerowrap.CtxWithField(ctx, "network", networkName) - log := zerowrap.FromCtx(ctx) - - networks, err := s.runtime.ListNetworks(ctx) - if err != nil { - return log.WrapErr(err, "failed to list networks") - } - - for _, n := range networks { - if n.Name == networkName && len(n.Containers) == 0 && n.Labels[domain.LabelManaged] == "true" { - if err := s.runtime.RemoveNetwork(ctx, networkName); err != nil { - return err - } - log.Info().Msg("cleaned up empty network") - break - } - } - - return nil -} - -func (s *Service) deployAttachments(ctx context.Context, domainName, networkName string) error { - // Collect attachments from both domain-specific and network group configs - attachments := s.resolveAttachmentsForDomain(domainName) - if len(attachments) == 0 { - return nil - } - - log := zerowrap.FromCtx(ctx) - var deployed []string // track successfully deployed attachment IDs for rollback - for _, svc := range attachments { - if err := s.deployAttachedService(ctx, domainName, svc, networkName); err != nil { - log.WrapErrWithFields(err, "failed to deploy attachment", map[string]any{zerowrap.FieldService: svc, "domain": domainName}) - - // Rollback: clean up already-deployed attachments - for _, id := range deployed { - if stopErr := s.runtime.StopContainer(ctx, id); stopErr != nil { - log.Warn().Err(stopErr).Str("attachment_id", id).Msg("rollback: failed to stop attachment") - } - if rmErr := s.runtime.RemoveContainer(ctx, id, true); rmErr != nil { - log.Warn().Err(rmErr).Str("attachment_id", id).Msg("rollback: failed to remove attachment") - } - } - // Deregister rolled-back attachments - if len(deployed) > 0 { - s.mu.Lock() - remaining := s.attachments[domainName] - filtered := remaining[:0] - rollbackSet := make(map[string]bool, len(deployed)) - for _, id := range deployed { - rollbackSet[id] = true - } - for _, id := range remaining { - if !rollbackSet[id] { - filtered = append(filtered, id) - } - } - s.attachments[domainName] = filtered - s.mu.Unlock() - } - - return fmt.Errorf("failed to deploy attachment %q (rolled back %d already-deployed)", svc, len(deployed)) - } - // Collect IDs of attachments tracked under this domain after successful deploy - s.mu.RLock() - ids := s.attachments[domainName] - if len(ids) > 0 { - latest := ids[len(ids)-1] - deployed = append(deployed, latest) - } - s.mu.RUnlock() - } - return nil -} - -// resolveAttachmentsForDomain returns attachments for a domain by checking: -// 1. Direct domain attachments (attachments[domain]) -// 2. Network group attachments (attachments[group] where domain is in network_groups[group]) -func (s *Service) resolveAttachmentsForDomain(domainName string) []string { - var attachments map[string][]string - var networkGroups map[string][]string - - if s.configProvider != nil { - snap := s.configProvider.GetAttachmentConfig() - attachments = snap.Attachments - networkGroups = snap.NetworkGroups - } else { - s.mu.RLock() - attachments = s.config.Attachments - networkGroups = s.config.NetworkGroups - s.mu.RUnlock() - } - - seen := make(map[string]bool) - var result []string - - // First, add domain-specific attachments - if domainAttachments, ok := attachments[domainName]; ok { - for _, img := range domainAttachments { - if !seen[img] { - seen[img] = true - result = append(result, img) - } - } - } - - // Then, find which network group this domain belongs to and add group attachments - for groupName, domains := range networkGroups { - if slices.Contains(domains, domainName) { - if groupAttachments, ok := attachments[groupName]; ok { - for _, img := range groupAttachments { - if !seen[img] { - seen[img] = true - result = append(result, img) - } - } - } - break // Domain can only be in one network group - } - } - - return result -} - -func (s *Service) getAttachmentsForDomain(ctx context.Context, domainName string) []domain.Attachment { - containers, err := s.runtime.ListContainers(ctx, true) - if err != nil { - return nil - } - - return s.filterAttachments(ctx, containers, domainName) -} - -// getAllAttachments fetches all containers once and returns attachments grouped by domain. -// This avoids N+1 queries when listing multiple routes. -func (s *Service) getAllAttachments(ctx context.Context) map[string][]domain.Attachment { - containers, err := s.runtime.ListContainers(ctx, true) - if err != nil { - return nil - } - - result := make(map[string][]domain.Attachment) - for _, container := range containers { - if container.Labels == nil { - continue - } - if container.Labels[domain.LabelAttachment] != "true" { - continue - } - ownerDomain := container.Labels[domain.LabelAttachedTo] - if ownerDomain == "" { - continue - } - image := container.Image - serviceName := container.Name - if labelImage, ok := container.Labels[domain.LabelImage]; ok && labelImage != "" { - image = labelImage - serviceName = extractServiceName(labelImage) - } - // Get the container's network for display in the routes table - network := "" - if networkName, err := s.runtime.GetContainerNetwork(ctx, container.ID); err == nil { - network = networkName - } - attachment := domain.Attachment{ - Name: serviceName, - Image: s.stripRegistryPrefix(image), - ContainerID: container.ID, - Status: container.Status, - Network: network, - Ports: append([]int(nil), container.Ports...), - } - result[ownerDomain] = append(result[ownerDomain], attachment) - } - - return result -} - -// filterAttachments extracts attachments for a specific domain from a container list. -func (s *Service) filterAttachments(ctx context.Context, containers []*domain.Container, domainName string) []domain.Attachment { - attachments := make([]domain.Attachment, 0) - for _, container := range containers { - if container.Labels == nil { - continue - } - if container.Labels[domain.LabelAttachment] != "true" { - continue - } - if container.Labels[domain.LabelAttachedTo] != domainName { - continue - } - image := container.Image - serviceName := container.Name - if labelImage, ok := container.Labels[domain.LabelImage]; ok && labelImage != "" { - image = labelImage - serviceName = extractServiceName(labelImage) - } - // Get the container's network for display in the routes table - network := "" - if networkName, err := s.runtime.GetContainerNetwork(ctx, container.ID); err == nil { - network = networkName - } - attachment := domain.Attachment{ - Name: serviceName, - Image: s.stripRegistryPrefix(image), - ContainerID: container.ID, - Status: container.Status, - Network: network, - Ports: append([]int(nil), container.Ports...), - } - attachments = append(attachments, attachment) - } - - return attachments -} - -func (s *Service) deployAttachedService(ctx context.Context, ownerDomain, serviceImage, networkName string) error { - log := zerowrap.FromCtx(ctx) - - // Parse service name from image (e.g., "my-postgres:latest" → "postgres") - serviceName := extractServiceName(serviceImage) - containerName := fmt.Sprintf("gordon-%s-%s", sanitizeName(ownerDomain), serviceName) - - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - "attachment": serviceName, - "container_name": containerName, - "owner_domain": ownerDomain, - }) - - // Check if already running (idempotent) - // Try new name first, then fall back to legacy name for backwards compatibility - existingContainer, err := s.resolveExistingAttachment(ctx, ownerDomain, serviceName) - if err != nil { - return err - } - preferredVolumes := namedVolumeMounts(existingContainer) - - if existingContainer != nil && existingContainer.Name == containerName { - shouldSkip, err := s.handleRunningAttachment(ctx, existingContainer, containerName, serviceImage) - if err != nil { - return err - } - if shouldSkip { - return nil - } - } - - log.Info().Str(zerowrap.FieldService, serviceImage).Msg("deploying attached service") - - config, err := s.buildAttachmentContainerConfig(ctx, ownerDomain, serviceName, serviceImage, containerName, networkName, preferredVolumes) - if err != nil { - return err - } - - if err := s.removeAttachmentForReplacement(ctx, existingContainer, containerName); err != nil { - return err - } - - container, err := s.runtime.CreateContainer(ctx, config) - if err != nil { - return log.WrapErr(err, "failed to create attachment container") - } - - // Start container - if err := s.runtime.StartContainer(ctx, container.ID); err != nil { - s.runtime.RemoveContainer(ctx, container.ID, true) - return log.WrapErr(err, "failed to start attachment container") - } - - // Wait for attachment to be ready before proceeding - if err := s.waitForAttachmentReady(ctx, container.ID, config); err != nil { - log.WrapErr(err, "attachment readiness check failed, cleaning up") - if stopErr := s.runtime.StopContainer(ctx, container.ID); stopErr != nil { - log.Warn().Err(stopErr).Str(zerowrap.FieldEntityID, container.ID).Msg("failed to stop unready attachment") - } - s.runtime.RemoveContainer(ctx, container.ID, true) - return fmt.Errorf("attachment %q not ready: %w", serviceName, err) - } - // Track attachment - s.mu.Lock() - s.attachments[ownerDomain] = append(s.attachments[ownerDomain], container.ID) - s.mu.Unlock() - - // Start log collection for attachment - s.startLogCollection(ctx, container.ID, containerName) - - log.Info().Str(zerowrap.FieldEntityID, container.ID).Msg("attachment deployed successfully") - return nil -} - -func (s *Service) buildAttachmentContainerConfig(ctx context.Context, ownerDomain, serviceName, serviceImage, containerName, networkName string, preferredVolumes map[string]namedVolumeMount) (*domain.ContainerConfig, error) { - imageRef, err := s.buildValidatedImageRef(ctx, serviceImage) - if err != nil { - return nil, err - } - actualImageRef, err := s.ensureImage(ctx, imageRef) - if err != nil { - return nil, err - } - - exposedPorts := s.attachmentExposedPorts(ctx, actualImageRef) - volumes, readOnlyVolumes, err := s.attachmentVolumes(ctx, containerName, actualImageRef, preferredVolumes) - if err != nil { - return nil, err - } - envVars, err := s.attachmentEnv(ctx, containerName, actualImageRef) - if err != nil { - return nil, err - } - - s.mu.RLock() - cfg := s.config - s.mu.RUnlock() - - config := &domain.ContainerConfig{ - Image: actualImageRef, - Name: containerName, - Hostname: serviceName, - Aliases: []string{serviceName}, - Ports: exposedPorts, - Env: envVars, - Volumes: volumes, - ReadOnlyVolumes: readOnlyVolumes, - NetworkMode: networkName, - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: ownerDomain, - domain.LabelEnvHash: hashEnvironment(envVars), - domain.LabelImage: serviceImage, - }, - MemoryLimit: cfg.DefaultMemoryLimit, - NanoCPUs: cfg.DefaultNanoCPUs, - PidsLimit: cfg.DefaultPidsLimit, - RestartPolicy: domain.RestartPolicyAlways, - } - applySecurityProfile(config, cfg) - return config, nil -} - -func (s *Service) attachmentExposedPorts(ctx context.Context, imageRef string) []int { - log := zerowrap.FromCtx(ctx) - exposedPorts, err := s.runtime.GetImageExposedPorts(ctx, imageRef) - if err != nil { - log.WrapErr(err, "failed to get exposed ports for attachment, using defaults") - return []int{} - } - return exposedPorts -} - -func (s *Service) attachmentVolumes(ctx context.Context, containerName, imageRef string, preferredVolumes map[string]namedVolumeMount) (map[string]string, map[string]string, error) { - volumes, err := s.setupVolumes(ctx, containerName, imageRef, preferredVolumes) - if err != nil { - return nil, nil, fmt.Errorf("failed to setup volumes for attachment %s with image %s: %w", containerName, imageRef, err) - } - readOnlyVolumes := make(map[string]string) - for path, preferred := range preferredVolumes { - if preferred.ReadOnly && volumes[path] == preferred.Name { - delete(volumes, path) - readOnlyVolumes[path] = preferred.Name - } - } - return volumes, readOnlyVolumes, nil -} - -type namedVolumeMount struct { - Name string - ReadOnly bool -} - -func namedVolumeMounts(container *domain.Container) map[string]namedVolumeMount { - volumes := make(map[string]namedVolumeMount) - if container == nil { - return volumes - } - for _, mount := range container.VolumeMounts { - if mount.Type == "volume" && mount.Name != "" && mount.Destination != "" { - volumes[mount.Destination] = namedVolumeMount{Name: mount.Name, ReadOnly: mount.ReadOnly} - } - } - return volumes -} - -func (s *Service) attachmentEnv(ctx context.Context, containerName, imageRef string) ([]string, error) { - envVars, err := s.loadEnvironment(ctx, nil, containerName, imageRef) - if err != nil { - return nil, fmt.Errorf("failed to load environment for attachment %s with image %s: %w", containerName, imageRef, err) - } - return envVars, nil -} - -func (s *Service) attachmentEnvDrifted(ctx context.Context, existing *domain.Container, containerName, serviceImage string) (bool, error) { - imageRef, err := s.buildValidatedImageRef(ctx, serviceImage) - if err != nil { - return false, err - } - currentEnv, err := s.loadEnvironment(ctx, nil, containerName, imageRef) - if err != nil { - return false, err - } - - currentHash := hashEnvironment(currentEnv) - existingHash, ok := existing.Labels[domain.LabelEnvHash] - if ok && existingHash == currentHash { - return false, nil - } - - return true, nil -} - -func (s *Service) handleRunningAttachment(ctx context.Context, existing *domain.Container, containerName, serviceImage string) (bool, error) { - if existing.Status != string(domain.ContainerStatusRunning) { - return false, nil - } - - log := zerowrap.FromCtx(ctx) - - // Check image drift - existingImage := existing.Labels[domain.LabelImage] - imageDrifted := existingImage != "" && existingImage != serviceImage - - // Check env drift - envDrifted, err := s.attachmentEnvDrifted(ctx, existing, containerName, serviceImage) - if err != nil { - return false, log.WrapErr(err, "failed to check attachment env drift") - } - - if !envDrifted && !imageDrifted { - log.Debug().Str("container_name", containerName).Msg("attachment already running with current env and image, skipping") - return true, nil - } - - if imageDrifted { - log.Info().Str("container_name", containerName).Str("existing_image", existingImage).Str("desired_image", serviceImage).Msg("attachment image changed, recreating") - } - if envDrifted { - log.Info().Str("container_name", containerName).Msg("attachment env changed, recreating") - } - - return false, nil -} - -func (s *Service) removeAttachmentForReplacement(ctx context.Context, existing *domain.Container, containerName string) error { - if existing == nil { - return nil - } - - log := zerowrap.FromCtx(ctx) - if existing.Status == string(domain.ContainerStatusRunning) { - log.Info().Str("container_name", containerName).Msg("stopping attachment container for replacement") - if err := s.runtime.StopContainer(ctx, existing.ID); err != nil { - return log.WrapErr(err, "failed to stop attachment for replacement") - } - } - log.Info().Str("container_name", containerName).Msg("removing attachment container for replacement") - if err := s.runtime.RemoveContainer(ctx, existing.ID, true); err != nil { - return log.WrapErr(err, "failed to remove attachment for replacement") - } - - return nil -} - -func (s *Service) resolveExistingAttachment(ctx context.Context, ownerDomain, serviceName string) (*domain.Container, error) { - log := zerowrap.FromCtx(ctx) - containerName := fmt.Sprintf("gordon-%s-%s", sanitizeName(ownerDomain), serviceName) - existingContainer := s.findContainerByName(ctx, containerName) - if existingContainer != nil { - if err := validateAttachmentOwnership(existingContainer, ownerDomain, containerName); err != nil { - return nil, err - } - return existingContainer, nil - } - - containerNameLegacy := fmt.Sprintf("gordon-%s-%s", sanitizeNameLegacy(ownerDomain), serviceName) - existingContainer = s.findContainerByName(ctx, containerNameLegacy) - if existingContainer == nil { - return nil, nil - } - - if err := validateAttachmentOwnership(existingContainer, ownerDomain, containerNameLegacy); err != nil { - return nil, err - } - log.Info().Str("container_name", containerNameLegacy).Msg("found attachment with legacy naming, will be replaced") - return existingContainer, nil -} - -func validateAttachmentOwnership(container *domain.Container, ownerDomain, containerName string) error { - if container.Labels[domain.LabelManaged] != "true" || - container.Labels[domain.LabelAttachment] != "true" || - container.Labels[domain.LabelAttachedTo] != ownerDomain { - return fmt.Errorf("%w: attachment name %q is owned by another workload", domain.ErrAttachmentOwnershipMismatch, containerName) - } - return nil -} - -// Utility functions - -func managedContainerName(domainName string) string { - return "gordon-" + domainName -} - -func normalizeImageRef(image string) string { - if isDigestRef(image) { - return image - } - - // Extract repository from image reference, handling port numbers correctly. - // Examples: - // nginx:latest -> docker.io/library/nginx - // user/repo:tag -> docker.io/user/repo - // localhost:5000/image:tag -> localhost:5000/image - // registry.example.com/image:tag -> registry.example.com/image - // - // Note: This function assumes valid Docker image references. Edge cases like - // "registry.com:5000" (registry with port but no image name) are not valid - // image references per Docker's naming conventions, which require at least - // one path component after the registry (e.g., "registry.com:5000/image"). - - // Find the tag separator: the last colon that isn't part of a port number. - // A colon is part of a port if there's no slash after it until the next colon. - repo := image - lastColon := strings.LastIndex(image, ":") - if lastColon != -1 { - afterColon := image[lastColon+1:] - // If there's no slash after the colon, it's the tag separator - if !strings.Contains(afterColon, "/") { - repo = image[:lastColon] - } - } - - if !strings.Contains(repo, "/") { - return "docker.io/library/" + repo - } - - if strings.Count(repo, "/") == 1 && !strings.Contains(strings.Split(repo, "/")[0], ".") && !strings.Contains(strings.Split(repo, "/")[0], ":") { - return "docker.io/" + repo - } - - return repo -} - -func generateVolumeName(prefix, domainName, volumePath string) string { - return fmt.Sprintf("%s-%s-%s", - prefix, - domain.StableResourceName(domainName), - domain.StableResourceName(strings.Trim(volumePath, "/"))) -} - -func legacyVolumeName(prefix, domainName, volumePath string) string { - return fmt.Sprintf("%s-%s-%s", - prefix, - strings.ReplaceAll(domainName, ".", "-"), - strings.ReplaceAll(strings.Trim(volumePath, "/"), "/", "-")) -} - -type legacyVolumeOwnership uint8 - -const ( - legacyVolumeOwnershipUnverified legacyVolumeOwnership = iota - legacyVolumeOwnershipOwned - legacyVolumeOwnershipForeign -) - -func (s *Service) legacyVolumeOwnership(ctx context.Context, volumeName, domainName string) (legacyVolumeOwnership, error) { - containers, err := s.runtime.ListContainers(ctx, true) - if err != nil { - return legacyVolumeOwnershipUnverified, fmt.Errorf("list containers for legacy volume ownership: %w", err) - } - ownerFound := false - foreignFound := false - for _, container := range containers { - if container.Labels[domain.LabelManaged] != "true" || !containerMountsVolume(container, volumeName) { - continue - } - if container.Name == domainName || container.Labels[domain.LabelRoute] == domainName || container.Labels[domain.LabelDomain] == domainName { - ownerFound = true - } else { - foreignFound = true - } - } - switch { - case ownerFound && !foreignFound: - return legacyVolumeOwnershipOwned, nil - case foreignFound && !ownerFound: - return legacyVolumeOwnershipForeign, nil - default: - return legacyVolumeOwnershipUnverified, nil - } -} - -func containerMountsVolume(container *domain.Container, volumeName string) bool { - for _, mount := range container.VolumeMounts { - if mount.Name == volumeName || mount.Source == volumeName { - return true - } - } - return false -} - -func mergeEnvironmentVariables(dockerfileEnv, userEnv []string) []string { - envMap := make(map[string]string) - - for _, env := range dockerfileEnv { - if k, v, ok := strings.Cut(env, "="); ok { - envMap[k] = v - } - } - - for _, env := range userEnv { - if k, v, ok := strings.Cut(env, "="); ok { - envMap[k] = v - } - } - - keys := make([]string, 0, len(envMap)) - for k := range envMap { - keys = append(keys, k) - } - slices.Sort(keys) - - result := make([]string, 0, len(envMap)) - for _, k := range keys { - result = append(result, k+"="+envMap[k]) - } - - return result -} - -// extractServiceName gets service name from image reference. -// "my-postgres:latest" → "postgres", "redis:7" → "redis" -func extractServiceName(image string) string { - // Remove tag - parts := strings.Split(image, ":") - name := parts[0] - - // Remove registry prefix if present - if strings.Contains(name, "/") { - nameParts := strings.Split(name, "/") - name = nameParts[len(nameParts)-1] - } - - // Remove common prefixes like "my-" - name = strings.TrimPrefix(name, "my-") - - return name -} - -// sanitizeName is a convenience wrapper around domain.SanitizeDomainForContainer. -func sanitizeName(d string) string { - return domain.SanitizeDomainForContainer(d) -} - -// sanitizeNameLegacy is a convenience wrapper around domain.SanitizeDomainForContainerLegacy. -func sanitizeNameLegacy(d string) string { - return domain.SanitizeDomainForContainerLegacy(d) -} - -func isContainerNotFoundError(err error) bool { - if err == nil { - return false - } - if errors.Is(err, domain.ErrContainerNotFound) { - return true - } - msg := strings.ToLower(err.Error()) - return strings.Contains(msg, "no such container") || - strings.Contains(msg, "no container with name or id") || - strings.Contains(msg, "container not found") -} - -func isConnectionRefusedError(err error) bool { - if err == nil { - return false - } - return strings.Contains(strings.ToLower(err.Error()), "connection refused") -} - -// findContainerByName finds a container by its name. -func (s *Service) findContainerByName(ctx context.Context, name string) *domain.Container { - containers, err := s.runtime.ListContainers(ctx, true) - if err != nil { - return nil - } - - for _, c := range containers { - if c.Name == name { - return c - } - } - return nil -} - -func (s *Service) waitForReady(ctx context.Context, containerID string, containerConfig *domain.ContainerConfig) error { - if err := s.pollContainerRunning(ctx, containerID); err != nil { - return err - } - - s.mu.RLock() - cfg := s.config - s.mu.RUnlock() - - readinessMode := cfg.ReadinessMode - if readinessMode == "" { - readinessMode = "delay" - } - - switch readinessMode { - case "delay": - return s.waitForReadyByDelay(ctx, containerID) - case "docker-health": - _, hasHealthcheck, err := s.runtime.GetContainerHealthStatus(ctx, containerID) - if err != nil { - return err - } - if !hasHealthcheck { - return errors.New("no healthcheck detected") - } - return s.waitForHealthy(ctx, containerID, cfg.HealthTimeout) - default: // auto - return s.readinessCascade(ctx, containerID, containerConfig, cfg) - } -} - -// readinessCascade auto-detects the strongest available readiness signal: -// 1. Docker healthcheck (if present) → wait for healthy status -// 2. HTTP probe (if gordon.health label set) → GET until 2xx/3xx -// 3. Default HTTP probe (GET / on exposed port) → wait for 2xx/3xx -// 4. TCP probe (if port info available) → connect until accepted -// 5. Delay fallback (last resort) → waitForReadyByDelay -func (s *Service) readinessCascade(ctx context.Context, containerID string, containerConfig *domain.ContainerConfig, cfg Config) error { - log := zerowrap.FromCtx(ctx) - - // 1. Docker healthcheck - _, hasHealthcheck, err := s.runtime.GetContainerHealthStatus(ctx, containerID) - if err != nil { - return err - } - if hasHealthcheck { - log.Info().Msg("readiness cascade: using Docker healthcheck") - return s.waitForHealthy(ctx, containerID, cfg.HealthTimeout) - } - - // 2. HTTP probe via gordon.health label - if probed, probeErr := s.tryHTTPProbe(ctx, containerID, containerConfig, cfg); probed { - return probeErr - } - - // 3. Default HTTP probe on root path (if port info available) - if probed, probeErr := s.tryDefaultHTTPProbe(ctx, containerID, containerConfig, cfg); probed { - return probeErr - } - - // 4. TCP probe (if port info available) - if probed, probeErr := s.tryTCPProbe(ctx, containerID, containerConfig, cfg); probed { - return probeErr - } - - // 5. Delay fallback - log.Info().Msg("readiness cascade: using delay fallback") - return s.waitForReadyByDelay(ctx, containerID) -} - -// waitForAttachmentReady runs a reduced readiness cascade for attachments: -// Docker healthcheck → TCP probe → delay fallback. -// HTTP probes are skipped — attachments are typically databases/caches. -func (s *Service) waitForAttachmentReady(ctx context.Context, containerID string, containerConfig *domain.ContainerConfig) error { - if err := s.pollContainerRunning(ctx, containerID); err != nil { - return err - } - - log := zerowrap.FromCtx(ctx) - - s.mu.RLock() - cfg := s.config - s.mu.RUnlock() - - // 1. Docker healthcheck (if present) - _, hasHealthcheck, err := s.runtime.GetContainerHealthStatus(ctx, containerID) - if err != nil { - return err - } - if hasHealthcheck { - log.Info().Msg("attachment readiness: using Docker healthcheck") - timeout := cfg.HealthTimeout - if timeout == 0 { - timeout = 90 * time.Second - } - return s.waitForHealthy(ctx, containerID, timeout) - } - - // 2. TCP probe (if port info available) - if containerConfig != nil && len(containerConfig.Ports) > 0 { - ip, port, probeErr := s.resolveProbeEndpoint(ctx, containerID, containerConfig) - if probeErr == nil && ip != "" && port > 0 { - addr := fmt.Sprintf("%s:%d", ip, port) - timeout := cfg.AttachmentReadinessTimeout - if timeout == 0 { - timeout = 30 * time.Second - } - log.Info().Str("addr", addr).Dur("timeout", timeout).Msg("attachment readiness: using TCP probe") - return tcpProbe(ctx, addr, timeout) - } - log.Debug().Err(probeErr).Msg("attachment readiness: TCP probe skipped, could not resolve endpoint") - } - - // 3. Delay fallback - log.Info().Msg("attachment readiness: using delay fallback") - return s.waitForReadyByDelay(ctx, containerID) -} - -// tryHTTPProbe attempts an HTTP probe if the gordon.health label is set. -// Returns (true, err) if the probe was attempted, (false, nil) if skipped. -func (s *Service) tryHTTPProbe(ctx context.Context, containerID string, containerConfig *domain.ContainerConfig, cfg Config) (bool, error) { - log := zerowrap.FromCtx(ctx) - if containerConfig == nil { - return false, nil - } - healthPath, ok := containerConfig.Labels[domain.LabelHealth] - if !ok || healthPath == "" { - return false, nil - } - ip, port, probeErr := s.resolveProbeEndpoint(ctx, containerID, containerConfig) - if probeErr != nil || ip == "" || port <= 0 { - log.Debug().Err(probeErr).Msg("readiness cascade: HTTP probe skipped, could not resolve container endpoint") - return false, nil - } - url := fmt.Sprintf("http://%s:%d%s", ip, port, healthPath) - timeout := cfg.HTTPProbeTimeout - if timeout == 0 { - timeout = 60 * time.Second - } - log.Info().Str("url", url).Dur("timeout", timeout).Msg("readiness cascade: using HTTP probe") - return true, httpProbe(ctx, url, timeout) -} - -// tryDefaultHTTPProbe attempts an HTTP GET on "/" using the container's -// exposed port. This is the default readiness check for web containers — -// it verifies the app is actually serving HTTP, not just accepting TCP. -// Returns (true, err) if the probe was attempted, (false, nil) if skipped. -func (s *Service) tryDefaultHTTPProbe(ctx context.Context, containerID string, containerConfig *domain.ContainerConfig, cfg Config) (bool, error) { - log := zerowrap.FromCtx(ctx) - if containerConfig == nil || len(containerConfig.Ports) == 0 { - return false, nil - } - ip, port, probeErr := s.resolveProbeEndpoint(ctx, containerID, containerConfig) - if probeErr != nil || ip == "" || port <= 0 { - log.Debug().Err(probeErr).Msg("readiness cascade: default HTTP probe skipped, could not resolve container endpoint") - return false, nil - } - url := fmt.Sprintf("http://%s:%d/", ip, port) - timeout := cfg.HTTPProbeTimeout - if timeout == 0 { - timeout = 60 * time.Second - } - log.Info().Str("url", url).Dur("timeout", timeout).Msg("readiness cascade: using default HTTP alive probe") - return true, httpAliveProbe(ctx, url, timeout) -} - -// tryTCPProbe attempts a TCP probe if port info is available. -// Returns (true, err) if the probe was attempted, (false, nil) if skipped. -func (s *Service) tryTCPProbe(ctx context.Context, containerID string, containerConfig *domain.ContainerConfig, cfg Config) (bool, error) { - log := zerowrap.FromCtx(ctx) - if containerConfig == nil || len(containerConfig.Ports) == 0 { - return false, nil - } - ip, port, probeErr := s.resolveProbeEndpoint(ctx, containerID, containerConfig) - if probeErr != nil || ip == "" || port <= 0 { - log.Debug().Err(probeErr).Msg("readiness cascade: TCP probe skipped, could not resolve container endpoint") - return false, nil - } - addr := fmt.Sprintf("%s:%d", ip, port) - timeout := cfg.TCPProbeTimeout - if timeout == 0 { - timeout = 30 * time.Second - } - log.Info().Str("addr", addr).Dur("timeout", timeout).Msg("readiness cascade: using TCP probe") - return true, tcpProbe(ctx, addr, timeout) -} - -// resolveContainerEndpoint returns a host-reachable address for probing. -// In rootless podman/Docker setups, container internal IPs are not routable -// from the host. We use the host port binding (127.0.0.1:) -// which is always reachable. Falls back to the container's internal IP -// only if no host port mapping exists (e.g. host-network mode). -func (s *Service) resolveProbeEndpoint(ctx context.Context, containerID string, containerConfig *domain.ContainerConfig) (string, int, error) { - log := zerowrap.FromCtx(ctx) - - internalPort := 0 - if containerConfig != nil { - if labels := containerConfig.Labels; labels != nil { - for _, key := range []string{domain.LabelProxyPort, domain.LabelPort} { - if portStr, ok := labels[key]; ok && portStr != "" { - port, err := strconv.Atoi(portStr) - if err == nil && port > 0 && port <= 65535 { - internalPort = port - break - } - log.Warn().Str("label", key).Str("port_value", portStr).Msg("invalid probe port label value") - } - } - } - } - - if internalPort == 0 { - _, resolvedPort, err := s.runtime.GetContainerNetworkInfo(ctx, containerID) - if err != nil { - return "", 0, err - } - internalPort = resolvedPort - } - - hostPort, hostErr := s.runtime.GetContainerPort(ctx, containerID, internalPort) - if hostErr == nil && hostPort > 0 { - log.Debug(). - Int("internal_port", internalPort). - Int("host_port", hostPort). - Msg("resolved container endpoint via host port binding") - return "127.0.0.1", hostPort, nil - } - - ip, _, fallbackErr := s.runtime.GetContainerNetworkInfo(ctx, containerID) - if fallbackErr != nil { - return "", 0, fallbackErr - } - log.Debug(). - Str("ip", ip). - Int("port", internalPort). - Msg("resolved container endpoint via internal IP (no host port binding)") - return ip, internalPort, nil -} - -// waitForReadyByDelay waits using the legacy running+delay strategy. -func (s *Service) waitForReadyByDelay(ctx context.Context, containerID string) error { - log := zerowrap.FromCtx(ctx) - - s.mu.RLock() - cfg := s.config - s.mu.RUnlock() - - delay := cfg.ReadinessDelay - if delay == 0 { - delay = 5 * time.Second - } - - log.Debug().Dur("delay", delay).Msg("waiting for container readiness") - select { - case <-time.After(delay): - case <-ctx.Done(): - return ctx.Err() - } - - // Verify still running after delay - running, err := s.runtime.IsContainerRunning(ctx, containerID) - if err != nil { - return err - } - if running { - return nil - } - - return s.waitForRecovery(ctx, containerID) -} - -func (s *Service) waitForHealthy(ctx context.Context, containerID string, timeout time.Duration) error { - if timeout == 0 { - timeout = 90 * time.Second - } - waitCtx, cancel := context.WithTimeout(ctx, timeout) - defer cancel() - - var lastStatus string - for { - status, hasHealthcheck, err := s.runtime.GetContainerHealthStatus(waitCtx, containerID) - if err != nil { - return err - } - if !hasHealthcheck { - return errors.New("no healthcheck detected") - } - // Treat empty health status as transitional startup. Some runtimes can - // temporarily report an empty status before first probe results. - if status == "" { - status = "starting" - } - if status == "healthy" { - return nil - } - if status == "unhealthy" { - return errors.New("container reported unhealthy") - } - lastStatus = status - - select { - case <-time.After(time.Second): - case <-waitCtx.Done(): - return fmt.Errorf("container healthcheck timeout after %s (last status: %s)", timeout, lastStatus) - } - } -} - -// pollContainerRunning polls until the container is running (max 30 seconds). -func (s *Service) pollContainerRunning(ctx context.Context, containerID string) error { - for i := 0; i < 30; i++ { - running, err := s.runtime.IsContainerRunning(ctx, containerID) - if err != nil { - return err - } - if running { - return nil - } - // Check if the container has already exited (crash) rather than still starting up. - if info, inspectErr := s.runtime.InspectContainer(ctx, containerID); inspectErr == nil && info != nil && info.Status == string(domain.ContainerStatusExited) { - return fmt.Errorf("container exited immediately with code %d: %w", info.ExitCode, domain.ErrContainerExited) - } - if i == 29 { - return fmt.Errorf("container did not start within 30 seconds") - } - select { - case <-time.After(time.Second): - case <-ctx.Done(): - return ctx.Err() - } - } - return fmt.Errorf("container did not start within 30 seconds") -} - -// waitForRecovery waits for a container to recover within the recovery window. -func (s *Service) waitForRecovery(ctx context.Context, containerID string) error { - log := zerowrap.FromCtx(ctx) - log.Warn(). - Dur("recovery_window", readinessRecoveryWindow). - Msg("container not running after readiness delay, waiting for recovery") - - deadline := time.Now().Add(readinessRecoveryWindow) - for time.Now().Before(deadline) { - select { - case <-time.After(time.Second): - case <-ctx.Done(): - return ctx.Err() - } - - running, err := s.runtime.IsContainerRunning(ctx, containerID) - if err != nil { - return err - } - if running { - log.Info().Msg("container recovered during readiness recovery window") - return nil - } - } - - return fmt.Errorf("container not running after readiness delay and recovery window") -} - -// publishContainerDeployed publishes a container.deployed event. -func (s *Service) publishContainerDeployed(ctx context.Context, domainName, containerID string) { - payload := &domain.ContainerEventPayload{ - ContainerID: containerID, - Domain: domainName, - Action: "deployed", - } - - if err := s.eventBus.Publish(domain.EventContainerDeployed, payload); err != nil { - log := zerowrap.FromCtx(ctx) - log.Warn().Err(err).Msg("failed to publish container deployed event") - } -} - -// rewriteToRegistryDomain rewrites an image reference to use the configured registry domain. -// e.g., "myapp:latest" -> "registry.example.com/myapp:latest" -func rewriteToRegistryDomain(imageRef, registryDomain string) string { - if registryDomain == "" { - return imageRef - } - - prefix := registryDomain + "/" - if strings.HasPrefix(imageRef, prefix) { - return imageRef - } - - return prefix + imageRef -} - -// rewriteToLocalRegistry rewrites Gordon-managed image references to use the -// local registry address. Current and legacy Gordon registry domains are -// stripped before prefixing localhost:/. Bare refs are prefixed. -// External registries are preserved unchanged. -func rewriteToLocalRegistry(imageRef, registryDomain string, legacyRegistryDomains []string, registryPort int) string { - if imageRef == "" { - return imageRef - } - - localRegistry := fmt.Sprintf("localhost:%d", registryPort) - localPrefix := localRegistry + "/" - if strings.HasPrefix(imageRef, localPrefix) { - return imageRef - } - - if domain.IsGordonRegistryImageRef(imageRef, registryDomain, legacyRegistryDomains) { - return localPrefix + domain.StripKnownGordonRegistry(imageRef, registryDomain, legacyRegistryDomains) - } - if hasExplicitRegistry(imageRef) { - return imageRef - } - - return localPrefix + imageRef } diff --git a/internal/usecase/container/service_poll_test.go b/internal/usecase/container/service_poll_test.go deleted file mode 100644 index 4c02df72f..000000000 --- a/internal/usecase/container/service_poll_test.go +++ /dev/null @@ -1,63 +0,0 @@ -package container - -import ( - "errors" - "testing" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/require" - - mocks "github.com/bnema/gordon/internal/boundaries/out/mocks" - "github.com/bnema/gordon/internal/domain" -) - -func TestPollContainerRunning(t *testing.T) { - t.Run("container starts running immediately", func(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{}, nil) - ctx := testContext() - - runtime.EXPECT().IsContainerRunning(ctx, "ctr-1").Return(true, nil).Once() - - err := svc.pollContainerRunning(ctx, "ctr-1") - require.NoError(t, err) - }) - - t.Run("container exits immediately with non-zero exit code", func(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{}, nil) - ctx := testContext() - - runtime.EXPECT().IsContainerRunning(ctx, "ctr-2").Return(false, nil).Once() - runtime.EXPECT().InspectContainer(ctx, "ctr-2").Return(&domain.Container{ - ID: "ctr-2", - Status: string(domain.ContainerStatusExited), - ExitCode: 127, - }, nil).Once() - - err := svc.pollContainerRunning(ctx, "ctr-2") - require.Error(t, err) - assert.Contains(t, err.Error(), "127") - assert.Contains(t, err.Error(), "exited") - assert.True(t, errors.Is(err, domain.ErrContainerExited)) - }) - - t.Run("container starts running after one failed poll", func(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{}, nil) - ctx := testContext() - - // First poll: not yet running, still in "created" state (not exited) - runtime.EXPECT().IsContainerRunning(ctx, "ctr-3").Return(false, nil).Once() - runtime.EXPECT().InspectContainer(ctx, "ctr-3").Return(&domain.Container{ - ID: "ctr-3", - Status: string(domain.ContainerStatusCreated), - }, nil).Once() - - // Second poll: running - runtime.EXPECT().IsContainerRunning(ctx, "ctr-3").Return(true, nil).Once() - - err := svc.pollContainerRunning(ctx, "ctr-3") - require.NoError(t, err) - }) -} diff --git a/internal/usecase/container/service_readiness_test.go b/internal/usecase/container/service_readiness_test.go deleted file mode 100644 index 71c7ce559..000000000 --- a/internal/usecase/container/service_readiness_test.go +++ /dev/null @@ -1,314 +0,0 @@ -package container - -import ( - "net" - "strings" - "testing" - "time" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/mock" - "github.com/stretchr/testify/require" - - "github.com/bnema/gordon/internal/boundaries/out/mocks" - "github.com/bnema/gordon/internal/domain" -) - -// reserveEphemeralAddr binds on 127.0.0.1:0, captures the assigned port, closes -// the listener, and returns the address string. The port is guaranteed to be -// unreachable (no process listening) when the probe runs. -func reserveEphemeralAddr(t *testing.T) (ip string, port int) { - t.Helper() - ln, err := net.Listen("tcp", "127.0.0.1:0") - require.NoError(t, err) - addr := ln.Addr().(*net.TCPAddr) - ln.Close() - return addr.IP.String(), addr.Port -} - -func TestService_WaitForReady_AutoFallsBackToDelayWhenNoHealthcheck(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{ - ReadinessMode: "auto", - ReadinessDelay: time.Millisecond, - }, nil) - - ctx := testContext() - containerID := "container-1" - - // Initial running poll - runtime.EXPECT().IsContainerRunning(mock.Anything, containerID).Return(true, nil).Once() - // Auto mode probes for healthcheck support first - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, containerID).Return("", false, nil).Once() - // Delay-mode fallback verification - runtime.EXPECT().IsContainerRunning(mock.Anything, containerID).Return(true, nil).Once() - - err := svc.waitForReady(ctx, containerID, nil) - assert.NoError(t, err) -} - -func TestService_WaitForReady_AutoCascadeUsesDefaultHTTPAliveProbeWhenNoHealthcheckAndNoHealthLabel(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{ - ReadinessMode: "auto", - HTTPProbeTimeout: 50 * time.Millisecond, - }, nil) - - ctx := testContext() - containerID := "container-1" - - containerConfig := &domain.ContainerConfig{ - Ports: []int{8080}, - Labels: map[string]string{}, - } - - // Reserve an ephemeral loopback port that is guaranteed to be unreachable. - probeIP, probePort := reserveEphemeralAddr(t) - - // Initial running poll - runtime.EXPECT().IsContainerRunning(mock.Anything, containerID).Return(true, nil).Once() - // No Docker healthcheck - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, containerID).Return("", false, nil).Once() - // Cascade resolves container endpoint for default HTTP alive probe - runtime.EXPECT().GetContainerNetworkInfo(mock.Anything, containerID).Return(probeIP, probePort, nil).Once() - // Host port binding resolution — return the same loopback addr so probe hits it - runtime.EXPECT().GetContainerPort(mock.Anything, containerID, probePort).Return(probePort, nil).Once() - - // Default HTTP alive probe will try to connect — which will fail since the port is closed. - - err := svc.waitForReady(ctx, containerID, containerConfig) - // Expect HTTP alive probe timeout (no server listening) - assert.Error(t, err) - assert.Contains(t, err.Error(), "HTTP alive probe timeout") -} - -func TestService_WaitForReady_AutoCascadeUsesHTTPProbeWhenHealthLabelSet(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{ - ReadinessMode: "auto", - HTTPProbeTimeout: 50 * time.Millisecond, - }, nil) - - ctx := testContext() - containerID := "container-1" - - containerConfig := &domain.ContainerConfig{ - Ports: []int{8080}, - Labels: map[string]string{ - domain.LabelHealth: "/healthz", - }, - } - - // Reserve an ephemeral loopback port that is guaranteed to be unreachable. - probeIP, probePort := reserveEphemeralAddr(t) - - // Initial running poll - runtime.EXPECT().IsContainerRunning(mock.Anything, containerID).Return(true, nil).Once() - // No Docker healthcheck - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, containerID).Return("", false, nil).Once() - // Cascade resolves container endpoint for HTTP probe - runtime.EXPECT().GetContainerNetworkInfo(mock.Anything, containerID).Return(probeIP, probePort, nil).Once() - // Host port binding resolution — return the same loopback addr so probe hits it - runtime.EXPECT().GetContainerPort(mock.Anything, containerID, probePort).Return(probePort, nil).Once() - - err := svc.waitForReady(ctx, containerID, containerConfig) - // Expect HTTP probe timeout (no server listening) - assert.Error(t, err) - assert.Contains(t, err.Error(), "HTTP probe timeout") -} - -func TestService_WaitForReady_AutoCascadeUsesHealthcheckWhenPresent(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{ - ReadinessMode: "auto", - HealthTimeout: 50 * time.Millisecond, - }, nil) - - ctx := testContext() - containerID := "container-1" - - containerConfig := &domain.ContainerConfig{ - Ports: []int{8080}, - Labels: map[string]string{ - domain.LabelHealth: "/healthz", - }, - } - - // Initial running poll - runtime.EXPECT().IsContainerRunning(mock.Anything, containerID).Return(true, nil).Once() - // Docker healthcheck IS present — cascade detects it, then waitForHealthy polls it - // First call: cascade detection (hasHealthcheck=true) - // Second call: waitForHealthy loop (status=healthy) - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, containerID).Return("starting", true, nil).Once() - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, containerID).Return("healthy", true, nil).Once() - - err := svc.waitForReady(ctx, containerID, containerConfig) - assert.NoError(t, err) -} - -func TestService_WaitForReady_AutoCascadeFallsToDelayWhenNoEndpoint(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{ - ReadinessMode: "auto", - ReadinessDelay: time.Millisecond, - }, nil) - - ctx := testContext() - containerID := "container-1" - - // Container has ports but network info is unavailable - containerConfig := &domain.ContainerConfig{ - Ports: []int{8080}, - Labels: map[string]string{}, - } - - // Initial running poll - runtime.EXPECT().IsContainerRunning(mock.Anything, containerID).Return(true, nil).Once() - // No Docker healthcheck - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, containerID).Return("", false, nil).Once() - // Network info fails for default HTTP alive probe, then again for TCP fallback before delay - runtime.EXPECT().GetContainerNetworkInfo(mock.Anything, containerID).Return("", 0, assert.AnError).Twice() - // Delay-mode fallback verification - runtime.EXPECT().IsContainerRunning(mock.Anything, containerID).Return(true, nil).Once() - - err := svc.waitForReady(ctx, containerID, containerConfig) - assert.NoError(t, err) -} - -func TestService_WaitForReady_ExplicitDelayModeIgnoresCascade(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{ - ReadinessMode: "delay", - ReadinessDelay: time.Millisecond, - }, nil) - - ctx := testContext() - containerID := "container-1" - - // Even with health label and ports, "delay" mode skips cascade entirely - containerConfig := &domain.ContainerConfig{ - Ports: []int{8080}, - Labels: map[string]string{ - domain.LabelHealth: "/healthz", - }, - } - - // Initial running poll - runtime.EXPECT().IsContainerRunning(mock.Anything, containerID).Return(true, nil).Once() - // No GetContainerHealthStatus call — delay mode skips it - // Delay-mode verification - runtime.EXPECT().IsContainerRunning(mock.Anything, containerID).Return(true, nil).Once() - - err := svc.waitForReady(ctx, containerID, containerConfig) - assert.NoError(t, err) -} - -func TestService_WaitForHealthy_EmptyStatusIsReportedAsStarting(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{}, nil) - - ctx := testContext() - containerID := "container-1" - - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, containerID).Return("", true, nil).Once() - - err := svc.waitForHealthy(ctx, containerID, 10*time.Millisecond) - assert.Error(t, err) - assert.True(t, strings.Contains(err.Error(), "last status: starting")) -} - -func TestService_WaitForReady_ProbeUsesGordonProxyPortLabel(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{ - ReadinessMode: "auto", - HTTPProbeTimeout: 50 * time.Millisecond, - }, nil) - - ctx := testContext() - containerID := "container-port-label" - labelPort := 3000 - - probeIP, probePort := reserveEphemeralAddr(t) - _ = probeIP - - // Container config has gordon.proxy.port label — should be used for probe endpoint - containerConfig := &domain.ContainerConfig{ - Ports: []int{22, 3000}, - Labels: map[string]string{ - domain.LabelProxyPort: "3000", - }, - } - - runtime.EXPECT().IsContainerRunning(mock.Anything, containerID).Return(true, nil).Once() - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, containerID).Return("", false, nil).Once() - // Should use label port (3000), NOT call GetContainerNetworkInfo - runtime.EXPECT().GetContainerPort(mock.Anything, containerID, labelPort).Return(probePort, nil).Once() - - err := svc.waitForReady(ctx, containerID, containerConfig) - // HTTP probe will timeout (no server), but the important thing is it used port 3000 - assert.Error(t, err) - assert.Contains(t, err.Error(), "HTTP alive probe timeout") -} - -func TestService_WaitForReady_ProbeUsesDeprecatedPortLabel(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{ - ReadinessMode: "auto", - HTTPProbeTimeout: 50 * time.Millisecond, - }, nil) - - ctx := testContext() - containerID := "container-proxy-port" - labelPort := 3000 - - probeIP, probePort := reserveEphemeralAddr(t) - _ = probeIP // probeIP not used directly but port is - - // Container config has deprecated gordon.port — should still work - containerConfig := &domain.ContainerConfig{ - Ports: []int{22, 3000}, - Labels: map[string]string{ - domain.LabelPort: "3000", - }, - } - - runtime.EXPECT().IsContainerRunning(mock.Anything, containerID).Return(true, nil).Once() - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, containerID).Return("", false, nil).Once() - runtime.EXPECT().GetContainerPort(mock.Anything, containerID, labelPort).Return(probePort, nil).Once() - - err := svc.waitForReady(ctx, containerID, containerConfig) - assert.Error(t, err) - assert.Contains(t, err.Error(), "HTTP alive probe timeout") -} - -func TestService_WaitForReady_ProxyPortTakesPriorityOverPort(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{ - ReadinessMode: "auto", - HTTPProbeTimeout: 50 * time.Millisecond, - }, nil) - - ctx := testContext() - containerID := "container-priority" - - probeIP, probePort := reserveEphemeralAddr(t) - _ = probeIP - - // Both labels set — gordon.proxy.port (8080) should win over gordon.port (3000) - containerConfig := &domain.ContainerConfig{ - Ports: []int{22, 3000, 8080}, - Labels: map[string]string{ - domain.LabelProxyPort: "8080", - domain.LabelPort: "3000", - }, - } - - runtime.EXPECT().IsContainerRunning(mock.Anything, containerID).Return(true, nil).Once() - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, containerID).Return("", false, nil).Once() - // Should use gordon.proxy.port=8080, NOT gordon.port=3000 - runtime.EXPECT().GetContainerPort(mock.Anything, containerID, 8080).Return(probePort, nil).Once() - - err := svc.waitForReady(ctx, containerID, containerConfig) - assert.Error(t, err) - assert.Contains(t, err.Error(), "HTTP alive probe timeout") -} diff --git a/internal/usecase/container/service_test.go b/internal/usecase/container/service_test.go index e7a1fb48c..16259c0b9 100644 --- a/internal/usecase/container/service_test.go +++ b/internal/usecase/container/service_test.go @@ -1,29 +1,14 @@ package container import ( - "bytes" "context" - "encoding/binary" - "errors" - "fmt" - "io" - "net" - "strings" - "sync" "testing" - "time" "github.com/bnema/zerowrap" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/mock" - "github.com/stretchr/testify/require" - "go.opentelemetry.io/otel" - sdkmetric "go.opentelemetry.io/otel/sdk/metric" - "go.opentelemetry.io/otel/sdk/metric/metricdata" - "github.com/bnema/gordon/internal/adapters/out/telemetry" - "github.com/bnema/gordon/internal/boundaries/out" - mocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/boundaries/out/mocks" "github.com/bnema/gordon/internal/domain" ) @@ -31,5214 +16,23 @@ func testContext() context.Context { return zerowrap.WithCtx(context.Background(), zerowrap.Default()) } -func dockerLogFrames(stream byte, lines ...string) io.ReadCloser { - var buf bytes.Buffer - for _, line := range lines { - payload := []byte(line) - header := make([]byte, 8) - header[0] = stream - binary.BigEndian.PutUint32(header[4:], uint32(len(payload))) - _, _ = buf.Write(header) - _, _ = buf.Write(payload) - } - - return io.NopCloser(bytes.NewReader(buf.Bytes())) -} - -// testMinDelayConfig returns a Config with all timing delays set to 1ms, -// suitable for unit tests that don't want to wait for real timeouts. -func testMinDelayConfig() Config { - return Config{ - AllowedRegistries: []string{"docker.io"}, - ReadinessDelay: time.Millisecond, - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - StabilizationDelay: time.Millisecond, - } -} - -func setupMetricsTest(t *testing.T) (*telemetry.Metrics, *sdkmetric.ManualReader) { - t.Helper() - - prev := otel.GetMeterProvider() - t.Cleanup(func() { - otel.SetMeterProvider(prev) - }) - - reader := sdkmetric.NewManualReader() - mp := sdkmetric.NewMeterProvider(sdkmetric.WithReader(reader)) - otel.SetMeterProvider(mp) - t.Cleanup(func() { - _ = mp.Shutdown(context.Background()) - }) - - m, err := telemetry.NewMetrics() - require.NoError(t, err) - return m, reader -} - -func managedMetricState(t *testing.T, reader *sdkmetric.ManualReader) (int64, int, []metricdata.DataPoint[int64]) { - t.Helper() - - var rm metricdata.ResourceMetrics - require.NoError(t, reader.Collect(context.Background(), &rm)) - - for _, sm := range rm.ScopeMetrics { - for _, m := range sm.Metrics { - if m.Name != "gordon.container.managed" { - continue - } - sum, ok := m.Data.(metricdata.Sum[int64]) - require.True(t, ok) - var total int64 - for _, dp := range sum.DataPoints { - total += dp.Value - } - return total, len(sum.DataPoints), sum.DataPoints - } - } - - return 0, 0, nil -} - -func TestService_NewService_WithConfigProvider(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - provider := mocks.NewMockAttachmentConfigProvider(t) - provider.EXPECT().GetAttachmentConfig().Return(out.AttachmentConfigSnapshot{ - Attachments: map[string][]string{"test.com": {"postgres:16"}}, - NetworkGroups: map[string][]string{"group1": {"test.com"}}, - }).Maybe() - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, provider) - assert.NotNil(t, svc) -} - -func TestService_ResolveAttachments_UsesLiveConfigProvider(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - provider := mocks.NewMockAttachmentConfigProvider(t) - provider.EXPECT().GetAttachmentConfig().Return(out.AttachmentConfigSnapshot{ - Attachments: map[string][]string{"app.example.com": {"postgres:16"}}, - NetworkGroups: map[string][]string{}, - }).Once() - provider.EXPECT().GetAttachmentConfig().Return(out.AttachmentConfigSnapshot{ - Attachments: map[string][]string{"app.example.com": {"postgres:16", "redis:7"}}, - NetworkGroups: map[string][]string{}, - }).Once() - - // Create service with EMPTY static config but a live provider - svc := NewService(runtime, envLoader, eventBus, nil, Config{ - NetworkIsolation: true, - NetworkPrefix: "gordon", - }, provider) - - // resolveAttachmentsForDomain should return the provider's data, not the empty snapshot - result := svc.resolveAttachmentsForDomain("app.example.com") - assert.Equal(t, []string{"postgres:16"}, result) - - result = svc.resolveAttachmentsForDomain("app.example.com") - assert.Equal(t, []string{"postgres:16", "redis:7"}, result) -} - -func TestService_ResolveAttachments_GroupBasedViaProvider(t *testing.T) { - provider := mocks.NewMockAttachmentConfigProvider(t) - provider.EXPECT().GetAttachmentConfig().Return(out.AttachmentConfigSnapshot{ - Attachments: map[string][]string{ - "shared-group": {"postgres:16"}, - }, - NetworkGroups: map[string][]string{ - "shared-group": {"app1.example.com", "app2.example.com"}, - }, - }).Twice() - - svc := NewService( - mocks.NewMockContainerRuntime(t), - mocks.NewMockEnvLoader(t), - mocks.NewMockEventPublisher(t), - nil, Config{NetworkIsolation: true, NetworkPrefix: "gordon"}, provider, - ) - - // Domain in group should see group attachments - result := svc.resolveAttachmentsForDomain("app1.example.com") - assert.Equal(t, []string{"postgres:16"}, result) - - // Domain not in any group should see nothing - result = svc.resolveAttachmentsForDomain("other.example.com") - assert.Empty(t, result) -} - -func TestService_GetNetworkForApp_UsesLiveConfigProvider(t *testing.T) { - provider := mocks.NewMockAttachmentConfigProvider(t) - provider.EXPECT().GetAttachmentConfig().Return(out.AttachmentConfigSnapshot{ - Attachments: map[string][]string{}, - NetworkGroups: map[string][]string{ - "shared-group": {"app1.example.com", "app2.example.com"}, - }, - }).Once() - provider.EXPECT().GetAttachmentConfig().Return(out.AttachmentConfigSnapshot{ - Attachments: map[string][]string{}, - NetworkGroups: map[string][]string{ - "shared-group": {"app1.example.com", "app2.example.com", "app3.example.com"}, - }, - }).Once() - provider.EXPECT().GetAttachmentConfig().Return(out.AttachmentConfigSnapshot{ - Attachments: map[string][]string{}, - NetworkGroups: map[string][]string{ - "shared-group": {"app1.example.com", "app2.example.com", "app3.example.com"}, - }, - }).Once() - - svc := NewService( - mocks.NewMockContainerRuntime(t), - mocks.NewMockEnvLoader(t), - mocks.NewMockEventPublisher(t), - nil, Config{NetworkIsolation: true, NetworkPrefix: "gordon"}, provider, - ) - - // Group and per-domain networks use collision-resistant names. - network := svc.getNetworkForApp("app1.example.com") - assert.Equal(t, svc.generateNetworkName("shared-group"), network) - - network = svc.getNetworkForApp("app3.example.com") - assert.Equal(t, svc.generateNetworkName("shared-group"), network) - - network = svc.getNetworkForApp("solo.example.com") - assert.Equal(t, svc.generateNetworkName("solo.example.com"), network) -} - -func TestService_ManagedContainersMetric_GlobalSeriesOnDeployReplaceAndRemove(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, nil) - ctx := testContext() - metrics, reader := setupMetricsTest(t) - svc.SetMetrics(metrics) - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil).Twice() - - // First deploy increments metric. - svc.activateDeployedContainer(ctx, "test.example.com", &domain.Container{ID: "container-1"}) - value, seriesCount, points := managedMetricState(t, reader) - assert.Equal(t, int64(1), value) - assert.Equal(t, 1, seriesCount) - assert.Equal(t, 0, points[0].Attributes.Len()) - - // Replacing an existing domain should not increment. - svc.activateDeployedContainer(ctx, "test.example.com", &domain.Container{ID: "container-2"}) - value, seriesCount, points = managedMetricState(t, reader) - assert.Equal(t, int64(1), value) - assert.Equal(t, 1, seriesCount) - assert.Equal(t, 0, points[0].Attributes.Len()) - - runtime.EXPECT().RemoveContainer(mock.Anything, "container-2", true).Return(nil) - require.NoError(t, svc.Remove(ctx, "container-2", true)) - - value, seriesCount, points = managedMetricState(t, reader) - assert.Equal(t, int64(0), value) - assert.Equal(t, 1, seriesCount) - assert.Equal(t, 0, points[0].Attributes.Len()) -} - -func TestService_SyncContainers_ManagedContainersMetricTracksDelta(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, nil) - ctx := testContext() - metrics, reader := setupMetricsTest(t) - svc.SetMetrics(metrics) - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - { - ID: "container-1", - Labels: map[string]string{ - "gordon.domain": "app1.example.com", - "gordon.managed": "true", - }, - }, - { - ID: "container-2", - Labels: map[string]string{ - "gordon.domain": "app2.example.com", - "gordon.managed": "true", - }, - }, - }, nil).Once() - require.NoError(t, svc.SyncContainers(ctx)) - - value, seriesCount, points := managedMetricState(t, reader) - assert.Equal(t, int64(2), value) - assert.Equal(t, 1, seriesCount) - assert.Equal(t, 0, points[0].Attributes.Len()) - - // No runtime change: metric should remain stable. - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - { - ID: "container-1", - Labels: map[string]string{ - "gordon.domain": "app1.example.com", - "gordon.managed": "true", - }, - }, - { - ID: "container-2", - Labels: map[string]string{ - "gordon.domain": "app2.example.com", - "gordon.managed": "true", - }, - }, - }, nil).Once() - require.NoError(t, svc.SyncContainers(ctx)) - - value, seriesCount, points = managedMetricState(t, reader) - assert.Equal(t, int64(2), value) - assert.Equal(t, 1, seriesCount) - assert.Equal(t, 0, points[0].Attributes.Len()) - - // One container removed in runtime: metric should decrement by one. - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - { - ID: "container-1", - Labels: map[string]string{ - "gordon.domain": "app1.example.com", - "gordon.managed": "true", - }, - }, - }, nil).Once() - require.NoError(t, svc.SyncContainers(ctx)) - - value, seriesCount, points = managedMetricState(t, reader) - assert.Equal(t, int64(1), value) - assert.Equal(t, 1, seriesCount) - assert.Equal(t, 0, points[0].Attributes.Len()) -} - -func TestService_Deploy_Success(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - NetworkIsolation: false, - VolumeAutoCreate: false, - ReadinessDelay: time.Millisecond, // Minimal delay for tests - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - } - - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:latest", - } - - // No container in memory — runtime resolution returns nothing - runtime.EXPECT().ListContainers(mock.Anything, false).Return([]*domain.Container{}, nil) - - // Setup mocks - no orphaned containers - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - - // Image is not found locally, needs to be pulled - runtime.EXPECT().ListImages(mock.Anything).Return([]string{}, nil) - runtime.EXPECT().PullImage(mock.Anything, "myapp:latest").Return(nil) - - // Get exposed ports - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:latest").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:latest").Return(nil, nil) - - // Load environment - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{"FOO=bar"}, nil) - - // Inspect image env - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:latest").Return([]string{"DEFAULT=value"}, nil) - - // No volume auto-create, so no volume calls - - // Create and start container - createdContainer := &domain.Container{ - ID: "container-123", - Name: "gordon-test.example.com", - Image: "myapp:latest", - Status: "created", - } - runtime.EXPECT().CreateContainer(mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")).Return(createdContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "container-123").Return(nil) - - // Wait for ready: IsContainerRunning (first check returns true) + verify after delay - runtime.EXPECT().IsContainerRunning(mock.Anything, "container-123").Return(true, nil).Times(2) - - // Re-inspect after start - runningContainer := &domain.Container{ - ID: "container-123", - Name: "gordon-test.example.com", - Image: "myapp:latest", - Status: "running", - Ports: []int{8080}, - } - runtime.EXPECT().InspectContainer(mock.Anything, "container-123").Return(runningContainer, nil) - - // Publish container deployed event - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - assert.NotNil(t, result) - assert.Equal(t, "container-123", result.ID) - assert.Equal(t, "running", result.Status) - - // Verify container is tracked - tracked, exists := svc.Get(ctx, "test.example.com") - assert.True(t, exists) - assert.Equal(t, "container-123", tracked.ID) -} - -func TestService_Deploy_ReadinessRecoveryWindow_AllowsTransientFlap(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - NetworkIsolation: false, - VolumeAutoCreate: false, - ReadinessDelay: time.Millisecond, - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - } - - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:latest", - } - - // No container in memory — runtime resolution returns nothing - runtime.EXPECT().ListContainers(mock.Anything, false).Return([]*domain.Container{}, nil) - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListImages(mock.Anything).Return([]string{}, nil) - runtime.EXPECT().PullImage(mock.Anything, "myapp:latest").Return(nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:latest").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:latest").Return(nil, nil) - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:latest").Return([]string{}, nil) - - createdContainer := &domain.Container{ - ID: "container-123", - Name: "gordon-test.example.com", - Image: "myapp:latest", - Status: "created", - } - runtime.EXPECT().CreateContainer(mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")).Return(createdContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "container-123").Return(nil) - - // Readiness check sequence: - // 1) Initial "started" check => running - // 2) Post-delay verification => transient not-running - // 3) Recovery-window poll => running again - runtime.EXPECT().IsContainerRunning(mock.Anything, "container-123").Return(true, nil).Once() - runtime.EXPECT().IsContainerRunning(mock.Anything, "container-123").Return(false, nil).Once() - runtime.EXPECT().IsContainerRunning(mock.Anything, "container-123").Return(true, nil).Once() - - runtime.EXPECT().InspectContainer(mock.Anything, "container-123").Return(&domain.Container{ - ID: "container-123", - Name: "gordon-test.example.com", - Image: "myapp:latest", - Status: "running", - Ports: []int{8080}, - }, nil) - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - assert.NotNil(t, result) - assert.Equal(t, "container-123", result.ID) -} - -func TestService_Deploy_ImagePullFailure(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{AllowedRegistries: []string{"docker.io"}} - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:latest", - } - - // No container in memory — runtime resolution returns nothing - runtime.EXPECT().ListContainers(mock.Anything, false).Return([]*domain.Container{}, nil) - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListImages(mock.Anything).Return([]string{}, nil) - runtime.EXPECT().PullImage(mock.Anything, "myapp:latest").Return(errors.New("image not found")) - - result, err := svc.Deploy(ctx, route) - - assert.Error(t, err) - assert.Nil(t, result) - var deployErr *domain.DeployFailureError - require.ErrorAs(t, err, &deployErr) - assert.Equal(t, "failed to deploy", deployErr.Error()) - assert.Equal(t, "failed to pull image", deployErr.Cause) - assert.Equal(t, "verify registry auth and confirm the image exists at the requested tag", deployErr.Hint) - assert.Equal(t, "gordon-test.example.com", deployErr.ContainerName) - assert.Empty(t, deployErr.ContainerID) - assert.ErrorIs(t, deployErr.Err, domain.ErrImagePullFailed) - assert.ErrorContains(t, deployErr.Err, "failed to pull image") -} - -func TestIsPullFailure_ErrImagePullFailed(t *testing.T) { - err := fmt.Errorf("wrapped: %w", domain.ErrImagePullFailed) - - assert.True(t, isPullFailure(err)) -} - -func TestService_Deploy_DoesNotMisclassifyUnauthorizedEnvLoadFailure(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, Config{AllowedRegistries: []string{"docker.io"}}, nil) - ctx := testContext() - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:latest", - } - - runtime.EXPECT().ListContainers(mock.Anything, false).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListImages(mock.Anything).Return([]string{"myapp:latest"}, nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:latest").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:latest").Return(nil, nil) - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return(nil, errors.New("unauthorized")) - - result, err := svc.Deploy(ctx, route) - - assert.Nil(t, result) - assert.ErrorContains(t, err, "failed to load environment variables: unauthorized") - var deployErr *domain.DeployFailureError - assert.False(t, errors.As(err, &deployErr)) -} - -func TestService_CaptureRecentContainerLogs_SkipsCanceledContext(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{}, nil) - - ctx, cancel := context.WithCancel(testContext()) - cancel() - - logs := svc.captureRecentContainerLogs(ctx, "container-123") - - assert.Nil(t, logs) -} - -func TestService_Deploy_ReadinessFailureReturnsDeployFailureWithLogs(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - ReadinessMode: "docker-health", - HealthTimeout: 10 * time.Millisecond, - ReadinessDelay: time.Millisecond, - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - } - - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - - route := domain.Route{Domain: "test.example.com", Image: "myapp:latest"} - - runtime.EXPECT().ListContainers(mock.Anything, false).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListImages(mock.Anything).Return([]string{}, nil) - runtime.EXPECT().PullImage(mock.Anything, "myapp:latest").Return(nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:latest").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:latest").Return(nil, nil) - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:latest").Return([]string{}, nil) - - newContainer := &domain.Container{ID: "new-container", Name: "gordon-test.example.com", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).Return(newContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "new-container").Return(nil) - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(true, nil).Once() - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "new-container").Return("starting", true, nil).Twice() - - logLines := make([]string, 0, 25) - for i := 1; i <= 25; i++ { - logLines = append(logLines, fmt.Sprintf("line-%02d\n", i)) - } - runtime.EXPECT().GetContainerLogs(mock.Anything, "new-container", false).Return(dockerLogFrames(1, logLines...), nil) - runtime.EXPECT().StopContainer(mock.Anything, "new-container").Return(nil) - runtime.EXPECT().RemoveContainer(mock.Anything, "new-container", true).Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.Nil(t, result) - var deployErr *domain.DeployFailureError - require.ErrorAs(t, err, &deployErr) - assert.Equal(t, "failed to deploy", deployErr.Summary) - assert.Equal(t, "container failed readiness check", deployErr.Cause) - assert.Equal(t, "check startup logs and confirm the health endpoint becomes ready within the configured timeout", deployErr.Hint) - assert.Equal(t, "gordon-test.example.com", deployErr.ContainerName) - assert.Equal(t, "new-container", deployErr.ContainerID) - require.Len(t, deployErr.Logs, 20) - assert.Equal(t, "line-06", deployErr.Logs[0]) - assert.Equal(t, "line-25", deployErr.Logs[19]) - assert.True(t, strings.Contains(err.Error(), "failed to deploy")) - assert.ErrorContains(t, deployErr.Err, "last status: starting") -} - -func TestService_Deploy_RequestCancellationStillCleansUpFailedCandidate(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - svc := NewService(runtime, envLoader, eventBus, nil, testMinDelayConfig(), nil) - - svc.containers["test.example.com"] = &domain.Container{ - ID: "old-container", - Name: "gordon-test.example.com", - Status: "running", - } - route := domain.Route{Domain: "test.example.com", Image: "myapp:v2"} - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListImages(mock.Anything).Return([]string{"myapp:v2"}, nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:v2").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:v2").Return(nil, nil) - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:v2").Return([]string{}, nil) - - candidate := &domain.Container{ID: "candidate", Name: "gordon-test.example.com-new", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).Return(candidate, nil) - - ctx, cancel := context.WithCancel(testContext()) - runtime.EXPECT().StartContainer(mock.Anything, candidate.ID).Run(func(context.Context, string) { - cancel() - }).Return(nil) - runtime.EXPECT().IsContainerRunning(mock.Anything, candidate.ID).Return(true, nil).Once() - - activeCleanupContext := mock.MatchedBy(func(ctx context.Context) bool { - _, hasDeadline := ctx.Deadline() - return ctx.Err() == nil && hasDeadline - }) - runtime.EXPECT().StopContainer(activeCleanupContext, candidate.ID).Return(nil).Once() - runtime.EXPECT().RemoveContainer(activeCleanupContext, candidate.ID, true).Return(nil).Once() - - result, err := svc.Deploy(ctx, route) - - assert.Nil(t, result) - assert.ErrorIs(t, err, context.Canceled) - tracked, exists := svc.Get(testContext(), route.Domain) - require.True(t, exists) - assert.Equal(t, "old-container", tracked.ID) -} - -func TestService_Deploy_StrictHealthModeWithoutHealthcheckReturnsHint(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - ReadinessMode: "docker-health", - ReadinessDelay: time.Millisecond, - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - } - - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - - route := domain.Route{Domain: "test.example.com", Image: "myapp:latest"} - - runtime.EXPECT().ListContainers(mock.Anything, false).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListImages(mock.Anything).Return([]string{}, nil) - runtime.EXPECT().PullImage(mock.Anything, "myapp:latest").Return(nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:latest").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:latest").Return(nil, nil) - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:latest").Return([]string{}, nil) - - newContainer := &domain.Container{ID: "new-container", Name: "gordon-test.example.com", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).Return(newContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "new-container").Return(nil) - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(true, nil).Once() - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "new-container").Return("", false, nil).Once() - runtime.EXPECT().GetContainerLogs(mock.Anything, "new-container", false).Return(dockerLogFrames(1, "booting\n"), nil) - runtime.EXPECT().StopContainer(mock.Anything, "new-container").Return(nil) - runtime.EXPECT().RemoveContainer(mock.Anything, "new-container", true).Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.Nil(t, result) - var deployErr *domain.DeployFailureError - require.ErrorAs(t, err, &deployErr) - assert.Equal(t, "container failed readiness check", deployErr.Cause) - assert.Equal(t, "add a Docker healthcheck or switch readiness mode", deployErr.Hint) - assert.Equal(t, []string{"booting"}, deployErr.Logs) - assert.Equal(t, "gordon-test.example.com", deployErr.ContainerName) - assert.Equal(t, "new-container", deployErr.ContainerID) - assert.ErrorContains(t, deployErr.Err, "no healthcheck detected") -} - -func TestService_PullImage_DoesNotSendServiceTokenToExternalRegistry(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - RegistryAuthEnabled: true, - ServiceTokenUsername: "gordon-service", - ServiceToken: "service-token", - } - - svc := NewService(runtime, nil, nil, nil, config, nil) - ctx := testContext() - - runtime.EXPECT().PullImage(mock.Anything, "registry.example.com/myapp:latest").Return(nil) - - err := svc.pullImage(ctx, "registry.example.com/myapp:latest", false) - - assert.NoError(t, err) -} - -func TestService_PullImage_InternalRetriesOnConnectionRefused(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - RegistryAuthEnabled: true, - InternalRegistryUsername: "internal", - InternalRegistryPassword: "secret", - } - - svc := NewService(runtime, nil, nil, nil, config, nil) - ctx := testContext() - - runtime.EXPECT().PullImageWithAuth(mock.Anything, "localhost:5000/myapp:latest", "internal", "secret"). - Return(errors.New("connection refused")). - Once() - runtime.EXPECT().PullImageWithAuth(mock.Anything, "localhost:5000/myapp:latest", "internal", "secret"). - Return(nil). - Once() - - err := svc.pullImage(ctx, "localhost:5000/myapp:latest", true) - - assert.NoError(t, err) -} - -func TestService_Deploy_ReplacesExistingContainer(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - cacheInvalidator := mocks.NewMockProxyCacheInvalidator(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - ReadinessDelay: time.Millisecond, // Minimal delay for tests - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - StabilizationDelay: time.Millisecond, - } - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - svc.SetProxyCacheInvalidator(cacheInvalidator) - ctx := testContext() - - // Pre-populate with existing container - existingContainer := &domain.Container{ - ID: "old-container", - Name: "gordon-test.example.com", - Status: "running", - } - svc.containers["test.example.com"] = existingContainer - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:v2", - } - - // Cleanup orphans - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - - // Image operations - runtime.EXPECT().ListImages(mock.Anything).Return([]string{}, nil) - runtime.EXPECT().PullImage(mock.Anything, "myapp:v2").Return(nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:v2").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:v2").Return(nil, nil) - - // Environment - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:v2").Return([]string{}, nil) - - // Create new container (with -new suffix for zero-downtime) - newContainer := &domain.Container{ID: "new-container", Name: "gordon-test.example.com-new", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { - return cfg.Name == "gordon-test.example.com-new" - })).Return(newContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "new-container").Return(nil) - - // Wait for ready (2 calls) + stabilization check (1 call) - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(true, nil).Times(3) - - // Inspect after ready - runtime.EXPECT().InspectContainer(mock.Anything, "new-container").Return(&domain.Container{ - ID: "new-container", - Status: "running", - }, nil) - - // Publish event - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - // Synchronous cache invalidation before old container cleanup - cacheInvalidator.EXPECT().InvalidateTarget(mock.Anything, "test.example.com").Return() - - // Now cleanup old container (after new one is ready + cache invalidated + stabilization + drain delay) - runtime.EXPECT().StopContainer(mock.Anything, "old-container").Return(nil) - runtime.EXPECT().RemoveContainer(mock.Anything, "old-container", true).Return(nil) - - // Rename new container to canonical name - runtime.EXPECT().RenameContainer(mock.Anything, "new-container", "gordon-test.example.com").Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - svc.WaitForCleanup() - assert.Equal(t, "new-container", result.ID) -} - -// TestService_Deploy_ResolvesExistingFromRuntime_WhenMemoryStale verifies that -// when in-memory state has no tracked container for a domain, Deploy queries -// the runtime and discovers the running container. This prevents orphan cleanup -// from killing the real active container after a Gordon restart. -func TestService_Deploy_ResolvesExistingFromRuntime_WhenMemoryStale(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - cacheInvalidator := mocks.NewMockProxyCacheInvalidator(t) - - config := testMinDelayConfig() - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - svc.SetProxyCacheInvalidator(cacheInvalidator) - ctx := testContext() - - // NO tracked container in memory — simulates Gordon restart - // (svc.containers is empty for "test.example.com") - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:v2", - } - - // resolveExistingContainer should query running containers from runtime - runtime.EXPECT().ListContainers(mock.Anything, false).Return([]*domain.Container{ - { - ID: "runtime-active-container", - Name: "gordon-test.example.com", - Status: "running", - Labels: map[string]string{ - domain.LabelDomain: "test.example.com", - domain.LabelManaged: "true", - }, - }, - }, nil) - - // Orphan cleanup lists ALL containers (including stopped) — the runtime-active - // container should be skipped because it is now recognized as the existing one - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - { - ID: "runtime-active-container", - Name: "gordon-test.example.com", - Status: "running", - Labels: map[string]string{ - domain.LabelDomain: "test.example.com", - domain.LabelManaged: "true", - }, - }, - }, nil) - - // Image operations - runtime.EXPECT().ListImages(mock.Anything).Return([]string{}, nil) - runtime.EXPECT().PullImage(mock.Anything, "myapp:v2").Return(nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:v2").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:v2").Return(nil, nil) - - // Environment - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:v2").Return([]string{}, nil) - - // Create new container with -new suffix (zero-downtime: existing was resolved from runtime) - newContainer := &domain.Container{ID: "new-container", Name: "gordon-test.example.com-new", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { - return cfg.Name == "gordon-test.example.com-new" - })).Return(newContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "new-container").Return(nil) - - // Wait for ready (2 calls) + stabilization check (1 call) - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(true, nil).Times(3) - - // Inspect after ready - runtime.EXPECT().InspectContainer(mock.Anything, "new-container").Return(&domain.Container{ - ID: "new-container", - Status: "running", - }, nil) - - // Publish event - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - // Synchronous cache invalidation - cacheInvalidator.EXPECT().InvalidateTarget(mock.Anything, "test.example.com").Return() - - // AFTER new container is ready: stop and remove old runtime-discovered container - // This is the key assertion — StopContainer on the runtime-active container - // happens during finalizePreviousContainer, NOT during orphan cleanup - runtime.EXPECT().StopContainer(mock.Anything, "runtime-active-container").Return(nil) - runtime.EXPECT().RemoveContainer(mock.Anything, "runtime-active-container", true).Return(nil) - - // Rename new container to canonical name - runtime.EXPECT().RenameContainer(mock.Anything, "new-container", "gordon-test.example.com").Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - svc.WaitForCleanup() - assert.Equal(t, "new-container", result.ID) - - // Verify new container is now tracked - tracked, exists := svc.Get(ctx, "test.example.com") - assert.True(t, exists) - assert.Equal(t, "new-container", tracked.ID) -} - -func TestService_Deploy_SkipRedundantDeploy_GetImageIDError(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := testMinDelayConfig() - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - expectedEnvHash := hashEnvironment([]string{}) - - // Existing container has ImageID, but GetImageID will fail - existingContainer := &domain.Container{ - ID: "existing-container", - Name: "gordon-test.example.com", - Image: "myapp:latest", - ImageID: "sha256:abc123", - Status: "running", - Labels: map[string]string{ - domain.LabelEnvHash: expectedEnvHash, - }, - } - svc.containers["test.example.com"] = existingContainer - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:latest", - } - - // prepareDeployResources - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListImages(mock.Anything).Return([]string{"myapp:latest"}, nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:latest").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:latest").Return(nil, nil) - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:latest").Return([]string{}, nil) - - // GetImageID fails => deploy proceeds normally (graceful degradation) - runtime.EXPECT().GetImageID(mock.Anything, "myapp:latest").Return("", errors.New("image inspect failed")) - - // Full deploy proceeds despite the error - newContainer := &domain.Container{ID: "new-container", Name: "gordon-test.example.com-new", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")).Return(newContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "new-container").Return(nil) - // Wait for ready (2 calls) + stabilization check (1 call) - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(true, nil).Times(3) - runtime.EXPECT().InspectContainer(mock.Anything, "new-container").Return(&domain.Container{ - ID: "new-container", Status: "running", - }, nil) - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - // Old container is finalized - runtime.EXPECT().StopContainer(mock.Anything, "existing-container").Return(nil) - runtime.EXPECT().RemoveContainer(mock.Anything, "existing-container", true).Return(nil) - runtime.EXPECT().RenameContainer(mock.Anything, "new-container", "gordon-test.example.com").Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - svc.WaitForCleanup() - assert.Equal(t, "new-container", result.ID) -} - -func TestService_Deploy_WithNetworkIsolation(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - NetworkIsolation: true, - NetworkPrefix: "gordon", - ReadinessDelay: time.Millisecond, // Minimal delay for tests - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - } - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:latest", - } - - // No container in memory — runtime resolution returns nothing - runtime.EXPECT().ListContainers(mock.Anything, false).Return([]*domain.Container{}, nil) - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListImages(mock.Anything).Return([]string{"myapp:latest"}, nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:latest").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:latest").Return(nil, nil) - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:latest").Return([]string{}, nil) - - // Network isolation - should check and create an owned network. - networkName := svc.generateNetworkName("test.example.com") - runtime.EXPECT().NetworkExists(mock.Anything, networkName).Return(false, nil) - runtime.EXPECT().CreateNetwork(mock.Anything, networkName, domain.NetworkConfig{ - Driver: "bridge", - Labels: map[string]string{domain.LabelManaged: "true"}, - }).Return(nil) - - container := &domain.Container{ID: "container-123", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { - return cfg.NetworkMode == networkName - })).Return(container, nil) - runtime.EXPECT().StartContainer(mock.Anything, "container-123").Return(nil) - - // Wait for ready - runtime.EXPECT().IsContainerRunning(mock.Anything, "container-123").Return(true, nil).Times(2) - - runtime.EXPECT().InspectContainer(mock.Anything, "container-123").Return(&domain.Container{ - ID: "container-123", - Status: "running", - }, nil) - - // Publish event - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - assert.NotNil(t, result) -} - -func TestService_Deploy_WithVolumeAutoCreate(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - VolumeAutoCreate: true, - VolumePrefix: "gordon", - ReadinessDelay: time.Millisecond, // Minimal delay for tests - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - } - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:latest", - } - - // No container in memory — runtime resolution returns nothing - runtime.EXPECT().ListContainers(mock.Anything, false).Return([]*domain.Container{}, nil) - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListImages(mock.Anything).Return([]string{"myapp:latest"}, nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:latest").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:latest").Return(nil, nil) - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:latest").Return([]string{}, nil) - - // Volume operations - dataVolume := generateVolumeName("gordon", "test.example.com", "/data") - configVolume := generateVolumeName("gordon", "test.example.com", "/config") - runtime.EXPECT().InspectImageVolumes(mock.Anything, "myapp:latest").Return([]string{"/data", "/config"}, nil) - runtime.EXPECT().VolumeExists(mock.Anything, legacyVolumeName("gordon", "test.example.com", "/data")).Return(false, nil) - runtime.EXPECT().VolumeExists(mock.Anything, dataVolume).Return(false, nil) - runtime.EXPECT().CreateVolume(mock.Anything, dataVolume).Return(nil) - runtime.EXPECT().VolumeExists(mock.Anything, legacyVolumeName("gordon", "test.example.com", "/config")).Return(false, nil) - runtime.EXPECT().VolumeExists(mock.Anything, configVolume).Return(true, nil) - - container := &domain.Container{ID: "container-123", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { - return len(cfg.Volumes) == 2 - })).Return(container, nil) - runtime.EXPECT().StartContainer(mock.Anything, "container-123").Return(nil) - - // Wait for ready - runtime.EXPECT().IsContainerRunning(mock.Anything, "container-123").Return(true, nil).Times(2) - - runtime.EXPECT().InspectContainer(mock.Anything, "container-123").Return(&domain.Container{ - ID: "container-123", - Status: "running", - }, nil) - - // Publish event - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - assert.NotNil(t, result) -} - -func TestService_Stop(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, nil) - ctx := testContext() - - runtime.EXPECT().StopContainer(mock.Anything, "container-123").Return(nil) - - err := svc.Stop(ctx, "container-123") - - assert.NoError(t, err) -} - -func TestService_Stop_Error(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, nil) - ctx := testContext() - - runtime.EXPECT().StopContainer(mock.Anything, "container-123").Return(errors.New("container not found")) - - err := svc.Stop(ctx, "container-123") - - assert.Error(t, err) - assert.Contains(t, err.Error(), "failed to stop container") -} - -func TestService_Remove(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, nil) - ctx := testContext() - - // Add tracked container - svc.containers["test.example.com"] = &domain.Container{ - ID: "container-123", - Name: "gordon-test.example.com", - } - - runtime.EXPECT().RemoveContainer(mock.Anything, "container-123", true).Return(nil) - - err := svc.Remove(ctx, "container-123", true) - - assert.NoError(t, err) - - // Verify container is no longer tracked - _, exists := svc.Get(ctx, "test.example.com") - assert.False(t, exists) -} - -func TestService_Remove_WithNetworkCleanup(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - NetworkIsolation: true, - NetworkPrefix: "gordon", - } - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - - svc.containers["test.example.com"] = &domain.Container{ - ID: "container-123", - Name: "gordon-test.example.com", - } - - runtime.EXPECT().RemoveContainer(mock.Anything, "container-123", true).Return(nil) - networkName := svc.generateNetworkName("test.example.com") - runtime.EXPECT().ListNetworks(mock.Anything).Return([]*domain.NetworkInfo{ - {Name: networkName, Containers: []string{}, Labels: map[string]string{domain.LabelManaged: "true"}}, - }, nil) - runtime.EXPECT().RemoveNetwork(mock.Anything, networkName).Return(nil) - - err := svc.Remove(ctx, "container-123", true) - - assert.NoError(t, err) -} - -func TestService_Get(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, nil) - ctx := testContext() - - container := &domain.Container{ID: "container-123"} - svc.containers["test.example.com"] = container - - result, exists := svc.Get(ctx, "test.example.com") - - assert.True(t, exists) - assert.Equal(t, "container-123", result.ID) -} - -func TestService_Get_NotFound(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, nil) - ctx := testContext() - - result, exists := svc.Get(ctx, "nonexistent.example.com") - - assert.False(t, exists) - assert.Nil(t, result) -} - -func TestService_List(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, nil) - ctx := testContext() - - svc.containers["app1.example.com"] = &domain.Container{ID: "container-1"} - svc.containers["app2.example.com"] = &domain.Container{ID: "container-2"} - - result := svc.List(ctx) - - assert.Len(t, result, 2) - assert.Equal(t, "container-1", result["app1.example.com"].ID) - assert.Equal(t, "container-2", result["app2.example.com"].ID) -} - -func TestService_HealthCheck(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, nil) - ctx := testContext() - - svc.containers["healthy.example.com"] = &domain.Container{ID: "healthy-container"} - svc.containers["unhealthy.example.com"] = &domain.Container{ID: "unhealthy-container"} - - runtime.EXPECT().IsContainerRunning(mock.Anything, "healthy-container").Return(true, nil) - runtime.EXPECT().IsContainerRunning(mock.Anything, "unhealthy-container").Return(false, nil) - - result := svc.HealthCheck(ctx) - - assert.True(t, result["healthy.example.com"]) - assert.False(t, result["unhealthy.example.com"]) -} - -func TestService_SyncContainers(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, nil) - ctx := testContext() - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - { - ID: "container-1", - Name: "gordon-app1.example.com", - Labels: map[string]string{ - "gordon.domain": "app1.example.com", - "gordon.managed": "true", - }, - }, - { - ID: "container-2", - Name: "gordon-app2.example.com", - Labels: map[string]string{ - "gordon.domain": "app2.example.com", - "gordon.managed": "true", - }, - }, - { - ID: "unmanaged-container", - Name: "some-other-app", - Labels: map[string]string{}, - }, - }, nil) - - err := svc.SyncContainers(ctx) - - assert.NoError(t, err) - assert.Len(t, svc.containers, 2) - assert.Contains(t, svc.containers, "app1.example.com") - assert.Contains(t, svc.containers, "app2.example.com") -} - -func TestService_SyncContainers_KeepsRestartingManagedContainers(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, nil) - ctx := testContext() - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - { - ID: "container-1", - Name: "gordon-app1.example.com", - Status: "restarting", - Labels: map[string]string{ - "gordon.domain": "app1.example.com", - "gordon.managed": "true", - }, - }, - }, nil) - - err := svc.SyncContainers(ctx) - - assert.NoError(t, err) - assert.Len(t, svc.containers, 1) - assert.Contains(t, svc.containers, "app1.example.com") -} - -func TestService_Shutdown(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, nil) - ctx := testContext() - - svc.containers["app1.example.com"] = &domain.Container{ID: "container-1"} - svc.containers["app2.example.com"] = &domain.Container{ID: "container-2"} - - // Shutdown no longer stops containers — they are left running for - // SyncContainers + AutoStart to pick back up on next boot. - err := svc.Shutdown(ctx) - - assert.NoError(t, err) - // Containers remain tracked (not stopped). - assert.Len(t, svc.containers, 2) -} - -func TestService_Restart_NotFound(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, nil) - ctx := testContext() - - err := svc.Restart(ctx, "nonexistent.example.com", false) - - assert.Error(t, err) - assert.True(t, errors.Is(err, domain.ErrContainerNotFound)) -} - -func TestService_Restart_Success(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, nil) - ctx := testContext() - - svc.containers["test.example.com"] = &domain.Container{ - ID: "container-123", - Name: "gordon-test.example.com", - Status: "running", - } - - runtime.EXPECT().RestartContainer(mock.Anything, "container-123").Return(nil) - - err := svc.Restart(ctx, "test.example.com", false) - - assert.NoError(t, err) -} - -func TestService_Restart_Error(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, nil) - ctx := testContext() - - svc.containers["test.example.com"] = &domain.Container{ - ID: "container-123", - Name: "gordon-test.example.com", - Status: "running", - } - - runtime.EXPECT().RestartContainer(mock.Anything, "container-123").Return(errors.New("restart failed")) - - err := svc.Restart(ctx, "test.example.com", false) - - assert.Error(t, err) - assert.Contains(t, err.Error(), "failed to restart container") -} - -func TestService_Restart_ReconcilesStaleContainerID(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, nil) - ctx := testContext() - - svc.containers["test.example.com"] = &domain.Container{ - ID: "stale-container-id", - Name: "gordon-test.example.com", - Status: "running", - } - - runtime.EXPECT().RestartContainer(mock.Anything, "stale-container-id"). - Return(errors.New("no container with name or ID \"stale-container-id\" found")). - Once() - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - { - ID: "fresh-container-id", - Name: "gordon-test.example.com", - Labels: map[string]string{ - "gordon.domain": "test.example.com", - "gordon.managed": "true", - }, - }, - }, nil).Once() - runtime.EXPECT().RestartContainer(mock.Anything, "fresh-container-id").Return(nil).Once() - - err := svc.Restart(ctx, "test.example.com", false) - - assert.NoError(t, err) -} - -func TestService_Restart_ReconcilesStaleContainerID_NotFoundAfterSync(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, nil) - ctx := testContext() - - svc.containers["test.example.com"] = &domain.Container{ - ID: "stale-container-id", - Name: "gordon-test.example.com", - Status: "running", - } - - runtime.EXPECT().RestartContainer(mock.Anything, "stale-container-id"). - Return(errors.New("no such container")). - Once() - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil).Once() - - err := svc.Restart(ctx, "test.example.com", false) - - assert.Error(t, err) - assert.True(t, errors.Is(err, domain.ErrContainerNotFound)) -} - -func TestService_Restart_WithAttachments(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, nil) - ctx := testContext() - - svc.containers["test.example.com"] = &domain.Container{ - ID: "container-123", - Name: "gordon-test.example.com", - Status: "running", - } - svc.attachments["test.example.com"] = []string{"container-attach-1", "container-attach-2"} - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - { - ID: "container-123", - Labels: map[string]string{ - domain.LabelDomain: "test.example.com", - domain.LabelManaged: "true", - }, - }, - { - ID: "container-attach-1", - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: "test.example.com", - }, - }, - { - ID: "container-attach-2", - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: "test.example.com", - }, - }, - }, nil).Maybe() - runtime.EXPECT().RestartContainer(mock.Anything, "container-123").Return(nil) - runtime.EXPECT().RestartContainer(mock.Anything, "container-attach-1").Return(nil) - runtime.EXPECT().RestartContainer(mock.Anything, "container-attach-2").Return(fmt.Errorf("boom")) - - err := svc.Restart(ctx, "test.example.com", true) - - assert.NoError(t, err) -} - -func TestService_Restart_WithAttachments_ErrorsWhenConfiguredButNotDeployed(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - provider := mocks.NewMockAttachmentConfigProvider(t) - provider.EXPECT().GetAttachmentConfig().Return(out.AttachmentConfigSnapshot{ - Attachments: map[string][]string{"test.example.com": {"postgres:16"}}, - NetworkGroups: map[string][]string{}, - }).Maybe() - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, provider) - - svc.containers["test.example.com"] = &domain.Container{ - ID: "container-123", - Name: "gordon-test.example.com", - Status: "running", - } - // No attachment containers tracked — s.attachments is empty - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil).Maybe() - - err := svc.Restart(testContext(), "test.example.com", true) - - assert.Error(t, err) - assert.True(t, errors.Is(err, domain.ErrAttachmentNotDeployed)) - // Main container should NOT have been restarted - runtime.AssertNotCalled(t, "RestartContainer", mock.Anything, mock.Anything) -} - -func TestService_Restart_WithAttachments_ErrorsWhenPartiallyDeployed(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - provider := mocks.NewMockAttachmentConfigProvider(t) - provider.EXPECT().GetAttachmentConfig().Return(out.AttachmentConfigSnapshot{ - Attachments: map[string][]string{"test.example.com": {"postgres:16", "redis:7"}}, - NetworkGroups: map[string][]string{}, - }).Maybe() - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, provider) - - svc.containers["test.example.com"] = &domain.Container{ - ID: "container-123", - Name: "gordon-test.example.com", - Status: "running", - } - // Only 1 of 2 attachments deployed - svc.attachments["test.example.com"] = []string{"attach-postgres"} - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - { - ID: "container-123", - Labels: map[string]string{ - domain.LabelDomain: "test.example.com", - domain.LabelManaged: "true", - }, - }, - { - ID: "attach-postgres", - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: "test.example.com", - }, - }, - }, nil).Maybe() - err := svc.Restart(testContext(), "test.example.com", true) - - assert.Error(t, err) - assert.True(t, errors.Is(err, domain.ErrAttachmentNotDeployed)) - assert.Contains(t, err.Error(), "1 missing") - // Main container should NOT have been restarted - runtime.AssertNotCalled(t, "RestartContainer", mock.Anything, mock.Anything) -} - -func TestService_Restart_WithAttachments_SucceedsWhenAllDeployed(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - provider := mocks.NewMockAttachmentConfigProvider(t) - provider.EXPECT().GetAttachmentConfig().Return(out.AttachmentConfigSnapshot{ - Attachments: map[string][]string{"test.example.com": {"postgres:16"}}, - NetworkGroups: map[string][]string{}, - }).Maybe() - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, provider) - - svc.containers["test.example.com"] = &domain.Container{ - ID: "container-123", - Name: "gordon-test.example.com", - Status: "running", - } - svc.attachments["test.example.com"] = []string{"attach-456"} - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - { - ID: "container-123", - Labels: map[string]string{ - domain.LabelDomain: "test.example.com", - domain.LabelManaged: "true", - }, - }, - { - ID: "attach-456", - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: "test.example.com", - }, - }, - }, nil).Maybe() - runtime.EXPECT().RestartContainer(mock.Anything, "container-123").Return(nil) - runtime.EXPECT().RestartContainer(mock.Anything, "attach-456").Return(nil) - - err := svc.Restart(testContext(), "test.example.com", true) - assert.NoError(t, err) -} - -func TestService_Restart_WithoutAttachmentFlag_IgnoresMissingAttachments(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - provider := mocks.NewMockAttachmentConfigProvider(t) - provider.EXPECT().GetAttachmentConfig().Return(out.AttachmentConfigSnapshot{ - Attachments: map[string][]string{"test.example.com": {"postgres:16"}}, - NetworkGroups: map[string][]string{}, - }).Maybe() - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, provider) - - svc.containers["test.example.com"] = &domain.Container{ - ID: "container-123", - Name: "gordon-test.example.com", - Status: "running", - } - - runtime.EXPECT().RestartContainer(mock.Anything, "container-123").Return(nil) - - err := svc.Restart(testContext(), "test.example.com", false) - assert.NoError(t, err) -} - -func TestService_Restart_WithAttachments_NoConfiguredNoDeployed_Succeeds(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - provider := mocks.NewMockAttachmentConfigProvider(t) - provider.EXPECT().GetAttachmentConfig().Return(out.AttachmentConfigSnapshot{ - Attachments: map[string][]string{}, - NetworkGroups: map[string][]string{}, - }).Maybe() - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, provider) - - svc.containers["test.example.com"] = &domain.Container{ - ID: "container-123", - Name: "gordon-test.example.com", - Status: "running", - } - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil).Maybe() - runtime.EXPECT().ListContainers(mock.Anything, false).Return([]*domain.Container{}, nil).Maybe() - - runtime.EXPECT().RestartContainer(mock.Anything, "container-123").Return(nil) - - err := svc.Restart(testContext(), "test.example.com", true) - assert.NoError(t, err) -} - -func TestNormalizeImageRef(t *testing.T) { - tests := []struct { - name string - input string - expected string - }{ - { - name: "simple image", - input: "nginx", - expected: "docker.io/library/nginx", - }, - { - name: "user/repo", - input: "user/myapp", - expected: "docker.io/user/myapp", - }, - { - name: "registry/repo", - input: "gcr.io/project/image", - expected: "gcr.io/project/image", - }, - { - name: "with tag", - input: "nginx:latest", - expected: "docker.io/library/nginx", - }, - { - name: "localhost with port", - input: "localhost:5000/myapp:latest", - expected: "localhost:5000/myapp", - }, - { - name: "localhost with port no tag", - input: "localhost:5000/myapp", - expected: "localhost:5000/myapp", - }, - { - name: "registry domain with port", - input: "registry.example.com:5000/image:v1.0", - expected: "registry.example.com:5000/image", - }, - { - name: "digest ref preserves digest", - input: "registry.example.com/image@sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", - expected: "registry.example.com/image@sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", - }, - { - name: "docker hub digest ref preserves digest", - input: "nginx@sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb", - expected: "nginx@sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb", - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - result := normalizeImageRef(tt.input) - assert.Equal(t, tt.expected, result) - }) - } -} - -func TestGenerateVolumeName(t *testing.T) { - tests := []struct { - name string - prefix string - domain string - volumePath string - expected string - }{ - { - name: "simple path", - prefix: "gordon", - domain: "app.example.com", - volumePath: "/data", - expected: "gordon-app__example__com-28059829b105-data-3a6eb0790f39", - }, - { - name: "nested path", - prefix: "gordon", - domain: "app.example.com", - volumePath: "/var/lib/data", - expected: "gordon-app__example__com-28059829b105-var--lib--data-323f0a44c820", - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - result := generateVolumeName(tt.prefix, tt.domain, tt.volumePath) - assert.Equal(t, tt.expected, result) - }) - } -} - -func TestService_SetupVolumes_ReusesLegacyVolume(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{VolumeAutoCreate: true, VolumePrefix: "gordon"}, nil) - legacyName := legacyVolumeName("gordon", "app.example.com", "/data") - - runtime.EXPECT().InspectImageVolumes(mock.Anything, "myapp:latest").Return([]string{"/data"}, nil) - runtime.EXPECT().VolumeExists(mock.Anything, legacyName).Return(true, nil) - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{{ - Labels: map[string]string{domain.LabelManaged: "true", domain.LabelRoute: "app.example.com"}, - VolumeMounts: []domain.ContainerVolumeMount{{Name: legacyName}}, - }}, nil) - - volumes, err := svc.setupVolumes(testContext(), "app.example.com", "myapp:latest", nil) - - require.NoError(t, err) - assert.Equal(t, map[string]string{"/data": legacyName}, volumes) -} - -func TestService_SetupVolumes_DoesNotReuseCollidingLegacyVolume(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{VolumeAutoCreate: true, VolumePrefix: "gordon"}, nil) - legacyName := legacyVolumeName("gordon", "app.example.com", "/data") - stableName := generateVolumeName("gordon", "app.example.com", "/data") - require.Equal(t, legacyName, legacyVolumeName("gordon", "app-example.com", "/data")) - - runtime.EXPECT().InspectImageVolumes(mock.Anything, "myapp:latest").Return([]string{"/data"}, nil) - runtime.EXPECT().VolumeExists(mock.Anything, legacyName).Return(true, nil) - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{{ - Labels: map[string]string{domain.LabelManaged: "true", domain.LabelRoute: "app-example.com"}, - VolumeMounts: []domain.ContainerVolumeMount{{Name: legacyName}}, - }}, nil) - runtime.EXPECT().VolumeExists(mock.Anything, stableName).Return(true, nil) - - volumes, err := svc.setupVolumes(testContext(), "app.example.com", "myapp:latest", nil) - - require.NoError(t, err) - assert.Equal(t, map[string]string{"/data": stableName}, volumes) -} - -func TestService_SetupVolumes_CreatesStableVolumeWhenLegacyVolumeIsForeign(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{VolumeAutoCreate: true, VolumePrefix: "gordon"}, nil) - legacyName := legacyVolumeName("gordon", "app.example.com", "/data") - stableName := generateVolumeName("gordon", "app.example.com", "/data") - - runtime.EXPECT().InspectImageVolumes(mock.Anything, "myapp:latest").Return([]string{"/data"}, nil).Once() - runtime.EXPECT().VolumeExists(mock.Anything, legacyName).Return(true, nil).Once() - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{{ - Labels: map[string]string{domain.LabelManaged: "true", domain.LabelRoute: "other.example.com"}, - VolumeMounts: []domain.ContainerVolumeMount{{Name: legacyName}}, - }}, nil).Once() - runtime.EXPECT().VolumeExists(mock.Anything, stableName).Return(false, nil).Once() - runtime.EXPECT().CreateVolume(mock.Anything, stableName).Return(nil).Once() - - volumes, err := svc.setupVolumes(testContext(), "app.example.com", "myapp:latest", nil) - - require.NoError(t, err) - assert.Equal(t, map[string]string{"/data": stableName}, volumes) -} - -func TestService_SetupVolumes_PreservesPreferredVolumesWhenAutoCreateDisabled(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{VolumeAutoCreate: false}, nil) - preferred := map[string]namedVolumeMount{ - "/data": {Name: "existing-data"}, - } - - runtime.EXPECT().VolumeExists(mock.Anything, "existing-data").Return(true, nil).Once() - - volumes, err := svc.setupVolumes(testContext(), "app.example.com", "myapp:latest", preferred) - - require.NoError(t, err) - assert.Equal(t, map[string]string{"/data": "existing-data"}, volumes) -} - -func TestService_SetupVolumes_RejectsUnverifiedLegacyVolumeWithoutStableReplacement(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{VolumeAutoCreate: true, VolumePrefix: "gordon"}, nil) - legacyName := legacyVolumeName("gordon", "app.example.com", "/data") - stableName := generateVolumeName("gordon", "app.example.com", "/data") - - runtime.EXPECT().InspectImageVolumes(mock.Anything, "myapp:latest").Return([]string{"/data"}, nil).Once() - runtime.EXPECT().VolumeExists(mock.Anything, legacyName).Return(true, nil).Once() - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil).Once() - runtime.EXPECT().VolumeExists(mock.Anything, stableName).Return(false, nil).Once() - - volumes, err := svc.setupVolumes(testContext(), "app.example.com", "myapp:latest", nil) - - require.ErrorContains(t, err, "refusing to create a replacement") - require.ErrorIs(t, err, domain.ErrVolumeOwnershipUnverified) - assert.Empty(t, volumes) -} - -func TestService_SetupVolumes_RejectsLegacyVolumeSharedWithForeignWorkload(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{VolumeAutoCreate: true, VolumePrefix: "gordon"}, nil) - legacyName := legacyVolumeName("gordon", "app.example.com", "/data") - stableName := generateVolumeName("gordon", "app.example.com", "/data") - mount := []domain.ContainerVolumeMount{{Name: legacyName}} - - runtime.EXPECT().InspectImageVolumes(mock.Anything, "myapp:latest").Return([]string{"/data"}, nil).Once() - runtime.EXPECT().VolumeExists(mock.Anything, legacyName).Return(true, nil).Once() - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - { - Labels: map[string]string{domain.LabelManaged: "true", domain.LabelRoute: "app.example.com"}, - VolumeMounts: mount, - }, - { - Labels: map[string]string{domain.LabelManaged: "true", domain.LabelRoute: "other.example.com"}, - VolumeMounts: mount, - }, - }, nil).Once() - runtime.EXPECT().VolumeExists(mock.Anything, stableName).Return(false, nil).Once() - - volumes, err := svc.setupVolumes(testContext(), "app.example.com", "myapp:latest", nil) - - require.ErrorIs(t, err, domain.ErrVolumeOwnershipUnverified) - assert.Empty(t, volumes) -} - -func TestValidateAttachmentOwnership(t *testing.T) { - owned := &domain.Container{Labels: map[string]string{ - domain.LabelManaged: "true", domain.LabelAttachment: "true", domain.LabelAttachedTo: "app.example.com", - }} - require.NoError(t, validateAttachmentOwnership(owned, "app.example.com", "gordon-app-postgres")) - - foreign := &domain.Container{Labels: map[string]string{ - domain.LabelManaged: "true", domain.LabelAttachment: "true", domain.LabelAttachedTo: "other.example.com", - }} - assert.ErrorIs(t, validateAttachmentOwnership(foreign, "app.example.com", "gordon-app-postgres"), domain.ErrAttachmentOwnershipMismatch) -} - -func TestMergeEnvironmentVariables(t *testing.T) { - dockerEnv := []string{"FOO=docker", "BAR=docker"} - userEnv := []string{"FOO=user", "BAZ=user"} - - result := mergeEnvironmentVariables(dockerEnv, userEnv) - - // User env should override docker env - assert.Contains(t, result, "FOO=user") - assert.Contains(t, result, "BAR=docker") - assert.Contains(t, result, "BAZ=user") -} - -func TestBuildImageRef(t *testing.T) { - tests := []struct { - name string - image string - registryAuthEnabled bool - registryDomain string - wantRef string - }{ - { - name: "adds registry domain prefix", - image: "myapp:latest", - registryAuthEnabled: true, - registryDomain: "registry.example.com", - wantRef: "registry.example.com/myapp:latest", - }, - { - name: "keeps existing registry domain prefix", - image: "registry.example.com/myapp:latest", - registryAuthEnabled: true, - registryDomain: "registry.example.com", - wantRef: "registry.example.com/myapp:latest", - }, - { - name: "skips external registry images", - image: "docker.io/library/nginx:latest", - registryAuthEnabled: true, - registryDomain: "registry.example.com", - wantRef: "docker.io/library/nginx:latest", - }, - { - name: "returns original when auth disabled", - image: "myapp:latest", - registryAuthEnabled: false, - registryDomain: "registry.example.com", - wantRef: "myapp:latest", - }, - { - name: "localhost domain adds prefix", - image: "myapp:latest", - registryAuthEnabled: true, - registryDomain: "localhost:5000", - wantRef: "localhost:5000/myapp:latest", - }, - { - name: "localhost domain keeps existing prefix", - image: "localhost:5000/myapp:latest", - registryAuthEnabled: true, - registryDomain: "localhost:5000", - wantRef: "localhost:5000/myapp:latest", - }, - { - name: "explicit host:port different than RegistryDomain", - image: "localhost:5001/myapp:latest", - registryAuthEnabled: true, - registryDomain: "localhost:5000", - wantRef: "localhost:5001/myapp:latest", - }, - { - name: "ghcr.io external registry", - image: "ghcr.io/owner/repo:latest", - registryAuthEnabled: true, - registryDomain: "localhost:5000", - wantRef: "ghcr.io/owner/repo:latest", - }, - { - name: "gcr.io external registry", - image: "gcr.io/project/image:v1", - registryAuthEnabled: true, - registryDomain: "registry.example.com", - wantRef: "gcr.io/project/image:v1", - }, - { - name: "quay.io external registry", - image: "quay.io/org/app:tag", - registryAuthEnabled: true, - registryDomain: "localhost:5000", - wantRef: "quay.io/org/app:tag", - }, - { - name: "localhost without port keeps existing prefix", - image: "localhost/myapp:latest", - registryAuthEnabled: true, - registryDomain: "localhost:5000", - wantRef: "localhost/myapp:latest", - }, - { - name: "registry domain with trailing slash", - image: "myapp:latest", - registryAuthEnabled: true, - registryDomain: "registry.example.com/", - wantRef: "registry.example.com/myapp:latest", - }, - { - name: "ipv6 registry without port", - image: "[fd00::1]/myapp:latest", - registryAuthEnabled: true, - registryDomain: "localhost:5000", - wantRef: "[fd00::1]/myapp:latest", - }, - { - name: "ipv6 registry with port", - image: "[fd00::1]:5000/myapp:latest", - registryAuthEnabled: true, - registryDomain: "localhost:5000", - wantRef: "[fd00::1]:5000/myapp:latest", - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - svc := &Service{ - config: Config{ - RegistryAuthEnabled: tt.registryAuthEnabled, - RegistryDomain: tt.registryDomain, - }, - } - gotRef := svc.buildImageRef(tt.image) - assert.Equal(t, tt.wantRef, gotRef, "unexpected image reference") - }) - } -} - -func TestRewriteToRegistryDomain(t *testing.T) { - tests := []struct { - name string - imageRef string - registryDomain string - wantRef string - }{ - { - name: "keeps registry domain prefix", - imageRef: "registry.example.com/myapp:latest", - registryDomain: "registry.example.com", - wantRef: "registry.example.com/myapp:latest", - }, - { - name: "prefixes registry domain", - imageRef: "myapp:v1.0", - registryDomain: "registry.example.com", - wantRef: "registry.example.com/myapp:v1.0", - }, - { - name: "prefixes external image", - imageRef: "docker.io/library/nginx:latest", - registryDomain: "registry.example.com", - wantRef: "registry.example.com/docker.io/library/nginx:latest", - }, - { - name: "empty registry domain returns original", - imageRef: "registry.example.com/myapp:latest", - registryDomain: "", - wantRef: "registry.example.com/myapp:latest", - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - gotRef := rewriteToRegistryDomain(tt.imageRef, tt.registryDomain) - assert.Equal(t, tt.wantRef, gotRef, "unexpected rewritten reference") - }) - } -} - -func TestRewriteToLocalRegistry(t *testing.T) { - tests := []struct { - name string - imageRef string - registryDomain string - legacyRegistryDomains []string - registryPort int - wantRef string - }{ - { - name: "rewrites registry domain prefix", - imageRef: "registry.example.com/myapp:latest", - registryDomain: "registry.example.com", - registryPort: 5000, - wantRef: "localhost:5000/myapp:latest", - }, - { - name: "prefixes local registry when no domain", - imageRef: "myapp:v1.0", - registryDomain: "registry.example.com", - registryPort: 5000, - wantRef: "localhost:5000/myapp:v1.0", - }, - { - name: "keeps existing localhost prefix", - imageRef: "localhost:5000/myapp:latest", - registryDomain: "registry.example.com", - registryPort: 5000, - wantRef: "localhost:5000/myapp:latest", - }, - { - name: "rewrites legacy registry domain prefix", - imageRef: "old-registry.example.com/myapp:latest", - registryDomain: "new-registry.example.com", - legacyRegistryDomains: []string{"old-registry.example.com"}, - registryPort: 5000, - wantRef: "localhost:5000/myapp:latest", - }, - { - name: "rewrites legacy registry domain prefix with explicit port and digest", - imageRef: "old-registry.example.com:5001/myapp@sha256:deadbeef", - registryDomain: "new-registry.example.com", - legacyRegistryDomains: []string{"old-registry.example.com:5001"}, - registryPort: 5000, - wantRef: "localhost:5000/myapp@sha256:deadbeef", - }, - { - name: "external image remains external", - imageRef: "docker.io/library/nginx:latest", - registryDomain: "new-registry.example.com", - legacyRegistryDomains: []string{"old-registry.example.com"}, - registryPort: 5000, - wantRef: "docker.io/library/nginx:latest", - }, - { - name: "hostile lookalike remains external", - imageRef: "old-registry.example.com.evil/myapp:latest", - registryDomain: "new-registry.example.com", - legacyRegistryDomains: []string{"old-registry.example.com"}, - registryPort: 5000, - wantRef: "old-registry.example.com.evil/myapp:latest", - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - gotRef := rewriteToLocalRegistry(tt.imageRef, tt.registryDomain, tt.legacyRegistryDomains, tt.registryPort) - assert.Equal(t, tt.wantRef, gotRef, "unexpected rewritten reference") - }) - } -} - -func TestService_Deploy_InternalDeployForcesPull(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - RegistryAuthEnabled: true, - RegistryDomain: "registry.example.com", - RegistryPort: 5000, - InternalRegistryUsername: "internal", - InternalRegistryPassword: "secret", - ReadinessDelay: time.Millisecond, - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - } - - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - - // Use internal deploy context (simulating image push event) - ctx := domain.WithInternalDeploy(testContext()) - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:latest", - } - - // No container in memory — runtime resolution returns nothing - runtime.EXPECT().ListContainers(mock.Anything, false).Return([]*domain.Container{}, nil) - - // Setup mocks - no orphaned containers - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - - // Even though image exists locally, internal deploy should force pull - // Note: ensureLocalImage returns false for internal deploy, skipping the ListImages check - runtime.EXPECT().PullImageWithAuth(mock.Anything, "localhost:5000/myapp:latest", "internal", "secret").Return(nil) - - // Tag image to canonical name - runtime.EXPECT().TagImage(mock.Anything, "localhost:5000/myapp:latest", "registry.example.com/myapp:latest").Return(nil) - runtime.EXPECT().UntagImage(mock.Anything, "localhost:5000/myapp:latest").Return(nil) - - // Get exposed ports - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "registry.example.com/myapp:latest").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "registry.example.com/myapp:latest").Return(nil, nil) - - // Load environment - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "registry.example.com/myapp:latest").Return([]string{}, nil) - - // Create and start container - createdContainer := &domain.Container{ - ID: "container-123", - Name: "gordon-test.example.com", - Status: "created", - } - runtime.EXPECT().CreateContainer(mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")).Return(createdContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "container-123").Return(nil) - - // Wait for ready - runtime.EXPECT().IsContainerRunning(mock.Anything, "container-123").Return(true, nil).Times(2) - - // Re-inspect after start - runningContainer := &domain.Container{ - ID: "container-123", - Name: "gordon-test.example.com", - Status: "running", - Ports: []int{8080}, - } - runtime.EXPECT().InspectContainer(mock.Anything, "container-123").Return(runningContainer, nil) - - // Publish event - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - assert.NotNil(t, result) - assert.Equal(t, "container-123", result.ID) -} - -func TestService_Deploy_InternalDeployForcesPull_LegacyRegistryHost(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - RegistryAuthEnabled: true, - RegistryDomain: "new-registry.example.com", - LegacyRegistryDomains: []string{"old-registry.example.com"}, - RegistryPort: 5000, - InternalRegistryUsername: "internal", - InternalRegistryPassword: "secret", - ReadinessDelay: time.Millisecond, - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - } - - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := domain.WithInternalDeploy(testContext()) - - route := domain.Route{ - Domain: "test.example.com", - Image: "old-registry.example.com/myapp:latest", - } - - runtime.EXPECT().ListContainers(mock.Anything, false).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - runtime.EXPECT().PullImageWithAuth(mock.Anything, "localhost:5000/myapp:latest", "internal", "secret").Return(nil) - runtime.EXPECT().TagImage(mock.Anything, "localhost:5000/myapp:latest", "old-registry.example.com/myapp:latest").Return(nil) - runtime.EXPECT().UntagImage(mock.Anything, "localhost:5000/myapp:latest").Return(nil) - - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "old-registry.example.com/myapp:latest").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "old-registry.example.com/myapp:latest").Return(nil, nil) - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "old-registry.example.com/myapp:latest").Return([]string{}, nil) - - createdContainer := &domain.Container{ - ID: "container-legacy", - Name: "gordon-test.example.com", - Status: "created", - } - runtime.EXPECT().CreateContainer(mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")).Return(createdContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "container-legacy").Return(nil) - runtime.EXPECT().IsContainerRunning(mock.Anything, "container-legacy").Return(true, nil).Times(2) - runtime.EXPECT().InspectContainer(mock.Anything, "container-legacy").Return(&domain.Container{ - ID: "container-legacy", - Name: "gordon-test.example.com", - Status: "running", - Ports: []int{8080}, - }, nil) - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - assert.NotNil(t, result) - assert.Equal(t, "container-legacy", result.ID) -} - -func TestService_AutoStart_StartsNewContainers(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - ReadinessDelay: time.Millisecond, - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - } - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - - routes := []domain.Route{ - {Domain: "app1.example.com", Image: "myapp1:latest"}, - {Domain: "app2.example.com", Image: "myapp2:latest"}, - } - - // No container in memory — runtime resolution returns nothing (one per route) - runtime.EXPECT().ListContainers(mock.Anything, false).Return([]*domain.Container{}, nil).Times(2) - - // Setup mocks for route deployments - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil).Times(2) - runtime.EXPECT().ListImages(mock.Anything).Return([]string{"myapp1:latest", "myapp2:latest"}, nil).Times(2) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, mock.AnythingOfType("string")).Return([]int{8080}, nil).Times(2) - runtime.EXPECT().GetImageLabels(mock.Anything, mock.AnythingOfType("string")).Return(nil, nil).Times(2) - envLoader.EXPECT().LoadEnv(mock.Anything, mock.AnythingOfType("string")).Return([]string{}, nil).Times(2) - runtime.EXPECT().InspectImageEnv(mock.Anything, mock.AnythingOfType("string")).Return([]string{}, nil).Times(2) - - // Create and start containers — readiness is skipped so no IsContainerRunning calls. - runtime.EXPECT().CreateContainer(mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")).Return(&domain.Container{ - ID: "container-1", Name: "gordon-app1.example.com", Status: "created", - }, nil).Once() - runtime.EXPECT().CreateContainer(mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")).Return(&domain.Container{ - ID: "container-2", Name: "gordon-app2.example.com", Status: "created", - }, nil).Once() - runtime.EXPECT().StartContainer(mock.Anything, mock.AnythingOfType("string")).Return(nil).Times(2) - runtime.EXPECT().InspectContainer(mock.Anything, mock.AnythingOfType("string")).Return(&domain.Container{ - Status: "running", - }, nil).Times(2) - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil).Times(2) - - err := svc.AutoStart(ctx, routes) - - assert.NoError(t, err) - assert.Len(t, svc.containers, 2) -} - -func TestService_AutoStart_SkipsExistingContainers(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - ReadinessDelay: time.Millisecond, - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - } - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - - // Pre-populate with existing container - svc.containers["app1.example.com"] = &domain.Container{ - ID: "existing-container", - } - - routes := []domain.Route{ - {Domain: "app1.example.com", Image: "myapp1:latest"}, // Already exists - {Domain: "app2.example.com", Image: "myapp2:latest"}, // New route - } - - // No container in memory for app2 — runtime resolution returns nothing - runtime.EXPECT().ListContainers(mock.Anything, false).Return([]*domain.Container{}, nil).Once() - - // Only deploy for app2 (app1 is skipped). Readiness is skipped — no IsContainerRunning. - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil).Once() - runtime.EXPECT().ListImages(mock.Anything).Return([]string{"myapp2:latest"}, nil).Once() - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp2:latest").Return([]int{8080}, nil).Once() - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp2:latest").Return(nil, nil).Once() - envLoader.EXPECT().LoadEnv(mock.Anything, "app2.example.com").Return([]string{}, nil).Once() - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp2:latest").Return([]string{}, nil).Once() - runtime.EXPECT().CreateContainer(mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")).Return(&domain.Container{ - ID: "container-2", Status: "created", - }, nil).Once() - runtime.EXPECT().StartContainer(mock.Anything, "container-2").Return(nil).Once() - runtime.EXPECT().InspectContainer(mock.Anything, "container-2").Return(&domain.Container{ - Status: "running", - }, nil).Once() - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil).Once() - - err := svc.AutoStart(ctx, routes) - - assert.NoError(t, err) - assert.Len(t, svc.containers, 2) -} - -func TestService_AutoStart_HandlesDeployErrors(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{AllowedRegistries: []string{"docker.io"}} - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - - routes := []domain.Route{ - {Domain: "app1.example.com", Image: "myapp1:latest"}, - } - - // No container in memory — runtime resolution returns nothing - runtime.EXPECT().ListContainers(mock.Anything, false).Return([]*domain.Container{}, nil) - - // Setup mocks for failure - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListImages(mock.Anything).Return([]string{}, nil) - runtime.EXPECT().PullImage(mock.Anything, "myapp1:latest").Return(errors.New("image not found")) - - err := svc.AutoStart(ctx, routes) - - // AutoStart should return error when some deployments fail - assert.Error(t, err) - assert.Contains(t, err.Error(), "auto-start completed with 1 errors") -} - -func TestService_AutoStart_UsesInternalDeployContext(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{ - RegistryDomain: "reg.example.com", - RegistryPort: 5000, - ReadinessDelay: time.Millisecond, - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - } - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - - // Mark context as internal deploy — this is what syncAndAutoStart should do - ctx := domain.WithInternalDeploy(testContext()) - - routes := []domain.Route{ - {Domain: "app1.example.com", Image: "reg.example.com/myapp:latest"}, - } - - // No container in memory — runtime resolution returns nothing - runtime.EXPECT().ListContainers(mock.Anything, false).Return([]*domain.Container{}, nil) - - // Key assertion: PullImage should be called with localhost:5000 rewrite, - // NOT the original reg.example.com/myapp:latest. - // Readiness is skipped — no IsContainerRunning calls. - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - runtime.EXPECT().PullImage(mock.Anything, "localhost:5000/myapp:latest").Return(nil) - runtime.EXPECT().TagImage(mock.Anything, "localhost:5000/myapp:latest", "reg.example.com/myapp:latest").Return(nil) - runtime.EXPECT().UntagImage(mock.Anything, "localhost:5000/myapp:latest").Return(nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "reg.example.com/myapp:latest").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "reg.example.com/myapp:latest").Return(nil, nil) - envLoader.EXPECT().LoadEnv(mock.Anything, "app1.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "reg.example.com/myapp:latest").Return([]string{}, nil) - runtime.EXPECT().CreateContainer(mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")).Return(&domain.Container{ - ID: "container-1", Name: "gordon-app1.example.com", Status: "created", - }, nil) - runtime.EXPECT().StartContainer(mock.Anything, "container-1").Return(nil) - runtime.EXPECT().InspectContainer(mock.Anything, "container-1").Return(&domain.Container{ - Status: "running", - }, nil) - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - err := svc.AutoStart(ctx, routes) - assert.NoError(t, err) -} - -// TestService_Deploy_OrphanCleanupSkipsTrackedContainer verifies that the orphan cleanup -// does not remove the currently tracked container during zero-downtime deployment. -// This is critical for preventing downtime - the old container must stay running -// until the new container is ready and traffic is switched. -func TestService_Deploy_OrphanCleanupSkipsTrackedContainer(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - cacheInvalidator := mocks.NewMockProxyCacheInvalidator(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - ReadinessDelay: time.Millisecond, // Minimal delay for tests - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - StabilizationDelay: time.Millisecond, - } - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - svc.SetProxyCacheInvalidator(cacheInvalidator) - ctx := testContext() - - // Pre-populate with existing tracked container - existingContainer := &domain.Container{ - ID: "tracked-container-123", - Name: "gordon-test.example.com", - Status: "running", - } - svc.containers["test.example.com"] = existingContainer - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:v2", - } - - // ListContainers returns the tracked container - this simulates the orphan check - // finding a container with the same name. The bug was that it would remove this - // container BEFORE the new one was ready, causing downtime. - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - { - ID: "tracked-container-123", - Name: "gordon-test.example.com", - Status: "running", - }, - }, nil) - - // Image operations - runtime.EXPECT().ListImages(mock.Anything).Return([]string{}, nil) - runtime.EXPECT().PullImage(mock.Anything, "myapp:v2").Return(nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:v2").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:v2").Return(nil, nil) - - // Environment - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:v2").Return([]string{}, nil) - - // Create new container with -new suffix for zero-downtime - newContainer := &domain.Container{ID: "new-container", Name: "gordon-test.example.com-new", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { - return cfg.Name == "gordon-test.example.com-new" - })).Return(newContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "new-container").Return(nil) - - // Wait for ready (2 calls) + stabilization check (1 call) - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(true, nil).Times(3) - - // Inspect after ready - runtime.EXPECT().InspectContainer(mock.Anything, "new-container").Return(&domain.Container{ - ID: "new-container", - Status: "running", - }, nil) - - // Publish event - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - // Synchronous cache invalidation - cacheInvalidator.EXPECT().InvalidateTarget(mock.Anything, "test.example.com").Return() - - // NOW (after new container is ready) the old container should be stopped and removed - // This is the correct zero-downtime sequence - not during orphan cleanup - runtime.EXPECT().StopContainer(mock.Anything, "tracked-container-123").Return(nil) - runtime.EXPECT().RemoveContainer(mock.Anything, "tracked-container-123", true).Return(nil) - - // Rename new container to canonical name - runtime.EXPECT().RenameContainer(mock.Anything, "new-container", "gordon-test.example.com").Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - svc.WaitForCleanup() - assert.Equal(t, "new-container", result.ID) - - // Verify new container is now tracked - tracked, exists := svc.Get(ctx, "test.example.com") - assert.True(t, exists) - assert.Equal(t, "new-container", tracked.ID) -} - -// TestService_Deploy_OrphanCleanupRemovesTrueOrphans verifies that stopped canonical -// containers with the same name but NOT tracked are properly removed as orphans. -// Running canonical containers are preserved (see NeverKillsRunningCanonical test). -func TestService_Deploy_OrphanCleanupRemovesTrueOrphans(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - ReadinessDelay: time.Millisecond, // Minimal delay for tests - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - } - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - - // NO tracked container - service is empty for this domain - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:v1", - } - - // No container in memory — runtime resolution finds no managed container - // (the orphan is stopped so it won't appear in running-only query) - runtime.EXPECT().ListContainers(mock.Anything, false).Return([]*domain.Container{}, nil) - - // ListContainers(true) returns a STOPPED orphaned container (same name, but not tracked) - // Stopped canonical containers are safe to remove during orphan cleanup - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - { - ID: "orphan-container", - Name: "gordon-test.example.com", - Status: "exited", - }, - }, nil) - - // Stopped orphan should be stopped and removed BEFORE we proceed - runtime.EXPECT().StopContainer(mock.Anything, "orphan-container").Return(nil) - runtime.EXPECT().RemoveContainer(mock.Anything, "orphan-container", true).Return(nil) - - // Image operations - runtime.EXPECT().ListImages(mock.Anything).Return([]string{}, nil) - runtime.EXPECT().PullImage(mock.Anything, "myapp:v1").Return(nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:v1").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:v1").Return(nil, nil) - - // Environment - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:v1").Return([]string{}, nil) - - // Create container (no -new suffix since no existing tracked container) - newContainer := &domain.Container{ID: "new-container", Name: "gordon-test.example.com", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { - return cfg.Name == "gordon-test.example.com" - })).Return(newContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "new-container").Return(nil) - - // Wait for ready - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(true, nil).Times(2) - - // Inspect after ready - runtime.EXPECT().InspectContainer(mock.Anything, "new-container").Return(&domain.Container{ - ID: "new-container", - Status: "running", - Ports: []int{8080}, - }, nil) - - // Publish event - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - assert.Equal(t, "new-container", result.ID) - - // Verify new container is tracked - tracked, exists := svc.Get(ctx, "test.example.com") - assert.True(t, exists) - assert.Equal(t, "new-container", tracked.ID) -} - -func TestService_Deploy_OrphanCleanupRemovesRunningStaleCandidate(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - cacheInvalidator := mocks.NewMockProxyCacheInvalidator(t) - - config := testMinDelayConfig() - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - svc.SetProxyCacheInvalidator(cacheInvalidator) - ctx := testContext() - - // The tracked canonical container is active. A candidate left running by an - // interrupted request must be removed before its temporary name is reused. - svc.containers["test.example.com"] = &domain.Container{ - ID: "tracked-container-123", - Name: "gordon-test.example.com", - Status: "running", - } - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:v2", - } - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - { - ID: "tracked-container-123", - Name: "gordon-test.example.com", - Status: "running", - }, - { - ID: "stale-new-container", - Name: "gordon-test.example.com-new", - Status: "running", - }, - }, nil) - runtime.EXPECT().StopContainer(mock.Anything, "stale-new-container").Return(nil).Once() - runtime.EXPECT().RemoveContainer(mock.Anything, "stale-new-container", true).Return(nil).Once() - - runtime.EXPECT().ListImages(mock.Anything).Return([]string{}, nil) - runtime.EXPECT().PullImage(mock.Anything, "myapp:v2").Return(nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:v2").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:v2").Return(nil, nil) - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:v2").Return([]string{}, nil) - - newContainer := &domain.Container{ID: "new-container", Name: "gordon-test.example.com-new", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { - return cfg.Name == "gordon-test.example.com-new" - })).Return(newContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "new-container").Return(nil) - // Wait for ready (2 calls) + stabilization check (1 call) - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(true, nil).Times(3) - runtime.EXPECT().InspectContainer(mock.Anything, "new-container").Return(&domain.Container{ - ID: "new-container", - Status: "running", - }, nil) - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - // Synchronous cache invalidation - cacheInvalidator.EXPECT().InvalidateTarget(mock.Anything, "test.example.com").Return() - - runtime.EXPECT().StopContainer(mock.Anything, "tracked-container-123").Return(nil) - runtime.EXPECT().RemoveContainer(mock.Anything, "tracked-container-123", true).Return(nil) - runtime.EXPECT().RenameContainer(mock.Anything, "new-container", "gordon-test.example.com").Return(nil) - - result, err := svc.Deploy(ctx, route) - assert.NoError(t, err) - svc.WaitForCleanup() - assert.NotNil(t, result) - assert.Equal(t, "new-container", result.ID) -} - -func TestService_Deploy_TrackedTempContainerUsesAlternateTempName(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - cacheInvalidator := mocks.NewMockProxyCacheInvalidator(t) - - config := testMinDelayConfig() - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - svc.SetProxyCacheInvalidator(cacheInvalidator) - ctx := testContext() - - // Simulate an interrupted zero-downtime deploy that left "-new" as the tracked container name. - svc.containers["test.example.com"] = &domain.Container{ - ID: "tracked-temp-container", - Name: "gordon-test.example.com-new", - Status: "running", - } - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:v2", - } - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - { - ID: "tracked-temp-container", - Name: "gordon-test.example.com-new", - Status: "running", - }, - }, nil) - - runtime.EXPECT().ListImages(mock.Anything).Return([]string{}, nil) - runtime.EXPECT().PullImage(mock.Anything, "myapp:v2").Return(nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:v2").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:v2").Return(nil, nil) - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:v2").Return([]string{}, nil) - - created := &domain.Container{ID: "new-container", Name: "gordon-test.example.com-next", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { - return cfg.Name == "gordon-test.example.com-next" - })).Return(created, nil) - runtime.EXPECT().StartContainer(mock.Anything, "new-container").Return(nil) - // Wait for ready (2 calls) + stabilization check (1 call) - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(true, nil).Times(3) - runtime.EXPECT().InspectContainer(mock.Anything, "new-container").Return(&domain.Container{ - ID: "new-container", - Status: "running", - }, nil) - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - // Synchronous cache invalidation - cacheInvalidator.EXPECT().InvalidateTarget(mock.Anything, "test.example.com").Return() - - runtime.EXPECT().StopContainer(mock.Anything, "tracked-temp-container").Return(nil) - runtime.EXPECT().RemoveContainer(mock.Anything, "tracked-temp-container", true).Return(nil) - runtime.EXPECT().RenameContainer(mock.Anything, "new-container", "gordon-test.example.com").Return(nil) - - result, err := svc.Deploy(ctx, route) - assert.NoError(t, err) - svc.WaitForCleanup() - assert.NotNil(t, result) - assert.Equal(t, "new-container", result.ID) -} - -func TestService_StripRegistryPrefix(t *testing.T) { - tests := []struct { - name string - registryDomain string - legacyRegistryDomains []string - image string - expected string - }{ - { - name: "strips registry prefix", - registryDomain: "reg.example.com", - image: "reg.example.com/myapp:latest", - expected: "myapp:latest", - }, - { - name: "strips registry prefix with trailing slash in domain", - registryDomain: "reg.example.com/", - image: "reg.example.com/myapp:v1.0", - expected: "myapp:v1.0", - }, - { - name: "strips legacy registry prefix", - registryDomain: "new-reg.example.com", - legacyRegistryDomains: []string{"old-reg.example.com"}, - image: "old-reg.example.com/myapp:latest", - expected: "myapp:latest", - }, - { - name: "preserves image without registry prefix", - registryDomain: "reg.example.com", - image: "nginx:latest", - expected: "nginx:latest", - }, - { - name: "preserves image with different registry", - registryDomain: "reg.example.com", - image: "gcr.io/project/image:latest", - expected: "gcr.io/project/image:latest", - }, - { - name: "handles empty registry domain", - registryDomain: "", - image: "reg.example.com/myapp:latest", - expected: "reg.example.com/myapp:latest", - }, - { - name: "handles nested paths", - registryDomain: "reg.example.com", - image: "reg.example.com/org/repo/app:latest", - expected: "org/repo/app:latest", - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{ - RegistryDomain: tt.registryDomain, - LegacyRegistryDomains: tt.legacyRegistryDomains, - } - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - - result := svc.stripRegistryPrefix(tt.image) - assert.Equal(t, tt.expected, result) - }) - } -} - -func TestSanitizeName(t *testing.T) { - tests := []struct { - domain string - expected string - description string - }{ - { - domain: "git.example.com", - expected: "git__example__com", - description: "Dots become double underscores", - }, - { - domain: "git-example.com", - expected: "git-example__com", - description: "Hyphens preserved, dots become underscores", - }, - { - domain: "app:8080.example.com", - expected: "app-_8080__example__com", - description: "Colons become hyphen-underscore", - }, - { - domain: "git.example.com:3000", - expected: "git__example__com-_3000", - description: "Multiple separators handled distinctly", - }, - { - domain: "simple.com", - expected: "simple__com", - description: "Simple domain", - }, - } - - for _, tt := range tests { - t.Run(tt.domain, func(t *testing.T) { - result := sanitizeName(tt.domain) - assert.Equal(t, tt.expected, result, "sanitization should match expected") - }) - } - - // Verify no collisions between potentially conflicting domains - t.Run("NoCollisions", func(t *testing.T) { - domains := []string{ - "git.example.com", - "git-example.com", - "app:8080.example.com", - "app-8080-example.com", - } - - results := make(map[string]string) - for _, d := range domains { - result := sanitizeName(d) - if original, exists := results[result]; exists { - t.Errorf("COLLISION: %q and %q both sanitize to %q", original, d, result) - } - results[result] = d - } - }) -} - -func TestService_Deploy_ConcurrentSameDomain(t *testing.T) { - // Verify that concurrent Deploy calls for the same domain are serialized - // and both succeed without container name conflicts. - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - cacheInvalidator := mocks.NewMockProxyCacheInvalidator(t) - - config := testMinDelayConfig() - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - svc.SetProxyCacheInvalidator(cacheInvalidator) - ctx := testContext() - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:latest", - } - - // Track call order to verify serialization - var callOrder []string - var callMu sync.Mutex - - // Setup mocks that will be called by both deploys sequentially. - // First deploy: no existing container → creates gordon-test.example.com - // Second deploy: sees first container → creates gordon-test.example.com-new - - // First deploy resolves from runtime (no container in memory yet) - runtime.EXPECT().ListContainers(mock.Anything, false).Return([]*domain.Container{}, nil).Once() - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil).Times(2) - runtime.EXPECT().ListImages(mock.Anything).Return([]string{"myapp:latest"}, nil).Times(2) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:latest").Return([]int{8080}, nil).Times(2) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:latest").Return(nil, nil).Times(2) - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return(nil, nil).Times(2) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:latest").Return(nil, nil).Times(2) - - createCall := 0 - runtime.EXPECT().CreateContainer(mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")). - RunAndReturn(func(_ context.Context, cfg *domain.ContainerConfig) (*domain.Container, error) { - callMu.Lock() - createCall++ - n := createCall - callOrder = append(callOrder, fmt.Sprintf("create-%d:%s", n, cfg.Name)) - callMu.Unlock() - return &domain.Container{ - ID: fmt.Sprintf("container-%d", n), - Name: cfg.Name, - Image: cfg.Image, - Status: "created", - }, nil - }) - - runtime.EXPECT().StartContainer(mock.Anything, mock.AnythingOfType("string")).Return(nil).Times(2) - // IsContainerRunning: 2x per deploy in waitForReady + 1x stabilization for the second deploy (which has existing) - runtime.EXPECT().IsContainerRunning(mock.Anything, mock.AnythingOfType("string")).Return(true, nil).Times(5) - - runtime.EXPECT().InspectContainer(mock.Anything, mock.AnythingOfType("string")). - RunAndReturn(func(_ context.Context, id string) (*domain.Container, error) { - // Return correct container name based on ID - // container-1 is the first deploy (canonical name) - // container-2 is the second deploy (with -new suffix) - name := "gordon-test.example.com" - if id == "container-2" { - name = "gordon-test.example.com-new" - } - return &domain.Container{ - ID: id, - Name: name, - Image: "myapp:latest", - Status: "running", - Ports: []int{8080}, - }, nil - }) - - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.Anything).Return(nil).Times(2) - - // Both deploys call InvalidateTarget (to ensure proxy picks up the new container). - cacheInvalidator.EXPECT().InvalidateTarget(mock.Anything, "test.example.com").Return().Times(2) - - // which means it will also stop+remove+rename the old container. - runtime.EXPECT().StopContainer(mock.Anything, mock.AnythingOfType("string")).Return(nil).Once() - runtime.EXPECT().RemoveContainer(mock.Anything, mock.AnythingOfType("string"), true).Return(nil).Once() - runtime.EXPECT().RenameContainer(mock.Anything, mock.AnythingOfType("string"), mock.AnythingOfType("string")).Return(nil).Once() - - // Launch two concurrent deploys - var wg sync.WaitGroup - errs := make([]error, 2) - - wg.Add(2) - for i := range 2 { - go func(idx int) { - defer wg.Done() - _, errs[idx] = svc.Deploy(ctx, route) - }(i) - } - wg.Wait() - svc.WaitForCleanup() - - // Both should succeed (second deploy waits for the first to finish) - assert.NoError(t, errs[0], "first deploy should succeed") - assert.NoError(t, errs[1], "second deploy should succeed") - - // Verify creates were serialized (not interleaved) - callMu.Lock() - assert.Len(t, callOrder, 2, "should have exactly 2 create calls") - // First deploy uses canonical name, second uses -new suffix (zero-downtime) - assert.Equal(t, "create-1:gordon-test.example.com", callOrder[0], "first create should use canonical name") - assert.Contains(t, callOrder[1], "gordon-test.example.com-new", "second create should use -new suffix") - callMu.Unlock() -} - -func TestService_Deploy_ContextCancellation(t *testing.T) { - // Verify that deploy lock acquisition respects context cancellation. - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - ReadinessDelay: time.Millisecond, - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - } - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:latest", - } - - // Test context cancelled before lock acquisition. - cancelledCtx, cancel := context.WithCancel(testContext()) - cancel() // Cancel immediately - - _, err := svc.Deploy(cancelledCtx, route) - assert.Error(t, err, "deploy should fail with cancelled context") - assert.ErrorIs(t, err, context.Canceled, "error should be context.Canceled") -} - -// TestService_Deploy_CacheInvalidationBeforeOldContainerStop verifies the fix for -// the proxy cache race condition: InvalidateTarget must be called synchronously -// BEFORE the old container is stopped, preventing 503 errors during deployment. -func TestService_Deploy_CacheInvalidationBeforeOldContainerStop(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - cacheInvalidator := mocks.NewMockProxyCacheInvalidator(t) - - config := testMinDelayConfig() - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - svc.SetProxyCacheInvalidator(cacheInvalidator) - ctx := testContext() - - // Pre-populate with existing container - existingContainer := &domain.Container{ - ID: "old-container", - Name: "gordon-test.example.com", - Status: "running", - } - svc.containers["test.example.com"] = existingContainer - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:v2", - } - - // Track ordering of cache invalidation and container stop - var callOrder []string - var orderMu sync.Mutex - - // Standard deploy mocks - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListImages(mock.Anything).Return([]string{"myapp:v2"}, nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:v2").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:v2").Return(nil, nil) - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:v2").Return([]string{}, nil) - - newContainer := &domain.Container{ID: "new-container", Name: "gordon-test.example.com-new", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")).Return(newContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "new-container").Return(nil) - // Wait for ready (2 calls) + stabilization check (1 call) - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(true, nil).Times(3) - runtime.EXPECT().InspectContainer(mock.Anything, "new-container").Return(&domain.Container{ - ID: "new-container", Status: "running", - }, nil) - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - // Track: cache invalidation should happen first - cacheInvalidator.EXPECT().InvalidateTarget(mock.Anything, "test.example.com"). - Run(func(_ context.Context, _ string) { - orderMu.Lock() - callOrder = append(callOrder, "invalidate_cache") - orderMu.Unlock() - }).Return() - - // Track: stop should happen after invalidation - runtime.EXPECT().StopContainer(mock.Anything, "old-container"). - RunAndReturn(func(_ context.Context, _ string) error { - orderMu.Lock() - callOrder = append(callOrder, "stop_old_container") - orderMu.Unlock() - return nil - }) - runtime.EXPECT().RemoveContainer(mock.Anything, "old-container", true).Return(nil) - runtime.EXPECT().RenameContainer(mock.Anything, "new-container", "gordon-test.example.com").Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - svc.WaitForCleanup() - assert.Equal(t, "new-container", result.ID) - - // Verify ordering: cache invalidation MUST happen before old container stop - orderMu.Lock() - defer orderMu.Unlock() - assert.Equal(t, []string{"invalidate_cache", "stop_old_container"}, callOrder, - "InvalidateTarget must be called before StopContainer to prevent 503 errors") -} - -// TestService_Deploy_NilCacheInvalidator verifies that deploy works gracefully -// when no cache invalidator is set (e.g., in tests or minimal configurations). -func TestService_Deploy_NilCacheInvalidator(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := testMinDelayConfig() - // Intentionally NOT setting cache invalidator - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - - // Pre-populate with existing container - svc.containers["test.example.com"] = &domain.Container{ - ID: "old-container", - Name: "gordon-test.example.com", - Status: "running", - } - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:v2", - } - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListImages(mock.Anything).Return([]string{"myapp:v2"}, nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:v2").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:v2").Return(nil, nil) - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:v2").Return([]string{}, nil) - - newContainer := &domain.Container{ID: "new-container", Name: "gordon-test.example.com-new", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")).Return(newContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "new-container").Return(nil) - // Wait for ready (2 calls) + stabilization check (1 call) - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(true, nil).Times(3) - runtime.EXPECT().InspectContainer(mock.Anything, "new-container").Return(&domain.Container{ - ID: "new-container", Status: "running", - }, nil) - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - runtime.EXPECT().StopContainer(mock.Anything, "old-container").Return(nil) - runtime.EXPECT().RemoveContainer(mock.Anything, "old-container", true).Return(nil) - runtime.EXPECT().RenameContainer(mock.Anything, "new-container", "gordon-test.example.com").Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - svc.WaitForCleanup() - assert.Equal(t, "new-container", result.ID) -} - -func TestService_Deploy_SkipsRedundantDeploy(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - ReadinessDelay: time.Millisecond, - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - } - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - expectedEnvHash := hashEnvironment([]string{}) - - // Pre-populate with existing container that has an ImageID (set by InspectContainer). - // This simulates the first deploy (from image.pushed event) having already completed. - existingContainer := &domain.Container{ - ID: "existing-container", - Name: "gordon-test.example.com", - Image: "myapp:latest", - ImageID: "sha256:abc123", - Status: "running", - Labels: map[string]string{ - domain.LabelEnvHash: expectedEnvHash, - }, - } - svc.containers["test.example.com"] = existingContainer - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:latest", - } - - // prepareDeployResources will run: orphan cleanup, image pull, etc. - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListImages(mock.Anything).Return([]string{"myapp:latest"}, nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:latest").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:latest").Return(nil, nil) - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:latest").Return([]string{}, nil) - - // skipRedundantDeploy: resolve image ID and compare with existing container - runtime.EXPECT().GetImageID(mock.Anything, "myapp:latest").Return("sha256:abc123", nil) - runtime.EXPECT().IsContainerRunning(mock.Anything, "existing-container").Return(true, nil) - - // No CreateContainer, StartContainer, or readiness calls should happen - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - assert.NotNil(t, result) - assert.Equal(t, "existing-container", result.ID) - - // Container should still be tracked - tracked, exists := svc.Get(ctx, "test.example.com") - assert.True(t, exists) - assert.Equal(t, "existing-container", tracked.ID) -} - -func TestService_Deploy_SkipsRedundantDeploy_WhenTrackedImageIDMissing(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - ReadinessDelay: time.Millisecond, - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - } - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - expectedEnvHash := hashEnvironment([]string{}) - - // Pre-populate with existing container missing ImageID (common after stale in-memory state). - existingContainer := &domain.Container{ - ID: "existing-container", - Name: "gordon-test.example.com", - Image: "myapp:latest", - Status: "running", - Labels: map[string]string{ - domain.LabelEnvHash: expectedEnvHash, - }, - } - svc.containers["test.example.com"] = existingContainer - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:latest", - } - - // prepareDeployResources - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListImages(mock.Anything).Return([]string{"myapp:latest"}, nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:latest").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:latest").Return(nil, nil) - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:latest").Return([]string{}, nil) - - // Existing container image ID is recovered from runtime inspection, - // then redundancy check can skip the deploy. - runtime.EXPECT().InspectContainer(mock.Anything, "existing-container").Return(&domain.Container{ - ID: "existing-container", - ImageID: "sha256:abc123", - Status: "running", - }, nil) - runtime.EXPECT().GetImageID(mock.Anything, "myapp:latest").Return("sha256:abc123", nil) - runtime.EXPECT().IsContainerRunning(mock.Anything, "existing-container").Return(true, nil) - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - assert.NotNil(t, result) - assert.Equal(t, "existing-container", result.ID) - - tracked, exists := svc.Get(ctx, "test.example.com") - assert.True(t, exists) - assert.Equal(t, "existing-container", tracked.ID) -} - -func TestService_Deploy_DoesNotSkipWhenImageIDDiffers(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - cacheInvalidator := mocks.NewMockProxyCacheInvalidator(t) - - config := testMinDelayConfig() - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - svc.SetProxyCacheInvalidator(cacheInvalidator) - ctx := testContext() - - // Existing container with a DIFFERENT image ID than what's being deployed - existingContainer := &domain.Container{ - ID: "old-container", - Name: "gordon-test.example.com", - Image: "myapp:latest", - ImageID: "sha256:old-image", - Status: "running", - } - svc.containers["test.example.com"] = existingContainer - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:latest", - } - - // prepareDeployResources - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListImages(mock.Anything).Return([]string{"myapp:latest"}, nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:latest").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:latest").Return(nil, nil) - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:latest").Return([]string{}, nil) - - // skipRedundantDeploy: image IDs differ, so deploy proceeds - runtime.EXPECT().GetImageID(mock.Anything, "myapp:latest").Return("sha256:new-image", nil) - - // Full deploy proceeds - newContainer := &domain.Container{ID: "new-container", Name: "gordon-test.example.com-new", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")).Return(newContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "new-container").Return(nil) - // Wait for ready (2 calls) + stabilization check (1 call) - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(true, nil).Times(3) - runtime.EXPECT().InspectContainer(mock.Anything, "new-container").Return(&domain.Container{ - ID: "new-container", Status: "running", - }, nil) - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - cacheInvalidator.EXPECT().InvalidateTarget(mock.Anything, "test.example.com").Return() - runtime.EXPECT().StopContainer(mock.Anything, "old-container").Return(nil) - runtime.EXPECT().RemoveContainer(mock.Anything, "old-container", true).Return(nil) - runtime.EXPECT().RenameContainer(mock.Anything, "new-container", "gordon-test.example.com").Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - svc.WaitForCleanup() - assert.Equal(t, "new-container", result.ID) -} - -func TestSyncContainersRebuildsAttachmentMap(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, nil) - ctx := testContext() - - // ListContainers returns one main container and one attachment container - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - { - ID: "main-container-1", - Labels: map[string]string{ - domain.LabelDomain: "app.example.com", - domain.LabelManaged: "true", - }, - }, - { - ID: "attachment-container-1", - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: "app.example.com", - }, - }, - }, nil).Once() - - require.NoError(t, svc.SyncContainers(ctx)) - - // Main container should be in s.containers - tracked, exists := svc.Get(ctx, "app.example.com") - require.True(t, exists, "main container should be tracked after SyncContainers") - assert.Equal(t, "main-container-1", tracked.ID) - - // Attachment container should be in s.attachments under the owner domain - svc.mu.RLock() - attachIDs := svc.attachments["app.example.com"] - svc.mu.RUnlock() - - require.Len(t, attachIDs, 1, "attachment map should contain one entry for app.example.com after SyncContainers") - assert.Equal(t, "attachment-container-1", attachIDs[0]) -} - -func TestService_Deploy_SkipRedundantDeploy_ContainerNotRunning(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := testMinDelayConfig() - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - expectedEnvHash := hashEnvironment([]string{}) - - // Existing container has same image ID but is NOT running (crashed) - existingContainer := &domain.Container{ - ID: "existing-container", - Name: "gordon-test.example.com", - Image: "myapp:latest", - ImageID: "sha256:abc123", - Status: "exited", - Labels: map[string]string{ - domain.LabelEnvHash: expectedEnvHash, - }, - } - svc.containers["test.example.com"] = existingContainer - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:latest", - } - - // prepareDeployResources - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListImages(mock.Anything).Return([]string{"myapp:latest"}, nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:latest").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:latest").Return(nil, nil) - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:latest").Return([]string{}, nil) - - // skipRedundantDeploy: same image ID but container not running => proceed with deploy - runtime.EXPECT().GetImageID(mock.Anything, "myapp:latest").Return("sha256:abc123", nil) - runtime.EXPECT().IsContainerRunning(mock.Anything, "existing-container").Return(false, nil) - - // Full deploy proceeds (replaces crashed container) - newContainer := &domain.Container{ID: "new-container", Name: "gordon-test.example.com-new", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")).Return(newContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "new-container").Return(nil) - // Wait for ready (2 calls) + stabilization check (1 call) - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(true, nil).Times(3) - runtime.EXPECT().InspectContainer(mock.Anything, "new-container").Return(&domain.Container{ - ID: "new-container", Status: "running", - }, nil) - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - // Old (exited) container is still finalized - runtime.EXPECT().StopContainer(mock.Anything, "existing-container").Return(nil) - runtime.EXPECT().RemoveContainer(mock.Anything, "existing-container", true).Return(nil) - runtime.EXPECT().RenameContainer(mock.Anything, "new-container", "gordon-test.example.com").Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - svc.WaitForCleanup() - assert.Equal(t, "new-container", result.ID) -} - -func TestService_Deploy_DoesNotSkip_WhenEnvHashDiffers(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := testMinDelayConfig() - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - expectedEnvHash := hashEnvironment([]string{}) - - existingContainer := &domain.Container{ - ID: "existing-container", - Name: "gordon-test.example.com", - Image: "myapp:latest", - ImageID: "sha256:abc123", - Status: "running", - Labels: map[string]string{ - domain.LabelEnvHash: "different-hash", - }, - } - svc.containers["test.example.com"] = existingContainer - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:latest", - } - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListImages(mock.Anything).Return([]string{"myapp:latest"}, nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:latest").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:latest").Return(nil, nil) - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:latest").Return([]string{}, nil) - - runtime.EXPECT().GetImageID(mock.Anything, "myapp:latest").Return("sha256:abc123", nil) - runtime.EXPECT().IsContainerRunning(mock.Anything, "existing-container").Return(true, nil) - - newContainer := &domain.Container{ID: "new-container", Name: "gordon-test.example.com-new", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { - return cfg.Labels[domain.LabelEnvHash] == expectedEnvHash - })).Return(newContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "new-container").Return(nil) - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(true, nil).Times(3) - runtime.EXPECT().InspectContainer(mock.Anything, "new-container").Return(&domain.Container{ - ID: "new-container", Status: "running", - }, nil) - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - runtime.EXPECT().StopContainer(mock.Anything, "existing-container").Return(nil) - runtime.EXPECT().RemoveContainer(mock.Anything, "existing-container", true).Return(nil) - runtime.EXPECT().RenameContainer(mock.Anything, "new-container", "gordon-test.example.com").Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - svc.WaitForCleanup() - assert.Equal(t, "new-container", result.ID) -} - -func TestService_Deploy_DoesNotSkip_WhenEnvHashMissing(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := testMinDelayConfig() - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - expectedEnvHash := hashEnvironment([]string{}) - - existingContainer := &domain.Container{ - ID: "existing-container", - Name: "gordon-test.example.com", - Image: "myapp:latest", - ImageID: "sha256:abc123", - Status: "running", - Labels: map[string]string{}, - } - svc.containers["test.example.com"] = existingContainer - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:latest", - } - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - runtime.EXPECT().ListImages(mock.Anything).Return([]string{"myapp:latest"}, nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:latest").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:latest").Return(nil, nil) - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:latest").Return([]string{}, nil) - - runtime.EXPECT().GetImageID(mock.Anything, "myapp:latest").Return("sha256:abc123", nil) - runtime.EXPECT().IsContainerRunning(mock.Anything, "existing-container").Return(true, nil) - - newContainer := &domain.Container{ID: "new-container", Name: "gordon-test.example.com-new", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { - return cfg.Labels[domain.LabelEnvHash] == expectedEnvHash - })).Return(newContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "new-container").Return(nil) - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(true, nil).Times(3) - runtime.EXPECT().InspectContainer(mock.Anything, "new-container").Return(&domain.Container{ - ID: "new-container", Status: "running", - }, nil) - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - runtime.EXPECT().StopContainer(mock.Anything, "existing-container").Return(nil) - runtime.EXPECT().RemoveContainer(mock.Anything, "existing-container", true).Return(nil) - runtime.EXPECT().RenameContainer(mock.Anything, "new-container", "gordon-test.example.com").Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - svc.WaitForCleanup() - assert.Equal(t, "new-container", result.ID) -} - -func TestService_attachmentEnvDrifted_ReturnsFalseWhenHashMatches(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, testMinDelayConfig(), nil) - ctx := testContext() - imageRef := svc.buildImageRef("postgres:16") - envVars := []string{"POSTGRES_PASSWORD=secret"} - envHash := hashEnvironment(envVars) - - envLoader.EXPECT().LoadEnv(mock.Anything, "gordon-app-example-com-postgres").Return(envVars, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, imageRef).Return([]string{}, nil) - - drifted, err := svc.attachmentEnvDrifted(ctx, &domain.Container{ - Labels: map[string]string{domain.LabelEnvHash: envHash}, - }, "gordon-app-example-com-postgres", "postgres:16") - - assert.NoError(t, err) - assert.False(t, drifted) -} - -func TestService_attachmentEnvDrifted_ReturnsTrueWhenHashDiffers(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, testMinDelayConfig(), nil) - ctx := testContext() - imageRef := svc.buildImageRef("postgres:16") - envVars := []string{"POSTGRES_PASSWORD=secret"} - - envLoader.EXPECT().LoadEnv(mock.Anything, "gordon-app-example-com-postgres").Return(envVars, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, imageRef).Return([]string{}, nil) - - drifted, err := svc.attachmentEnvDrifted(ctx, &domain.Container{ - Labels: map[string]string{domain.LabelEnvHash: "stale-hash"}, - }, "gordon-app-example-com-postgres", "postgres:16") - - assert.NoError(t, err) - assert.True(t, drifted) -} - -func TestService_attachmentEnvDrifted_ReturnsTrueWhenHashMissing(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, testMinDelayConfig(), nil) - ctx := testContext() - imageRef := svc.buildImageRef("postgres:16") - envVars := []string{"POSTGRES_PASSWORD=secret"} - - envLoader.EXPECT().LoadEnv(mock.Anything, "gordon-app-example-com-postgres").Return(envVars, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, imageRef).Return([]string{}, nil) - - drifted, err := svc.attachmentEnvDrifted(ctx, &domain.Container{ - Labels: map[string]string{}, - }, "gordon-app-example-com-postgres", "postgres:16") - - assert.NoError(t, err) - assert.True(t, drifted) -} - -// TestService_Deploy_OrphanCleanup_NeverKillsRunningCanonical verifies that orphan -// cleanup never stops a running container with the canonical name. Only stopped -// canonical containers and temp containers (-new/-next) should be cleaned up. -// The active canonical container is resolved via resolveExistingContainer (it has the -// managed label), so the new container is created with the "-new" suffix — Docker -// disallows creating a container with the same name as an existing running container. -func TestService_Deploy_OrphanCleanup_NeverKillsRunningCanonical(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := testMinDelayConfig() - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - - // The canonical container is running and has the managed label — it will be - // discovered by resolveExistingContainer's slow path (no in-memory state). - existingContainer := &domain.Container{ - ID: "active-canonical", - Name: "gordon-test.example.com", - Status: "running", - Labels: map[string]string{domain.LabelManaged: "true"}, - } - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:v2", - } - - // resolveExistingContainer: slow path finds the running canonical container via label. - runtime.EXPECT().ListContainers(mock.Anything, false).Return([]*domain.Container{existingContainer}, nil) - - // cleanupOrphanedContainers: finds the running canonical AND a stale -new leftover. - // The canonical container must NOT be killed because it's running. - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - existingContainer, - { - ID: "stale-leftover", - Name: "gordon-test.example.com-new", - Status: "exited", - }, - }, nil) - - // Only the stale -new container should be stopped and removed during orphan cleanup. - runtime.EXPECT().StopContainer(mock.Anything, "stale-leftover").Return(nil).Once() - runtime.EXPECT().RemoveContainer(mock.Anything, "stale-leftover", true).Return(nil).Once() - - // Image operations: pull new image (existing.Image is empty so redundancy check is skipped) - runtime.EXPECT().ListImages(mock.Anything).Return([]string{}, nil) - runtime.EXPECT().PullImage(mock.Anything, "myapp:v2").Return(nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:v2").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:v2").Return(nil, nil) - - // Environment - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:v2").Return([]string{}, nil) - - // Create container — existing is the canonical container, so the new one gets "-new" suffix. - // Docker enforces unique container names; using "-new" avoids a name collision. - newContainer := &domain.Container{ID: "new-container", Name: "gordon-test.example.com-new", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { - return cfg.Name == "gordon-test.example.com-new" - })).Return(newContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "new-container").Return(nil) - - // Wait for ready (delay mode): pollContainerRunning (1×) + waitForReadyByDelay (1×) - // Post-switch stabilization: 1× — total 3 calls. - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(true, nil).Times(3) - - // Inspect after ready - runtime.EXPECT().InspectContainer(mock.Anything, "new-container").Return(&domain.Container{ - ID: "new-container", - Status: "running", - Ports: []int{8080}, - }, nil) - - // Rename new container to canonical name after stabilization. - runtime.EXPECT().RenameContainer(mock.Anything, "new-container", "gordon-test.example.com").Return(nil).Once() - - // Stop and remove old canonical container. - runtime.EXPECT().StopContainer(mock.Anything, "active-canonical").Return(nil).Once() - runtime.EXPECT().RemoveContainer(mock.Anything, "active-canonical", true).Return(nil).Once() - - // Publish event - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - svc.WaitForCleanup() - assert.Equal(t, "new-container", result.ID) - - // Verify new container is tracked - tracked, exists := svc.Get(ctx, "test.example.com") - assert.True(t, exists) - assert.Equal(t, "new-container", tracked.ID) -} - -func TestService_CleanupOrphanedContainers_SkipsRestartingContainer(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, Config{}, nil) - ctx := testContext() - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{ - { - ID: "restarting-container", - Name: "gordon-test.example.com", - Status: "restarting", - }, - }, nil) - - err := svc.cleanupOrphanedContainers(ctx, "test.example.com", "") - - assert.NoError(t, err) -} - -// TestService_Deploy_RollbackOnPostSwitchCrash verifies that if the new container crashes -// during the post-switch stabilization window, traffic is automatically rolled back to the -// old container: the old container is restored as tracked, the proxy cache is re-invalidated, -// and the failed new container is cleaned up. -func TestService_Deploy_RollbackOnPostSwitchCrash(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - cacheInvalidator := mocks.NewMockProxyCacheInvalidator(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - ReadinessDelay: time.Millisecond, - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - StabilizationDelay: time.Millisecond, - } - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - svc.SetProxyCacheInvalidator(cacheInvalidator) - ctx := testContext() - - // Pre-populate with existing container - existing := &domain.Container{ - ID: "old-container", - Name: "gordon-test.example.com", - Status: "running", - } - svc.containers["test.example.com"] = existing - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:v2", - } - - // Cleanup orphans - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - - // Image operations - runtime.EXPECT().ListImages(mock.Anything).Return([]string{}, nil) - runtime.EXPECT().PullImage(mock.Anything, "myapp:v2").Return(nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:v2").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:v2").Return(nil, nil) - - // Environment - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:v2").Return([]string{}, nil) - - // Create new container - newContainer := &domain.Container{ID: "new-container", Name: "gordon-test.example.com-new", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { - return cfg.Name == "gordon-test.example.com-new" - })).Return(newContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "new-container").Return(nil) - - // Readiness: new container is running during readiness checks (2 calls) - // Stabilization: new container has crashed (returns false) - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(true, nil).Times(2) - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(false, nil).Once() - - // Rollback: verify old container is still running before restoring - runtime.EXPECT().IsContainerRunning(mock.Anything, "old-container").Return(true, nil).Once() - - // Inspect after ready - runtime.EXPECT().InspectContainer(mock.Anything, "new-container").Return(&domain.Container{ - ID: "new-container", - Name: "gordon-test.example.com-new", - Status: "running", - }, nil) - - // Publish event (happens during activateDeployedContainer) - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - // InvalidateTarget called TWICE: once during activate, once during rollback - cacheInvalidator.EXPECT().InvalidateTarget(mock.Anything, "test.example.com").Return().Times(2) - - // Cleanup failed new container during rollback - runtime.EXPECT().StopContainer(mock.Anything, "new-container").Return(nil) - runtime.EXPECT().RemoveContainer(mock.Anything, "new-container", true).Return(nil) - - // StopContainer on old-container should NOT be called (old stays running) - // (no expectation for StopContainer("old-container")) - - result, err := svc.Deploy(ctx, route) - - // Deploy returns the old (existing) container after rollback - assert.NoError(t, err) - assert.NotNil(t, result) - assert.Equal(t, "old-container", result.ID, "should return old container after rollback") - - // Verify old container is restored as tracked - trackedRB, existsRB := svc.Get(ctx, "test.example.com") - assert.True(t, existsRB) - assert.Equal(t, "old-container", trackedRB.ID, "old container should be restored after rollback") -} - -// TestService_Deploy_StabilizationSuccess verifies that when the new container remains -// healthy during the stabilization window, deployment proceeds normally and the old -// container is finalized (stopped and removed). -func TestService_Deploy_StabilizationSuccess(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - cacheInvalidator := mocks.NewMockProxyCacheInvalidator(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - ReadinessDelay: time.Millisecond, - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - StabilizationDelay: time.Millisecond, - } - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - svc.SetProxyCacheInvalidator(cacheInvalidator) - ctx := testContext() - - // Pre-populate with existing container - existing := &domain.Container{ - ID: "old-container", - Name: "gordon-test.example.com", - Status: "running", - } - svc.containers["test.example.com"] = existing - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:v2", - } - - // Cleanup orphans - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - - // Image operations - runtime.EXPECT().ListImages(mock.Anything).Return([]string{}, nil) - runtime.EXPECT().PullImage(mock.Anything, "myapp:v2").Return(nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:v2").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:v2").Return(nil, nil) - - // Environment - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:v2").Return([]string{}, nil) - - // Create new container - newContainer := &domain.Container{ID: "new-container", Name: "gordon-test.example.com-new", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { - return cfg.Name == "gordon-test.example.com-new" - })).Return(newContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "new-container").Return(nil) - - // Readiness (2 calls) + stabilization check (1 call) — all return running - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(true, nil).Times(3) - - // Inspect after ready - runtime.EXPECT().InspectContainer(mock.Anything, "new-container").Return(&domain.Container{ - ID: "new-container", - Name: "gordon-test.example.com-new", - Status: "running", - }, nil) - - // Publish event - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - // Cache invalidation (once, during activate) - cacheInvalidator.EXPECT().InvalidateTarget(mock.Anything, "test.example.com").Return() - - // Old container finalized normally - runtime.EXPECT().StopContainer(mock.Anything, "old-container").Return(nil) - runtime.EXPECT().RemoveContainer(mock.Anything, "old-container", true).Return(nil) - runtime.EXPECT().RenameContainer(mock.Anything, "new-container", "gordon-test.example.com").Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - svc.WaitForCleanup() - assert.NotNil(t, result) - assert.Equal(t, "new-container", result.ID, "should return new container on successful stabilization") - - // Verify new container is tracked - trackedSS, existsSS := svc.Get(ctx, "test.example.com") - assert.True(t, existsSS) - assert.Equal(t, "new-container", trackedSS.ID) -} - -// TestService_Deploy_ZeroDowntime_OldNeverStoppedBeforeNewReady verifies the strict -// ordering invariant: StopContainer("old-container") must NEVER happen before -// InspectContainer("new-container") completes (the last step of readiness). -// Uses testify's NotBefore() to enforce mock call ordering. -func TestService_Deploy_ZeroDowntime_OldNeverStoppedBeforeNewReady(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - cacheInvalidator := mocks.NewMockProxyCacheInvalidator(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - ReadinessDelay: time.Millisecond, - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - StabilizationDelay: time.Millisecond, - } - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - svc.SetProxyCacheInvalidator(cacheInvalidator) - ctx := testContext() - - // Pre-populate with existing tracked container - existingContainer := &domain.Container{ - ID: "old-container", - Name: "gordon-test.example.com", - Status: "running", - } - svc.containers["test.example.com"] = existingContainer - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:v2", - } - - // Cleanup orphans - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - - // Image operations - runtime.EXPECT().ListImages(mock.Anything).Return([]string{}, nil) - runtime.EXPECT().PullImage(mock.Anything, "myapp:v2").Return(nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:v2").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:v2").Return(nil, nil) - - // Environment - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:v2").Return([]string{}, nil) - - // Create new container with -new suffix for zero-downtime - newContainer := &domain.Container{ID: "new-container", Name: "gordon-test.example.com-new", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { - return cfg.Name == "gordon-test.example.com-new" - })).Return(newContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "new-container").Return(nil) - - // Readiness: pollContainerRunning (1 call) + waitForReadyByDelay verification (1 call) - // Stabilization check (1 call) - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(true, nil).Times(3) - - // KEY ORDERING CONSTRAINT: InspectContainer("new-container") is the last step - // of readiness inside createStartedContainer. StopContainer("old-container") - // must happen AFTER this call completes. - inspectCall := runtime.EXPECT().InspectContainer(mock.Anything, "new-container"). - Return(&domain.Container{ - ID: "new-container", - Name: "gordon-test.example.com-new", - Status: "running", - }, nil) - - // Publish event - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - // Synchronous cache invalidation - cacheInvalidator.EXPECT().InvalidateTarget(mock.Anything, "test.example.com").Return() - - // STRICT ORDERING: StopContainer on old container must NOT happen before - // InspectContainer on new container (readiness completion) - runtime.EXPECT().StopContainer(mock.Anything, "old-container"). - Return(nil). - NotBefore(inspectCall.Call) - - runtime.EXPECT().RemoveContainer(mock.Anything, "old-container", true).Return(nil) - runtime.EXPECT().RenameContainer(mock.Anything, "new-container", "gordon-test.example.com").Return(nil) - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - svc.WaitForCleanup() - assert.NotNil(t, result) - assert.Equal(t, "new-container", result.ID) - - // Verify new container is tracked - tracked, exists := svc.Get(ctx, "test.example.com") - assert.True(t, exists) - assert.Equal(t, "new-container", tracked.ID) -} - -// TestService_Deploy_ZeroDowntime_ReadinessFailure_OldUntouched verifies that when a -// new container fails readiness (IsContainerRunning returns false), the old container -// is completely untouched: never stopped, never removed. Deploy returns an error and -// the old container remains as the tracked container. -func TestService_Deploy_ZeroDowntime_ReadinessFailure_OldUntouched(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - ReadinessDelay: time.Millisecond, - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - StabilizationDelay: time.Millisecond, - } - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - - // Pre-populate with existing tracked container - existingContainer := &domain.Container{ - ID: "old-container", - Name: "gordon-test.example.com", - Status: "running", - } - svc.containers["test.example.com"] = existingContainer - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:v2", - } - - // Cleanup orphans - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - - // Image operations - runtime.EXPECT().ListImages(mock.Anything).Return([]string{}, nil) - runtime.EXPECT().PullImage(mock.Anything, "myapp:v2").Return(nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:v2").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:v2").Return(nil, nil) - - // Environment - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:v2").Return([]string{}, nil) - - // Create new container - newContainer := &domain.Container{ID: "new-container", Name: "gordon-test.example.com-new", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { - return cfg.Name == "gordon-test.example.com-new" - })).Return(newContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "new-container").Return(nil) - - // Readiness FAILS: pollContainerRunning succeeds (container starts) but - // waitForReadyByDelay verification finds it not running. - // pollContainerRunning: returns true (container initially starts) - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(true, nil).Once() - // waitForReadyByDelay post-delay verification: returns false (container crashed) - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container").Return(false, nil).Once() - // Recovery window poll: return error to fail fast instead of waiting 30s - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-container"). - Return(false, errors.New("container exited")).Once() - runtime.EXPECT().GetContainerLogs(mock.Anything, "new-container", false).Return(dockerLogFrames(1, "crash line\n"), nil) - - // cleanupFailedContainer: stop and remove the failed new container - runtime.EXPECT().StopContainer(mock.Anything, "new-container").Return(nil) - runtime.EXPECT().RemoveContainer(mock.Anything, "new-container", true).Return(nil) - - // NOTE: StopContainer("old-container") and RemoveContainer("old-container", true) - // are intentionally NOT mocked. If either is called, testify will panic with - // "unexpected method call" — verifying the old container is never touched. - - result, err := svc.Deploy(ctx, route) - - // Deploy should return an error (readiness failure) - assert.Error(t, err) - assert.Nil(t, result) - var deployErr *domain.DeployFailureError - require.ErrorAs(t, err, &deployErr) - assert.Equal(t, "failed to deploy", deployErr.Error()) - assert.Equal(t, "container failed readiness check", deployErr.Cause) - assert.Equal(t, []string{"crash line"}, deployErr.Logs) - assert.Equal(t, "gordon-test.example.com-new", deployErr.ContainerName) - assert.Equal(t, "new-container", deployErr.ContainerID) - assert.ErrorContains(t, deployErr.Err, "container exited") - - // Old container must still be tracked — completely untouched - tracked, exists := svc.Get(ctx, "test.example.com") - assert.True(t, exists, "old container should still be tracked after readiness failure") - assert.Equal(t, "old-container", tracked.ID, "old container ID should be unchanged") -} - -// TestService_Deploy_ZeroDowntime_NoExisting_FirstDeploy is a sanity check: -// first deploy with no existing container should work normally without any -// finalize/drain logic being triggered. -func TestService_Deploy_ZeroDowntime_NoExisting_FirstDeploy(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - ReadinessDelay: time.Millisecond, - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - StabilizationDelay: time.Millisecond, - } - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - - // NO existing container — this is a first deploy - - route := domain.Route{ - Domain: "test.example.com", - Image: "myapp:v2", - } - - // resolveExistingContainer: no container in memory, runtime returns nothing - runtime.EXPECT().ListContainers(mock.Anything, false).Return([]*domain.Container{}, nil) - - // Cleanup orphans — nothing found - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - - // Image operations - runtime.EXPECT().ListImages(mock.Anything).Return([]string{}, nil) - runtime.EXPECT().PullImage(mock.Anything, "myapp:v2").Return(nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "myapp:v2").Return([]int{8080}, nil) - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:v2").Return(nil, nil) - - // Environment - envLoader.EXPECT().LoadEnv(mock.Anything, "test.example.com").Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "myapp:v2").Return([]string{}, nil) - - // Create container — canonical name (no -new suffix since no existing container) - newContainer := &domain.Container{ID: "first-container", Name: "gordon-test.example.com", Status: "created"} - runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { - return cfg.Name == "gordon-test.example.com" - })).Return(newContainer, nil) - runtime.EXPECT().StartContainer(mock.Anything, "first-container").Return(nil) - - // Readiness: pollContainerRunning (1 call) + waitForReadyByDelay verification (1 call) - // No stabilization check for first deploy (hasExisting is false) - runtime.EXPECT().IsContainerRunning(mock.Anything, "first-container").Return(true, nil).Times(2) - - // Inspect after readiness - runtime.EXPECT().InspectContainer(mock.Anything, "first-container").Return(&domain.Container{ - ID: "first-container", - Name: "gordon-test.example.com", - Status: "running", - Ports: []int{8080}, - }, nil) - - // Publish event - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - // NOTE: No StopContainer, RemoveContainer, or RenameContainer expectations — - // first deploy has no previous container to finalize, and no stabilization check. - - result, err := svc.Deploy(ctx, route) - - assert.NoError(t, err) - assert.NotNil(t, result) - assert.Equal(t, "first-container", result.ID) - assert.Equal(t, "running", result.Status) - - // Verify container is tracked - tracked, exists := svc.Get(ctx, "test.example.com") - assert.True(t, exists) - assert.Equal(t, "first-container", tracked.ID) -} - -func TestService_Deploy_PropagatesImageLabelsToContainerConfig(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := Config{ - AllowedRegistries: []string{"docker.io"}, - NetworkIsolation: false, - VolumeAutoCreate: false, - ReadinessDelay: time.Millisecond, - DrainDelay: time.Millisecond, - DrainDelayConfigured: true, - } - - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - - route := domain.Route{ - Domain: "gitea.example.com", - Image: "gitea/gitea:latest", - } - - // No container in memory — runtime resolution returns nothing - runtime.EXPECT().ListContainers(mock.Anything, false).Return([]*domain.Container{}, nil) - // No orphaned containers - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil) - - // Image is found locally - runtime.EXPECT().ListImages(mock.Anything).Return([]string{"gitea/gitea:latest"}, nil) - - // Exposed ports — image exposes both SSH and HTTP - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "gitea/gitea:latest").Return([]int{22, 3000}, nil) - - // Image labels include gordon.proxy.port and gordon.health - runtime.EXPECT().GetImageLabels(mock.Anything, "gitea/gitea:latest").Return(map[string]string{ - domain.LabelProxyPort: "3000", - domain.LabelHealth: "/healthz", - "some.other.label": "ignored", - }, nil) - - // Load environment - envLoader.EXPECT().LoadEnv(mock.Anything, "gitea.example.com").Return(nil, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "gitea/gitea:latest").Return(nil, nil) - - // Create container — ASSERT that labels include propagated image labels - createdContainer := &domain.Container{ - ID: "c-gitea-123", - Name: "gordon-gitea.example.com", - Image: "gitea/gitea:latest", - Status: "created", - } - runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { - // Verify gordon.proxy.port was propagated from image labels - if cfg.Labels[domain.LabelProxyPort] != "3000" { - return false - } - // Verify gordon.health was propagated from image labels - if cfg.Labels[domain.LabelHealth] != "/healthz" { - return false - } - // Verify non-gordon labels were NOT propagated - if _, exists := cfg.Labels["some.other.label"]; exists { - return false - } - // Verify standard container labels are still present - if cfg.Labels[domain.LabelManaged] != "true" { - return false - } - return true - })).Return(createdContainer, nil) - - runtime.EXPECT().StartContainer(mock.Anything, "c-gitea-123").Return(nil) - - // Readiness: delay mode (minimal) - runtime.EXPECT().IsContainerRunning(mock.Anything, "c-gitea-123").Return(true, nil).Times(2) - - // Re-inspect - runningContainer := &domain.Container{ - ID: "c-gitea-123", - Name: "gordon-gitea.example.com", - Image: "gitea/gitea:latest", - Status: "running", - Ports: []int{22, 3000}, - } - runtime.EXPECT().InspectContainer(mock.Anything, "c-gitea-123").Return(runningContainer, nil) - - // Publish event - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil) - - result, err := svc.Deploy(ctx, route) - - require.NoError(t, err) - assert.Equal(t, "c-gitea-123", result.ID) - assert.Equal(t, "running", result.Status) -} - -func TestLoadEnvironment_PreResolvedEnv(t *testing.T) { - ctx := testContext() - - mockEnvLoader := mocks.NewMockEnvLoader(t) - mockRuntime := mocks.NewMockContainerRuntime(t) - - // EnvLoader should NOT be called when pre-resolved env is provided - mockRuntime.EXPECT().InspectImageEnv(ctx, "myapp:latest"). - Return([]string{"FROM_DOCKERFILE=base"}, nil) - - svc := NewService(mockRuntime, mockEnvLoader, mocks.NewMockEventPublisher(t), nil, Config{}, nil) - - preResolved := []string{"DB_HOST=localhost", "API_KEY=secret"} - result, err := svc.loadEnvironment(ctx, preResolved, "ignored-domain", "myapp:latest") - require.NoError(t, err) - - assert.Contains(t, result, "DB_HOST=localhost") - assert.Contains(t, result, "API_KEY=secret") - assert.Contains(t, result, "FROM_DOCKERFILE=base") -} - -func TestLoadEnvironment_NoPreResolvedEnv(t *testing.T) { - ctx := testContext() - - mockEnvLoader := mocks.NewMockEnvLoader(t) - mockRuntime := mocks.NewMockContainerRuntime(t) - - mockEnvLoader.EXPECT().LoadEnv(ctx, "myapp.example.com"). - Return([]string{"DB_HOST=prod"}, nil) - mockRuntime.EXPECT().InspectImageEnv(ctx, "myapp:latest"). - Return([]string{"FROM_DOCKERFILE=base"}, nil) - - svc := NewService(mockRuntime, mockEnvLoader, mocks.NewMockEventPublisher(t), nil, Config{}, nil) - - result, err := svc.loadEnvironment(ctx, nil, "myapp.example.com", "myapp:latest") - require.NoError(t, err) - - assert.Contains(t, result, "DB_HOST=prod") - assert.Contains(t, result, "FROM_DOCKERFILE=base") -} - -func TestService_WaitForAttachmentReady_TCPProbe(t *testing.T) { - // Start a real TCP listener - ln, err := net.Listen("tcp", "127.0.0.1:0") - require.NoError(t, err) - defer ln.Close() - go func() { - for { - conn, err := ln.Accept() - if err != nil { - return - } - conn.Close() - } - }() - addr := ln.Addr().(*net.TCPAddr) - - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{ - AttachmentReadinessTimeout: 5 * time.Second, - }, nil) - - ctx := testContext() - containerID := "attachment-1" - containerConfig := &domain.ContainerConfig{ - Ports: []int{5432}, - } - - // pollContainerRunning - runtime.EXPECT().IsContainerRunning(mock.Anything, containerID).Return(true, nil).Once() - // No healthcheck - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, containerID).Return("", false, nil).Once() - // Resolve endpoint to our real listener - runtime.EXPECT().GetContainerNetworkInfo(mock.Anything, containerID).Return(addr.IP.String(), addr.Port, nil).Once() - runtime.EXPECT().GetContainerPort(mock.Anything, containerID, addr.Port).Return(addr.Port, nil).Once() - - err = svc.waitForAttachmentReady(ctx, containerID, containerConfig) - assert.NoError(t, err) -} - -func TestService_WaitForAttachmentReady_Timeout(t *testing.T) { - // Reserve a port with no listener - ln, err := net.Listen("tcp", "127.0.0.1:0") - require.NoError(t, err) - addr := ln.Addr().(*net.TCPAddr) - ln.Close() // Close so nothing listens - +func TestService_ListNetworks(t *testing.T) { runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{ - AttachmentReadinessTimeout: 300 * time.Millisecond, - }, nil) - + svc := NewService(runtime, nil, nil, Config{NetworkPrefix: "gordon"}) ctx := testContext() - containerID := "attachment-1" - containerConfig := &domain.ContainerConfig{ - Ports: []int{5432}, - } - - runtime.EXPECT().IsContainerRunning(mock.Anything, containerID).Return(true, nil).Once() - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, containerID).Return("", false, nil).Once() - runtime.EXPECT().GetContainerNetworkInfo(mock.Anything, containerID).Return(addr.IP.String(), addr.Port, nil).Once() - runtime.EXPECT().GetContainerPort(mock.Anything, containerID, addr.Port).Return(addr.Port, nil).Once() - - err = svc.waitForAttachmentReady(ctx, containerID, containerConfig) - assert.Error(t, err) - assert.Contains(t, err.Error(), "TCP probe timeout") -} -func TestService_WaitForAttachmentReady_FallsBackToDelay(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{ - ReadinessDelay: time.Millisecond, + runtime.EXPECT().ListNetworks(mock.Anything).Return([]*domain.NetworkInfo{ + {Name: "gordon-app", Labels: map[string]string{domain.LabelManaged: "true"}}, + {Name: "bridge"}, + {Name: "gordon-shared", Labels: map[string]string{domain.LabelManaged: "true"}}, + {Name: "gordon-unmanaged"}, }, nil) - ctx := testContext() - containerID := "attachment-1" - // No ports → skip TCP probe → fall back to delay - containerConfig := &domain.ContainerConfig{} - - runtime.EXPECT().IsContainerRunning(mock.Anything, containerID).Return(true, nil).Once() - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, containerID).Return("", false, nil).Once() - // Delay fallback checks IsContainerRunning again - runtime.EXPECT().IsContainerRunning(mock.Anything, containerID).Return(true, nil).Once() + networks, err := svc.ListNetworks(ctx) - err := svc.waitForAttachmentReady(ctx, containerID, containerConfig) assert.NoError(t, err) -} - -func TestDeployAttachedService_ReusesVerifiedLegacyVolumeFromStoppedAttachment(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := testMinDelayConfig() - config.VolumeAutoCreate = true - config.VolumePrefix = "gordon" - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - - ownerDomain := "app.example.com" - serviceImage := "postgres:16" - networkName := "gordon-net" - containerName := fmt.Sprintf("gordon-%s-postgres", domain.SanitizeDomainForContainer(ownerDomain)) - volumePath := "/var/lib/postgresql" - legacyVolume := legacyVolumeName(config.VolumePrefix, containerName, volumePath) - existing := &domain.Container{ - ID: "stopped-postgres", - Name: containerName, - Status: string(domain.ContainerStatusExited), - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: ownerDomain, - }, - VolumeMounts: []domain.ContainerVolumeMount{{ - Name: legacyVolume, - Type: "volume", - Destination: volumePath, - ReadOnly: true, - }}, - } - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{existing}, nil).Once() - runtime.EXPECT().RemoveContainer(mock.Anything, existing.ID, true).Return(nil).Once() - runtime.EXPECT().ListImages(mock.Anything).Return([]string{serviceImage}, nil).Once() - runtime.EXPECT().GetImageExposedPorts(mock.Anything, serviceImage).Return([]int{}, nil).Once() - runtime.EXPECT().InspectImageVolumes(mock.Anything, serviceImage).Return([]string{volumePath}, nil).Once() - runtime.EXPECT().VolumeExists(mock.Anything, legacyVolume).Return(true, nil).Once() - envLoader.EXPECT().LoadEnv(mock.Anything, containerName).Return([]string{}, nil).Once() - runtime.EXPECT().InspectImageEnv(mock.Anything, serviceImage).Return([]string{}, nil).Once() - runtime.EXPECT().CreateContainer(mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")). - Run(func(_ context.Context, cfg *domain.ContainerConfig) { - assert.NotContains(t, cfg.Volumes, volumePath) - assert.Equal(t, map[string]string{volumePath: legacyVolume}, cfg.ReadOnlyVolumes) - }). - Return(&domain.Container{ID: "new-postgres", Name: containerName}, nil).Once() - runtime.EXPECT().StartContainer(mock.Anything, "new-postgres").Return(nil).Once() - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-postgres").Return(true, nil).Times(2) - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "new-postgres").Return("", false, nil).Once() - - require.NoError(t, svc.deployAttachedService(ctx, ownerDomain, serviceImage, networkName)) -} - -func TestDeployAttachedService_KeepsStoppedAttachmentWhenVolumeValidationFails(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := testMinDelayConfig() - config.VolumeAutoCreate = true - config.VolumePrefix = "gordon" - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - - ownerDomain := "app.example.com" - serviceImage := "postgres:16" - containerName := fmt.Sprintf("gordon-%s-postgres", domain.SanitizeDomainForContainer(ownerDomain)) - existing := &domain.Container{ - ID: "stopped-postgres", - Name: containerName, - Status: string(domain.ContainerStatusExited), - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: ownerDomain, - }, - VolumeMounts: []domain.ContainerVolumeMount{{ - Name: "missing-data", - Type: "volume", - Destination: "/var/lib/postgresql", - }}, - } - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{existing}, nil).Once() - runtime.EXPECT().ListImages(mock.Anything).Return([]string{serviceImage}, nil).Once() - runtime.EXPECT().GetImageExposedPorts(mock.Anything, serviceImage).Return([]int{}, nil).Once() - runtime.EXPECT().VolumeExists(mock.Anything, "missing-data").Return(false, nil).Once() - - err := svc.deployAttachedService(testContext(), ownerDomain, serviceImage, "gordon-net") - - require.ErrorContains(t, err, "previously mounted volume") - require.ErrorIs(t, err, domain.ErrVolumeNotFound) -} - -func TestDeployAttachedService_KeepsRunningAttachmentWhenVolumeValidationFails(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := testMinDelayConfig() - config.VolumeAutoCreate = true - config.VolumePrefix = "gordon" - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - - ownerDomain := "app.example.com" - serviceImage := "postgres:16" - containerName := fmt.Sprintf("gordon-%s-postgres", domain.SanitizeDomainForContainer(ownerDomain)) - existing := &domain.Container{ - ID: "running-postgres", - Name: containerName, - Status: string(domain.ContainerStatusRunning), - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: ownerDomain, - domain.LabelImage: "postgres:15", - domain.LabelEnvHash: hashEnvironment(nil), - }, - VolumeMounts: []domain.ContainerVolumeMount{{ - Name: "missing-data", - Type: "volume", - Destination: "/var/lib/postgresql", - }}, - } - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{existing}, nil).Once() - envLoader.EXPECT().LoadEnv(mock.Anything, containerName).Return([]string{}, nil).Once() - runtime.EXPECT().InspectImageEnv(mock.Anything, serviceImage).Return([]string{}, nil).Once() - runtime.EXPECT().ListImages(mock.Anything).Return([]string{serviceImage}, nil).Once() - runtime.EXPECT().GetImageExposedPorts(mock.Anything, serviceImage).Return([]int{}, nil).Once() - runtime.EXPECT().VolumeExists(mock.Anything, "missing-data").Return(false, nil).Once() - - err := svc.deployAttachedService(testContext(), ownerDomain, serviceImage, "gordon-net") - - require.ErrorContains(t, err, "previously mounted volume") - require.ErrorIs(t, err, domain.ErrVolumeNotFound) -} - -func TestDeployAttachedService_ReusesVolumeFromLegacyNamedAttachment(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - config := testMinDelayConfig() - config.VolumeAutoCreate = true - config.VolumePrefix = "gordon" - svc := NewService(runtime, envLoader, eventBus, nil, config, nil) - ctx := testContext() - - ownerDomain := "app.example.com" - serviceImage := "postgres:16" - serviceName := "postgres" - networkName := "gordon-net" - containerName := fmt.Sprintf("gordon-%s-%s", domain.SanitizeDomainForContainer(ownerDomain), serviceName) - legacyContainerName := fmt.Sprintf("gordon-%s-%s", domain.SanitizeDomainForContainerLegacy(ownerDomain), serviceName) - volumePath := "/var/lib/postgresql" - legacyVolume := legacyVolumeName(config.VolumePrefix, legacyContainerName, volumePath) - existing := &domain.Container{ - ID: "legacy-postgres", - Name: legacyContainerName, - Status: string(domain.ContainerStatusRunning), - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: ownerDomain, - }, - VolumeMounts: []domain.ContainerVolumeMount{{ - Name: legacyVolume, - Type: "volume", - Destination: volumePath, - }}, - } - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{existing}, nil).Times(2) - runtime.EXPECT().StopContainer(mock.Anything, existing.ID).Return(nil).Once() - runtime.EXPECT().RemoveContainer(mock.Anything, existing.ID, true).Return(nil).Once() - runtime.EXPECT().ListImages(mock.Anything).Return([]string{serviceImage}, nil).Once() - runtime.EXPECT().GetImageExposedPorts(mock.Anything, serviceImage).Return([]int{}, nil).Once() - runtime.EXPECT().InspectImageVolumes(mock.Anything, serviceImage).Return([]string{volumePath}, nil).Once() - runtime.EXPECT().VolumeExists(mock.Anything, legacyVolume).Return(true, nil).Once() - envLoader.EXPECT().LoadEnv(mock.Anything, containerName).Return([]string{}, nil).Once() - runtime.EXPECT().InspectImageEnv(mock.Anything, serviceImage).Return([]string{}, nil).Once() - runtime.EXPECT().CreateContainer(mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")). - Run(func(_ context.Context, cfg *domain.ContainerConfig) { - assert.Equal(t, map[string]string{volumePath: legacyVolume}, cfg.Volumes) - }). - Return(&domain.Container{ID: "new-postgres", Name: containerName}, nil).Once() - runtime.EXPECT().StartContainer(mock.Anything, "new-postgres").Return(nil).Once() - runtime.EXPECT().IsContainerRunning(mock.Anything, "new-postgres").Return(true, nil).Times(2) - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "new-postgres").Return("", false, nil).Once() - - require.NoError(t, svc.deployAttachedService(ctx, ownerDomain, serviceImage, networkName)) -} - -func TestDeployAttachedService_SetsAliasOnContainerConfig(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - envLoader := mocks.NewMockEnvLoader(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(runtime, envLoader, eventBus, nil, testMinDelayConfig(), nil) - ctx := testContext() - - ownerDomain := "app.example.com" - serviceImage := "postgres:16" - networkName := "gordon-net" - containerName := fmt.Sprintf("gordon-%s-postgres", domain.SanitizeDomainForContainer(ownerDomain)) - - // resolveExistingAttachment: findContainerByName for new name, then legacy name — both miss - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil).Times(2) - - // ensureImage: pullRefForDeploy returns imageRef unchanged (no registry configured); - // ensureLocalImage checks ListImages and finds it locally. - runtime.EXPECT().ListImages(mock.Anything).Return([]string{"postgres:16"}, nil) - - // GetImageExposedPorts - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "postgres:16").Return([]int{5432}, nil) - - // loadEnvironment - envLoader.EXPECT().LoadEnv(mock.Anything, containerName).Return([]string{}, nil) - runtime.EXPECT().InspectImageEnv(mock.Anything, "postgres:16").Return([]string{}, nil) - - // setupVolumes: VolumeAutoCreate is false (default), so no volume calls needed - - // CreateContainer — capture the config and verify Aliases - runtime.EXPECT().CreateContainer(mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")). - Run(func(_ context.Context, cfg *domain.ContainerConfig) { - assert.Equal(t, []string{"postgres"}, cfg.Aliases, "attachment container must have network aliases for DNS resolution") - assert.Equal(t, "postgres", cfg.Hostname) - assert.Equal(t, networkName, cfg.NetworkMode) - assert.Equal(t, domain.RestartPolicyAlways, cfg.RestartPolicy) - }). - Return(&domain.Container{ID: "pg-container-1", Name: containerName}, nil) - - // StartContainer - runtime.EXPECT().StartContainer(mock.Anything, "pg-container-1").Return(nil) - - // waitForAttachmentReady: pollContainerRunning - runtime.EXPECT().IsContainerRunning(mock.Anything, "pg-container-1").Return(true, nil).Times(2) - // No Docker healthcheck - runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "pg-container-1").Return("", false, nil) - // TCP probe: resolve endpoint fails -> falls back to delay - runtime.EXPECT().GetContainerNetworkInfo(mock.Anything, "pg-container-1").Return("", 0, errors.New("no network")) - - err := svc.deployAttachedService(ctx, ownerDomain, serviceImage, networkName) - require.NoError(t, err) - - // Verify the attachment is tracked - svc.mu.RLock() - attachIDs := svc.attachments[ownerDomain] - svc.mu.RUnlock() - require.Len(t, attachIDs, 1) - assert.Equal(t, "pg-container-1", attachIDs[0]) -} - -func TestService_BuildValidatedImageRef_RegistryPolicy(t *testing.T) { - svc := NewService(nil, nil, nil, nil, Config{ - RegistryAuthEnabled: true, - RegistryDomain: "registry.example.com", - AllowedRegistries: []string{"docker.io"}, - }, nil) - - ref, err := svc.buildValidatedImageRef(context.Background(), "myapp:latest") - require.NoError(t, err) - assert.Equal(t, "registry.example.com/myapp:latest", ref) - - ref, err = svc.buildValidatedImageRef(context.Background(), "registry.example.com/myapp:latest") - require.NoError(t, err) - assert.Equal(t, "registry.example.com/myapp:latest", ref) - - ref, err = svc.buildValidatedImageRef(context.Background(), "docker.io/library/nginx:latest") - require.NoError(t, err) - assert.Equal(t, "docker.io/library/nginx:latest", ref) - - for _, image := range []string{ - "ghcr.io/acme/app:latest", - "localhost:5000/app:latest", - "127.0.0.1:5000/app:latest", - "10.0.0.2:5000/app:latest", - "169.254.169.254/app:latest", - } { - t.Run(image, func(t *testing.T) { - _, err := svc.buildValidatedImageRef(context.Background(), image) - require.Error(t, err) - }) - } -} - -func TestService_BuildValidatedImageRef_RejectsExternalRegistryWhenAllowlistEmpty(t *testing.T) { - svc := NewService(nil, nil, nil, nil, Config{}, nil) - - _, err := svc.buildValidatedImageRef(context.Background(), "ghcr.io/acme/app:latest") - require.Error(t, err) - assert.Contains(t, err.Error(), "images.allowed_registries") - - err = svc.validateImagePullRef(context.Background(), "nginx:latest") - require.Error(t, err) - assert.Contains(t, err.Error(), "docker.io") -} - -func TestService_BuildValidatedImageRef_RequireDigestOnlyForExternalAllowedRegistries(t *testing.T) { - svc := NewService(nil, nil, nil, nil, Config{ - RegistryAuthEnabled: true, - RegistryDomain: "registry.example.com", - RequireImageDigest: true, - AllowedRegistries: []string{"docker.io"}, - }, nil) - - _, err := svc.buildValidatedImageRef(context.Background(), "docker.io/library/nginx:latest") - require.Error(t, err) - - err = svc.validateImagePullRef(context.Background(), "nginx:latest") - require.Error(t, err) - - _, err = svc.buildValidatedImageRef(context.Background(), "docker.io/library/nginx@sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa") - require.NoError(t, err) - - _, err = svc.buildValidatedImageRef(context.Background(), "docker.io/library/nginx@sha256:not-a-digest") - require.Error(t, err) - - ref, err := svc.buildValidatedImageRef(context.Background(), "myapp:latest") - require.NoError(t, err) - assert.Equal(t, "registry.example.com/myapp:latest", ref) - - _, err = svc.buildValidatedImageRef(context.Background(), "registry.example.com/myapp:latest") - require.NoError(t, err) -} - -func TestService_BuildValidatedImageRef_AllowedRegistryMatchesPort(t *testing.T) { - svc := NewService(nil, nil, nil, nil, Config{AllowedRegistries: []string{"docker.io:5000"}}, nil) - - _, err := svc.buildValidatedImageRef(context.Background(), "docker.io:5000/app:latest") - require.NoError(t, err) - - _, err = svc.buildValidatedImageRef(context.Background(), "docker.io:5001/app:latest") - require.Error(t, err) -} - -func TestService_CreateNetworkIfNeeded_UsesInternalOptionWhenConfigured(t *testing.T) { - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, nil, nil, Config{NetworkInternal: true}, nil) - - runtime.EXPECT().NetworkExists(mock.Anything, "gordon-app").Return(false, nil) - runtime.EXPECT().CreateNetwork(mock.Anything, "gordon-app", domain.NetworkConfig{ - Driver: "bridge", - Internal: true, - Labels: map[string]string{domain.LabelManaged: "true"}, - }).Return(nil) - - require.NoError(t, svc.createNetworkIfNeeded(testContext(), "gordon-app")) -} - -func TestService_BuildContainerConfig_StrictSecurityProfile(t *testing.T) { - svc := NewService(nil, nil, nil, nil, Config{SecurityProfile: "strict"}, nil) - - cfg := svc.buildContainerConfig(containerConfigInput{Domain: "app.example.com", Image: "app:latest", ImageRef: "app:latest"}) - - assert.True(t, cfg.ReadOnlyRootFS) - assert.Equal(t, []string{"ALL"}, cfg.CapDrop) - assert.Equal(t, []string{"NET_BIND_SERVICE"}, cfg.CapAdd) -} - -func TestService_BuildContainerConfig_CompatSecurityProfileDefault(t *testing.T) { - svc := NewService(nil, nil, nil, nil, Config{}, nil) - - cfg := svc.buildContainerConfig(containerConfigInput{Domain: "app.example.com", Image: "app:latest", ImageRef: "app:latest"}) - - assert.False(t, cfg.ReadOnlyRootFS) - assert.Nil(t, cfg.CapDrop) - assert.Nil(t, cfg.CapAdd) -} - -func TestService_BuildContainerConfig_SetsRestartPolicyAlways(t *testing.T) { - svc := NewService(nil, nil, nil, nil, Config{}, nil) - - cfg := svc.buildContainerConfig(containerConfigInput{Domain: "app.example.com", Image: "app:latest", ImageRef: "app:latest"}) - - assert.Equal(t, domain.RestartPolicyAlways, cfg.RestartPolicy) -} - -func TestService_ReconcileRemovedRoute_RemovesOnlyMainRouteContainers(t *testing.T) { - ctx := testContext() - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, mocks.NewMockEventPublisher(t), nil, Config{}, nil) - - mainContainer := &domain.Container{ - ID: "route-container", - Name: "gordon-app.example.com", - Status: string(domain.ContainerStatusRunning), - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelRoute: "app.example.com", - domain.LabelDomain: "app.example.com", - }, - } - attachment := &domain.Container{ - ID: "attachment-container", - Name: "gordon-app-example-com-postgres", - Status: string(domain.ContainerStatusRunning), - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: "app.example.com", - }, - } - - svc.mu.Lock() - svc.containers["app.example.com"] = mainContainer - svc.attachments["app.example.com"] = []string{attachment.ID} - svc.managedCount = 1 - svc.mu.Unlock() - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{mainContainer, attachment}, nil).Once() - runtime.EXPECT().StopContainer(mock.Anything, mainContainer.ID).Return(nil).Once() - runtime.EXPECT().RemoveContainer(mock.Anything, mainContainer.ID, true).Return(nil).Once() - - report, err := svc.ReconcileRemovedRoute(ctx, "app.example.com") - require.NoError(t, err) - require.NotNil(t, report) - - assert.Equal(t, "app.example.com", report.Domain) - require.Len(t, report.RemovedContainers, 1) - assert.Equal(t, mainContainer.ID, report.RemovedContainers[0].ID) - require.Len(t, report.PreservedAttachments, 1) - assert.Equal(t, attachment.ID, report.PreservedAttachments[0].ContainerID) - assert.Contains(t, report.Hints, "attachments preserved; remove or purge attachments explicitly if they are no longer needed") - - _, exists := svc.Get(ctx, "app.example.com") - assert.False(t, exists, "removed route must be cleared from in-memory tracking so monitor cannot restart it") - - svc.mu.RLock() - _, attachmentsTracked := svc.attachments["app.example.com"] - svc.mu.RUnlock() - assert.True(t, attachmentsTracked, "route cleanup preserves attachment tracking for follow-up cleanup") -} - -func TestService_ReconcileRemovedRoute_IsIdempotentWhenNoContainerExists(t *testing.T) { - ctx := testContext() - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, mocks.NewMockEventPublisher(t), nil, Config{}, nil) - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{}, nil).Once() - - report, err := svc.ReconcileRemovedRoute(ctx, "missing.example.com") - require.NoError(t, err) - require.NotNil(t, report) - - assert.Equal(t, "missing.example.com", report.Domain) - assert.Empty(t, report.RemovedContainers) - assert.Contains(t, report.Warnings, "no active route container found for removed route") -} - -func TestService_PreviewRemovedRouteCleanup_ReportsRuntimeStateWithoutMutation(t *testing.T) { - ctx := testContext() - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, mocks.NewMockEventPublisher(t), nil, Config{}, nil) - - mainContainer := &domain.Container{ - ID: "route-container", - Name: "gordon-app.example.com", - Status: string(domain.ContainerStatusRunning), - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelRoute: "app.example.com", - domain.LabelDomain: "app.example.com", - }, - } - attachment := &domain.Container{ - ID: "attachment-container", - Name: "gordon-app-example-com-postgres", - Status: string(domain.ContainerStatusRunning), - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: "app.example.com", - }, - } - - svc.mu.Lock() - svc.containers["app.example.com"] = mainContainer - svc.attachments["app.example.com"] = []string{attachment.ID} - svc.managedCount = 1 - svc.mu.Unlock() - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{mainContainer, attachment}, nil).Once() - - report, err := svc.PreviewRemovedRouteCleanup(ctx, "app.example.com") - require.NoError(t, err) - require.NotNil(t, report) - - require.Len(t, report.OrphanedEntities, 1) - assert.Equal(t, "route_container", report.OrphanedEntities[0].Kind) - assert.Equal(t, mainContainer.ID, report.OrphanedEntities[0].ID) - require.Len(t, report.PreservedAttachments, 1) - assert.Equal(t, attachment.ID, report.PreservedAttachments[0].ContainerID) - assert.Contains(t, report.Hints, "attachments preserved; remove or purge attachments explicitly if they are no longer needed") - - _, exists := svc.Get(ctx, "app.example.com") - assert.True(t, exists, "preview should not clear route tracking") -} - -func TestService_ReconcileRemovedRoute_CanonicalizesDomainBeforeCleanup(t *testing.T) { - ctx := testContext() - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, mocks.NewMockEventPublisher(t), nil, Config{}, nil) - - mainContainer := &domain.Container{ - ID: "route-container", - Name: "gordon-app.example.com", - Status: string(domain.ContainerStatusRunning), - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelRoute: "app.example.com", - domain.LabelDomain: "app.example.com", - }, - } - svc.mu.Lock() - svc.containers["app.example.com"] = mainContainer - svc.managedCount = 1 - svc.mu.Unlock() - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{mainContainer}, nil).Once() - runtime.EXPECT().StopContainer(mock.Anything, mainContainer.ID).Return(nil).Once() - runtime.EXPECT().RemoveContainer(mock.Anything, mainContainer.ID, true).Return(nil).Once() - - report, err := svc.ReconcileRemovedRoute(ctx, "App.EXAMPLE.com") - require.NoError(t, err) - require.NotNil(t, report) - assert.Equal(t, "app.example.com", report.Domain) - require.Len(t, report.RemovedContainers, 1) - - _, exists := svc.Get(ctx, "app.example.com") - assert.False(t, exists) -} - -func TestService_ReconcileRemovedRoute_StopsLogCollectionForRemovedContainer(t *testing.T) { - ctx := testContext() - runtime := mocks.NewMockContainerRuntime(t) - logWriter := mocks.NewMockContainerLogWriter(t) - svc := NewService(runtime, nil, mocks.NewMockEventPublisher(t), logWriter, Config{}, nil) - - mainContainer := &domain.Container{ - ID: "route-container", - Name: "gordon-app.example.com", - Status: string(domain.ContainerStatusRunning), - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelRoute: "app.example.com", - }, - } - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{mainContainer}, nil).Once() - logWriter.EXPECT().StopLogging(mainContainer.ID).Return(nil).Once() - runtime.EXPECT().StopContainer(mock.Anything, mainContainer.ID).Return(nil).Once() - runtime.EXPECT().RemoveContainer(mock.Anything, mainContainer.ID, true).Return(nil).Once() - - _, err := svc.ReconcileRemovedRoute(ctx, "app.example.com") - require.NoError(t, err) -} - -func TestService_ReconcileRemovedRoute_InvalidatesProxyCacheAndMetric(t *testing.T) { - ctx := testContext() - runtime := mocks.NewMockContainerRuntime(t) - eventBus := mocks.NewMockEventPublisher(t) - cacheInvalidator := mocks.NewMockProxyCacheInvalidator(t) - svc := NewService(runtime, nil, eventBus, nil, Config{}, nil) - - metrics, reader := setupMetricsTest(t) - svc.SetMetrics(metrics) - - mainContainer := &domain.Container{ - ID: "route-container", - Name: "gordon-app.example.com", - Status: string(domain.ContainerStatusRunning), - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelRoute: "app.example.com", - }, - } - eventBus.EXPECT().Publish(domain.EventContainerDeployed, mock.AnythingOfType("*domain.ContainerEventPayload")).Return(nil).Once() - svc.activateDeployedContainer(ctx, "app.example.com", mainContainer) - svc.SetProxyCacheInvalidator(cacheInvalidator) - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{mainContainer}, nil).Once() - cacheInvalidator.EXPECT().InvalidateTarget(mock.Anything, "app.example.com").Return().Once() - runtime.EXPECT().StopContainer(mock.Anything, mainContainer.ID).Return(nil).Once() - runtime.EXPECT().RemoveContainer(mock.Anything, mainContainer.ID, true).Return(nil).Once() - - _, err := svc.ReconcileRemovedRoute(ctx, "app.example.com") - require.NoError(t, err) - - value, _, _ := managedMetricState(t, reader) - assert.Equal(t, int64(0), value) -} - -func TestService_ListOrphanedAttachments_DetectsRunningAttachmentNoLongerConfigured(t *testing.T) { - ctx := testContext() - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, mocks.NewMockEventPublisher(t), nil, Config{Attachments: map[string][]string{ - "app.example.com": {"redis:7"}, - }}, nil) - - orphan := &domain.Container{ - ID: "postgres-1", - Name: "gordon-app-example-com-postgres", - Image: "postgres:16", - Status: string(domain.ContainerStatusRunning), - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: "app.example.com", - domain.LabelImage: "postgres:16", - }, - } - configured := &domain.Container{ - ID: "redis-1", - Name: "gordon-app-example-com-redis", - Image: "redis:7", - Status: string(domain.ContainerStatusRunning), - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: "app.example.com", - domain.LabelImage: "redis:7", - }, - } - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{orphan, configured}, nil).Once() - - orphans, err := svc.ListOrphanedAttachments(ctx) - require.NoError(t, err) - require.Len(t, orphans, 1) - assert.Equal(t, orphan.ID, orphans[0].ContainerID) - assert.Equal(t, "postgres:16", orphans[0].Image) - assert.Contains(t, orphans[0].Reason, "no longer configured") -} - -func TestService_CleanupOrphanedAttachments_DryRunPreservesRunningContainer(t *testing.T) { - ctx := testContext() - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, mocks.NewMockEventPublisher(t), nil, Config{}, nil) - orphan := &domain.Container{ - ID: "postgres-1", - Name: "gordon-app-example-com-postgres", - Image: "postgres:16", - Status: string(domain.ContainerStatusRunning), - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: "app.example.com", - domain.LabelImage: "postgres:16", - }, - } - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{orphan}, nil).Once() - - report, err := svc.CleanupOrphanedAttachments(ctx, "", false) - require.NoError(t, err) - require.NotNil(t, report) - require.Len(t, report.PreservedAttachments, 1) - assert.Equal(t, orphan.ID, report.PreservedAttachments[0].ContainerID) - runtime.AssertNotCalled(t, "StopContainer", mock.Anything, orphan.ID) - runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, orphan.ID, mock.Anything) -} - -func TestService_CleanupOrphanedAttachments_StopRemovesContainerButPreservesData(t *testing.T) { - ctx := testContext() - runtime := mocks.NewMockContainerRuntime(t) - logWriter := mocks.NewMockContainerLogWriter(t) - svc := NewService(runtime, nil, mocks.NewMockEventPublisher(t), logWriter, Config{}, nil) - running := &domain.Container{ - ID: "postgres-1", - Name: "gordon-app-example-com-postgres", - Image: "postgres:16", - Status: string(domain.ContainerStatusRunning), - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: "app.example.com", - domain.LabelImage: "postgres:16", - }, - } - stopped := &domain.Container{ - ID: "redis-1", - Name: "gordon-app-example-com-redis", - Image: "redis:7", - Status: string(domain.ContainerStatusExited), - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: "app.example.com", - domain.LabelImage: "redis:7", - }, - } - svc.mu.Lock() - svc.attachments["app.example.com"] = []string{running.ID, stopped.ID} - svc.mu.Unlock() - - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{running, stopped}, nil).Once() - logWriter.EXPECT().StopLogging(running.ID).Return(nil).Once() - runtime.EXPECT().StopContainer(mock.Anything, running.ID).Return(nil).Once() - runtime.EXPECT().RemoveContainer(mock.Anything, running.ID, true).Return(nil).Once() - logWriter.EXPECT().StopLogging(stopped.ID).Return(nil).Once() - runtime.EXPECT().RemoveContainer(mock.Anything, stopped.ID, true).Return(nil).Once() - - report, err := svc.CleanupOrphanedAttachments(ctx, "", true) - require.NoError(t, err) - require.NotNil(t, report) - require.Len(t, report.RemovedContainers, 2) - assert.Equal(t, running.ID, report.RemovedContainers[0].ID) - assert.Equal(t, stopped.ID, report.RemovedContainers[1].ID) - assert.Contains(t, report.Hints, "attachment volumes and data were preserved") - - svc.mu.RLock() - ids := svc.attachments["app.example.com"] - svc.mu.RUnlock() - assert.Empty(t, ids) -} - -func TestService_ListOrphanedAttachments_IncludesStoppedAttachments(t *testing.T) { - ctx := testContext() - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, mocks.NewMockEventPublisher(t), nil, Config{}, nil) - stopped := &domain.Container{ - ID: "postgres-1", - Name: "gordon-app-example-com-postgres", - Image: "postgres:16", - Status: string(domain.ContainerStatusExited), - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: "app.example.com", - domain.LabelImage: "postgres:16", - }, - } - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{stopped}, nil).Once() - - orphans, err := svc.ListOrphanedAttachments(ctx) - require.NoError(t, err) - require.Len(t, orphans, 1) - assert.Equal(t, stopped.ID, orphans[0].ContainerID) - assert.Equal(t, string(domain.ContainerStatusExited), orphans[0].Status) -} - -func TestService_ListOrphanedAttachments_IncludesOwnerAndHonorsGroupConfig(t *testing.T) { - ctx := testContext() - runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(runtime, nil, mocks.NewMockEventPublisher(t), nil, Config{ - NetworkIsolation: true, - NetworkGroups: map[string][]string{ - "backend": {"app.example.com"}, - }, - Attachments: map[string][]string{ - "backend": {"postgres:16"}, - }, - }, nil) - configuredByGroup := &domain.Container{ - ID: "postgres-1", - Name: "gordon-app-example-com-postgres", - Image: "postgres:16", - Status: string(domain.ContainerStatusRunning), - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: "app.example.com", - domain.LabelImage: "postgres:16", - }, - } - orphan := &domain.Container{ - ID: "redis-1", - Name: "gordon-app-example-com-redis", - Image: "redis:7", - Status: string(domain.ContainerStatusRunning), - Labels: map[string]string{ - domain.LabelManaged: "true", - domain.LabelAttachment: "true", - domain.LabelAttachedTo: "app.example.com", - domain.LabelImage: "redis:7", - }, - } - runtime.EXPECT().ListContainers(mock.Anything, true).Return([]*domain.Container{configuredByGroup, orphan}, nil).Once() - - orphans, err := svc.ListOrphanedAttachments(ctx) - require.NoError(t, err) - require.Len(t, orphans, 1) - assert.Equal(t, "redis-1", orphans[0].ContainerID) - assert.Equal(t, "app.example.com", orphans[0].Owner) + assert.Len(t, networks, 2) + assert.Equal(t, "gordon-app", networks[0].Name) + assert.Equal(t, "gordon-shared", networks[1].Name) + assert.NotContains(t, []string{networks[0].Name, networks[1].Name}, "gordon-unmanaged") } diff --git a/internal/usecase/deployment/backend_binds.go b/internal/usecase/deployment/backend_binds.go new file mode 100644 index 000000000..33a266eea --- /dev/null +++ b/internal/usecase/deployment/backend_binds.go @@ -0,0 +1,156 @@ +package deployment + +import ( + "context" + "fmt" + "sort" + + "github.com/bnema/gordon/internal/domain" +) + +// backendPorts collects every interface container port for loopback +// publication: public HTTP + TCP interfaces on tcp plus UDP interfaces on +// udp, plus an explicit TCP readiness port. Internal HTTP ports are +// excluded: they are reached only over the private network and must keep +// no host binding. Each publish is on 127.0.0.1 ephemeral (never public). +// Deduplicated by (protocol, container port). +func backendPorts(spec domain.AppService) []domain.ContainerBackendPort { + seen := map[domain.ContainerBackendPort]struct{}{} + var ports []domain.ContainerBackendPort + add := func(port int, protocol domain.NetworkProtocol) { + if port <= 0 { + return + } + key := domain.ContainerBackendPort{ContainerPort: port, Protocol: protocol} + if _, ok := seen[key]; !ok { + seen[key] = struct{}{} + ports = append(ports, key) + } + } + for _, h := range spec.HTTP { + if !h.IsPublic() { + continue + } + add(h.Port, domain.NetworkProtocolTCP) + } + for _, t := range spec.TCP { + add(t.Port, domain.NetworkProtocolTCP) + } + for _, u := range spec.UDP { + add(u.Port, domain.NetworkProtocolUDP) + } + // Readiness probes are TCP-only (validation rejects UDP-only + // services with tcp/http readiness); the readiness port matches a + // declared TCP container port. An internal-only port is never + // published, so readiness metadata cannot create a host binding for + // it. + if spec.Readiness.Port > 0 && !spec.InternallyOnlyPort(spec.Readiness.Port) { + add(spec.Readiness.Port, domain.NetworkProtocolTCP) + } + sort.Slice(ports, func(i, j int) bool { + if ports[i].Protocol != ports[j].Protocol { + return ports[i].Protocol < ports[j].Protocol + } + return ports[i].ContainerPort < ports[j].ContainerPort + }) + return ports +} + +// backendPublishes maps backend ports to 127.0.0.1 ephemeral publishes. +func backendPublishes(spec domain.AppService) []domain.ContainerPortPublish { + ports := backendPorts(spec) + publishes := make([]domain.ContainerPortPublish, 0, len(ports)) + for _, port := range ports { + publishes = append(publishes, domain.ContainerPortPublish{ + HostIP: "127.0.0.1", + HostPort: 0, + ContainerPort: port.ContainerPort, + Protocol: port.Protocol, + }) + } + return publishes +} + +// readBackendBinds resolves each published container port to its +// 127.0.0.1 host port and registers the Gordon-generated claims +// (owner gordon-backend) in the global checkpoint. Missing binds fail: +// an unbound backend can serve neither readiness nor proxy traffic. +// A registration conflict removes the CANDIDATE container and fails — +// callers must only pass freshly created candidates here, never an +// existing ACTIVE container (see inspectBackendBinds). +func (s *Service) readBackendBinds(ctx context.Context, app, service, containerID string, ports []domain.ContainerBackendPort) (map[int]int, map[int]int, error) { + if len(ports) == 0 { + return map[int]int{}, map[int]int{}, nil + } + observed, err := s.deps.Runtime.GetContainerBackendBinds(ctx, containerID, ports) + if err != nil { + return nil, nil, fmt.Errorf("deployment: backend binds: %w", err) + } + binds, udpBinds := splitBackendBinds(observed) + claims := backendClaims(app, service, containerID, observed) + if len(claims) == 0 { + return binds, udpBinds, nil + } + if err := s.deps.State.RegisterBackendBinds(ctx, claims); err != nil { + s.retireCandidate(ctx, app, service, containerID) + return nil, nil, err + } + return binds, udpBinds, nil +} + +// inspectBackendBinds re-inspects the loopback publishes of an EXISTING +// container and registers the Gordon-generated claims. Unlike +// readBackendBinds it never stops or removes the container: a +// registration conflict fails without touching the ACTIVE workload. +// Missing binds fail: an unbound backend must not stay routable. +func (s *Service) inspectBackendBinds(ctx context.Context, app, service, containerID string, ports []domain.ContainerBackendPort) (map[int]int, map[int]int, error) { + if len(ports) == 0 { + return map[int]int{}, map[int]int{}, nil + } + observed, err := s.deps.Runtime.GetContainerBackendBinds(ctx, containerID, ports) + if err != nil { + return nil, nil, fmt.Errorf("deployment: backend binds: %w", err) + } + binds, udpBinds := splitBackendBinds(observed) + claims := backendClaims(app, service, containerID, observed) + if len(claims) == 0 { + return binds, udpBinds, nil + } + if err := s.deps.State.RegisterBackendBinds(ctx, claims); err != nil { + return nil, nil, err + } + return binds, udpBinds, nil +} + +// splitBackendBinds partitions observed binds by protocol. TCP feeds +// readiness and HTTP/TCP proxying; UDP feeds UDP relaying. +func splitBackendBinds(observed []domain.ContainerBackendBind) (map[int]int, map[int]int) { + binds := map[int]int{} + udpBinds := map[int]int{} + for _, bind := range observed { + if bind.Protocol == domain.NetworkProtocolUDP { + udpBinds[bind.ContainerPort] = bind.HostPort + } else { + binds[bind.ContainerPort] = bind.HostPort + } + } + return binds, udpBinds +} + +// backendClaims maps observed binds to Gordon-generated loopback claims +// with the exact protocol (tcp/udp) for the global checkpoint. +func backendClaims(app, service, containerID string, observed []domain.ContainerBackendBind) []domain.AppListenerReservation { + claims := make([]domain.AppListenerReservation, 0, len(observed)) + for _, bind := range observed { + claims = append(claims, domain.AppListenerReservation{ + Proto: string(bind.Protocol), + IP: "127.0.0.1", + Port: bind.HostPort, + Service: service, + App: app, + Owner: domain.OwnerGordonBackend, + ContainerID: containerID, + }) + } + return claims +} diff --git a/internal/usecase/deployment/bind_policy_internal_test.go b/internal/usecase/deployment/bind_policy_internal_test.go new file mode 100644 index 000000000..a29536edd --- /dev/null +++ b/internal/usecase/deployment/bind_policy_internal_test.go @@ -0,0 +1,329 @@ +package deployment + +import ( + "context" + "fmt" + "os" + "path/filepath" + "testing" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" +) + +func bindPolicyFor(source, root string, readOnly bool, apps, services []string) domain.AppBindPolicy { + return domain.AppBindPolicy{ + Name: "config", + Source: source, + Root: root, + ReadOnly: readOnly, + AllowedApps: apps, + AllowedServices: services, + } +} + +func TestResolveServiceBinds_ExactTranslation(t *testing.T) { + root := t.TempDir() + source := filepath.Join(root, "binds", "config") + require.NoError(t, os.MkdirAll(source, 0o755)) + realSource, err := filepath.EvalSymlinks(source) + require.NoError(t, err) + + svc := NewService(Deps{}, zerowrap.Default()). + WithBindPolicies(map[string]domain.AppBindPolicy{ + "config": bindPolicyFor(source, root, false, []string{"blog"}, []string{"web"}), + }) + + got, err := svc.resolveServiceBinds("blog", domain.AppService{ + Name: "web", + Binds: []domain.AppBind{{Name: "config", Path: "/etc/app.conf", ReadOnly: true}}, + }) + require.NoError(t, err) + require.Len(t, got, 1) + assert.Equal(t, domain.ContainerBind{ + Name: "config", + Source: realSource, + Destination: "/etc/app.conf", + ReadOnly: true, + }, got[0]) +} + +func TestResolveServiceBinds_ReadOnlyNonWeakening(t *testing.T) { + root := t.TempDir() + source := filepath.Join(root, "binds") + require.NoError(t, os.MkdirAll(source, 0o755)) + + svc := NewService(Deps{}, zerowrap.Default()). + WithBindPolicies(map[string]domain.AppBindPolicy{ + "config": bindPolicyFor(source, root, true, []string{"blog"}, []string{"web"}), + }) + + got, err := svc.resolveServiceBinds("blog", domain.AppService{ + Name: "web", + Binds: []domain.AppBind{{Name: "config", Path: "/etc/app.conf"}}, + }) + require.NoError(t, err) + require.Len(t, got, 1) + assert.True(t, got[0].ReadOnly, "a read-only policy must not be weakened by the manifest") +} + +func TestResolveServiceBinds_UnknownAndRevokedPolicy(t *testing.T) { + spec := domain.AppService{ + Name: "web", + Binds: []domain.AppBind{{Name: "config", Path: "/etc/app.conf"}}, + } + + t.Run("unknown policy", func(t *testing.T) { + svc := NewService(Deps{}, zerowrap.Default()) + _, err := svc.resolveServiceBinds("blog", spec) + require.ErrorIs(t, err, domain.ErrBindPolicy) + assert.Contains(t, err.Error(), "blog") + assert.Contains(t, err.Error(), "web") + assert.Contains(t, err.Error(), "config") + }) + + t.Run("revoked policy", func(t *testing.T) { + root := t.TempDir() + source := filepath.Join(root, "binds") + require.NoError(t, os.MkdirAll(source, 0o755)) + svc := NewService(Deps{}, zerowrap.Default()). + WithBindPolicies(map[string]domain.AppBindPolicy{ + "config": bindPolicyFor(source, root, false, []string{"blog"}, []string{"web"}), + }) + _, err := svc.resolveServiceBinds("blog", spec) + require.NoError(t, err) + + svc.SetBindPolicies(nil) + _, err = svc.resolveServiceBinds("blog", spec) + require.ErrorIs(t, err, domain.ErrBindPolicy) + }) + + t.Run("refused policy omits source", func(t *testing.T) { + const secretSource = "/srv/gordon/secret-source" + svc := NewService(Deps{}, zerowrap.Default()). + WithBindPolicies(map[string]domain.AppBindPolicy{ + "config": bindPolicyFor(secretSource, "/srv/gordon", false, []string{"other"}, []string{"web"}), + }) + _, err := svc.resolveServiceBinds("blog", spec) + require.ErrorIs(t, err, domain.ErrBindPolicy) + assert.NotContains(t, err.Error(), secretSource, "errors must never leak the host source") + }) +} + +// TestCreateAndStart_RefusesRevokedBindBeforeRuntimeMutation proves the +// immediate-pre-create re-resolution fails before any runtime call when the +// policy is missing or revoked. +func TestCreateAndStart_RefusesRevokedBindBeforeRuntimeMutation(t *testing.T) { + runtime := outmocks.NewMockContainerRuntime(t) + svc := NewService(Deps{Runtime: runtime}, zerowrap.Default()) + p := pinnedService{ + name: "web", + spec: domain.AppService{ + Name: "web", + Binds: []domain.AppBind{{Name: "config", Path: "/etc/app.conf"}}, + }, + } + + _, err := svc.createAndStart(context.Background(), "blog", "rev-1", p, "op-1", nil) + + require.ErrorIs(t, err, domain.ErrBindPolicy) + assert.Contains(t, err.Error(), "config") + runtime.AssertNotCalled(t, "CreateContainer") + runtime.AssertNotCalled(t, "CreateVolume") + runtime.AssertNotCalled(t, "StartContainer") + runtime.AssertNotCalled(t, "ConnectContainerToNetwork") + runtime.AssertExpectations(t) +} + +// TestCreateContainer_BindBearingRuntimeErrorIsRedactedNotBindPolicy proves a +// CreateContainer failure on a bind-bearing service is reported as a generic +// redacted runtime error, never as a bind policy violation, and never leaks +// the resolved host source embedded in the runtime error. +func TestCreateContainer_BindBearingRuntimeErrorIsRedactedNotBindPolicy(t *testing.T) { + const secretSource = "/srv/gordon/secret-source" + runtime := outmocks.NewMockContainerRuntime(t) + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything). + Return(nil, fmt.Errorf("mount %s: permission denied", secretSource)).Once() + svc := NewService(Deps{Runtime: runtime}, zerowrap.Default()) + + _, err := svc.createContainer(context.Background(), "web", &domain.ContainerConfig{ + Binds: []domain.ContainerBind{{Name: "config", Source: secretSource, Destination: "/etc/app.conf"}}, + }) + + require.Error(t, err) + require.NotErrorIs(t, err, domain.ErrBindPolicy, "a runtime failure is not a policy refusal") + assert.NotContains(t, err.Error(), secretSource, "errors must never leak the resolved host source") + assert.NotContains(t, err.Error(), "permission denied", "no runtime text may reach the caller") + runtime.AssertExpectations(t) +} + +func TestCheckImageVolumes_IncludesBindDestinations(t *testing.T) { + t.Run("bind destination maps image volume", func(t *testing.T) { + runtime := outmocks.NewMockContainerRuntime(t) + runtime.EXPECT().InspectImageVolumes(context.Background(), "img:1").Return([]string{"/data"}, nil).Once() + svc := NewService(Deps{Runtime: runtime}, zerowrap.Default()) + + err := svc.checkImageVolumes(context.Background(), domain.AppService{ + Name: "web", + Binds: []domain.AppBind{{Name: "config", Path: "/data"}}, + }, "img:1") + + require.NoError(t, err) + }) + + t.Run("unmapped image volume still refused", func(t *testing.T) { + runtime := outmocks.NewMockContainerRuntime(t) + runtime.EXPECT().InspectImageVolumes(context.Background(), "img:1").Return([]string{"/data"}, nil).Once() + svc := NewService(Deps{Runtime: runtime}, zerowrap.Default()) + + err := svc.checkImageVolumes(context.Background(), domain.AppService{Name: "web"}, "img:1") + + require.ErrorIs(t, err, domain.ErrAppUnmanagedImageVolume) + }) +} + +// TestSingleWriterRequiredAndImplicitHTTPReadiness proves the two +// independent service rules: singleWriterRequired marks the stateful +// services whose superseded generation must stay recovery-inhibited, and +// implicitHTTPProbe decides which undeclared readiness checks keep the +// implicit HTTP probe. +func TestSingleWriterRequiredAndImplicitHTTPReadiness(t *testing.T) { + httpSpec := func() domain.AppService { + return domain.AppService{ + Name: "web", + HTTP: []domain.AppHTTPInterface{{Host: "app.example.com", Port: 8080, TLS: domain.AppTLSAuto}}, + } + } + + t.Run("stateless public HTTP service keeps the implicit HTTP probe", func(t *testing.T) { + assert.True(t, implicitHTTPProbe(httpSpec())) + assert.Equal(t, domain.AppReadinessHTTP, deploymentReadiness(httpSpec()).Readiness.Type) + assert.False(t, singleWriterRequired(httpSpec())) + }) + + t.Run("undeclared probe is not guessed for internal, mixed, or L4 services", func(t *testing.T) { + internal := domain.AppService{ + Name: "api", + HTTP: []domain.AppHTTPInterface{{Port: 8080, Visibility: domain.AppVisibilityInternal}}, + } + assert.False(t, implicitHTTPProbe(internal), + "an internal-only interface is not reachable through a loopback publish, so no HTTP probe is guessed") + assert.Empty(t, deploymentReadiness(internal).Readiness.Type) + assert.False(t, singleWriterRequired(internal)) + + mixed := httpSpec() + mixed.HTTP = append(mixed.HTTP, domain.AppHTTPInterface{Port: 9090, Visibility: domain.AppVisibilityInternal}) + assert.False(t, implicitHTTPProbe(mixed)) + assert.Empty(t, deploymentReadiness(mixed).Readiness.Type) + assert.False(t, singleWriterRequired(mixed)) + + l4Only := domain.AppService{ + Name: "db", + TCP: []domain.AppTCPInterface{{Entrypoint: "tcp", Port: 5432, Publish: "5432"}}, + } + assert.False(t, implicitHTTPProbe(l4Only), "an HTTP GET proves nothing about an L4 workload") + assert.Empty(t, deploymentReadiness(l4Only).Readiness.Type) + }) + + t.Run("declared probes are never replaced by the implicit HTTP probe", func(t *testing.T) { + for _, declared := range []string{domain.AppReadinessHTTP, domain.AppReadinessTCP, domain.AppReadinessLog} { + spec := httpSpec() + spec.Readiness = domain.AppReadiness{Type: declared} + assert.False(t, implicitHTTPProbe(spec), "%s is an explicit declaration", declared) + assert.Equal(t, declared, deploymentReadiness(spec).Readiness.Type) + } + }) + + t.Run("an explicit none on a probeable service keeps the implicit HTTP probe", func(t *testing.T) { + spec := httpSpec() + spec.Readiness = domain.AppReadiness{Type: domain.AppReadinessNone} + assert.Equal(t, domain.AppReadinessHTTP, deploymentReadiness(spec).Readiness.Type) + }) + + t.Run("an explicit none on an L4 service stays unchecked", func(t *testing.T) { + spec := domain.AppService{ + Name: "db", + TCP: []domain.AppTCPInterface{{Entrypoint: "tcp", Port: 5432, Publish: "5432"}}, + Readiness: domain.AppReadiness{Type: domain.AppReadinessNone}, + } + assert.Equal(t, domain.AppReadinessNone, deploymentReadiness(spec).Readiness.Type) + }) + + t.Run("writable bind is single-writer", func(t *testing.T) { + spec := httpSpec() + spec.Binds = []domain.AppBind{{Name: "config", Path: "/etc/app.conf"}} + assert.False(t, implicitHTTPProbe(spec), "a stateful service must declare its own readiness") + assert.True(t, singleWriterRequired(spec)) + }) + + t.Run("read-only bind is also single-writer", func(t *testing.T) { + spec := httpSpec() + spec.Binds = []domain.AppBind{{Name: "config", Path: "/etc/app.conf", ReadOnly: true}} + assert.False(t, implicitHTTPProbe(spec)) + assert.True(t, singleWriterRequired(spec)) + }) + + t.Run("volume is single-writer", func(t *testing.T) { + spec := httpSpec() + spec.Volumes = []domain.AppVolume{{Name: "data", Path: "/data"}} + assert.False(t, implicitHTTPProbe(spec)) + assert.True(t, singleWriterRequired(spec)) + }) +} + +// TestRestartOneService_RevokedBindFailsWithoutRuntimeMutation proves an +// in-place restart fails closed when the bind policy is missing/revoked, +// before touching the runtime. +func TestRestartOneService_RevokedBindFailsWithoutRuntimeMutation(t *testing.T) { + runtime := outmocks.NewMockContainerRuntime(t) + svc := NewService(Deps{Runtime: runtime}, zerowrap.Default()) + eff := domain.AppEffectiveService{ + Container: "c1", + Spec: domain.AppService{ + Name: "web", + Binds: []domain.AppBind{{Name: "config", Path: "/etc/app.conf"}}, + }, + } + + op := &domain.AppOperation{Op: "op-1", Steps: []domain.AppOperationStep{{ + ID: "service.web.restart", State: domain.AppStepPending, Before: eff.Container, + }}} + result := svc.restartOneService(context.Background(), "blog", "op-1", "web", eff, op, 0) + + assert.Equal(t, domain.AppStepFailed, op.Steps[0].State) + assert.Contains(t, op.Steps[0].Error, "config") + assert.Equal(t, "failed", result.Result) + runtime.AssertNotCalled(t, "RestartContainer") + runtime.AssertExpectations(t) +} + +// TestEnsureServiceRunning_RevokedBindFailsWithoutRuntimeMutation proves the +// boot/start recovery path fails closed before starting a container when the +// bind policy is missing/revoked. +func TestEnsureServiceRunning_RevokedBindFailsWithoutRuntimeMutation(t *testing.T) { + runtime := outmocks.NewMockContainerRuntime(t) + svc := NewService(Deps{Runtime: runtime}, zerowrap.Default()) + eff := domain.AppEffectiveService{ + Container: "c1", + Spec: domain.AppService{ + Name: "web", + Binds: []domain.AppBind{{Name: "config", Path: "/etc/app.conf"}}, + }, + } + op := &domain.AppOperation{Op: "op-1", Steps: []domain.AppOperationStep{{ + ID: "service.web.start", State: domain.AppStepPending, Before: eff.Container, + }}} + result := &LifecycleResult{Services: map[string]ServiceResult{}} + + svc.ensureServiceRunning(context.Background(), "blog", "op-1", "web", eff, op, 0, result) + + assert.Equal(t, domain.AppStepFailed, op.Steps[0].State) + assert.Contains(t, op.Steps[0].Error, "config") + runtime.AssertNotCalled(t, "StartContainer") + runtime.AssertExpectations(t) +} diff --git a/internal/usecase/deployment/candidate.go b/internal/usecase/deployment/candidate.go new file mode 100644 index 000000000..8ee9edcd3 --- /dev/null +++ b/internal/usecase/deployment/candidate.go @@ -0,0 +1,369 @@ +package deployment + +import ( + "context" + "errors" + "fmt" + "sort" + "strings" + + "github.com/bnema/gordon/internal/domain" +) + +// startedCandidate is one freshly started replacement generation and its +// observed loopback backend binds (container port -> host port). +type startedCandidate struct { + Container *domain.Container + TCPBinds map[int]int + UDPBinds map[int]int +} + +// createAndStart builds the runtime config and starts the replacement. +// Preflight already pulled and inspected the pinned runtime image. +// Every TCP-capable interface port is published on 127.0.0.1 ephemeral +// plus every UDP interface port on 127.0.0.1/udp ephemeral: readiness +// and the proxy dial these loopback binds rootless-first, +// never container IPs. The runtime keeps native restarts; Gordon +// reconciles intent at boot and in the monitor (accepted decision: +// native runtime restarts plus daemon reconciliation). +func (s *Service) createAndStart(ctx context.Context, app, revision string, p pinnedService, opID string, journal candidateJournal) (startedCandidate, error) { + // Re-resolve binds from the current policy immediately before any + // runtime mutation: a bind revoked since preflight must fail here, + // before volume ownership or container creation. + resolvedBinds, err := s.resolveServiceBinds(app, p.spec) + if err != nil { + return startedCandidate{}, err + } + // Re-resolve devices from the current policy under the same + // fail-closed rule: revoked grants never reach the runtime. + resolvedDevices, err := s.resolveServiceDevices(app, p.spec) + if err != nil { + return startedCandidate{}, err + } + env, err := s.serviceEnv(ctx, app, p) + if err != nil { + return startedCandidate{}, err + } + image := runtimeImageRef(p) + // The incarnation network is verified or created before any + // container exists, so a workload is only ever started on a + // Gordon-owned network that no other app shares. + appID, err := s.ensureIncarnationID(ctx, app) + if err != nil { + return startedCandidate{}, err + } + nets := s.resolveAppNetworks(appID, p.sharedNetworks) + if err := s.ensureAppNetworks(ctx, app, appID, nets); err != nil { + return startedCandidate{}, err + } + volumes := map[string]string{} + readOnlyVolumes := map[string]string{} + for _, vol := range p.spec.Volumes { + runtimeName := domain.RuntimeVolumeName(app, p.spec.Name, vol.Name) + // Durable ownership is reserved BEFORE the runtime volume + // exists: a crash in between leaves a protected record, never + // an unowned volume that prune could later adopt. The + // incarnation ID is already validated above, so the labels + // carry the same ID without another ownership load. + if err := s.reserveVolumeOwnership(ctx, app, appID, p.spec.Name, vol.Name, runtimeName); err != nil { + return startedCandidate{}, err + } + if err := s.deps.Runtime.CreateVolume(ctx, runtimeName, volumeProvenanceLabels(app, appID, p.spec.Name, revision)); err != nil { + // CreateVolume is idempotent at the adapter; existence was + // checked at preflight, so only real backend errors fail here. + return startedCandidate{}, fmt.Errorf("deployment: create volume %q: %w", vol.Name, err) + } + // A declared read-only mount must reach the runtime as read-only: + // a writable mount would let the service modify protected data. + if vol.ReadOnly { + readOnlyVolumes[vol.Path] = runtimeName + continue + } + volumes[vol.Path] = runtimeName + } + config := &domain.ContainerConfig{ + Image: image, + Name: domain.LogicalServiceIdentity(app, p.spec.Name) + "--" + shortOp(opID), + Env: env, + Entrypoint: append([]string(nil), p.spec.Command...), + Volumes: volumes, + ReadOnlyVolumes: readOnlyVolumes, + Binds: resolvedBinds, + CDIDevices: resolvedDevices, + Labels: appLabels(app, p.spec.Name, revision), + AutoRemove: false, + RestartPolicy: domain.RestartPolicyAlways, + PortPublishes: backendPublishes(p.spec), + NetworkMode: nets.private, + Hostname: p.spec.Name, + Aliases: []string{p.spec.Name}, + MemoryLimit: s.deps.Limits.MemoryBytes, + NanoCPUs: s.deps.Limits.NanoCPUs, + PidsLimit: s.deps.Limits.PidsLimit, + } + created, err := s.createContainer(ctx, p.name, config) + if err != nil { + return startedCandidate{}, err + } + // The candidate ID is durably journaled before the container is started + // or joins a shared network, so an interrupted deployment can be cleaned + // up instead of leaving an untracked running generation. + if err := s.recordCandidate(ctx, app, p.spec.Name, created.ID, journal); err != nil { + return startedCandidate{}, err + } + if err := s.connectSharedNetworks(ctx, created.ID, nets); err != nil { + s.retireCandidate(ctx, app, p.spec.Name, created.ID) + return startedCandidate{Container: created}, err + } + if err := s.deps.Runtime.StartContainer(ctx, created.ID); err != nil { + s.retireCandidate(ctx, app, p.spec.Name, created.ID) + return startedCandidate{Container: created}, fmt.Errorf("deployment: start container: %w", err) + } + binds, udpBinds, err := s.readBackendBinds(ctx, app, p.spec.Name, created.ID, backendPorts(p.spec)) + if err != nil { + s.retireCandidate(ctx, app, p.spec.Name, created.ID) + return startedCandidate{Container: created}, err + } + return startedCandidate{Container: created, TCPBinds: binds, UDPBinds: udpBinds}, nil +} + +// runtimeImageRef resolves the exact runtime image reference of one pinned +// service: the image preflight resolved, else the manifest reference pinned to +// its digest. +func runtimeImageRef(p pinnedService) string { + if p.runtimeImage != "" { + return p.runtimeImage + } + if p.digest == "" { + return p.spec.Image + } + return stripImageTag(p.spec.Image) + "@" + p.digest +} + +// recordCandidate durably journals a freshly created candidate before it is +// started. A candidate whose ID cannot be recorded must not keep running: +// recovery could never find it, so it is removed and the failure surfaced. +func (s *Service) recordCandidate(ctx context.Context, app, service, containerID string, journal candidateJournal) error { + if journal == nil { + return nil + } + if err := journal(ctx, containerID); err != nil { + retired := s.retireContainer(ctx, app, retireOptions{Service: service, Force: true}, containerID) + if !retired.Gone { + return fmt.Errorf("%w (candidate cleanup: %s)", err, cleanupDetail(retired)) + } + return err + } + return nil +} + +func (s *Service) createContainer(ctx context.Context, service string, config *domain.ContainerConfig) (*domain.Container, error) { + created, err := s.deps.Runtime.CreateContainer(ctx, config) + if err == nil { + return created, nil + } + // The engine-unsupported sentinel survives redaction so callers can + // map it to the structured runtime-unsupported envelope. It carries + // no host inventory (engine family/version only), and the device + // gate runs only for device-bearing creates, so this branch also + // covers services that declare both binds and devices. + if errors.Is(err, domain.ErrRuntimeUnsupported) { + return nil, fmt.Errorf("deployment: create container for service %q with administrative devices: %w", service, domain.ErrRuntimeUnsupported) + } + if len(config.Binds) > 0 { + // A CreateContainer failure is a runtime error, not a bind policy + // violation: policy was already enforced by resolveServiceBinds + // before this call. Runtime errors may embed the resolved host + // path, so redact the whole cause instead of mislabelling it and + // keep it out of operation journals and API/CLI responses. + return nil, fmt.Errorf("deployment: create container for service %q with administrative mounts: runtime error redacted", service) + } + if len(config.CDIDevices) > 0 { + // Same redaction rule for device-bearing creates: runtime errors + // may embed resolved CDI IDs, which are host inventory. Policy + // was already enforced by resolveServiceDevices before this call. + return nil, fmt.Errorf("deployment: create container for service %q with administrative devices: runtime error redacted", service) + } + return nil, fmt.Errorf("deployment: create container: %w", err) +} + +// reserveVolumeOwnership durably records one app-owned volume as +// attached BEFORE the runtime volume is created, so a crash between the +// two leaves a protected record instead of an unowned volume. The +// caller passes the validated incarnation ID from ensureIncarnationID, +// which already assigned it, so no reload is needed for the labels. +func (s *Service) reserveVolumeOwnership(ctx context.Context, app, appID, service, volumeName, runtimeName string) error { + ownership, err := s.deps.State.LoadOwnership(ctx, app) + if err != nil { + return fmt.Errorf("deployment: load ownership: %w", err) + } + if ownership.App == "" { + ownership.App = app + } + if ownership.ID == "" { + ownership.ID = appID + } + entry := domain.AppOwnedVolume{ + Name: volumeName, + Service: service, + RuntimeName: runtimeName, + State: domain.AppResourceAttached, + } + replaced := false + for i, existing := range ownership.Volumes { + if existing.RuntimeName == runtimeName { + ownership.Volumes[i] = entry + replaced = true + break + } + } + if !replaced { + ownership.Volumes = append(ownership.Volumes, entry) + } + if err := s.deps.State.SaveOwnership(ctx, ownership); err != nil { + return fmt.Errorf("deployment: reserve volume ownership: %w", err) + } + return nil +} + +// volumeProvenanceLabels is the label set stamped on an app-created +// volume. Ownership is never inferred from the volume name, so the +// labels must carry the full provenance: app, incarnation UUID, and +// service. The managed marker is stamped here as well as merged by the +// adapter, so provenance never depends on adapter behaviour alone. +func volumeProvenanceLabels(app, appID, service, revision string) map[string]string { + labels := map[string]string{ + domain.LabelManaged: "true", + domain.LabelApp: app, + domain.LabelAppService: service, + domain.LabelAppRevision: revision, + } + if appID != "" { + labels[domain.LabelAppID] = appID + } + return labels +} + +// appLabels stamps engine ownership on every created container. +func appLabels(app, service, revision string) map[string]string { + return map[string]string{ + domain.LabelManaged: "true", + domain.LabelApp: app, + domain.LabelAppService: service, + domain.LabelAppRevision: revision, + } +} + +// shortOp keeps container names short but unique per op. +func shortOp(opID string) string { + trimmed := strings.TrimPrefix(opID, "op-") + if len(trimmed) > 12 { + trimmed = trimmed[:12] + } + if trimmed == "" { + return "op" + } + return trimmed +} + +// serviceEnv resolves the service environment: the captured revision's +// app-wide public env on every service, plus that service's own secret +// values (memory only, UUID-keyed paths). Keys are emitted sorted and +// deterministic. Overlapping public/secret keys are invalid with no +// precedence; preflight rejects them before any mutation. +func (s *Service) serviceEnv(ctx context.Context, app string, p pinnedService) ([]string, error) { + spec := p.spec + envMap := make(map[string]string, len(p.appEnv)+len(spec.Secrets)) + for key, value := range p.appEnv { + envMap[key] = value + } + keys := make([]string, 0, len(spec.Secrets)) + for envKey := range spec.Secrets { + keys = append(keys, envKey) + } + sort.Strings(keys) + if len(keys) == 0 { + return sortedEnv(envMap), nil + } + id, err := s.appSecretID(ctx, app) + if err != nil { + return nil, err + } + for _, envKey := range keys { + if _, overlap := envMap[envKey]; overlap { + return nil, fmt.Errorf("deployment: env key %q collides with a secret key in service %q: %w", envKey, spec.Name, domain.ErrInvalidAppSpec) + } + path := domain.AppSecretPathForID(id, app, spec.Name, spec.Secrets[envKey]) + value, err := s.deps.Secrets.GetSecret(ctx, path) + if err != nil { + return nil, fmt.Errorf("deployment: secret %s (env %s) missing: %w", path, envKey, domain.ErrAppSecretMissing) + } + envMap[envKey] = value + } + return sortedEnv(envMap), nil +} + +// sortedEnv renders a KEY=value list in sorted key order. Values stay in +// memory for container creation only; nothing is persisted or logged. +func sortedEnv(envMap map[string]string) []string { + if len(envMap) == 0 { + return nil + } + keys := make([]string, 0, len(envMap)) + for key := range envMap { + keys = append(keys, key) + } + sort.Strings(keys) + env := make([]string, 0, len(keys)) + for _, key := range keys { + env = append(env, key+"="+envMap[key]) + } + return env +} + +// pullImage fetches the pinned image from the installation registry. +// External refs pull anonymously; installation-registry refs use the +// internal credentials. A missing registry config disables pulling +// (tests and pre-pulled environments). +func (s *Service) pullImage(ctx context.Context, image string) (string, error) { + if err := s.validateImageSource(image, ""); err != nil { + return "", err + } + if s.deps.Registry.Domain == "" { + return image, nil + } + if !s.deps.ImagePolicy.IsInstallationImage(image) { + if err := s.deps.Runtime.PullImage(ctx, image); err != nil { + return "", fmt.Errorf("deployment: pull image %q: %w", image, err) + } + return image, nil + } + _, remainder, ok := strings.Cut(image, "/") + if !ok || !strings.Contains(remainder, "@sha256:") { + return "", fmt.Errorf("deployment: installation image must include an exact sha256 digest: %w", domain.ErrAppImageNotAllowed) + } + pullImage := s.deps.Registry.PullAddress + "/" + remainder + request := domain.ImagePullRequest{Reference: pullImage, Username: s.deps.Registry.Username, Password: s.deps.Registry.Password, Transport: domain.ImagePullTransportHTTP} + if err := s.deps.Runtime.PullImageWithOptions(ctx, request); err != nil { + return "", fmt.Errorf("deployment: pull installation image %q via configured local HTTP transport %q: %w", image, s.deps.Registry.PullAddress, err) + } + digest := remainder[strings.LastIndex(remainder, "@")+1:] + if err := s.deps.Runtime.VerifyImageDigest(ctx, pullImage, digest); err != nil { + return "", fmt.Errorf("deployment: verify pulled installation image %q: %w", image, err) + } + return pullImage, nil +} + +// stripImageTag removes any tag or digest suffix from an image reference. +// colon is a registry port (localhost:15500/e2e/web:v1 strips to +// localhost:15500/e2e/web). +func stripImageTag(ref string) string { + if base, _, ok := strings.Cut(ref, "@"); ok { + return base + } + lastSlash := strings.LastIndex(ref, "/") + if idx := strings.LastIndex(ref, ":"); idx > lastSlash { + return ref[:idx] + } + return ref +} diff --git a/internal/usecase/deployment/coordinator.go b/internal/usecase/deployment/coordinator.go new file mode 100644 index 000000000..139c55f7f --- /dev/null +++ b/internal/usecase/deployment/coordinator.go @@ -0,0 +1,60 @@ +package deployment + +import ( + "context" + "sync" +) + +// appCoordinator serializes workload mutations per app. User mutations +// and boot acquire the app lock blocking; periodic reconciliation +// probes it with TryAcquire and skips busy apps instead of waiting. +// Entries are never deleted so a lock cannot be re-created under a +// holder; the map stays bounded by known app names. +type appCoordinator struct { + mu sync.Mutex + locks map[string]chan struct{} +} + +func newAppCoordinator() *appCoordinator { + return &appCoordinator{locks: map[string]chan struct{}{}} +} + +// lockFor returns the app's lock channel, creating it free (token +// present) on first use. The caller must hold mu. +func (c *appCoordinator) lockFor(app string) chan struct{} { + ch, ok := c.locks[app] + if !ok { + ch = make(chan struct{}, 1) + ch <- struct{}{} + c.locks[app] = ch + } + return ch +} + +// acquire blocks until the app lock is held or ctx ends. The returned +// release must be called exactly once by the holder. +func (c *appCoordinator) acquire(ctx context.Context, app string) (func(), error) { + c.mu.Lock() + ch := c.lockFor(app) + c.mu.Unlock() + select { + case <-ctx.Done(): + return nil, ctx.Err() + case <-ch: + return func() { ch <- struct{}{} }, nil + } +} + +// tryAcquire takes the app lock without blocking. ok is false when the +// app is busy; the caller must skip the app in that case. +func (c *appCoordinator) tryAcquire(app string) (release func(), ok bool) { + c.mu.Lock() + ch := c.lockFor(app) + c.mu.Unlock() + select { + case <-ch: + return func() { ch <- struct{}{} }, true + default: + return nil, false + } +} diff --git a/internal/usecase/deployment/coordinator_internal_test.go b/internal/usecase/deployment/coordinator_internal_test.go new file mode 100644 index 000000000..9cd10edd9 --- /dev/null +++ b/internal/usecase/deployment/coordinator_internal_test.go @@ -0,0 +1,379 @@ +package deployment + +import ( + "context" + "io" + "strings" + "sync" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +func TestAppCoordinator_SerializesOneApp(t *testing.T) { + coord := newAppCoordinator() + release, err := coord.acquire(context.Background(), "blog") + require.NoError(t, err) + + acquired := make(chan struct{}) + go func() { + release2, err := coord.acquire(context.Background(), "blog") + if err == nil { + close(acquired) + release2() + } + }() + + select { + case <-acquired: + t.Fatal("second acquisition ran while the app lock was held") + case <-time.After(50 * time.Millisecond): + } + release() + select { + case <-acquired: + case <-time.After(time.Second): + t.Fatal("second acquisition never ran after release") + } +} + +func TestAppCoordinator_AllowsParallelApps(t *testing.T) { + coord := newAppCoordinator() + releaseBlog, err := coord.acquire(context.Background(), "blog") + require.NoError(t, err) + defer releaseBlog() + + releaseWiki, err := coord.acquire(context.Background(), "wiki") + require.NoError(t, err) + defer releaseWiki() + + // Distinct apps must not block each other. + _, ok := coord.tryAcquire("other") + assert.True(t, ok) +} + +func TestAppCoordinator_TryAcquireSkipsBusyApp(t *testing.T) { + coord := newAppCoordinator() + release, ok := coord.tryAcquire("blog") + require.True(t, ok) + + _, ok = coord.tryAcquire("blog") + assert.False(t, ok, "a periodic pass must skip a busy app instead of waiting") + + release() + _, ok = coord.tryAcquire("blog") + assert.True(t, ok, "the lock is reusable after release") +} + +func TestAppCoordinator_HonorsCancellation(t *testing.T) { + coord := newAppCoordinator() + release, ok := coord.tryAcquire("blog") + require.True(t, ok) + + ctx, cancel := context.WithCancel(context.Background()) + cancel() + _, err := coord.acquire(ctx, "blog") + require.ErrorIs(t, err, context.Canceled) + release() +} + +func TestAppCoordinator_ConcurrentUseIsRaceFree(t *testing.T) { + coord := newAppCoordinator() + var wg sync.WaitGroup + var counter int64 + var mu sync.Mutex + for range 8 { + wg.Add(1) + go func() { + defer wg.Done() + for range 50 { + release, err := coord.acquire(context.Background(), "blog") + if err != nil { + return + } + mu.Lock() + counter++ + mu.Unlock() + release() + } + }() + } + wg.Wait() + assert.Equal(t, int64(400), counter) +} + +func TestRecoveryBackoff_ThreeAttemptsThenEscalatingDelays(t *testing.T) { + now := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + backoff := newRecoveryBackoff(func() time.Time { return now }) + key := recoveryKey{app: "blog", service: "web", container: "c-1"} + + for attempt := 1; attempt <= 3; attempt++ { + assert.True(t, backoff.allow(key), "attempt %d must be permitted", attempt) + backoff.recordFailure(key) + } + assert.False(t, backoff.allow(key), "the fourth attempt is delayed by one minute") + + expected := []time.Duration{time.Minute, 2 * time.Minute, 4 * time.Minute, 8 * time.Minute, 15 * time.Minute} + for i, delay := range expected { + now = now.Add(delay) + assert.True(t, backoff.allow(key), "delay %d must expire", i) + backoff.recordFailure(key) + assert.False(t, backoff.allow(key)) + } + // Further failures stay capped at fifteen minutes. + now = now.Add(15 * time.Minute) + assert.True(t, backoff.allow(key)) + backoff.recordFailure(key) + now = now.Add(14 * time.Minute) + assert.False(t, backoff.allow(key), "the cap holds at fifteen minutes") + now = now.Add(time.Minute) + assert.True(t, backoff.allow(key)) +} + +func TestRecoveryBackoff_EscalationSurvivesWithoutStableObservation(t *testing.T) { + now := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + backoff := newRecoveryBackoff(func() time.Time { return now }) + key := recoveryKey{app: "blog", service: "web", container: "c-1"} + + for range 3 { + backoff.recordFailure(key) + } + require.False(t, backoff.allow(key)) + + // A long idle period permits the delayed attempt, but the escalation + // itself only resets after five continuously stable minutes. + now = now.Add(30 * time.Minute) + require.True(t, backoff.allow(key)) + backoff.recordFailure(key) + assert.False(t, backoff.allow(key), "the failure budget is not escaped by waiting") +} + +func TestRecoveryBackoff_StableObservationResetsHistory(t *testing.T) { + now := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + backoff := newRecoveryBackoff(func() time.Time { return now }) + key := recoveryKey{app: "blog", service: "web", container: "c-1"} + + for range 3 { + backoff.recordFailure(key) + } + require.False(t, backoff.allow(key)) + + now = now.Add(30 * time.Second) + backoff.observeStable(key) + assert.Equal(t, 3, recoveredFailures(backoff, key), "one healthy observation is not enough") + + now = now.Add(stableReadyWindow - time.Second) + backoff.observeStable(key) + assert.Equal(t, 3, recoveredFailures(backoff, key), "history resets only after five stable minutes") + + now = now.Add(2 * time.Second) + backoff.observeStable(key) + assert.Equal(t, 0, recoveredFailures(backoff, key), "five stable minutes reset the failure history") + assert.True(t, backoff.allow(key)) +} + +// recoveredFailures reports the escalation counter of one generation; 0 +// means the history was dropped. +func recoveredFailures(backoff *recoveryBackoff, key recoveryKey) int { + backoff.mu.Lock() + defer backoff.mu.Unlock() + entry, ok := backoff.entries[key] + if !ok { + return 0 + } + return entry.failures +} + +func TestRecoveryBackoff_FailureClearsStability(t *testing.T) { + now := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + backoff := newRecoveryBackoff(func() time.Time { return now }) + key := recoveryKey{app: "blog", service: "web", container: "c-1"} + + backoff.observeStable(key) + now = now.Add(4 * time.Minute) + backoff.recordFailure(key) + now = now.Add(2 * time.Minute) + backoff.observeStable(key) + assert.True(t, backoff.allow(key), "the streak restarted at the failure") +} + +func TestRecoveryBackoff_IsGenerationScoped(t *testing.T) { + now := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + backoff := newRecoveryBackoff(func() time.Time { return now }) + failing := recoveryKey{app: "blog", service: "web", container: "c-1"} + for range 4 { + backoff.recordFailure(failing) + } + assert.False(t, backoff.allow(failing)) + assert.True(t, backoff.allow(recoveryKey{app: "blog", service: "web", container: "c-2"}), + "a replacement generation starts with a fresh budget") + assert.True(t, backoff.allow(recoveryKey{app: "blog", service: "worker", container: "c-1"})) + assert.True(t, backoff.allow(recoveryKey{app: "wiki", service: "web", container: "c-1"})) +} + +func TestRecoveryBackoff_ForgetDropsHistory(t *testing.T) { + now := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + backoff := newRecoveryBackoff(func() time.Time { return now }) + key := recoveryKey{app: "blog", service: "web", container: "c-1"} + for range 4 { + backoff.recordFailure(key) + } + require.False(t, backoff.allow(key)) + backoff.forget(key) + assert.True(t, backoff.allow(key)) +} + +func TestRecoveryBackoff_UnhealthyStreak(t *testing.T) { + backoff := newRecoveryBackoff(time.Now) + key := recoveryKey{app: "blog", service: "web", container: "c-1"} + + assert.Equal(t, 1, backoff.observeUnhealthy(key)) + assert.Equal(t, 2, backoff.observeUnhealthy(key)) + backoff.clearUnhealthy(key) + assert.Equal(t, 1, backoff.observeUnhealthy(key), "a non-unhealthy observation resets the streak") +} + +func TestPublicationInhibition_MarkAndClear(t *testing.T) { + var inhibition publicationInhibition + assert.False(t, inhibition.inhibited("blog", "web")) + inhibition.mark("blog", "web") + assert.True(t, inhibition.inhibited("blog", "web")) + assert.False(t, inhibition.inhibited("blog", "worker")) + assert.False(t, inhibition.inhibited("wiki", "web")) + inhibition.clear("blog", "web") + assert.False(t, inhibition.inhibited("blog", "web")) + inhibition.clear("blog", "web") // clearing twice is harmless +} + +func TestExecutionTracker_DetectsNewExecution(t *testing.T) { + var tracker executionTracker + key := recoveryKey{app: "blog", service: "web", container: "c-1"} + first := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + + assert.True(t, tracker.isNewExecution(key, first), + "an unseen execution must be verified because the runtime may have restarted it while Gordon was absent") + tracker.record(key, first) + assert.False(t, tracker.isNewExecution(key, first), "a stable execution is not new") + assert.True(t, tracker.isNewExecution(key, first.Add(time.Minute)), "a native restart is a new execution") + tracker.forget(key) + assert.True(t, tracker.isNewExecution(key, first)) +} + +func TestRecoveryBackoff_PendingAttemptIsCharged(t *testing.T) { + now := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + backoff := newRecoveryBackoff(func() time.Time { return now }) + key := recoveryKey{app: "blog", service: "web", container: "c-1"} + + assert.False(t, backoff.attemptPending(key)) + backoff.markAttempt(key) + assert.True(t, backoff.attemptPending(key)) + backoff.observeStable(key) + assert.True(t, backoff.attemptPending(key), + "one stable observation does not confirm an intervention: only five stable minutes do") + backoff.observeStable(key) + assert.True(t, backoff.attemptPending(key)) + now = now.Add(stableReadyWindow + time.Second) + backoff.now = func() time.Time { return now } + backoff.observeStable(key) + assert.False(t, backoff.attemptPending(key), "the stability window confirms and resets the generation") +} + +func TestRecoveryBackoff_UnstableObservationInvalidatesTheWindow(t *testing.T) { + now := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + backoff := newRecoveryBackoff(func() time.Time { return now }) + key := recoveryKey{app: "blog", service: "web", container: "c-1"} + + for range 3 { + backoff.recordFailure(key) + } + now = now.Add(time.Minute) + backoff.observeStable(key) + now = now.Add(4 * time.Minute) + backoff.invalidateStable(key) + now = now.Add(time.Minute) + backoff.observeStable(key) + assert.Equal(t, 3, recoveredFailures(backoff, key), + "an unhealthy or unverifiable observation restarts the five-minute window") +} + +func TestRecoveryBackoff_StableObservationNeverCreatesHistory(t *testing.T) { + backoff := newRecoveryBackoff(time.Now) + key := recoveryKey{app: "blog", service: "web", container: "c-1"} + backoff.observeStable(key) + assert.Equal(t, 0, recoveredFailures(backoff, key)) + backoff.mu.Lock() + _, tracked := backoff.entries[key] + backoff.mu.Unlock() + assert.False(t, tracked, "healthy services must not accumulate monitor state") +} + +func TestPublicationInhibition_RetainsActiveServices(t *testing.T) { + var inhibition publicationInhibition + inhibition.mark("blog", "web") + inhibition.mark("blog", "worker") + inhibition.mark("wiki", "web") + assert.Equal(t, []string{"web", "worker"}, inhibition.pending("blog")) + assert.Equal(t, []string{"web"}, inhibition.pending("wiki")) + + inhibition.retainAppServices("blog", map[string]struct{}{"web": {}}) + + assert.Equal(t, []string{"web"}, inhibition.pending("blog"), + "an active service remains inhibited until withdrawal succeeds") + assert.Equal(t, []string{"web"}, inhibition.pending("wiki")) +} + +// TestWaitLogReady_RejectsAnUnknownExecutionBoundary proves the probe +// fails closed when the runtime reports no execution start: without a +// boundary it would scan the container's whole history and could match a +// marker from a previous execution. +func TestWaitLogReady_RejectsAnUnknownExecutionBoundary(t *testing.T) { + deps := ProbeDeps{ + containerStart: func(context.Context, string) (time.Time, error) { return time.Time{}, nil }, + logStream: func(context.Context, string, time.Time) (io.ReadCloser, error) { + return io.NopCloser(strings.NewReader("ready\n")), nil + }, + } + spec := domain.AppService{Readiness: domain.AppReadiness{Type: domain.AppReadinessLog, Contains: "ready"}} + err := waitLogReadyWithDeps(context.Background(), deps, "c-1", spec) + require.Error(t, err) + assert.Contains(t, err.Error(), "runtime reported none") +} + +// TestWaitLogReady_UsesTheExactContainerAndExecution proves the probe +// reads the passed container ID and the current execution boundary. +func TestWaitLogReady_UsesTheExactContainerAndExecution(t *testing.T) { + startedAt := time.Date(2026, 9, 10, 12, 0, 0, 123456789, time.UTC) + var gotID string + var gotSince time.Time + deps := ProbeDeps{ + containerStart: func(context.Context, string) (time.Time, error) { return startedAt, nil }, + logStream: func(_ context.Context, containerID string, since time.Time) (io.ReadCloser, error) { + gotID, gotSince = containerID, since + return io.NopCloser(strings.NewReader("2026-09-10T12:00:01Z ready\n")), nil + }, + } + spec := domain.AppService{Readiness: domain.AppReadiness{Type: domain.AppReadinessLog, Contains: "ready"}} + require.NoError(t, waitLogReadyWithDeps(context.Background(), deps, "c-exact", spec)) + assert.Equal(t, "c-exact", gotID) + assert.Equal(t, startedAt, gotSince) +} + +// TestWaitLogReady_StreamFailureIsActionable proves a transport failure +// is wrapped with the container it belongs to. +func TestWaitLogReady_StreamFailureIsActionable(t *testing.T) { + deps := ProbeDeps{ + containerStart: func(context.Context, string) (time.Time, error) { return time.Now(), nil }, + logStream: func(context.Context, string, time.Time) (io.ReadCloser, error) { + return nil, assert.AnError + }, + } + spec := domain.AppService{Readiness: domain.AppReadiness{Type: domain.AppReadinessLog, Contains: "ready"}} + err := waitLogReadyWithDeps(context.Background(), deps, "c-1", spec) + require.Error(t, err) + assert.Contains(t, err.Error(), "log readiness stream for c-1") + assert.ErrorIs(t, err, assert.AnError) +} diff --git a/internal/usecase/deployment/crash_recovery_test.go b/internal/usecase/deployment/crash_recovery_test.go new file mode 100644 index 000000000..8fb4d424e --- /dev/null +++ b/internal/usecase/deployment/crash_recovery_test.go @@ -0,0 +1,664 @@ +package deployment_test + +import ( + "context" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/out/appstate" + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/deployment" +) + +// seedRevision materializes one revision of an app, so a later recovery can +// rebuild that generation from its pinned digest. +func seedRevision(t *testing.T, ctx context.Context, store *appstate.Store, intentID, supersedes string, rev domain.AppDesiredRevision) { + t.Helper() + require.NoError(t, store.StageApply(ctx, domain.AppApplyIntent{ + Intent: intentID, App: rev.App, Revision: rev.Revision, Supersedes: supersedes, + SourceSHA256: "0f1e2d", Spec: rev.Spec, CreatedAt: time.Now().UTC(), + })) + require.NoError(t, store.CommitApply(ctx, rev.App, intentID)) + require.NoError(t, store.MaterializeApply(ctx, rev.App, intentID)) +} + +// seedReplaceableApp seeds one app whose ACTIVE generation (rev-0) has a +// materialized superseding revision (rev-1). The published container c-old is +// gone from the runtime: the caller decides what the runtime reports for it. +func seedReplaceableApp(t *testing.T, ctx context.Context, store *appstate.Store) (domain.AppDesiredRevision, domain.AppDesiredRevision) { + t.Helper() + spec := webService() + spec.Secrets = map[string]string{} + spec.Readiness = domain.AppReadiness{Type: domain.AppReadinessHTTP, Path: "/healthz", Timeout: time.Second} + activeRev := testRevision("blog", spec) + activeRev.Revision = "rev-0" + require.NoError(t, store.SaveIntent(ctx, domain.AppStopIntent{App: "blog"})) + require.NoError(t, store.SaveActive(ctx, domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": { + Container: "c-old", EffectiveRevision: "rev-0", Image: spec.Image, + Digest: restartTestDigest, Spec: spec, BackendBinds: map[int]int{8080: 18080}, + }, + }})) + require.NoError(t, store.SaveOwnership(ctx, domain.AppOwnership{App: "blog", ID: "app-blog"})) + seedRevision(t, ctx, store, "intent-0", "", activeRev) + desired := testRevision("blog", spec) + seedRevision(t, ctx, store, "intent-1", "rev-0", desired) + return activeRev, desired +} + +// TestDeploy_CandidateIsJournaledBeforeItStarts proves the crash window is +// closed as far as a container ID allows: the candidate ID reaches the +// durable journal before the container is started, and the journal is +// finalized as a success only after publication. +func TestDeploy_CandidateIsJournaledBeforeItStarts(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + runtime := outmocks.NewMockContainerRuntime(t) + images := outmocks.NewMockImageResolver(t) + secrets := outmocks.NewMockSecretProvider(t) + seedReplaceableApp(t, ctx, store) + + images.EXPECT().ResolveDigest(mock.Anything, "docker.io/example/web:1.4.2").Return(restartTestDigest, nil).Once() + runtime.EXPECT().InspectImageVolumes(mock.Anything, "docker.io/example/web:1.4.2").Return(nil, nil).Once() + expectNetworkProvision(runtime, "app-blog", 1) + runtime.EXPECT().StopContainer(mock.Anything, "c-old", mock.Anything).Return(nil).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-old", false).Return(nil).Once() + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).Return(&domain.Container{ID: "c-new", Name: "web"}, nil).Once() + + journaledBeforeStart := false + runtime.EXPECT().StartContainer(mock.Anything, "c-new").RunAndReturn(func(context.Context, string) error { + op, ok, err := store.LoadLatestOperation(context.Background(), "blog") + require.NoError(t, err) + require.True(t, ok, "the replacement is journaled before it starts") + require.False(t, op.Terminal(), "the journal is still in flight while the candidate starts") + step, found := opStep(op, "service.web.replace") + journaledBeforeStart = found && step.After == "c-new" && step.State == domain.AppStepPending + return nil + }).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-new", mock.Anything).Return( + []domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 18081, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + + svc := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 200, nil }, + func(context.Context, string) error { return nil }, + )) + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog", Op: "op-journal"}) + require.NoError(t, err) + require.NotNil(t, result) + assert.Equal(t, "deployed", result.Services["web"].Result) + assert.True(t, journaledBeforeStart, + "the candidate ID is durable before the container is started") + + op, err := store.LoadOperation(context.Background(), "blog", "op-journal") + require.NoError(t, err) + assert.Equal(t, domain.AppOutcomeSuccess, op.Outcome) + step, found := opStep(op, "service.web.replace") + require.True(t, found) + assert.Equal(t, domain.AppStepSucceeded, step.State) + assert.Equal(t, "c-new", step.After) +} + +// crashDuringReplacement runs one replacement deploy and interrupts it +// between creating the candidate and publishing it. The candidate is left +// running, ACTIVE still names the superseded generation, and the operation +// journal stays in flight: the state a process crash or a failed cleanup +// leaves behind. +func crashDuringReplacement(t *testing.T, store *appstate.Store, opKey string) { + t.Helper() + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + runtime := outmocks.NewMockContainerRuntime(t) + images := outmocks.NewMockImageResolver(t) + secrets := outmocks.NewMockSecretProvider(t) + + images.EXPECT().ResolveDigest(mock.Anything, "docker.io/example/web:1.4.2").Return(restartTestDigest, nil).Once() + runtime.EXPECT().InspectImageVolumes(mock.Anything, "docker.io/example/web:1.4.2").Return(nil, nil).Once() + expectNetworkProvision(runtime, "app-blog", 1) + runtime.EXPECT().StopContainer(mock.Anything, "c-old", mock.Anything).Return(nil).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-old", false).Return(nil).Once() + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).Return(&domain.Container{ID: "c-orphan", Name: "web"}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-orphan").Return(nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-orphan", mock.Anything).Return( + []domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 18081, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + // The canceled context also refuses the candidate cleanup, so the + // candidate outlives the operation exactly as a crash would leave it. + runtime.EXPECT().RemoveContainer(mock.Anything, "c-orphan", true).RunAndReturn( + func(removeCtx context.Context, _ string, _ bool) error { + return removeCtx.Err() + }).Once() + + probeStarted := make(chan struct{}) + svc := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(probeCtx context.Context, _, _ string) (int, error) { + select { + case <-probeStarted: + default: + close(probeStarted) + } + <-probeCtx.Done() + return 0, probeCtx.Err() + }, + func(context.Context, string) error { return nil }, + )) + done := make(chan error, 1) + go func() { + _, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog", Op: opKey}) + done <- err + }() + select { + case <-probeStarted: + case <-time.After(5 * time.Second): + t.Fatal("timed out waiting for the readiness probe") + } + cancel() + require.Error(t, <-done) +} + +// TestBootRecovery_RemovesUnpublishedCandidateBeforeRebuildingPublished +// proves the crash window between creating a replacement and publishing it is +// recoverable: an interruption leaves the candidate journaled but unpublished, +// and boot recovery removes that candidate before it rebuilds the generation +// recorded in ACTIVE, so two generations of one service never run together. +func TestBootRecovery_RemovesUnpublishedCandidateBeforeRebuildingPublished(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + images := outmocks.NewMockImageResolver(t) + secrets := outmocks.NewMockSecretProvider(t) + seedReplaceableApp(t, ctx, store) + crashDuringReplacement(t, store, "op-crash") + + // The crash left the candidate durable but unpublished, and ACTIVE still + // names the superseded generation. + interrupted, err := store.LoadOperation(context.Background(), "blog", "op-crash") + require.NoError(t, err) + assert.False(t, interrupted.Terminal(), "an interrupted operation stays in flight until recovery finalizes it") + crashStep, found := opStep(interrupted, "service.web.replace") + require.True(t, found) + assert.Equal(t, domain.AppStepPending, crashStep.State) + assert.Equal(t, "c-orphan", crashStep.After, "the created candidate is durably recorded") + active, ok, err := store.LoadActive(context.Background(), "blog") + require.NoError(t, err) + require.True(t, ok) + assert.Equal(t, "c-old", active.Services["web"].Container) + + // Boot recovery: the orphan is removed before the published generation is + // rebuilt from its pinned revision. + var order []string + runtime2 := outmocks.NewMockContainerRuntime(t) + runtime2.EXPECT().RemoveContainer(mock.Anything, "c-orphan", true).RunAndReturn( + func(context.Context, string, bool) error { + order = append(order, "remove-orphan") + return nil + }).Once() + runtime2.EXPECT().IsContainerRunning(mock.Anything, "c-old").Return(false, nil).Once() + runtime2.EXPECT().StartContainer(mock.Anything, "c-old").Return(domain.ErrContainerNotFound).Once() + runtime2.EXPECT().StopContainer(mock.Anything, "c-old", mock.Anything).Return(domain.ErrContainerNotFound).Once() + runtime2.EXPECT().RemoveContainer(mock.Anything, "c-old", false).Return(domain.ErrContainerNotFound).Once() + expectNetworkProvision(runtime2, "app-blog", 1) + runtime2.EXPECT().CreateContainer(mock.Anything, mock.Anything).RunAndReturn( + func(context.Context, *domain.ContainerConfig) (*domain.Container, error) { + order = append(order, "rebuild-published-generation") + return &domain.Container{ID: "c-rebuilt", Name: "web"}, nil + }).Once() + runtime2.EXPECT().StartContainer(mock.Anything, "c-rebuilt").Return(nil).Once() + runtime2.EXPECT().GetContainerBackendBinds(mock.Anything, "c-rebuilt", mock.Anything).Return( + []domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 18082, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + + rebooted := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime2, Images: images, Secrets: secrets, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime2, + func(context.Context, string, string) (int, error) { return 200, nil }, + func(context.Context, string) error { return nil }, + )) + // Boot recovery runs on a fresh process context: the crashed deploy's + // context is gone with it. + require.NoError(t, rebooted.ReconcileBoot(ctx)) + + assert.Equal(t, []string{"remove-orphan", "rebuild-published-generation"}, order, + "the unpublished candidate is gone before the published generation is rebuilt") + runtime2.AssertNotCalled(t, "StartContainer", mock.Anything, "c-orphan") + runtime2.AssertNotCalled(t, "RestartContainer", mock.Anything, "c-orphan", mock.Anything) + + // The interrupted operation is finalized as failed, so a repeat of its key + // replays an explicit failure instead of a permanent in-flight claim. + finalized, err := store.LoadOperation(context.Background(), "blog", "op-crash") + require.NoError(t, err) + require.True(t, finalized.Terminal(), "boot recovery finalizes the interrupted operation") + assert.Equal(t, domain.AppOutcomeFailed, finalized.Outcome) + finalStep, found := opStep(finalized, "service.web.replace") + require.True(t, found) + assert.Equal(t, domain.AppStepFailed, finalStep.State) + assert.Contains(t, finalStep.Error, "unpublished candidate was removed") + + // ACTIVE now names the rebuilt generation: recovery converged the app + // instead of leaving it pointing at a container that no longer exists. + converged, ok, err := store.LoadActive(context.Background(), "blog") + require.NoError(t, err) + require.True(t, ok) + assert.Equal(t, "c-rebuilt", converged.Services["web"].Container) + assert.Equal(t, "rev-0", converged.Services["web"].EffectiveRevision) +} + +// TestDeploy_ReconcilesInterruptedPredecessorBeforeCreatingAReplacement +// proves the live case: when a cleanup failed and Gordon kept running, the +// next mutation of the app removes the leftover candidate before it creates a +// new generation, so two generations of one service never run together even +// without a restart. It also finalizes the interrupted journal. +func TestDeploy_ReconcilesInterruptedPredecessorBeforeCreatingAReplacement(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + images := outmocks.NewMockImageResolver(t) + secrets := outmocks.NewMockSecretProvider(t) + seedReplaceableApp(t, ctx, store) + crashDuringReplacement(t, store, "op-crash") + + var order []string + runtime := outmocks.NewMockContainerRuntime(t) + runtime.EXPECT().RemoveContainer(mock.Anything, "c-orphan", true).RunAndReturn( + func(context.Context, string, bool) error { + order = append(order, "remove-leftover-candidate") + return nil + }).Once() + images.EXPECT().ResolveDigest(mock.Anything, "docker.io/example/web:1.4.2").Return(restartTestDigest, nil).Once() + runtime.EXPECT().InspectImageVolumes(mock.Anything, "docker.io/example/web:1.4.2").Return(nil, nil).Once() + runtime.EXPECT().StopContainer(mock.Anything, "c-old", mock.Anything).Return(domain.ErrContainerNotFound).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-old", false).Return(domain.ErrContainerNotFound).Once() + expectNetworkProvision(runtime, "app-blog", 1) + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).RunAndReturn( + func(context.Context, *domain.ContainerConfig) (*domain.Container, error) { + order = append(order, "create-new-generation") + return &domain.Container{ID: "c-next", Name: "web"}, nil + }).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-next").Return(nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-next", mock.Anything).Return( + []domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 18082, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + + svc := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 200, nil }, + func(context.Context, string) error { return nil }, + )) + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog", Op: "op-next"}) + require.NoError(t, err) + require.NotNil(t, result) + assert.Equal(t, "deployed", result.Services["web"].Result) + assert.Equal(t, []string{"remove-leftover-candidate", "create-new-generation"}, order) + + active, ok, err := store.LoadActive(ctx, "blog") + require.NoError(t, err) + require.True(t, ok) + assert.Equal(t, "c-next", active.Services["web"].Container) + + interrupted, err := store.LoadOperation(ctx, "blog", "op-crash") + require.NoError(t, err) + require.True(t, interrupted.Terminal(), "the interrupted operation is finalized") + assert.Equal(t, domain.AppOutcomeFailed, interrupted.Outcome) + step, found := opStep(interrupted, "service.web.replace") + require.True(t, found) + assert.Equal(t, domain.AppStepFailed, step.State) +} + +// TestBootRecovery_RemovesCandidateOfAnInterruptedFirstDeployment proves the +// same recovery works when the interrupted operation is the app's FIRST +// deployment: there is no ACTIVE generation to converge, and the leftover +// candidate is still removed and the journal finalized. +func TestBootRecovery_RemovesCandidateOfAnInterruptedFirstDeployment(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + images := outmocks.NewMockImageResolver(t) + secrets := outmocks.NewMockSecretProvider(t) + require.NoError(t, store.SaveIntent(ctx, domain.AppStopIntent{App: "blog"})) + require.NoError(t, store.SaveOperation(ctx, domain.AppOperation{ + Op: "op-seeded", Kind: "deploy", App: "blog", StartedAt: time.Now().UTC(), + Steps: []domain.AppOperationStep{ + {ID: "preflight", State: domain.AppStepSucceeded}, + {ID: "service.web.replace", State: domain.AppStepPending, Service: "web", Before: "c-old", After: "c-orphan"}, + }, + })) + + runtime := outmocks.NewMockContainerRuntime(t) + runtime.EXPECT().RemoveContainer(mock.Anything, "c-orphan", true).Return(nil).Once() + + svc := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()) + require.NoError(t, svc.ReconcileBoot(ctx)) + + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, "c-orphan") + + finalized, err := store.LoadOperation(ctx, "blog", "op-seeded") + require.NoError(t, err) + require.True(t, finalized.Terminal()) + assert.Equal(t, domain.AppOutcomeFailed, finalized.Outcome) + step, found := opStep(finalized, "service.web.replace") + require.True(t, found) + assert.Equal(t, domain.AppStepFailed, step.State) + assert.Contains(t, step.Error, "unpublished candidate was removed") +} + +// TestRestart_FailedRebuildClearsTheInhibitionOfTheMissingContainer proves a +// restart that falls back to a rebuild does not leave a durable inhibition for +// a container that is already gone: nothing could ever revive it, and the +// marker would refuse every later start, recovery pass, and restart. +func TestRestart_FailedRebuildClearsTheInhibitionOfTheMissingContainer(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + images := outmocks.NewMockImageResolver(t) + secrets := outmocks.NewMockSecretProvider(t) + runtime := outmocks.NewMockContainerRuntime(t) + + spec := webService() + spec.Secrets = map[string]string{} + spec.Volumes = []domain.AppVolume{{Name: "data", Path: "/data"}} + spec.Readiness = domain.AppReadiness{Type: domain.AppReadinessHTTP, Path: "/healthz", Timeout: 50 * time.Millisecond} + rev := testRevision("blog", spec) + rev.Revision = "rev-0" + require.NoError(t, store.SaveIntent(ctx, domain.AppStopIntent{App: "blog"})) + require.NoError(t, store.SaveActive(ctx, domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": { + Container: "c-old", EffectiveRevision: "rev-0", Image: spec.Image, + Digest: restartTestDigest, Spec: spec, BackendBinds: map[int]int{8080: 18080}, + }, + }})) + require.NoError(t, store.SaveOwnership(ctx, domain.AppOwnership{App: "blog", ID: "app-blog"})) + seedRevision(t, ctx, store, "intent-0", "", rev) + + runtime.EXPECT().RestartContainer(mock.Anything, "c-old", mock.Anything).Return(domain.ErrContainerNotFound).Once() + runtime.EXPECT().StopContainer(mock.Anything, "c-old", mock.Anything).Return(domain.ErrContainerNotFound).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-old", false).Return(domain.ErrContainerNotFound).Once() + expectNetworkProvision(runtime, "app-blog", 1) + runtime.EXPECT().CreateVolume(mock.Anything, mock.Anything, mock.Anything).Return(nil).Once() + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).Return(&domain.Container{ID: "c-rebuild", Name: "web"}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-rebuild").Return(nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-rebuild", mock.Anything).Return( + []domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 18081, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + runtime.EXPECT().GetContainerLogs(mock.Anything, "c-rebuild", false).Return(nil, assert.AnError).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-rebuild", true).Return(nil).Once() + + svc := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 503, nil }, + func(context.Context, string) error { return assert.AnError }, + )) + + result, err := svc.Restart(ctx, "blog", "web", "op-restart-failed") + require.Error(t, err) + require.NotNil(t, result) + assert.Equal(t, "failed", result.Services["web"].Result) + + inhibitions, loadErr := store.LoadRecoveryInhibitions(ctx, "blog") + require.NoError(t, loadErr) + assert.Empty(t, inhibitions, + "the inhibition of a container proven gone must not survive a failed rebuild") +} + +// TestBootRecovery_RetriesLeftoverOfATerminalFailedOperation proves a +// leftover candidate recorded by an operation that already reached a terminal +// failure is still converged: the journal keeps the candidate ID, and the next +// recovery pass removes it. +func TestBootRecovery_RetriesLeftoverOfATerminalFailedOperation(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + images := outmocks.NewMockImageResolver(t) + secrets := outmocks.NewMockSecretProvider(t) + require.NoError(t, store.SaveIntent(ctx, domain.AppStopIntent{App: "blog", Stopped: true})) + require.NoError(t, store.SaveActive(ctx, domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-old", EffectiveRevision: "rev-0", Spec: headlessSpec()}, + }})) + require.NoError(t, store.SaveOperation(ctx, domain.AppOperation{ + Op: "op-failed", Kind: "deploy", App: "blog", StartedAt: time.Now().UTC(), + Outcome: domain.AppOutcomeFailed, + Steps: []domain.AppOperationStep{ + {ID: "preflight", State: domain.AppStepSucceeded}, + {ID: "service.web.replace", State: domain.AppStepFailed, Service: "web", Before: "c-old", After: "c-orphan", Error: "readiness"}, + }, + })) + + runtime := outmocks.NewMockContainerRuntime(t) + runtime.EXPECT().RemoveContainer(mock.Anything, "c-orphan", true).Return(nil).Once() + runtime.EXPECT().InspectContainer(mock.Anything, "c-old").Return(&domain.Container{ID: "c-old", Status: "running"}, nil).Once() + runtime.EXPECT().StopContainer(mock.Anything, "c-old", mock.Anything).Return(nil).Once() + + svc := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()) + require.NoError(t, svc.ReconcileBoot(ctx)) + + runtime.AssertCalled(t, "RemoveContainer", mock.Anything, "c-orphan", true) + + // The removal is recorded: the step no longer retries the candidate on a + // later pass, and the removed container id stays visible for audit. + convergedOp, err := store.LoadOperation(ctx, "blog", "op-failed") + require.NoError(t, err) + step, found := opStep(convergedOp, "service.web.replace") + require.True(t, found) + assert.Empty(t, step.After, "a converged candidate must not be retried on the next pass") + assert.Contains(t, step.Detail, "c-orphan", "the removed candidate stays in the audit trail") +} + +// TestBootRecovery_InterruptionAfterTheLastStepIsASuccess proves an operation +// whose effects all ran but whose final outcome write was interrupted is +// classified as the success it was, and the published generation is kept. +func TestBootRecovery_InterruptionAfterTheLastStepIsASuccess(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + images := outmocks.NewMockImageResolver(t) + secrets := outmocks.NewMockSecretProvider(t) + require.NoError(t, store.SaveIntent(ctx, domain.AppStopIntent{App: "blog", Stopped: true})) + require.NoError(t, store.SaveActive(ctx, domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-published", EffectiveRevision: "rev-0", Spec: headlessSpec()}, + }})) + require.NoError(t, store.SaveOperation(ctx, domain.AppOperation{ + Op: "op-unfinished", Kind: "deploy", App: "blog", StartedAt: time.Now().UTC(), + Steps: []domain.AppOperationStep{ + {ID: "preflight", State: domain.AppStepSucceeded}, + {ID: "service.web.replace", State: domain.AppStepSucceeded, Service: "web", Before: "c-old", After: "c-published"}, + }, + })) + + runtime := outmocks.NewMockContainerRuntime(t) + runtime.EXPECT().InspectContainer(mock.Anything, "c-published").Return(&domain.Container{ID: "c-published", Status: "exited"}, nil).Once() + + svc := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()) + require.NoError(t, svc.ReconcileBoot(ctx)) + + reconciled, err := store.LoadOperation(ctx, "blog", "op-unfinished") + require.NoError(t, err) + require.True(t, reconciled.Terminal()) + assert.Equal(t, domain.AppOutcomeSuccess, reconciled.Outcome) + step, found := opStep(reconciled, "service.web.replace") + require.True(t, found) + assert.Equal(t, domain.AppStepSucceeded, step.State, "a step that already succeeded is left as recorded") + runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, "c-published", mock.Anything) +} + +// TestBootRecovery_FailsClosedWhenCandidateCannotBeRemoved proves +// reconciliation never proceeds while an orphan candidate may still run: the +// removal failure aborts the mutation, the interrupted journal is not +// finalized, and no newer operation is created to mask the leftover. +func TestBootRecovery_FailsClosedWhenCandidateCannotBeRemoved(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + images := outmocks.NewMockImageResolver(t) + secrets := outmocks.NewMockSecretProvider(t) + seedReplaceableApp(t, ctx, store) + require.NoError(t, store.SaveOperation(ctx, domain.AppOperation{ + Op: "op-crash", Kind: "deploy", App: "blog", StartedAt: time.Now().UTC(), + Steps: []domain.AppOperationStep{ + {ID: "service.web.replace", State: domain.AppStepPending, Service: "web", Before: "c-old", After: "c-orphan"}, + }, + })) + + runtime := outmocks.NewMockContainerRuntime(t) + runtime.EXPECT().RemoveContainer(mock.Anything, "c-orphan", true).Return(assert.AnError) + + svc := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()) + + err := svc.ReconcileBoot(ctx) + require.Error(t, err) + assert.Contains(t, err.Error(), "could not be removed") + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) + + // The leftover stays journaled and non-terminal, so it is still the latest + // operation. + leftover, err := store.LoadOperation(ctx, "blog", "op-crash") + require.NoError(t, err) + assert.False(t, leftover.Terminal(), "a leftover that could not be removed is never finalized") + + // A later mutation fails closed on the same leftover and opens no journal + // that could mask it. + _, err = svc.Deploy(ctx, deployment.DeployInput{App: "blog", Op: "op-next"}) + require.Error(t, err) + assert.Contains(t, err.Error(), "could not be removed") + _, err = store.LoadOperation(ctx, "blog", "op-next") + require.ErrorIs(t, err, domain.ErrAppOperationNotFound) +} + +// TestBootRecovery_ClearsReplacementInhibitionAndRebuildsFromActive proves +// the crash window is fully recoverable for a volume-owning service: an +// interruption after the superseded container was retired leaves the +// replacement-pending inhibition behind, boot recovery removes the +// never-published candidate, clears that stale inhibition, and rebuilds the +// generation ACTIVE records from its pinned revision. +func TestBootRecovery_ClearsReplacementInhibitionAndRebuildsFromActive(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + images := outmocks.NewMockImageResolver(t) + secrets := outmocks.NewMockSecretProvider(t) + + spec := webService() + spec.Secrets = map[string]string{} + spec.Volumes = []domain.AppVolume{{Name: "data", Path: "/data"}} + spec.Readiness = domain.AppReadiness{Type: domain.AppReadinessHTTP, Path: "/healthz", Timeout: time.Second} + rev := testRevision("blog", spec) + rev.Revision = "rev-0" + require.NoError(t, store.SaveIntent(ctx, domain.AppStopIntent{App: "blog"})) + require.NoError(t, store.SaveActive(ctx, domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": { + Container: "c-old", EffectiveRevision: "rev-0", Image: spec.Image, + Digest: restartTestDigest, Spec: spec, BackendBinds: map[int]int{8080: 18080}, + }, + }})) + require.NoError(t, store.SaveOwnership(ctx, domain.AppOwnership{App: "blog", ID: "app-blog"})) + seedRevision(t, ctx, store, "intent-0", "", rev) + // A replacement interrupted after it retired c-old and created its + // candidate: ACTIVE still names c-old, the candidate is journaled but + // unpublished, and c-old carries the replacement-pending inhibition. + require.NoError(t, store.SaveOperation(ctx, domain.AppOperation{ + Op: "op-crash", Kind: "deploy", App: "blog", StartedAt: time.Now().UTC(), + Steps: []domain.AppOperationStep{ + {ID: "preflight", State: domain.AppStepSucceeded}, + {ID: "service.web.replace", State: domain.AppStepPending, Service: "web", Before: "c-old", After: "c-orphan"}, + }, + })) + require.NoError(t, store.SaveRecoveryInhibition(ctx, domain.AppRecoveryInhibition{ + App: "blog", Service: "web", ContainerID: "c-old", + Reason: domain.AppInhibitReplacementPending, Operation: "op-crash", + })) + + runtime := outmocks.NewMockContainerRuntime(t) + runtime.EXPECT().RemoveContainer(mock.Anything, "c-orphan", true).Return(nil).Once() + runtime.EXPECT().IsContainerRunning(mock.Anything, "c-old").Return(false, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-old").Return(domain.ErrContainerNotFound).Once() + runtime.EXPECT().StopContainer(mock.Anything, "c-old", mock.Anything).Return(domain.ErrContainerNotFound).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-old", false).Return(domain.ErrContainerNotFound).Once() + expectNetworkProvision(runtime, "app-blog", 1) + runtime.EXPECT().CreateVolume(mock.Anything, mock.Anything, mock.Anything).Return(nil).Once() + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).Return(&domain.Container{ID: "c-rebuilt", Name: "web"}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-rebuilt").Return(nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-rebuilt", mock.Anything).Return( + []domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 18082, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + + svc := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 200, nil }, + func(context.Context, string) error { return nil }, + )) + require.NoError(t, svc.ReconcileBoot(ctx)) + + inhibitions, err := store.LoadRecoveryInhibitions(ctx, "blog") + require.NoError(t, err) + assert.Empty(t, inhibitions, "the stale replacement inhibition must not survive recovery") + + active, ok, err := store.LoadActive(ctx, "blog") + require.NoError(t, err) + require.True(t, ok) + assert.Equal(t, "c-rebuilt", active.Services["web"].Container) + assert.Equal(t, "rev-0", active.Services["web"].EffectiveRevision) + + finalized, err := store.LoadOperation(ctx, "blog", "op-crash") + require.NoError(t, err) + require.True(t, finalized.Terminal()) + assert.Equal(t, domain.AppOutcomeFailed, finalized.Outcome) +} + +// TestRestart_RebuildsMissingActiveContainer proves restart is not a dead end +// when the recorded container is gone: the service is rebuilt from the pinned +// ACTIVE digest and published again. +func TestRestart_RebuildsMissingActiveContainer(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + runtime := outmocks.NewMockContainerRuntime(t) + images := outmocks.NewMockImageResolver(t) + secrets := outmocks.NewMockSecretProvider(t) + seedReplaceableApp(t, ctx, store) + + runtime.EXPECT().RestartContainer(mock.Anything, "c-old", mock.Anything).Return(domain.ErrContainerNotFound).Once() + runtime.EXPECT().StopContainer(mock.Anything, "c-old", mock.Anything).Return(domain.ErrContainerNotFound).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-old", false).Return(domain.ErrContainerNotFound).Once() + expectNetworkProvision(runtime, "app-blog", 1) + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).Return(&domain.Container{ID: "c-rebuilt", Name: "web"}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-rebuilt").Return(nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-rebuilt", mock.Anything).Return( + []domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 18082, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + + svc := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 200, nil }, + func(context.Context, string) error { return nil }, + )) + + result, err := svc.Restart(ctx, "blog", "web", "op-restart-missing") + require.NoError(t, err) + require.NotNil(t, result) + assert.Equal(t, "deployed", result.Services["web"].Result) + assert.Equal(t, "c-rebuilt", result.Services["web"].After, + "the missing generation is rebuilt from the pinned ACTIVE digest") + + active, ok, err := store.LoadActive(context.Background(), "blog") + require.NoError(t, err) + require.True(t, ok) + assert.Equal(t, "c-rebuilt", active.Services["web"].Container) + assert.Equal(t, "rev-0", active.Services["web"].EffectiveRevision, + "the rebuild reuses the pinned revision") + + op, err := store.LoadOperation(context.Background(), "blog", "op-restart-missing") + require.NoError(t, err) + assert.Equal(t, domain.AppOutcomeSuccess, op.Outcome) + step, found := opStep(op, "service.web.restart") + require.True(t, found) + assert.Equal(t, domain.AppStepSucceeded, step.State) + assert.Equal(t, "c-rebuilt", step.After) +} diff --git a/internal/usecase/deployment/deploy_mock_test.go b/internal/usecase/deployment/deploy_mock_test.go new file mode 100644 index 000000000..11063712c --- /dev/null +++ b/internal/usecase/deployment/deploy_mock_test.go @@ -0,0 +1,836 @@ +package deployment_test + +import ( + "context" + "errors" + "io" + "strings" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/deployment" +) + +func mockDeps(t *testing.T) (*outmocks.MockAppState, *outmocks.MockContainerRuntime, *outmocks.MockImageResolver, *outmocks.MockSecretProvider) { + t.Helper() + state := outmocks.NewMockAppState(t) + // Every app mutation reconciles an interrupted predecessor first. These + // tests exercise the mutation itself, so they see no interrupted + // operation; tests that seed one use their own expectation or a real + // store. + state.EXPECT().LoadLatestOperation(mock.Anything, mock.Anything). + Return(domain.AppOperation{}, false, nil).Maybe() + // ExecuteDeploy always reloads the claimed operation from the store by + // app/op identity. Mock-backed execution sees the freshly claimed + // non-terminal journal shape (a single pending preflight step) unless a + // test overrides it with expectStoredOperation. + state.EXPECT().LoadOperation(mock.Anything, mock.Anything, mock.Anything). + RunAndReturn(func(_ context.Context, app, opID string) (domain.AppOperation, error) { + return domain.AppOperation{ + Kind: "deploy", App: app, Op: opID, + Steps: []domain.AppOperationStep{{ID: "preflight", State: domain.AppStepPending}}, + }, nil + }).Maybe() + return state, + outmocks.NewMockContainerRuntime(t), + outmocks.NewMockImageResolver(t), + outmocks.NewMockSecretProvider(t) +} + +// expectStoredOperation makes one store record authoritative for the next +// ExecuteDeploy reload. The operation identity is echoed from the request so +// an unkeyed claim matches too. It replaces the catch-all default installed +// by mockDeps, letting a test prove that a terminal or mismatched store +// record overrides a stale in-process journal. +func expectStoredOperation(state *outmocks.MockAppState, op domain.AppOperation) { + kept := state.ExpectedCalls[:0] + for _, call := range state.ExpectedCalls { + if call.Method != "LoadOperation" { + kept = append(kept, call) + } + } + state.ExpectedCalls = kept + state.EXPECT().LoadOperation(mock.Anything, op.App, mock.Anything). + RunAndReturn(func(_ context.Context, app, opID string) (domain.AppOperation, error) { + op.App = app + op.Op = opID + return op, nil + }) +} + +func mockRevision() domain.AppDesiredRevision { + svc := webService() + svc.Image = "docker.io/example/web:1.4.2" + svc.Readiness.Timeout = time.Second + return domain.AppDesiredRevision{ + Revision: "rev-1", App: "blog", + Spec: domain.AppSpec{Name: "blog", Env: map[string]string{}, Services: []domain.AppService{svc}}, + } +} + +// expectDeployPreflight wires recover + revision + digest + secrets + +// image volumes + checkpoint + ownership + the two journal writes that +// frame execution (preflight table, then per-service/outcome updates). +func expectDeployPreflight( + state *outmocks.MockAppState, + runtime *outmocks.MockContainerRuntime, + images *outmocks.MockImageResolver, + secrets *outmocks.MockSecretProvider, + rev domain.AppDesiredRevision, +) { + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + images.EXPECT().ResolveDigest(mock.Anything, rev.Spec.Services[0].Image).Return("sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", nil) + secrets.EXPECT().GetSecret(mock.Anything, "gordon/apps/app-blog/web/database-url").Return("x", nil) + runtime.EXPECT().InspectImageVolumes(mock.Anything, rev.Spec.Services[0].Image).Return(nil, nil) + state.EXPECT().LoadCheckpoint(mock.Anything).Return(domain.AppStoreCheckpoint{}, nil) + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-blog"}, nil) + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) +} + +// expectNetworkProvision registers the two runtime calls every +// create/recovery path makes when the incarnation network does not yet +// exist: inspect existing networks (none), then create the private one. +func expectNetworkProvision(runtime *outmocks.MockContainerRuntime, appID string, times int) { + runtime.EXPECT().ListNetworks(mock.Anything).Return([]*domain.NetworkInfo{}, nil).Times(times) + runtime.EXPECT().CreateNetwork(mock.Anything, domain.AppPrivateNetworkName("gordon", appID), mock.Anything).Return(nil).Times(times) +} + +func TestDeploy_PreflightFailureReturnsJournaledOperation(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + state.EXPECT().Recover(mock.Anything).Return(nil).Once() + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil).Once() + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil).Twice() + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-blog"}, nil).Once() + images.EXPECT().ResolveDigest(mock.Anything, rev.Spec.Services[0].Image).Return("", assert.AnError).Once() + // The store record carries the revision the claim was started for. + expectStoredOperation(state, domain.AppOperation{ + Kind: "deploy", App: "blog", InputRevision: rev.Revision, + Steps: []domain.AppOperationStep{{ID: "preflight", State: domain.AppStepPending}}, + }) + + svc := deployment.NewService(deployment.Deps{State: state, Runtime: runtime, Images: images, Secrets: secrets}, zerowrap.Default()) + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + + require.ErrorIs(t, err, domain.ErrAppImageUnresolvable) + require.NotNil(t, result) + assert.NotEmpty(t, result.Op) + assert.Equal(t, "blog", result.App) + assert.Equal(t, rev.Revision, result.Revision) + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) +} + +func TestDeploy_Mockery_PullsPinnedImageWithRegistryAuth(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + svcSpec := webService() + svcSpec.Image = "registry.example.com/blog/web:1.4.2" + svcSpec.HTTP = nil + svcSpec.Readiness = domain.AppReadiness{} + rev := domain.AppDesiredRevision{ + Revision: "rev-1", App: "blog", + Spec: domain.AppSpec{Name: "blog", Env: map[string]string{}, Services: []domain.AppService{svcSpec}}, + } + + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + images.EXPECT().ResolveDigest(mock.Anything, svcSpec.Image).Return("sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", nil) + secrets.EXPECT().GetSecret(mock.Anything, "gordon/apps/app-blog/web/database-url").Return("x", nil) + pullRef := "127.0.0.1:5000/blog/web@sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + runtime.EXPECT().PullImageWithOptions(mock.Anything, domain.ImagePullRequest{Reference: pullRef, Username: "gordon", Password: "s3cret", Transport: domain.ImagePullTransportHTTP}).Return(nil).Once() + runtime.EXPECT().VerifyImageDigest(mock.Anything, pullRef, "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa").Return(nil).Once() + runtime.EXPECT().InspectImageVolumes(mock.Anything, "127.0.0.1:5000/blog/web@sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa").Return(nil, nil) + state.EXPECT().LoadCheckpoint(mock.Anything).Return(domain.AppStoreCheckpoint{}, nil) + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-blog"}, nil) + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", + Services: map[string]domain.AppEffectiveService{}, + }, true, nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + expectNetworkProvision(runtime, "app-blog", 1) + + // The pinned ref was pulled during preflight before image-volume inspection. + runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { + return cfg.Image == "127.0.0.1:5000/blog/web@sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" && + assert.Equal(t, svcSpec.Command, cfg.Entrypoint) && assert.Empty(t, cfg.Cmd) + })).Return(&domain.Container{ID: "c-new", Name: "web"}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-new").Return(nil).Once() + // No TCP-capable interfaces (HTTP nil, readiness empty): no backend + // binds published, no GetContainerBackendBinds call. + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", + Services: map[string]domain.AppEffectiveService{}, + }, true, nil) + state.EXPECT().SaveActive(mock.Anything, mock.Anything).Return(nil) + state.EXPECT().SaveOwnership(mock.Anything, mock.Anything).Return(nil) + + svc := deployment.NewService( + deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, + Registry: deployment.RegistryConfig{ + Domain: "registry.example.com", + PullAddress: "127.0.0.1:5000", + Username: "gordon", + Password: "s3cret", + }, + }, + zerowrap.Default(), + ).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 200, nil }, + func(context.Context, string) error { return nil }, + )) + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + require.NoError(t, err) + require.NotNil(t, result) + assert.Equal(t, "deployed", result.Services["web"].Result) + assert.Empty(t, result.Services["web"].BackendBinds) +} + +// TestDeploy_Mockery_HTTPSuccessRecordsLoopbackBinds proves the rootless-first +// readiness path: the engine publishes the HTTP container port on +// 127.0.0.1 ephemeral, probes the recorded loopback bind (never a +// container IP), and records the binds for the active record + proxy. +func TestDeploy_Mockery_HTTPSuccessRecordsLoopbackBinds(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + expectDeployPreflight(state, runtime, images, secrets, rev) + expectNetworkProvision(runtime, "app-blog", 1) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", Services: map[string]domain.AppEffectiveService{}, + }, true, nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).Return(&domain.Container{ID: "c-new", Name: "web"}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-new").Return(nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-new", mock.Anything).Return([]domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 18080, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + state.EXPECT().RegisterBackendBinds(mock.Anything, mock.MatchedBy(func(claims []domain.AppListenerReservation) bool { + return len(claims) == 1 && claims[0].Port == 18080 && claims[0].Owner == domain.OwnerGordonBackend && claims[0].ContainerID == "c-new" + })).Return(nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", Services: map[string]domain.AppEffectiveService{}, + }, true, nil) + state.EXPECT().SaveActive(mock.Anything, mock.MatchedBy(func(a domain.AppActive) bool { + return a.Services["web"].BackendBinds[8080] == 18080 + })).Return(nil).Once() + state.EXPECT().SaveOwnership(mock.Anything, mock.Anything).Return(nil) + + var probedURL string + svc := deployment.NewService( + deployment.Deps{State: state, Runtime: runtime, Images: images, Secrets: secrets}, + zerowrap.Default(), + ).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(_ context.Context, url, _ string) (int, error) { probedURL = url; return 200, nil }, + func(context.Context, string) error { return nil }, + )) + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + require.NoError(t, err) + require.NotNil(t, result) + assert.Equal(t, "deployed", result.Services["web"].Result) + assert.Equal(t, map[int]int{8080: 18080}, result.Services["web"].BackendBinds) + assert.Contains(t, probedURL, "127.0.0.1:18080", "readiness dials the loopback bind, never a container IP") + runtime.AssertNotCalled(t, "GetContainerNetworkInfo", mock.Anything, mock.Anything) +} + +// TestDeploy_Mockery_HTTPReplacesOldGenerationBeforePublish proves the +// sequential ordering: the superseded container is stopped and removed +// before the replacement is created, and ACTIVE names the replacement only +// after it exists. A public HTTP service without volumes is replaced exactly +// like every other service. +func TestDeploy_Mockery_HTTPReplacesOldGenerationBeforePublish(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + expectDeployPreflight(state, runtime, images, secrets, rev) + expectNetworkProvision(runtime, "app-blog", 1) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", + Services: map[string]domain.AppEffectiveService{ + "web": {EffectiveRevision: "rev-0", Image: "img:0", Container: "c-old"}, + }, + }, true, nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + + var replacedAt, createdAt, publishedAt int + var step int + runtime.EXPECT().StopContainer(mock.Anything, "c-old", mock.Anything).RunAndReturn(func(context.Context, string, time.Duration) error { + step++ + replacedAt = step + return nil + }).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-old", false).Return(nil).Once() + state.EXPECT().ReleaseBackendBinds(mock.Anything, "blog", "c-old").Return(nil).Once() + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).RunAndReturn(func(context.Context, *domain.ContainerConfig) (*domain.Container, error) { + step++ + createdAt = step + return &domain.Container{ID: "c-new", Name: "web"}, nil + }).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-new").Return(nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-new", mock.Anything).Return([]domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 18080, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + state.EXPECT().RegisterBackendBinds(mock.Anything, mock.MatchedBy(func(claims []domain.AppListenerReservation) bool { + return len(claims) == 1 && claims[0].Port == 18080 && claims[0].Owner == domain.OwnerGordonBackend && claims[0].ContainerID == "c-new" + })).Return(nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", + Services: map[string]domain.AppEffectiveService{ + "web": {EffectiveRevision: "rev-0", Image: "img:0", Container: "c-old"}, + }, + }, true, nil) + state.EXPECT().SaveActive(mock.Anything, mock.Anything).RunAndReturn(func(context.Context, domain.AppActive) error { + step++ + publishedAt = step + return nil + }).Once() + state.EXPECT().SaveOwnership(mock.Anything, mock.Anything).Return(nil) + // The superseded generation stops being recovery-inhibited only + // once the replacement is published. + state.EXPECT().ClearRecoveryInhibition(mock.Anything, "blog", "web", "c-old").Return(nil).Once() + + svc := deployment.NewService( + deployment.Deps{State: state, Runtime: runtime, Images: images, Secrets: secrets}, + zerowrap.Default(), + ).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 200, nil }, + func(context.Context, string) error { return nil }, + )) + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + require.NoError(t, err) + require.NotNil(t, result) + assert.Equal(t, "deployed", result.Services["web"].Result) + assert.Less(t, replacedAt, createdAt, "the superseded generation is gone before the replacement is created") + assert.Greater(t, publishedAt, createdAt, "ACTIVE is published after the replacement exists") + assert.Empty(t, result.CleanupWarnings) +} + +// TestDeploy_Mockery_HTTPReadinessFailureLeavesNoOldGeneration proves a +// failed replacement never falls back to the generation it already retired: +// the superseded container stays gone, the unready candidate is removed, and +// nothing is published in its place. +func TestDeploy_Mockery_HTTPReadinessFailureLeavesNoOldGeneration(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + expectDeployPreflight(state, runtime, images, secrets, rev) + expectNetworkProvision(runtime, "app-blog", 1) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", + Services: map[string]domain.AppEffectiveService{ + "web": {EffectiveRevision: "rev-0", Image: "img:0", Container: "c-old"}, + }, + }, true, nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + + // The superseded generation is retired before the replacement is created, + // then the replacement fails readiness fast. + runtime.EXPECT().StopContainer(mock.Anything, "c-old", mock.Anything).Return(nil).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-old", false).Return(nil).Once() + state.EXPECT().ReleaseBackendBinds(mock.Anything, "blog", "c-old").Return(nil).Once() + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).Return(&domain.Container{ID: "c-new", Name: "web"}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-new").Return(nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-new", mock.Anything).Return([]domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 18080, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + state.EXPECT().RegisterBackendBinds(mock.Anything, mock.MatchedBy(func(claims []domain.AppListenerReservation) bool { + return len(claims) == 1 && claims[0].Port == 18080 && claims[0].Owner == domain.OwnerGordonBackend && claims[0].ContainerID == "c-new" + })).Return(nil).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-new", true).Return(nil).Once() + // The failed candidate's claims are released: a candidate that never + // became effective must not keep a loopback reservation. + state.EXPECT().ReleaseBackendBinds(mock.Anything, "blog", "c-new").Return(nil).Once() + runtime.EXPECT().GetContainerLogs(mock.Anything, "c-new", false).Return(io.NopCloser(strings.NewReader("boom\n")), nil).Once() + + svc := deployment.NewService( + deployment.Deps{State: state, Runtime: runtime, Images: images, Secrets: secrets}, + zerowrap.Default(), + ).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 500, nil }, + func(context.Context, string) error { return errors.New("refused") }, + )) + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + require.Error(t, err) + require.NotNil(t, result) + assert.Equal(t, "failed", result.Services["web"].Result) + assert.NotContains(t, result.Services["web"].Error, "logs:", "raw logs must never be embedded in the public error") + require.Len(t, result.Services["web"].Diagnostics, 1, "failure diagnostics are kept separately") + assert.Equal(t, "boom", result.Services["web"].Diagnostics[0]) + runtime.AssertNotCalled(t, "StopContainer", mock.Anything, "c-new", mock.Anything) + runtime.AssertNotCalled(t, "RemoveVolume", mock.Anything, mock.Anything, mock.Anything) + state.AssertNotCalled(t, "SaveActive", mock.Anything, mock.Anything) + state.AssertNotCalled(t, "ClearRecoveryInhibition", mock.Anything, mock.Anything, mock.Anything, mock.Anything) +} + +func TestDeploy_Mockery_VolumeReplacementMarksUnsafe(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + + // The GC barrier must be held across resource acquisition through + // the durable ownership publication that protects the volume. + barrier := outmocks.NewMockGCBarrier(t) + gcLease := outmocks.NewMockGCLease(t) + gcLeaseHeld := false + ownershipSaves := 0 + barrier.EXPECT().AcquireShared(mock.Anything).Run(func(context.Context) { + gcLeaseHeld = true + }).Return(gcLease, nil).Once() + gcLease.EXPECT().Release().Run(func() { + gcLeaseHeld = false + }).Once() + + svcSpec := webService() + svcSpec.HTTP = nil + svcSpec.TCP = []domain.AppTCPInterface{{Entrypoint: "tcp", Port: 9000, Publish: "9000"}} + svcSpec.Readiness = domain.AppReadiness{Type: "none", Timeout: time.Second} + svcSpec.Volumes = []domain.AppVolume{{Name: "d", Path: "/data"}} + svcSpec.Secrets = map[string]string{} + rev := domain.AppDesiredRevision{ + Revision: "rev-1", App: "blog", + Spec: domain.AppSpec{Name: "blog", Env: map[string]string{}, Services: []domain.AppService{svcSpec}}, + } + + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + images.EXPECT().ResolveDigest(mock.Anything, svcSpec.Image).Return("sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", nil) + runtime.EXPECT().InspectImageVolumes(mock.Anything, svcSpec.Image).Return(nil, nil) + state.EXPECT().LoadCheckpoint(mock.Anything).Return(domain.AppStoreCheckpoint{}, nil) + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-uuid-blog", Volumes: []domain.AppOwnedVolume{{Name: "d", Service: "web", RuntimeName: "gordon-blog--web--vol--d", State: domain.AppResourceAttached}}}, nil).Times(4) + runtime.EXPECT().VolumeExists(mock.Anything, "gordon-blog--web--vol--d").Return(true, nil) + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", + Services: map[string]domain.AppEffectiveService{ + "web": {EffectiveRevision: "rev-0", Image: "img:0", Container: "c-old"}, + }, + }, true, nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + + var order []string + runtime.EXPECT().StopContainer(mock.Anything, "c-old", mock.Anything).RunAndReturn(func(context.Context, string, time.Duration) error { + order = append(order, "stop-old") + return nil + }).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-old", false).RunAndReturn(func(context.Context, string, bool) error { + order = append(order, "remove-old") + return nil + }).Once() + // The superseded writer's confirmed disappearance releases its claims. + state.EXPECT().ReleaseBackendBinds(mock.Anything, "blog", "c-old").Return(nil).Once() + expectNetworkProvision(runtime, "app-uuid-blog", 1) + // The volume-owning replacement must be inhibited before it can + // write, and released only after the new generation is published. + state.EXPECT().SaveRecoveryInhibition(mock.Anything, mock.MatchedBy(func(inhibition domain.AppRecoveryInhibition) bool { + return inhibition.App == "blog" && inhibition.Service == "web" && + inhibition.ContainerID == "c-old" && inhibition.Reason == domain.AppInhibitReplacementPending + })).RunAndReturn(func(context.Context, domain.AppRecoveryInhibition) error { + order = append(order, "inhibit") + return nil + }).Once() + state.EXPECT().ClearRecoveryInhibition(mock.Anything, "blog", "web", "c-old").Return(nil).Once() + runtime.EXPECT().CreateVolume(mock.Anything, "gordon-blog--web--vol--d", mock.MatchedBy(func(labels map[string]string) bool { + return labels[domain.LabelApp] == "blog" && + labels[domain.LabelAppID] == "app-uuid-blog" && + labels[domain.LabelAppService] == "web" && + labels[domain.LabelAppRevision] == "rev-1" + })).Run(func(context.Context, string, map[string]string) { + require.True(t, gcLeaseHeld, "volume creation must run inside the shared GC lease") + // The durable ownership reservation must already be written: + // a crash after this point leaves a protected record, never an + // unowned volume that prune could later adopt. + require.GreaterOrEqual(t, ownershipSaves, 1, + "ownership must be reserved before the runtime volume exists") + }).Return(nil).Once() + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).RunAndReturn(func(context.Context, *domain.ContainerConfig) (*domain.Container, error) { + order = append(order, "create-new") + return &domain.Container{ID: "c-new", Name: "web"}, nil + }).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-new").Return(nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-new", mock.Anything).Return([]domain.ContainerBackendBind{{ContainerPort: 9000, HostPort: 19000, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + state.EXPECT().RegisterBackendBinds(mock.Anything, mock.MatchedBy(func(claims []domain.AppListenerReservation) bool { + return len(claims) == 1 && claims[0].Port == 19000 && claims[0].Owner == domain.OwnerGordonBackend && claims[0].ContainerID == "c-new" + })).Return(nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{}}, true, nil) + state.EXPECT().SaveActive(mock.Anything, mock.Anything).Return(nil).Once() + restartUnsafeSaved := false + state.EXPECT().SaveOwnership(mock.Anything, mock.Anything).Return(nil).Run(func(_ context.Context, ownership domain.AppOwnership) { + require.True(t, gcLeaseHeld, "ownership publication must run inside the shared GC lease") + ownershipSaves++ + if ownership.Services["web"].RestartUnsafe { + restartUnsafeSaved = true + } + }) + + svc := deployment.NewService( + deployment.Deps{State: state, Runtime: runtime, Images: images, Secrets: secrets}, + zerowrap.Default(), + ).WithProbeDeps(deployment.NewProbeDeps(runtime)).WithGCBarrier(barrier) + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + require.NoError(t, err) + require.NotNil(t, result) + assert.Equal(t, []string{"inhibit", "stop-old", "remove-old", "create-new"}, order, + "the durable inhibition is written before the superseded writer can be stopped, and the writer is gone before the replacement can write") + assert.True(t, result.Services["web"].RestartUnsafe) + assert.True(t, restartUnsafeSaved, "the terminal ownership write must record the restart-unsafe service") + assert.GreaterOrEqual(t, ownershipSaves, 2, "the volume reservation and the terminal ownership write must both be durable") + assert.False(t, gcLeaseHeld, "the shared GC lease must be released after publication") + runtime.AssertNotCalled(t, "RemoveVolume", mock.Anything, mock.Anything, mock.Anything) +} + +// TestDeploy_Mockery_MixedTCPUDPPublishesBothLoopbacks proves a mixed +// TCP+UDP service publishes both protocols on 127.0.0.1 ephemeral, +// resolves them in one grouped inspection (same container port number +// on both protocols yields two distinct binds), and persists both maps +// with exact-protocol claims. +func TestDeploy_Mockery_MixedTCPUDPPublishesBothLoopbacks(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + svcSpec := webService() + svcSpec.HTTP = nil + svcSpec.TCP = []domain.AppTCPInterface{{Entrypoint: "tcp", Port: 9000, Publish: "9000"}} + svcSpec.UDP = []domain.AppUDPInterface{{Entrypoint: "udp", Port: 9000, Publish: "9000"}} + svcSpec.Readiness = domain.AppReadiness{Type: "none", Timeout: time.Second} + svcSpec.Secrets = map[string]string{} + rev := domain.AppDesiredRevision{ + Revision: "rev-1", App: "blog", + Spec: domain.AppSpec{Name: "blog", Env: map[string]string{}, Services: []domain.AppService{svcSpec}}, + } + + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + images.EXPECT().ResolveDigest(mock.Anything, svcSpec.Image).Return("sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", nil) + runtime.EXPECT().InspectImageVolumes(mock.Anything, svcSpec.Image).Return(nil, nil) + state.EXPECT().LoadCheckpoint(mock.Anything).Return(domain.AppStoreCheckpoint{}, nil) + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-blog"}, nil) + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", Services: map[string]domain.AppEffectiveService{}, + }, true, nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + expectNetworkProvision(runtime, "app-blog", 1) + + runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { + if len(cfg.PortPublishes) != 2 { + return false + } + byProto := map[domain.NetworkProtocol]domain.ContainerPortPublish{} + for _, publish := range cfg.PortPublishes { + if publish.HostIP != "127.0.0.1" || publish.HostPort != 0 || publish.ContainerPort != 9000 { + return false + } + byProto[publish.Protocol] = publish + } + return len(byProto) == 2 + })).Return(&domain.Container{ID: "c-new", Name: "web"}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-new").Return(nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-new", mock.MatchedBy(func(ports []domain.ContainerBackendPort) bool { + if len(ports) != 2 { + return false + } + seen := map[domain.ContainerBackendPort]bool{} + for _, port := range ports { + seen[port] = true + } + return seen[domain.ContainerBackendPort{ContainerPort: 9000, Protocol: domain.NetworkProtocolTCP}] && + seen[domain.ContainerBackendPort{ContainerPort: 9000, Protocol: domain.NetworkProtocolUDP}] + })).Return([]domain.ContainerBackendBind{ + {ContainerPort: 9000, HostPort: 19000, Protocol: domain.NetworkProtocolTCP}, + {ContainerPort: 9000, HostPort: 19001, Protocol: domain.NetworkProtocolUDP}, + }, nil).Once() + state.EXPECT().RegisterBackendBinds(mock.Anything, mock.MatchedBy(func(claims []domain.AppListenerReservation) bool { + if len(claims) != 2 { + return false + } + byProto := map[string]domain.AppListenerReservation{} + for _, claim := range claims { + if claim.Owner != domain.OwnerGordonBackend || claim.ContainerID != "c-new" || claim.IP != "127.0.0.1" { + return false + } + byProto[claim.Proto] = claim + } + return byProto["tcp"].Port == 19000 && byProto["udp"].Port == 19001 + })).Return(nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{}}, true, nil) + state.EXPECT().SaveActive(mock.Anything, mock.MatchedBy(func(a domain.AppActive) bool { + svc := a.Services["web"] + return svc.BackendBinds[9000] == 19000 && svc.UDPBackendBinds[9000] == 19001 + })).Return(nil).Once() + state.EXPECT().SaveOwnership(mock.Anything, mock.Anything).Return(nil).Once() + + svc := deployment.NewService( + deployment.Deps{State: state, Runtime: runtime, Images: images, Secrets: secrets}, + zerowrap.Default(), + ).WithProbeDeps(deployment.NewProbeDeps(runtime)) + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + require.NoError(t, err) + require.NotNil(t, result) + assert.Equal(t, "deployed", result.Services["web"].Result) + assert.Equal(t, map[int]int{9000: 19000}, result.Services["web"].BackendBinds) + assert.Equal(t, map[int]int{9000: 19001}, result.Services["web"].UDPBackendBinds) +} + +// TestDeploy_Mockery_InjectsAppEnvAndSecrets proves the captured revision's +// app-wide public env reaches every service container alongside that +// service's own resolved secrets, sorted and deterministic. Public-only +// and empty-value cases are covered; resolved secret values never enter +// saved state (SaveActive/SaveOwnership carry no KEY=value payloads). +func TestDeploy_Mockery_InjectsAppEnvAndSecrets(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + svcSpec := webService() + svcSpec.HTTP = nil + svcSpec.Readiness = domain.AppReadiness{} + rev := domain.AppDesiredRevision{ + Revision: "rev-1", App: "blog", + Spec: domain.AppSpec{ + Name: "blog", + Env: map[string]string{"APP_ENV": "production", "EMPTY_OK": ""}, + Services: []domain.AppService{ + svcSpec, + { + Name: "worker", Image: "docker.io/example/worker:1.4.2", + StopGrace: 10 * time.Second, + Readiness: domain.AppReadiness{Timeout: 30 * time.Second}, + Secrets: map[string]string{}, + }, + }, + }, + } + + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + images.EXPECT().ResolveDigest(mock.Anything, rev.Spec.Services[0].Image).Return("sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", nil).Once() + images.EXPECT().ResolveDigest(mock.Anything, rev.Spec.Services[1].Image).Return("sha256:"+strings.Repeat("b", 64), nil).Once() + secrets.EXPECT().GetSecret(mock.Anything, "gordon/apps/app-blog/web/database-url").Return("s3cr3t", nil).Twice() + runtime.EXPECT().InspectImageVolumes(mock.Anything, rev.Spec.Services[0].Image).Return(nil, nil).Once() + runtime.EXPECT().InspectImageVolumes(mock.Anything, rev.Spec.Services[1].Image).Return(nil, nil).Once() + state.EXPECT().LoadCheckpoint(mock.Anything).Return(domain.AppStoreCheckpoint{}, nil).Once() + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-blog"}, nil) + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", Services: map[string]domain.AppEffectiveService{}, + }, true, nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + expectNetworkProvision(runtime, "app-blog", 2) + + var webEnv, workerEnv []string + runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { + if len(cfg.Env) != 3 { + return false + } + webEnv = append([]string(nil), cfg.Env...) + return true + })).Return(&domain.Container{ID: "c-web", Name: "web"}, nil).Once() + runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { + workerEnv = append([]string(nil), cfg.Env...) + return len(cfg.Env) == 2 + })).Return(&domain.Container{ID: "c-worker", Name: "worker"}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, mock.Anything).Return(nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", Services: map[string]domain.AppEffectiveService{}, + }, true, nil) + state.EXPECT().SaveActive(mock.Anything, mock.MatchedBy(func(a domain.AppActive) bool { + for _, svc := range a.Services { + for _, kv := range []string{svc.Image, svc.Digest, svc.Container} { + if kv == "s3cr3t" || kv == "DATABASE_URL=s3cr3t" { + return false + } + } + } + return true + })).Return(nil) + state.EXPECT().SaveOwnership(mock.Anything, mock.Anything).Return(nil) + + svc := deployment.NewService( + deployment.Deps{State: state, Runtime: runtime, Images: images, Secrets: secrets}, + zerowrap.Default(), + ).WithProbeDeps(deployment.NewProbeDeps(runtime)) + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + require.NoError(t, err) + require.NotNil(t, result) + assert.Equal(t, []string{"APP_ENV=production", "DATABASE_URL=s3cr3t", "EMPTY_OK="}, webEnv) + assert.Equal(t, []string{"APP_ENV=production", "EMPTY_OK="}, workerEnv) +} + +// TestStart_RunningContainerRefreshesShiftedBinds proves boot/start bind +// verification: a running container whose ephemeral loopback bind shifted +// (native restart while the daemon was away) gets its ACTIVE record +// updated before the proxy can dial the stale bind. The ACTIVE +// container is never stopped or removed by re-inspection. +func TestStart_RunningContainerRefreshesShiftedBinds(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) + state.EXPECT().SaveIntent(mock.Anything, mock.Anything).Return(nil) + active := domain.AppActive{ + App: "blog", + Services: map[string]domain.AppEffectiveService{ + "web": { + EffectiveRevision: "rev-1", + Image: rev.Spec.Services[0].Image, + Digest: "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + Container: "c-old", + Spec: rev.Spec.Services[0], + BackendBinds: map[int]int{8080: 32770}, + }, + }, + } + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + runtime.EXPECT().IsContainerRunning(mock.Anything, "c-old").Return(true, nil) + // Native restart shifted the ephemeral bind: 32770 -> 32771. + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-old", mock.Anything).Return([]domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 32771, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + state.EXPECT().RegisterBackendBinds(mock.Anything, mock.MatchedBy(func(claims []domain.AppListenerReservation) bool { + return len(claims) == 1 && claims[0].Port == 32771 && claims[0].Owner == domain.OwnerGordonBackend && claims[0].ContainerID == "c-old" + })).Return(nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().SaveActive(mock.Anything, mock.MatchedBy(func(a domain.AppActive) bool { + return a.Services["web"].BackendBinds[8080] == 32771 + })).Return(nil).Once() + + svc := deployment.NewService( + deployment.Deps{State: state, Runtime: runtime, Images: images, Secrets: secrets}, + zerowrap.Default(), + ).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 200, nil }, + func(context.Context, string) error { return nil }, + )) + result, err := svc.Start(ctx, "blog", "") + require.NoError(t, err) + require.NotNil(t, result) + require.Contains(t, result.Services, "web") + assert.Equal(t, "deployed", result.Services["web"].Result) + assert.Equal(t, map[int]int{8080: 32771}, result.Services["web"].BackendBinds) + runtime.AssertNotCalled(t, "StopContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, mock.Anything, mock.Anything) +} + +// TestStart_BindVerificationFailureFailsClosed proves an unverified bind +// is never served: when re-inspection fails on a running container, the +// service step fails (fail-closed) without touching the container, and +// the stale recorded bind is withdrawn from ACTIVE so the next traffic +// rebuild cannot re-expose a dead or recycled loopback port. +func TestStart_BindVerificationFailureFailsClosed(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) + state.EXPECT().SaveIntent(mock.Anything, mock.Anything).Return(nil) + active := domain.AppActive{ + App: "blog", + Services: map[string]domain.AppEffectiveService{ + "web": { + EffectiveRevision: "rev-1", + Image: rev.Spec.Services[0].Image, + Digest: "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + Container: "c-old", + Spec: rev.Spec.Services[0], + BackendBinds: map[int]int{8080: 32770}, + }, + }, + } + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + runtime.EXPECT().IsContainerRunning(mock.Anything, "c-old").Return(true, nil) + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-old", mock.Anything).Return(nil, assert.AnError).Once() + + traffic := &recordingTraffic{} + svc := deployment.NewService( + deployment.Deps{State: state, Runtime: runtime, Images: images, Secrets: secrets, Traffic: traffic}, + zerowrap.Default(), + ) + result, err := svc.Start(ctx, "blog", "") + require.NoError(t, err) + require.NotNil(t, result) + assert.Equal(t, "failed", result.Services["web"].Result) + runtime.AssertNotCalled(t, "StopContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, mock.Anything, mock.Anything) + // The stale bind is withdrawn through the canonical boundary + // (fail-closed projection); deployment writes no ACTIVE binds itself. + assert.Equal(t, []string{"blog/web", "blog/web"}, traffic.withdrawn()) + state.AssertNotCalled(t, "SaveActive", mock.Anything, mock.Anything) +} + +// TestRestart_MixedServiceRefreshesBothProtocols proves a runtime restart +// re-inspects TCP and UDP binds together: shifted ephemeral ports on +// both protocols persist to ACTIVE and surface in the result. +func TestRestart_MixedServiceRefreshesBothProtocols(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + svcSpec := webService() + svcSpec.HTTP = nil + svcSpec.TCP = []domain.AppTCPInterface{{Entrypoint: "tcp", Port: 9000, Publish: "9000"}} + svcSpec.UDP = []domain.AppUDPInterface{{Entrypoint: "udp", Port: 9000, Publish: "9000"}} + svcSpec.Readiness = domain.AppReadiness{Type: "none", Timeout: time.Second} + svcSpec.Secrets = map[string]string{} + active := domain.AppActive{ + App: "blog", + Services: map[string]domain.AppEffectiveService{ + "web": { + EffectiveRevision: "rev-1", + Image: svcSpec.Image, + Digest: "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + Container: "c-old", + Spec: svcSpec, + BackendBinds: map[int]int{9000: 32770}, + UDPBackendBinds: map[int]int{9000: 32780}, + }, + }, + } + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + runtime.EXPECT().RestartContainer(mock.Anything, "c-old", mock.Anything).Return(nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-old", mock.Anything).Return([]domain.ContainerBackendBind{ + {ContainerPort: 9000, HostPort: 32771, Protocol: domain.NetworkProtocolTCP}, + {ContainerPort: 9000, HostPort: 32781, Protocol: domain.NetworkProtocolUDP}, + }, nil).Once() + state.EXPECT().RegisterBackendBinds(mock.Anything, mock.MatchedBy(func(claims []domain.AppListenerReservation) bool { + if len(claims) != 2 { + return false + } + byProto := map[string]int{} + for _, claim := range claims { + byProto[claim.Proto] = claim.Port + } + return byProto["tcp"] == 32771 && byProto["udp"] == 32781 + })).Return(nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().SaveActive(mock.Anything, mock.MatchedBy(func(a domain.AppActive) bool { + svc := a.Services["web"] + return svc.BackendBinds[9000] == 32771 && svc.UDPBackendBinds[9000] == 32781 + })).Return(nil).Once() + + svc := deployment.NewService( + deployment.Deps{State: state, Runtime: runtime, Images: images, Secrets: secrets}, + zerowrap.Default(), + ) + result, err := svc.Restart(ctx, "blog", "", "") + require.NoError(t, err) + require.NotNil(t, result) + assert.Equal(t, "deployed", result.Services["web"].Result) + assert.Equal(t, map[int]int{9000: 32771}, result.Services["web"].BackendBinds) + assert.Equal(t, map[int]int{9000: 32781}, result.Services["web"].UDPBackendBinds) +} diff --git a/internal/usecase/deployment/deploy_unchanged_test.go b/internal/usecase/deployment/deploy_unchanged_test.go new file mode 100644 index 000000000..eaf6c937a --- /dev/null +++ b/internal/usecase/deployment/deploy_unchanged_test.go @@ -0,0 +1,264 @@ +package deployment_test + +import ( + "context" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/out/appstate" + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/deployment" +) + +// unchangedSeed shapes the ACTIVE record and the desired revision of an app +// whose defaults match exactly (same image, digest, spec, env, binds). +type unchangedSeed struct { + binds map[int]int + desiredEnv map[string]string + inhibitedBy string +} + +// seedUnchangedApp seeds an app whose ACTIVE service runs exactly what the +// desired revision would create, unless the seed alters one input. +func seedUnchangedApp(t *testing.T, ctx context.Context, store *appstate.Store, seeds ...unchangedSeed) { + t.Helper() + seed := unchangedSeed{binds: map[int]int{8080: 18080}} + if len(seeds) > 0 { + seed = seeds[0] + } + spec := webService() + spec.Secrets = map[string]string{} + spec.Readiness = domain.AppReadiness{Type: domain.AppReadinessHTTP, Path: "/healthz", Timeout: time.Second} + activeRev := testRevision("blog", spec) + activeRev.Revision = "rev-0" + require.NoError(t, store.SaveIntent(ctx, domain.AppStopIntent{App: "blog"})) + require.NoError(t, store.SaveActive(ctx, domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": { + Container: "c-old", EffectiveRevision: "rev-0", Image: spec.Image, ActivatedBy: "op-created", + Digest: restartTestDigest, Spec: activeRev.Spec.Services[0], BackendBinds: seed.binds, + }, + }})) + require.NoError(t, store.SaveOwnership(ctx, domain.AppOwnership{App: "blog", ID: "app-blog"})) + seedRevision(t, ctx, store, "intent-0", "", activeRev) + desired := testRevision("blog", spec) + if seed.desiredEnv != nil { + desired.Spec.Env = seed.desiredEnv + } + seedRevision(t, ctx, store, "intent-1", "rev-0", desired) + if seed.inhibitedBy != "" { + require.NoError(t, store.SaveRecoveryInhibition(ctx, domain.AppRecoveryInhibition{ + App: "blog", Service: "web", ContainerID: "c-old", Reason: domain.AppInhibitReplacementPending, Operation: seed.inhibitedBy, + })) + } +} + +// expectReplacement wires one full replacement of c-old by c-new. +func expectReplacement(runtime *outmocks.MockContainerRuntime) { + expectNetworkProvision(runtime, "app-blog", 1) + runtime.EXPECT().StopContainer(mock.Anything, "c-old", mock.Anything).Return(nil).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-old", false).Return(nil).Once() + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).Return(&domain.Container{ID: "c-new", Name: "web"}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-new").Return(nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-new", mock.Anything).Return( + []domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 18081, Protocol: domain.NetworkProtocolTCP}}, nil).Once() +} + +// unchangedDeployService wires a deploy of an app whose ACTIVE service +// already matches the desired revision. +func unchangedDeployService(t *testing.T, seeds ...unchangedSeed) (*deployment.Service, *outmocks.MockContainerRuntime, *outmocks.MockImageResolver, context.Context, *appstate.Store) { + t.Helper() + ctx := context.Background() + store := newTestStore(t) + runtime := outmocks.NewMockContainerRuntime(t) + images := outmocks.NewMockImageResolver(t) + secrets := outmocks.NewMockSecretProvider(t) + seedUnchangedApp(t, ctx, store, seeds...) + images.EXPECT().ResolveDigest(mock.Anything, "docker.io/example/web:1.4.2").Return(restartTestDigest, nil).Once() + runtime.EXPECT().InspectImageVolumes(mock.Anything, "docker.io/example/web:1.4.2").Return(nil, nil).Once() + svc := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 200, nil }, + func(context.Context, string) error { return nil }, + )) + return svc, runtime, images, ctx, store +} + +// TestDeploy_SkipsServiceAlreadyRunningTheSameImage proves a deploy whose +// resolved digest, spec, and env match the running service keeps the +// container and reports it unchanged instead of recreating it. +func TestDeploy_SkipsServiceAlreadyRunningTheSameImage(t *testing.T) { + svc, runtime, _, ctx, store := unchangedDeployService(t) + runtime.EXPECT().InspectContainer(mock.Anything, "c-old"). + Return(&domain.Container{ID: "c-old", Status: string(domain.ContainerStatusRunning)}, nil).Once() + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog", Op: "op-same"}) + require.NoError(t, err) + assert.Equal(t, deployment.ServiceResultUnchanged, result.Services["web"].Result) + assert.Equal(t, "c-old", result.Services["web"].After, "the running container is kept") + runtime.AssertNotCalled(t, "StopContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) + + op, err := store.LoadOperation(ctx, "blog", "op-same") + require.NoError(t, err) + assert.Equal(t, domain.AppOutcomeSuccess, op.Outcome) + step, found := opStep(op, "service.web.replace") + require.True(t, found) + assert.Equal(t, domain.AppStepSucceeded, step.State) + assert.Contains(t, step.Detail, "unchanged: already running docker.io/example/web:1.4.2") + + active, ok, err := store.LoadActive(ctx, "blog") + require.NoError(t, err) + require.True(t, ok) + assert.Equal(t, "c-old", active.Services["web"].Container) + assert.Equal(t, map[int]int{8080: 18080}, active.Services["web"].BackendBinds, "the kept container stays routable") + assert.Equal(t, "op-created", active.Services["web"].ActivatedBy, "a kept container keeps its original activation") + assert.True(t, active.Converged, "an unchanged deploy still converges ACTIVE on the deployed revision") +} + +// TestDeploy_ReplacesSameDigestWhenTheServiceIsNotServing proves the skip +// never keeps a container that is not fully published: missing routing +// binds (a withdrawn generation), a recovery inhibition left by a failed +// replacement, or a changed app env all force a replacement. +func TestDeploy_ReplacesSameDigestWhenTheServiceIsNotServing(t *testing.T) { + tests := []struct { + name string + seed unchangedSeed + }{ + {name: "withdrawn binds", seed: unchangedSeed{}}, + {name: "recovery inhibited", seed: unchangedSeed{binds: map[int]int{8080: 18080}, inhibitedBy: "op-failed"}}, + {name: "app env changed", seed: unchangedSeed{binds: map[int]int{8080: 18080}, desiredEnv: map[string]string{"MODE": "prod"}}}, + } + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + svc, runtime, _, ctx, _ := unchangedDeployService(t, tc.seed) + expectReplacement(runtime) + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog", Op: "op-replace"}) + require.NoError(t, err) + assert.Equal(t, "deployed", result.Services["web"].Result) + assert.Equal(t, "c-new", result.Services["web"].After) + }) + } +} + +// TestDeploy_ReplacesWhenTheRunningContainerIsNotRunning proves the skip +// never trusts ACTIVE alone: a stopped or missing container is replaced. +func TestDeploy_ReplacesWhenTheRunningContainerIsNotRunning(t *testing.T) { + svc, runtime, _, ctx, _ := unchangedDeployService(t) + runtime.EXPECT().InspectContainer(mock.Anything, "c-old"). + Return(&domain.Container{ID: "c-old", Status: "exited"}, nil).Once() + expectReplacement(runtime) + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog", Op: "op-exited"}) + require.NoError(t, err) + assert.Equal(t, "deployed", result.Services["web"].Result) + assert.Equal(t, "c-new", result.Services["web"].After) +} + +// TestDeploy_ReplacesWhenTheDigestChanged proves a new image digest behind +// the same tag is always deployed. +func TestDeploy_ReplacesWhenTheDigestChanged(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + runtime := outmocks.NewMockContainerRuntime(t) + images := outmocks.NewMockImageResolver(t) + secrets := outmocks.NewMockSecretProvider(t) + seedUnchangedApp(t, ctx, store) + newDigest := "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb" + images.EXPECT().ResolveDigest(mock.Anything, "docker.io/example/web:1.4.2").Return(newDigest, nil).Once() + runtime.EXPECT().InspectImageVolumes(mock.Anything, "docker.io/example/web:1.4.2").Return(nil, nil).Once() + expectNetworkProvision(runtime, "app-blog", 1) + runtime.EXPECT().StopContainer(mock.Anything, "c-old", mock.Anything).Return(nil).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-old", false).Return(nil).Once() + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).Return(&domain.Container{ID: "c-new", Name: "web"}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-new").Return(nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-new", mock.Anything).Return( + []domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 18081, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + + svc := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 200, nil }, + func(context.Context, string) error { return nil }, + )) + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog", Op: "op-new-digest"}) + require.NoError(t, err) + assert.Equal(t, "deployed", result.Services["web"].Result) + runtime.AssertNotCalled(t, "InspectContainer", mock.Anything, "c-old") + + active, ok, err := store.LoadActive(ctx, "blog") + require.NoError(t, err) + require.True(t, ok) + assert.Equal(t, newDigest, active.Services["web"].Digest) +} + +// TestDeploy_RecreatesWhenASecretValueChanged proves deploy applies new +// secret values: a running container created with an old value is replaced, +// and one created with the current value is kept. +func TestDeploy_RecreatesWhenASecretValueChanged(t *testing.T) { + tests := []struct { + name string + runningEnv []string + wantResult string + }{ + {name: "secret changed", runningEnv: []string{"DB_PASSWORD=old", "PATH=/bin"}, wantResult: "deployed"}, + {name: "secret current", runningEnv: []string{"DB_PASSWORD=new", "PATH=/bin"}, wantResult: deployment.ServiceResultUnchanged}, + } + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + svc, runtime, secrets, ctx := secretDeployService(t) + secrets.EXPECT().GetSecret(mock.Anything, mock.Anything).Return("new", nil) + runtime.EXPECT().InspectContainer(mock.Anything, "c-old"). + Return(&domain.Container{ID: "c-old", Status: string(domain.ContainerStatusRunning), Env: tc.runningEnv}, nil).Once() + if tc.wantResult == "deployed" { + expectReplacement(runtime) + } + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog", Op: "op-secret"}) + require.NoError(t, err) + assert.Equal(t, tc.wantResult, result.Services["web"].Result) + }) + } +} + +// secretDeployService wires an unchanged app whose service reads one secret. +func secretDeployService(t *testing.T) (*deployment.Service, *outmocks.MockContainerRuntime, *outmocks.MockSecretProvider, context.Context) { + t.Helper() + ctx := context.Background() + store := newTestStore(t) + runtime := outmocks.NewMockContainerRuntime(t) + images := outmocks.NewMockImageResolver(t) + secrets := outmocks.NewMockSecretProvider(t) + spec := webService() + spec.Secrets = map[string]string{"DB_PASSWORD": "db-password"} + spec.Readiness = domain.AppReadiness{Type: domain.AppReadinessHTTP, Path: "/healthz", Timeout: time.Second} + activeRev := testRevision("blog", spec) + activeRev.Revision = "rev-0" + require.NoError(t, store.SaveIntent(ctx, domain.AppStopIntent{App: "blog"})) + require.NoError(t, store.SaveActive(ctx, domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": { + Container: "c-old", EffectiveRevision: "rev-0", Image: spec.Image, ActivatedBy: "op-created", + Digest: restartTestDigest, Spec: activeRev.Spec.Services[0], BackendBinds: map[int]int{8080: 18080}, + }, + }})) + require.NoError(t, store.SaveOwnership(ctx, domain.AppOwnership{App: "blog", ID: "app-blog"})) + seedRevision(t, ctx, store, "intent-0", "", activeRev) + seedRevision(t, ctx, store, "intent-1", "rev-0", testRevision("blog", spec)) + images.EXPECT().ResolveDigest(mock.Anything, "docker.io/example/web:1.4.2").Return(restartTestDigest, nil).Once() + runtime.EXPECT().InspectImageVolumes(mock.Anything, "docker.io/example/web:1.4.2").Return(nil, nil).Once() + svc := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 200, nil }, + func(context.Context, string) error { return nil }, + )) + return svc, runtime, secrets, ctx +} diff --git a/internal/usecase/deployment/device_policy_internal_test.go b/internal/usecase/deployment/device_policy_internal_test.go new file mode 100644 index 000000000..b6c7b8c20 --- /dev/null +++ b/internal/usecase/deployment/device_policy_internal_test.go @@ -0,0 +1,305 @@ +package deployment + +import ( + "context" + "errors" + "fmt" + "testing" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" +) + +func devicePolicyFor(apps, services []string) domain.AppDevicePolicy { + return domain.AppDevicePolicy{ + Name: "test_gpu", + CDI: []string{"example.com/gpu=GPU-test-uuid"}, + AllowedApps: apps, + AllowedServices: services, + } +} + +func TestResolveServiceDevices_ExactResolution(t *testing.T) { + svc := NewService(Deps{}, zerowrap.Default()). + WithDevicePolicies(map[string]domain.AppDevicePolicy{ + "test_gpu": devicePolicyFor([]string{"blog"}, []string{"web"}), + }) + + got, err := svc.resolveServiceDevices("blog", domain.AppService{ + Name: "web", + Devices: []string{"test_gpu"}, + }) + require.NoError(t, err) + assert.Equal(t, []string{"example.com/gpu=GPU-test-uuid"}, got) +} + +func TestResolveServiceDevices_EmptyIsNil(t *testing.T) { + svc := NewService(Deps{}, zerowrap.Default()) + got, err := svc.resolveServiceDevices("blog", domain.AppService{Name: "web"}) + require.NoError(t, err) + assert.Nil(t, got) +} + +func TestResolveServiceDevices_UnknownAndRevokedPolicy(t *testing.T) { + spec := domain.AppService{ + Name: "web", + Devices: []string{"test_gpu"}, + } + + t.Run("unknown policy", func(t *testing.T) { + svc := NewService(Deps{}, zerowrap.Default()) + _, err := svc.resolveServiceDevices("blog", spec) + require.ErrorIs(t, err, domain.ErrDevicePolicy) + assert.Contains(t, err.Error(), "blog") + assert.Contains(t, err.Error(), "web") + assert.Contains(t, err.Error(), "test_gpu") + }) + + t.Run("revoked policy", func(t *testing.T) { + svc := NewService(Deps{}, zerowrap.Default()). + WithDevicePolicies(map[string]domain.AppDevicePolicy{ + "test_gpu": devicePolicyFor([]string{"blog"}, []string{"web"}), + }) + _, err := svc.resolveServiceDevices("blog", spec) + require.NoError(t, err) + + svc.SetDevicePolicies(nil) + _, err = svc.resolveServiceDevices("blog", spec) + require.ErrorIs(t, err, domain.ErrDevicePolicy) + }) + + t.Run("refused policy omits device IDs", func(t *testing.T) { + svc := NewService(Deps{}, zerowrap.Default()). + WithDevicePolicies(map[string]domain.AppDevicePolicy{ + "test_gpu": devicePolicyFor([]string{"other"}, []string{"web"}), + }) + _, err := svc.resolveServiceDevices("blog", spec) + require.ErrorIs(t, err, domain.ErrDevicePolicy) + assert.NotContains(t, err.Error(), "GPU-test-uuid", "errors must never leak CDI IDs") + }) +} + +// TestCreateAndStart_RefusesRevokedDeviceBeforeRuntimeMutation proves the +// immediate-pre-create re-resolution fails before any runtime call when the +// device policy is missing or revoked. +func TestCreateAndStart_RefusesRevokedDeviceBeforeRuntimeMutation(t *testing.T) { + runtime := outmocks.NewMockContainerRuntime(t) + svc := NewService(Deps{Runtime: runtime}, zerowrap.Default()) + p := pinnedService{ + name: "web", + spec: domain.AppService{ + Name: "web", + Devices: []string{"test_gpu"}, + }, + } + + _, err := svc.createAndStart(context.Background(), "blog", "rev-1", p, "op-1", nil) + + require.ErrorIs(t, err, domain.ErrDevicePolicy) + assert.Contains(t, err.Error(), "test_gpu") + runtime.AssertNotCalled(t, "CreateContainer") + runtime.AssertNotCalled(t, "CreateVolume") + runtime.AssertNotCalled(t, "StartContainer") + runtime.AssertNotCalled(t, "ConnectContainerToNetwork") + runtime.AssertExpectations(t) +} + +// TestCreateAndStart_PassesResolvedCDIIDsToRuntime proves the deploy path +// carries authorized policy resolution into ContainerConfig.CDIDevices: +// only the granted service receives IDs, helpers and ordinary services +// receive none. +func TestCreateAndStart_PassesResolvedCDIIDsToRuntime(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + + private := domain.AppPrivateNetworkName("gordon", "app-1") + state.EXPECT().LoadOwnership(mock.Anything, "blog"). + Return(domain.AppOwnership{App: "blog", ID: "app-1"}, nil).Once() + runtime.EXPECT().ListNetworks(mock.Anything).Return([]*domain.NetworkInfo{}, nil).Once() + runtime.EXPECT().CreateNetwork(mock.Anything, private, mock.Anything).Return(nil).Once() + runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { + return assert.Equal(t, []string{"example.com/gpu=GPU-test-uuid"}, cfg.CDIDevices) + })).Return(&domain.Container{ID: "c-1", Name: "web"}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-1").Return(nil).Once() + + svc := NewService(Deps{State: state, Runtime: runtime, Networks: NetworkConfig{Prefix: "gordon"}}, zerowrap.Default()). + WithDevicePolicies(map[string]domain.AppDevicePolicy{ + "test_gpu": devicePolicyFor([]string{"blog"}, []string{"web"}), + }) + candidate, err := svc.createAndStart(ctx, "blog", "rev-1", pinnedService{ + name: "web", + spec: domain.AppService{Name: "web", Devices: []string{"test_gpu"}}, + }, "op-1", nil) + created := candidate.Container + require.NoError(t, err) + require.NotNil(t, created) + runtime.AssertExpectations(t) +} + +// TestCreateContainer_DeviceBearingRuntimeErrorIsRedactedNotDevicePolicy +// proves a CreateContainer failure on a device-bearing service is reported +// as a generic redacted runtime error, never as a device policy violation, +// and never leaks the resolved CDI IDs embedded in the runtime error. +func TestCreateContainer_DeviceBearingRuntimeErrorIsRedactedNotDevicePolicy(t *testing.T) { + const deviceID = "example.com/gpu=GPU-test-uuid" + runtime := outmocks.NewMockContainerRuntime(t) + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything). + Return(nil, fmt.Errorf("device %s: permission denied", deviceID)).Once() + svc := NewService(Deps{Runtime: runtime}, zerowrap.Default()) + + _, err := svc.createContainer(context.Background(), "web", &domain.ContainerConfig{ + CDIDevices: []string{deviceID}, + }) + + require.Error(t, err) + require.NotErrorIs(t, err, domain.ErrDevicePolicy, "a runtime failure is not a policy refusal") + assert.NotContains(t, err.Error(), deviceID, "errors must never leak resolved CDI IDs") + assert.NotContains(t, err.Error(), "permission denied", "no runtime text may reach the caller") + runtime.AssertExpectations(t) +} + +// TestRestartOneService_RevokedDeviceFailsWithoutRuntimeMutation proves an +// in-place restart fails closed when the device policy is missing/revoked, +// before touching the runtime. +func TestRestartOneService_RevokedDeviceFailsWithoutRuntimeMutation(t *testing.T) { + runtime := outmocks.NewMockContainerRuntime(t) + svc := NewService(Deps{Runtime: runtime}, zerowrap.Default()) + eff := domain.AppEffectiveService{ + Container: "c1", + Spec: domain.AppService{ + Name: "web", + Devices: []string{"test_gpu"}, + }, + } + + op := &domain.AppOperation{Op: "op-1", Steps: []domain.AppOperationStep{{ + ID: "service.web.restart", State: domain.AppStepPending, Before: eff.Container, + }}} + result := svc.restartOneService(context.Background(), "blog", "op-1", "web", eff, op, 0) + + assert.Equal(t, domain.AppStepFailed, op.Steps[0].State) + assert.Contains(t, op.Steps[0].Error, "test_gpu") + assert.Equal(t, "failed", result.Result) + runtime.AssertNotCalled(t, "RestartContainer") + runtime.AssertExpectations(t) +} + +// TestEnsureServiceRunning_RevokedDeviceFailsWithoutRuntimeMutation proves +// the boot/start recovery path fails closed before starting a container +// when the device policy is missing/revoked. +func TestEnsureServiceRunning_RevokedDeviceFailsWithoutRuntimeMutation(t *testing.T) { + runtime := outmocks.NewMockContainerRuntime(t) + svc := NewService(Deps{Runtime: runtime}, zerowrap.Default()) + eff := domain.AppEffectiveService{ + Container: "c1", + Spec: domain.AppService{ + Name: "web", + Devices: []string{"test_gpu"}, + }, + } + op := &domain.AppOperation{Op: "op-1", Steps: []domain.AppOperationStep{{ + ID: "service.web.start", State: domain.AppStepPending, Before: eff.Container, + }}} + result := &LifecycleResult{Services: map[string]ServiceResult{}} + + svc.ensureServiceRunning(context.Background(), "blog", "op-1", "web", eff, op, 0, result) + + assert.Equal(t, domain.AppStepFailed, op.Steps[0].State) + assert.Contains(t, op.Steps[0].Error, "test_gpu") + runtime.AssertNotCalled(t, "StartContainer") + runtime.AssertExpectations(t) +} + +// TestPreflightServices_ProbesEngineOncePerRevision proves preflight calls +// SupportsCDIDevices exactly once for a multi-service device-bearing +// revision, and that pinned services carry their logical device requests. +func TestPreflightServices_ProbesEngineOncePerRevision(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + images := outmocks.NewMockImageResolver(t) + secrets := outmocks.NewMockSecretProvider(t) + + rev := domain.AppDesiredRevision{ + Revision: "rev-1", + App: "blog", + Spec: domain.AppSpec{ + Name: "blog", + Env: map[string]string{}, + Services: []domain.AppService{ + {Name: "web", Image: "img:1", Devices: []string{"test_gpu"}}, + {Name: "worker", Image: "img:1", Devices: []string{"test_gpu"}}, + }, + }, + } + state.EXPECT().LoadOwnership(mock.Anything, "blog"). + Return(domain.AppOwnership{App: "blog", ID: "app-1"}, nil).Once() + images.EXPECT().ResolveDigest(mock.Anything, "img:1").Return("sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", nil).Twice() + runtime.EXPECT().InspectImageVolumes(mock.Anything, "img:1").Return(nil, nil).Twice() + runtime.EXPECT().SupportsCDIDevices(mock.Anything).Return(nil).Once() + state.EXPECT().LoadCheckpoint(mock.Anything).Return(domain.AppStoreCheckpoint{}, nil).Once() + + svc := NewService(Deps{State: state, Runtime: runtime, Images: images, Secrets: secrets}, zerowrap.Default()). + WithDevicePolicies(map[string]domain.AppDevicePolicy{ + "test_gpu": { + Name: "test_gpu", + CDI: []string{"example.com/gpu=GPU-test-uuid"}, + AllowedApps: []string{"blog"}, + AllowedServices: []string{"web", "worker"}, + }, + }) + pinned, err := svc.preflightServices(ctx, "blog", rev, "") + require.NoError(t, err) + require.Len(t, pinned, 2) + runtime.AssertExpectations(t) +} + +// TestCreateContainer_RuntimeUnsupportedSurvivesRedaction proves the +// engine-unsupported sentinel from a device-bearing create survives the +// host-inventory redaction so callers map it to the structured +// runtime-unsupported envelope. +func TestCreateContainer_RuntimeUnsupportedSurvivesRedaction(t *testing.T) { + runtime := outmocks.NewMockContainerRuntime(t) + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything). + Return(nil, domain.ErrRuntimeUnsupported).Once() + svc := NewService(Deps{Runtime: runtime}, zerowrap.Default()) + + _, err := svc.createContainer(context.Background(), "web", &domain.ContainerConfig{ + CDIDevices: []string{"example.com/gpu=GPU-test-uuid"}, + }) + + require.Error(t, err) + require.ErrorIs(t, err, domain.ErrRuntimeUnsupported) + assert.NotContains(t, err.Error(), "GPU-test-uuid", "the sanitized error must not echo CDI IDs") + runtime.AssertExpectations(t) +} + +// TestSanitizedEngineProbeError proves the capability-probe error keeps +// cancellation meaningful, collapses every other cause to the unsupported +// sentinel alone, and never echoes the adapter text. +func TestSanitizedEngineProbeError(t *testing.T) { + t.Run("cancellation is preserved and reported as cancellation", func(t *testing.T) { + err := sanitizedEngineProbeError("web", fmt.Errorf("dial unix /run/podman/podman.sock: %w", context.Canceled)) + require.ErrorIs(t, err, context.Canceled) + assert.False(t, errors.Is(err, domain.ErrRuntimeUnsupported), "an aborted probe is not an incapable engine") + assert.NotContains(t, err.Error(), "podman.sock") + }) + + t.Run("deadline is preserved", func(t *testing.T) { + err := sanitizedEngineProbeError("web", fmt.Errorf("probe: %w", context.DeadlineExceeded)) + require.ErrorIs(t, err, context.DeadlineExceeded) + assert.NotContains(t, err.Error(), "probe") + }) + + t.Run("other causes collapse to the unsupported sentinel", func(t *testing.T) { + err := sanitizedEngineProbeError("web", errors.New("dial unix /run/podman/podman.sock: connection refused")) + require.ErrorIs(t, err, domain.ErrRuntimeUnsupported) + assert.NotContains(t, err.Error(), "podman.sock") + }) +} diff --git a/internal/usecase/deployment/diagnostics.go b/internal/usecase/deployment/diagnostics.go new file mode 100644 index 000000000..e1e8520c0 --- /dev/null +++ b/internal/usecase/deployment/diagnostics.go @@ -0,0 +1,77 @@ +package deployment + +import ( + "bufio" + "context" + "io" + "strings" + + "github.com/bnema/gordon/internal/domain" +) + +// logTail returns the bounded recent log tail of the failing container. +// The read is bounded by the caller's deadline and by an internal cap so +// a slow or hung log stream cannot stall the operation. +func (s *Service) logTail(ctx context.Context, containerID string) []string { + if containerID == "" || ctx.Err() != nil { + return nil + } + tailCtx, cancel := context.WithTimeout(ctx, failLogTailTimeout) + defer cancel() + stream, err := s.deps.Runtime.GetContainerLogs(tailCtx, containerID, false) + if err != nil { + return nil + } + defer func() { _ = stream.Close() }() + lines, err := tailLines(stream, failLogTailLines) + if err != nil { + return nil + } + return lines +} + +// redactDiagnostics replaces every value of this service's secrets in the +// given log lines. Secret values are read into memory only. When a secret +// cannot be read, diagnostics are dropped entirely: unredacted application +// output is never persisted. +func (s *Service) redactDiagnostics(ctx context.Context, app string, p pinnedService, lines []string) []string { + if len(lines) == 0 { + return nil + } + if len(p.spec.Secrets) == 0 || s.deps.Secrets == nil { + return lines + } + id, err := s.appSecretID(ctx, app) + if err != nil { + return nil + } + redacted := append([]string(nil), lines...) + for _, name := range p.spec.Secrets { + path := domain.AppSecretPathForID(id, app, p.spec.Name, name) + value, err := s.deps.Secrets.GetSecret(ctx, path) + if err != nil || value == "" { + return nil + } + for i, line := range redacted { + redacted[i] = strings.ReplaceAll(line, value, "[redacted]") + } + } + return redacted +} + +// tailLines keeps the last N lines; logs are untrusted app output. +func tailLines(r io.Reader, n int) ([]string, error) { + scanner := bufio.NewScanner(r) + scanner.Buffer(make([]byte, 64*1024), 1024*1024) + var lines []string + for scanner.Scan() { + lines = append(lines, scanner.Text()) + if len(lines) > n { + lines = lines[len(lines)-n:] + } + } + if err := scanner.Err(); err != nil { + return nil, err + } + return lines, nil +} diff --git a/internal/usecase/deployment/execute.go b/internal/usecase/deployment/execute.go new file mode 100644 index 000000000..d3383bfd2 --- /dev/null +++ b/internal/usecase/deployment/execute.go @@ -0,0 +1,872 @@ +package deployment + +import ( + "context" + "fmt" + "maps" + "reflect" + "sort" + "time" + + "github.com/bnema/zerowrap" + + "github.com/bnema/gordon/internal/domain" +) + +// failLogTailLines bounds the log tail attached to service failures. +const failLogTailLines = 50 + +// failLogTailTimeout bounds the diagnostics read of a failing container. +const failLogTailTimeout = 10 * time.Second + +// Deploy executes a preflighted revision: fail-fast across services in +// sorted name order. Every service is replaced sequentially — withdraw +// traffic, retire the superseded container, start the replacement and wait +// for its readiness probe — so a deployment may briefly interrupt the +// service, and two Gordon-managed generations of one service never run at +// the same time. Volumes are never deleted: no RemoveVolume call, no +// volume-deletion flags on container removal. +// +// It is the synchronous composition of the two engine phases: StartDeploy +// claims the key and journals the operation, then ExecuteDeploy runs the +// effects for the owner only. +func (s *Service) Deploy(ctx context.Context, input DeployInput) (*DeployResult, error) { + started, err := s.StartDeploy(ctx, input) + if err != nil { + return nil, err + } + if !started.Owned { + // The key already answered this request: its stored journal is the + // result and no workload is touched again. + return journaledOrNil(input, &started.Claim.Journal, started.ReplayError()) + } + return s.ExecuteDeploy(ctx, started.Claim) +} + +// DeployClaim is the durable identity of one started deploy operation. It is +// the hand-off between the two phases: StartDeploy returns it and +// ExecuteDeploy consumes it. App and Op are the durable identity, so a +// caller that reconstructs the claim from them alone (for example after a +// restart, with no in-process state) leaves Journal and Resolved empty and +// ExecuteDeploy loads and resolves both from the store. +type DeployClaim struct { + App string + Op string + Service string + Revision string + // Resolved is the revision StartDeploy resolved for this claim. Empty + // when the claim was reconstructed from identity alone. + Resolved domain.AppDesiredRevision + // Journal is the claimed operation held in process. Its Op is empty when + // the claim was reconstructed from identity alone. + Journal domain.AppOperation +} + +// StartDeployResult reports whether this caller owns the claimed operation. +// Owned is true only for the single caller that claimed Op. Every idempotent +// replay (the same key already answered) reports Owned false, and the stored +// journal in Claim.Journal is the result. Only an owner may pass the claim to +// ExecuteDeploy. +type StartDeployResult struct { + Claim DeployClaim + Owned bool +} + +// ReplayError is the explicit disposition of a claim this caller does not +// own: nil for a terminal success, and a conflict for an in-flight, +// interrupted, or failed operation, so a replay is never mistaken for a +// second execution. It is nil for an owned operation. +func (r StartDeployResult) ReplayError() error { + if r.Owned { + return nil + } + return replayError(r.Claim.Journal) +} + +// StartDeploy performs the first phase of a deploy. Under the app lock it +// converges any interrupted predecessor, resolves the request to a revision +// and runs the targeted-deploy convergence checks, atomically claims the +// request key, and persists the non-terminal journal. It performs no image +// pull, pinning, or workload mutation, so a crash right after it leaves a +// durable claim that reconciliation can converge. +// +// The returned claim distinguishes the newly owned operation (Owned true) +// from an idempotent replay (Owned false): the same key is never handed to +// two owners, and a replay never executes effects again. +func (s *Service) StartDeploy(ctx context.Context, input DeployInput) (*StartDeployResult, error) { + // Planning and claiming acquire no runtime resource, so StartDeploy takes + // only the per-app coordinator. ExecuteDeploy takes the GC shared lease + // across resource acquisition and publication, where it is required. + release, err := s.coord.acquire(ctx, input.App) + if err != nil { + return nil, err + } + defer release() + return s.startDeployLocked(ctx, input) +} + +func (s *Service) startDeployLocked(ctx context.Context, input DeployInput) (*StartDeployResult, error) { + ctx = zerowrap.CtxWithFields(ctx, map[string]any{ + zerowrap.FieldLayer: "usecase", + zerowrap.FieldUseCase: "StartDeploy", + "app": input.App, + }) + log := zerowrap.FromCtx(ctx) + + if err := s.deps.State.Recover(ctx); err != nil { + return nil, fmt.Errorf("deployment: recover before deploy: %w", err) + } + // A mutation of this app never starts on top of an interrupted one: its + // never-published candidates must be gone before this mutation creates a + // generation of the same service. ACTIVE is materialized first, so the + // published generation cannot be mistaken for a leftover. + if err := s.reconcileInterruptedDeploy(ctx, input.App); err != nil { + return nil, err + } + rev, op, owned, err := s.claimDeploymentLocked(ctx, input) + if err != nil { + return nil, err + } + if owned { + // The claimed operation is now live in this process. It is + // registered before the caller releases the coordinator, so a + // foreground mutation that takes the lock in the gap before + // ExecuteDeploy cannot reconcile the claim away. + s.markOperationLive(input.App, op.Op) + } else { + log.Info().Str("op", op.Op).Msg("deployment: replayed an already claimed operation") + } + return &StartDeployResult{ + Claim: DeployClaim{ + App: input.App, Op: op.Op, Service: input.Service, Revision: input.Revision, + Resolved: rev, Journal: op, + }, + Owned: owned, + }, nil +} + +// ExecuteDeploy performs the second phase of a deploy: it reacquires the app +// lock, loads and verifies the claimed journal by app/op identity, runs the +// image pull/pin and the sequential replacement, and records the terminal +// outcome. A claim that is already terminal, or that does not match its +// identity, is never executed: it replays its stored outcome instead. An +// operation that stops mid-execution (shutdown or cancellation) leaves its +// non-terminal journal behind for reconciliation, so it can never run twice. +// +// The operation stays live until this returns: the store record, not the +// in-process claim, is what executes, and the live marker is cleared only +// after the outcome was recorded or a recoverable non-terminal claim was +// left behind, while the app lock is still held. +func (s *Service) ExecuteDeploy(ctx context.Context, claim DeployClaim) (*DeployResult, error) { + release, err := s.acquireAppContext(ctx, claim.App) + if err != nil { + // The claim is left as persisted: recoverable, but no longer owned + // here, so a later reconciliation can converge it. + s.clearOperationLive(claim.App, claim.Op) + return nil, err + } + defer release() + // Registered after release, so it runs first: the marker is dropped + // before the app lock is released, and never while a live goroutine may + // still persist state. + defer s.clearOperationLive(claim.App, claim.Op) + return s.executeLocked(ctx, claim) +} + +// AbandonDeploy settles a claim this process owns but will never execute, such +// as when daemon shutdown races the hand-off between StartDeploy and +// ExecuteDeploy. It drops the in-process live marker and converges the still +// non-terminal journal as an interrupted operation, so no durable claim stays +// marked live forever and a later reconciliation reads a terminal outcome +// instead of a permanently in-flight key. It takes the app lock but performs +// no workload effect. +func (s *Service) AbandonDeploy(ctx context.Context, claim DeployClaim) error { + if claim.App == "" || claim.Op == "" { + return fmt.Errorf("deployment: abandon requires an app and operation identity: %w", domain.ErrAppStateConflict) + } + // The marker is dropped before any fallible step: even if convergence + // fails, the claim is no longer owned here, so a later reconciliation + // (foreground or boot) can converge it instead of skipping it as live. + s.clearOperationLive(claim.App, claim.Op) + release, err := s.coord.acquire(ctx, claim.App) + if err != nil { + return err + } + defer release() + return s.reconcileInterruptedDeploy(ctx, claim.App) +} + +// executeLocked runs the claimed, non-terminal operation. The caller holds +// the app lock. Every failure that happens before a terminal outcome is +// recorded (load active, revision, service step, traffic) leaves the journal +// with its step state, never a silent success. +func (s *Service) executeLocked(ctx context.Context, claim DeployClaim) (*DeployResult, error) { + input := DeployInput{App: claim.App, Revision: claim.Revision, Service: claim.Service} + ctx = zerowrap.CtxWithFields(ctx, map[string]any{ + zerowrap.FieldLayer: "usecase", + zerowrap.FieldUseCase: "ExecuteDeploy", + "app": claim.App, + "op": claim.Op, + }) + log := zerowrap.FromCtx(ctx) + + op, err := s.loadClaim(ctx, claim) + if err != nil { + return nil, err + } + if op.Terminal() { + // The claim was finalized between the phases: never execute a settled + // key, replay its stored outcome instead. + return journaledDeployResult(input, &op), replayError(op) + } + // The claimed revision is authoritative for the whole execution: it was + // resolved when the key was claimed and is what the journal records, so a + // desired-state change between the two phases must never redirect this + // operation to a different revision than the one it pinned. + rev, err := s.claimRevision(ctx, claim, op) + if err != nil { + s.failOperation(ctx, &op, err, "error") + return journaledDeployResult(input, &op), err + } + pinned, err := s.pinPreflightLocked(ctx, input.App, input.Service, rev, &op) + if err != nil { + return journaledDeployResult(input, &op), err + } + sort.Slice(pinned, func(i, j int) bool { return pinned[i].name < pinned[j].name }) + + active, _, err := s.deps.State.LoadActive(ctx, input.App) + if err != nil { + loadErr := fmt.Errorf("deployment: load active: %w", err) + s.failOperation(ctx, &op, loadErr, "error") + return journaledDeployResult(input, &op), loadErr + } + + result := &DeployResult{ + Op: op.Op, + App: input.App, + Revision: rev.Revision, + Services: map[string]ServiceResult{}, + } + // A targeted deploy is intentionally a partial plan: services omitted + // from pinned remain active. Full deploys reconcile actual removals. + if input.Service == "" { + if err := s.reconcileRemovalsForDeploy(ctx, input.App, active, pinned, &op, result, log); err != nil { + return result, err + } + } + for i, p := range pinned { + if err := s.runServiceStep(ctx, input.App, rev.Revision, i, p, &op, active, result, log); err != nil { + return result, err + } + } + op.Outcome = ComputeOutcome(result.Services) + op.Warnings = journalWarnings(collectCleanupWarnings(result.Services)) + result.CleanupWarnings = collectCleanupWarnings(result.Services) + if saveErr := s.deps.State.SaveOperation(ctx, op); saveErr != nil { + log.Warn().Err(saveErr).Msg("deployment: failed to record deploy outcome") + } + return result, nil +} + +// loadClaim verifies the claimed journal by app/op identity. The store is +// authoritative: the operation is always reloaded by app/op ID, so a stale +// in-process journal can never be executed (for example when another actor +// finalized the key between the claim and execution). A journal that does +// not match the identity, or is not a deploy, is refused. +func (s *Service) loadClaim(ctx context.Context, claim DeployClaim) (domain.AppOperation, error) { + if claim.App == "" || claim.Op == "" { + return domain.AppOperation{}, fmt.Errorf("deployment: execute requires an app and operation identity: %w", domain.ErrAppStateConflict) + } + op, err := s.deps.State.LoadOperation(ctx, claim.App, claim.Op) + if err != nil { + return domain.AppOperation{}, fmt.Errorf("deployment: load claimed operation %q: %w", claim.Op, err) + } + if op.App != claim.App || op.Op != claim.Op { + return domain.AppOperation{}, fmt.Errorf("deployment: claim does not identify app %q operation %q: %w", claim.App, claim.Op, domain.ErrAppStateConflict) + } + if op.Kind != "deploy" { + return domain.AppOperation{}, fmt.Errorf("deployment: operation %q is %q, not a deploy: %w", claim.Op, op.Kind, domain.ErrAppStateConflict) + } + return op, nil +} + +// claimRevision returns the revision StartDeploy captured. A claim +// reconstructed from identity alone resolves the revision its journal +// recorded, so a resumed execution pins the revision it was started for. +func (s *Service) claimRevision(ctx context.Context, claim DeployClaim, op domain.AppOperation) (domain.AppDesiredRevision, error) { + if claim.Resolved.Revision != "" { + return claim.Resolved, nil + } + return s.resolveRevision(ctx, DeployInput{App: claim.App, Revision: op.InputRevision, Service: claim.Service}) +} + +// runServiceStep executes one pinned service in the journal: replace it, +// publish its ACTIVE entry, then publish traffic for the new container. The +// step is recorded as succeeded only after the runtime effect, the ACTIVE +// publication, and the traffic apply all succeeded, so a journaled success +// is never ahead of what is routed. +func (s *Service) runServiceStep( + ctx context.Context, + app, revision string, + index int, + p pinnedService, + op *domain.AppOperation, + active domain.AppActive, + result *DeployResult, + log zerowrap.Logger, +) error { + before := "" + if eff, ok := active.Services[p.name]; ok { + before = eff.Container + } + stepID := "service." + p.name + ".replace" + // The step is journaled before the replacement runs and carries the + // candidate's container ID as soon as it exists, so an interrupted deploy + // leaves a trace boot recovery can clean up instead of an untracked + // generation. + step := &op.Steps[index+1] + *step = domain.AppOperationStep{ + ID: stepID, Service: p.name, + Digest: p.digest, Image: p.runtimeImage, + Before: before, State: domain.AppStepPending, + } + var svcResult ServiceResult + if current, ok := s.unchangedService(ctx, app, p, active); ok { + // Same image, spec, environment, and secret values already running: + // keep the container. + svcResult = ServiceResult{ + Result: ServiceResultUnchanged, EffectiveRevision: revision, + Before: before, After: before, + BackendBinds: current.BackendBinds, UDPBackendBinds: current.UDPBackendBinds, + RestartUnsafe: singleWriterRequired(p.spec), + } + step.Detail = domain.AppServiceUnchanged + ": already running " + p.spec.Image + " (" + p.digest + ")" + } else { + svcResult = s.deployService(ctx, app, revision, p, op.Op, before, activeStopGrace(active, p.name), s.journalCandidate(op, index+1)) + } + result.Services[p.name] = svcResult + // A recorded candidate is authoritative while the step is unresolved: a + // replacement that failed after creating its container must stay traceable + // for reconciliation. + if svcResult.After != "" { + step.After = svcResult.After + } + fail := func(err error) error { + step.State = domain.AppStepFailed + step.Error = err.Error() + step.Diagnostics = svcResult.Diagnostics + op.Warnings = journalWarnings(collectCleanupWarnings(result.Services)) + op.Outcome = ComputeOutcome(result.Services) + if saveErr := s.deps.State.SaveOperation(ctx, *op); saveErr != nil { + log.Warn().Err(saveErr).Msg("deployment: failed to record service failure") + } + return err + } + if svcResult.Result == "failed" { + // Fail fast: later services stay unchanged. + return fail(fmt.Errorf("deployment: service %q failed at %s: %s: %w", + p.name, stepID, svcResult.Error, domain.ErrAppStateConflict)) + } + // Publish the per-service effective state: the new binds become the + // service's recorded backend before the proxy is repointed at them. + if err := s.publishService(ctx, app, revision, p, svcResult, op.Op); err != nil { + svcResult.Result = "failed" + svcResult.Error = err.Error() + result.Services[p.name] = svcResult + return fail(err) + } + // Rebuild and publish the traffic graph. ACTIVE already names the new + // container, but the operation is a failure until routing accepted it, + // so a rejected graph must never be journaled as success. + if err := s.refreshTraffic(ctx, app); err != nil { + svcResult.Result = "failed" + svcResult.Error = err.Error() + result.Services[p.name] = svcResult + return fail(err) + } + result.Services[p.name] = svcResult + step.State = domain.AppStepSucceeded + step.Error = "" + if saveErr := s.deps.State.SaveOperation(ctx, *op); saveErr != nil { + log.Warn().Err(saveErr).Msg("deployment: failed to checkpoint service progress") + } + return nil +} + +// unchangedService reports whether the service already runs exactly what p +// would create, in a healthy published state: the same digest, service spec, +// app environment, and shared networks, in a running container that is not +// withdrawn or recovery-inhibited and whose recorded binds cover every +// backend port, and whose environment already holds the current secret +// values. Services with host binds or devices are never skipped: their +// resolved sources depend on host policy that ACTIVE does not record. +func (s *Service) unchangedService(ctx context.Context, app string, p pinnedService, active domain.AppActive) (domain.AppEffectiveService, bool) { + eff, ok := active.Services[p.name] + if !ok || eff.Container == "" || eff.Digest == "" || eff.Digest != p.digest { + return eff, false + } + reason := s.unchangedRefusal(ctx, app, p, eff) + if reason != "" { + log := zerowrap.FromCtx(ctx) + log.Debug().Str("app", app).Str("service", p.name).Str("reason", reason). + Msg("deployment: same digest, replacing service") + return eff, false + } + return eff, true +} + +// unchangedRefusal returns why a service running the same digest must still +// be replaced, or "" when it can be kept. +func (s *Service) unchangedRefusal(ctx context.Context, app string, p pinnedService, eff domain.AppEffectiveService) string { + switch { + case len(p.spec.Binds) > 0 || len(p.spec.Devices) > 0: + return "host binds or devices" + case !domain.SameAppService(p.spec, eff.Spec): + return "spec changed" + case !bindsCover(eff, backendPorts(p.spec)): + return "backend binds missing" + case s.publication.inhibited(app, p.name): + return "withdrawal pending" + } + previous, err := s.deps.State.LoadRevision(ctx, app, eff.EffectiveRevision) + if err != nil { + return "load effective revision: " + err.Error() + } + if !maps.Equal(previous.Spec.Env, p.appEnv) { + return "app env changed" + } + if !reflect.DeepEqual(domain.AppServiceSharedNetworks(previous.Spec, p.name), p.sharedNetworks) { + return "shared networks changed" + } + if inhibited, err := s.recoveryInhibited(ctx, app, p.name, eff.Container); err != nil || inhibited { + return "recovery inhibited" + } + container, err := s.deps.Runtime.InspectContainer(ctx, eff.Container) + if err != nil || container.Status != string(domain.ContainerStatusRunning) { + return "container not running" + } + if !s.envCurrent(ctx, app, p, container.Env) { + return "environment or secrets changed" + } + return "" +} + +// envCurrent reports whether the running container was created with the +// environment p would get now, including current secret values. The +// runtime env also holds image-declared keys, so every desired entry must +// be present; removed keys are caught by the spec and app env comparisons. +// Values are compared in memory only and never logged. +func (s *Service) envCurrent(ctx context.Context, app string, p pinnedService, running []string) bool { + desired, err := s.serviceEnv(ctx, app, p) + if err != nil { + return false + } + have := make(map[string]struct{}, len(running)) + for _, entry := range running { + have[entry] = struct{}{} + } + for _, entry := range desired { + if _, ok := have[entry]; !ok { + return false + } + } + return true +} + +// bindsCover reports whether the recorded loopback binds publish every +// backend port the spec requires, so a kept container stays routable. +func bindsCover(eff domain.AppEffectiveService, ports []domain.ContainerBackendPort) bool { + for _, port := range ports { + binds := eff.BackendBinds + if port.Protocol == domain.NetworkProtocolUDP { + binds = eff.UDPBackendBinds + } + if binds[port.ContainerPort] == 0 { + return false + } + } + return true +} + +// failOperation records a terminal failure on an already claimed +// journal, so a keyed repeat replays a terminal outcome instead of +// reading a permanently in-flight claim. stepID names the failing phase +// (a service step, a traffic publication, or a bare error). Journal-write +// failures are logged, never returned: the caller already holds a more +// specific error, and the next mutation's recovery pass converges state. +func (s *Service) failOperation(ctx context.Context, op *domain.AppOperation, err error, stepID string) { + op.Outcome = domain.AppOutcomeFailed + op.Steps = append(op.Steps, domain.AppOperationStep{ + ID: stepID, State: domain.AppStepFailed, Error: err.Error(), + }) + if saveErr := s.deps.State.SaveOperation(ctx, *op); saveErr != nil { + log := zerowrap.FromCtx(ctx) + log.Warn().Err(saveErr).Str("op", op.Op).Msg("deployment: failed to record operation failure") + } +} + +// failTrafficPublication records that the final graph apply was rejected. +// The workload steps already ran and stay accurate, but the operation is a +// failure: routing did not accept the published state, so a later replay +// of the same key must never read success. +func (s *Service) failTrafficPublication(ctx context.Context, op *domain.AppOperation, err error) { + s.failOperation(ctx, op, err, "traffic.publish") +} + +// journaledOrNil reports a preflight that already answered the request +// key: the stored journal when there is one, the bare error otherwise. +func journaledOrNil(input DeployInput, op *domain.AppOperation, err error) (*DeployResult, error) { + if op == nil { + return nil, err + } + return journaledDeployResult(input, op), err +} + +func journaledDeployResult(input DeployInput, op *domain.AppOperation) *DeployResult { + return &DeployResult{ + Op: op.Op, + App: input.App, + Revision: op.InputRevision, + Outcome: op.Outcome, + } +} + +// candidateJournal durably records a freshly created replacement container +// before it can be started or join a network. An interrupted deployment then +// leaves a trace boot recovery removes, instead of an untracked running +// generation that could overlap the one recovery rebuilds. +// +// The window between the runtime creating a container and this write is the +// irreducible one: a container ID cannot be recorded before it exists. +// sync/atomic is not needed here: the journal write is synchronous. +type candidateJournal func(ctx context.Context, containerID string) error + +// journalCandidate records one created candidate in the operation journal. +// The step is addressed by index and read at write time, so a later append to +// the operation's steps can never leave the recorder writing to a stale +// element. A nil operation (an internal path with no journal) records nothing. +func (s *Service) journalCandidate(op *domain.AppOperation, index int) candidateJournal { + if op == nil || index < 0 || index >= len(op.Steps) { + return nil + } + return func(ctx context.Context, containerID string) error { + op.Steps[index].After = containerID + if err := s.deps.State.SaveOperation(ctx, *op); err != nil { + return fmt.Errorf("deployment: record candidate %s: %w", containerID, err) + } + return nil + } +} + +// deployService replaces one service sequentially and returns its terminal +// result. before is the ACTIVE container being superseded (empty on a first +// deployment); beforeGrace is its effective stop grace. +// +// The order is fixed: withdraw from traffic and confirm it, durably inhibit +// a single-writer generation, retire the superseded container and confirm +// it is gone, then create and start the replacement and wait for its +// readiness probe. Nothing is created before the previous generation is +// confirmed gone, so two generations never overlap; a withdrawal or +// retirement that cannot be confirmed aborts without creating a candidate. +func (s *Service) deployService(ctx context.Context, app, revision string, p pinnedService, opID, before string, beforeGrace time.Duration, journal candidateJournal) ServiceResult { + // Revalidate the current authorization and source before withdrawing or + // retiring the serving generation. A reload between preflight and this + // service step must fail without mutating the existing workload. + if _, err := s.resolveServiceBinds(app, p.spec); err != nil { + return s.failResult(revision, before, "", err) + } + if _, err := s.resolveServiceDevices(app, p.spec); err != nil { + return s.failResult(revision, before, "", err) + } + // Withdraw the service from traffic first and confirm it: a withdrawal + // that cannot be applied must block every container mutation, so no + // unverified generation stays routable. A first deployment has no + // published generation to withdraw. + if before != "" { + if err := s.withdrawForRecovery(ctx, app, p.name); err != nil { + return s.failResult(revision, before, "", err) + } + } + // A volume- or bind-owning replacement may write while the old + // generation still exists and could be revived by native restart policy. + // Inhibit that generation durably BEFORE the write can happen, so boot or + // periodic recovery can never restart the old writer on top of the new + // one. Cleared once the safe generation is published (publishService) or + // the operator removes the app. + if before != "" && singleWriterRequired(p.spec) { + if err := s.inhibitRecovery(ctx, app, p.name, before, domain.AppInhibitReplacementPending, opID); err != nil { + return s.failResult(revision, before, "", err) + } + } + if before != "" { + // The superseded generation must be confirmed gone before a new + // generation starts: a failed retirement aborts the replacement + // instead of allowing overlapping generations, and the generation + // keeps its inhibition and its claims. + retired := s.retireContainer(ctx, app, retireOptions{ + Service: p.name, Grace: beforeGrace, + }, before) + if !retired.Gone { + return s.failResult(revision, before, "", fmt.Errorf("deployment: retire superseded container %s: %s", before, cleanupDetail(retired))) + } + } + candidate, err := s.createAndStart(ctx, app, revision, p, opID, journal) + if err != nil { + result := s.failResult(revision, before, "", err) + if candidate.Container != nil { + // The candidate exists: keep its ID in the terminal result so the + // journal and the operator can still trace it. + result.After = candidate.Container.ID + } + return result + } + created := candidate.Container + if err := s.waitServiceReady(ctx, app, created.ID, deploymentReadiness(p.spec), candidate.TCPBinds); err != nil { + // Capture redacted diagnostics while the failed replacement still + // exists, then remove only that candidate. The superseded container + // is already gone and is never recreated. Cleanup is best effort: a + // canceled context can leave the candidate behind, which is then + // reported as a leftover warning instead of a false success. + tail := s.redactDiagnostics(ctx, app, p, s.logTail(ctx, created.ID)) + cleanup := s.retireCandidate(ctx, app, p.name, created.ID) + return ServiceResult{ + Result: "failed", + EffectiveRevision: revision, + Before: before, + After: created.ID, + RestartUnsafe: singleWriterRequired(p.spec), + Error: err.Error(), + Diagnostics: tail, + CleanupWarnings: cleanup, + } + } + return ServiceResult{ + Result: "deployed", + EffectiveRevision: revision, + Before: before, + After: created.ID, + RestartUnsafe: singleWriterRequired(p.spec), + BackendBinds: candidate.TCPBinds, + UDPBackendBinds: candidate.UDPBinds, + } +} + +// singleWriterRequired reports services that must never have two generations +// running concurrently. Persistent volumes and any bind (especially a +// writable one) may be written by both, so they are treated identically. +func singleWriterRequired(spec domain.AppService) bool { + return len(spec.Volumes) > 0 || len(spec.Binds) > 0 +} + +// refreshTraffic rebuilds the proxy host index and applies the full +// HTTP/L4 graph after activation. A failure is fatal to the caller: +// ACTIVE is published but the new service is not routable, so the +// operation must not report success. Any service of this app whose +// fail-closed withdrawal was never applied is withdrawn again first, so +// a later publication never runs on top of unproven forwarding. +func (s *Service) refreshTraffic(ctx context.Context, app string) error { + if s.deps.Traffic == nil { + return nil + } + for _, service := range s.publication.pending(app) { + if err := s.withdrawForRecovery(ctx, app, service); err != nil { + return err + } + } + if err := s.deps.Traffic.RebuildTraffic(ctx); err != nil { + return fmt.Errorf("deployment: rebuild traffic for %q: %w", app, err) + } + return nil +} + +// publishService writes the per-service effective record once the +// replacement passed readiness. +func (s *Service) publishService(ctx context.Context, app, revision string, p pinnedService, result ServiceResult, opID string) error { + if app == "" { + return fmt.Errorf("deployment: publish active: empty app: %w", domain.ErrAppStateConflict) + } + active, ok, err := s.deps.State.LoadActive(ctx, app) + if err != nil { + return fmt.Errorf("deployment: load active: %w", err) + } + if !ok { + active = domain.AppActive{App: app, Services: map[string]domain.AppEffectiveService{}} + } + if active.Services == nil { + active.Services = map[string]domain.AppEffectiveService{} + } + activatedBy, activatedAt := opID, time.Now().UTC() + if previous, ok := active.Services[p.name]; ok && result.Result == ServiceResultUnchanged { + // A kept container keeps the activation of the operation that created it. + activatedBy, activatedAt = previous.ActivatedBy, previous.ActivatedAt + } + active.Services[p.name] = domain.AppEffectiveService{ + EffectiveRevision: revision, + ActivatedBy: activatedBy, + ActivatedAt: activatedAt, + Image: p.spec.Image, + Digest: p.digest, + Container: result.After, + Spec: p.spec, + BackendBinds: result.BackendBinds, + UDPBackendBinds: result.UDPBackendBinds, + } + active.ConvergedRevision = revision + converged := true + for _, svc := range active.Services { + if svc.EffectiveRevision != revision { + converged = false + break + } + } + active.Converged = converged + if converged { + active.Networks = append([]domain.AppSharedNetwork(nil), p.appNetworks...) + } + if err := s.deps.State.SaveActive(ctx, active); err != nil { + return fmt.Errorf("deployment: publish active: %w", err) + } + if err := s.recordOwnership(ctx, app, p); err != nil { + return err + } + // The superseded generation is now safely replaced: clearing its + // inhibition is what makes this generation authoritative. A failed + // publication leaves the marker in place. + if result.Before != "" && result.Before != result.After { + if err := s.clearRecoveryInhibition(ctx, app, p.name, result.Before); err != nil { + return err + } + } + return nil +} + +// recordOwnership stamps volume/secret/network ownership after publication. +func (s *Service) recordOwnership(ctx context.Context, app string, p pinnedService) error { + ownership, err := s.deps.State.LoadOwnership(ctx, app) + if err != nil { + return err + } + if ownership.App == "" { + ownership.App = app + } + // ownership.ID arrives from LoadOwnership (store-assigned UUID); + // empty only for legacy records predating UUIDs. + volumes := append([]domain.AppOwnedVolume(nil), ownership.Volumes...) + seen := map[string]int{} + for i, vol := range volumes { + seen[vol.Name] = i + } + for _, vol := range p.spec.Volumes { + entry := domain.AppOwnedVolume{ + Name: vol.Name, + Service: p.spec.Name, + RuntimeName: domain.RuntimeVolumeName(app, p.spec.Name, vol.Name), + State: domain.AppResourceAttached, + } + if idx, ok := seen[vol.Name]; ok { + volumes[idx] = entry + } else { + seen[vol.Name] = len(volumes) + volumes = append(volumes, entry) + } + } + ownership.Volumes = volumes + secrets := append([]domain.AppOwnedSecret(nil), ownership.Secrets...) + secretSeen := map[string]int{} + for i, secret := range secrets { + secretSeen[secret.Service+"\x00"+secret.Env] = i + } + for envKey, name := range p.spec.Secrets { + entry := domain.AppOwnedSecret{ + Service: p.spec.Name, + Env: envKey, + Name: name, + Path: domain.AppSecretPathForID(ownership.ID, app, p.spec.Name, name), + State: domain.AppResourceAttached, + } + key := p.spec.Name + "\x00" + envKey + if idx, ok := secretSeen[key]; ok { + secrets[idx] = entry + } else { + secretSeen[key] = len(secrets) + secrets = append(secrets, entry) + } + } + ownership.Secrets = secrets + ownership.Images = recordImageOwnership(ownership.Images, p) + recordNetworkOwnership(&ownership, s.deps.Networks.Prefix, p.sharedNetworks) + if ownership.Services == nil { + ownership.Services = map[string]domain.AppServiceRecovery{} + } + ownership.Services[p.spec.Name] = domain.AppServiceRecovery{RestartUnsafe: singleWriterRequired(p.spec)} + if err := s.deps.State.SaveOwnership(ctx, ownership); err != nil { + return fmt.Errorf("deployment: record ownership: %w", err) + } + return nil +} + +// recordImageOwnership stamps one service's pinned image into the app's +// owned-image list. Positive durable ownership is what later authorizes +// runtime image prune; image labels alone never do. +func recordImageOwnership(existing []domain.AppOwnedImage, p pinnedService) []domain.AppOwnedImage { + entry := domain.AppOwnedImage{ + Service: p.spec.Name, + Reference: p.spec.Image, + Digest: p.digest, + State: domain.AppResourceAttached, + } + out := append([]domain.AppOwnedImage(nil), existing...) + for i, image := range out { + if image.Service == entry.Service { + out[i] = entry + return out + } + } + return append(out, entry) +} + +// failResult builds a preflight/create failure before any container exists. +func (s *Service) failResult(revision, before, after string, err error) ServiceResult { + return ServiceResult{ + Result: "failed", + EffectiveRevision: revision, + Before: before, + After: after, + Error: err.Error(), + } +} + +// deploymentReadiness resolves the readiness check a deploy or restart +// runs. Readiness never depends on how a service is replaced: a declared +// http, tcp, or log check is used as declared, and an undeclared check +// ("none" or unset) keeps the implicit HTTP probe on a service Gordon can +// probe over the published loopback bind — at least one effective-public +// HTTP interface, no L4 interface, no volume, and no bind. An undeclared +// probe on a stateful or L4 workload stays unchecked instead: an HTTP GET on +// the published port proves nothing there, so Gordon never guesses one. +func deploymentReadiness(spec domain.AppService) domain.AppService { + if !implicitHTTPProbe(spec) { + return spec + } + spec.Readiness.Type = domain.AppReadinessHTTP + return spec +} + +// implicitHTTPProbe reports whether a service that declares no readiness +// check receives the implicit HTTP probe when a deploy or restart starts it. +// Recovery verification of an already running generation keeps the declared +// readiness unchanged: it re-checks a generation it did not replace. +func implicitHTTPProbe(spec domain.AppService) bool { + if spec.Readiness.Type != "" && spec.Readiness.Type != domain.AppReadinessNone { + return false + } + if len(spec.HTTP) == 0 || len(spec.TCP) > 0 || len(spec.UDP) > 0 || len(spec.Volumes) > 0 || len(spec.Binds) > 0 { + return false + } + for _, h := range spec.HTTP { + if !h.IsPublic() { + return false + } + } + return true +} diff --git a/internal/usecase/deployment/execute_internal_test.go b/internal/usecase/deployment/execute_internal_test.go new file mode 100644 index 000000000..9b47307ff --- /dev/null +++ b/internal/usecase/deployment/execute_internal_test.go @@ -0,0 +1,22 @@ +package deployment + +import "testing" + +func TestStripImageTag(t *testing.T) { + cases := []struct { + ref string + want string + }{ + {"localhost:15500/e2e/web:v1", "localhost:15500/e2e/web"}, + {"localhost:15500/e2e/web", "localhost:15500/e2e/web"}, + {"reg.example.com/blog/web:1.4.2", "reg.example.com/blog/web"}, + {"web:v1", "web"}, + {"web", "web"}, + {"reg.example.com/blog/web@sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", "reg.example.com/blog/web"}, + } + for _, tc := range cases { + if got := stripImageTag(tc.ref); got != tc.want { + t.Errorf("stripImageTag(%q) = %q, want %q", tc.ref, got, tc.want) + } + } +} diff --git a/internal/usecase/deployment/image_policy_internal_test.go b/internal/usecase/deployment/image_policy_internal_test.go new file mode 100644 index 000000000..4bd65de68 --- /dev/null +++ b/internal/usecase/deployment/image_policy_internal_test.go @@ -0,0 +1,75 @@ +package deployment + +import ( + "context" + "strings" + "testing" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" +) + +func imagePolicyService(policy domain.ImageSourcePolicy) *Service { + return NewService(Deps{ImagePolicy: policy}, zerowrap.Default()) +} + +// TestValidateImageSource_RejectsPrivateRegistryDigest proves the image +// policy runs on the shared preflight path used by deploy, boot, restart, +// and recovery, so a digest-pinned local/private reference is refused +// before any pull. +func TestValidateImageSource_RejectsPrivateRegistryDigest(t *testing.T) { + digest := "sha256:" + strings.Repeat("a", 64) + svc := imagePolicyService(domain.ImageSourcePolicy{ + AllowedRegistries: []string{"registry.example.com"}, + InstallationRegistry: "gordon.example.com", + }) + + assert.ErrorIs(t, svc.validateImageSource("127.0.0.1:12345/private", digest), domain.ErrAppImageNotAllowed) + assert.ErrorIs(t, svc.validateImageSource("169.254.169.254/metadata", digest), domain.ErrAppImageNotAllowed) + assert.ErrorIs(t, svc.validateImageSource("other.example.com/team/app", digest), domain.ErrAppImageNotAllowed) + require.NoError(t, svc.validateImageSource("registry.example.com/team/app", digest)) + require.NoError(t, svc.validateImageSource("gordon.example.com/blog/web", digest)) +} + +func TestPullImage_RechecksRegistryPolicyImmediatelyBeforePull(t *testing.T) { + runtime := outmocks.NewMockContainerRuntime(t) + svc := NewService(Deps{ + Runtime: runtime, + Registry: RegistryConfig{Domain: "gordon.example.com"}, + ImagePolicy: domain.ImageSourcePolicy{InstallationRegistry: "gordon.example.com"}, + }, zerowrap.Default()) + + _, err := svc.pullImage(context.Background(), "registry.example.com/team/app@sha256:"+strings.Repeat("a", 64)) + + assert.ErrorIs(t, err, domain.ErrAppImageNotAllowed) +} + +func TestValidateImageSource_RequireDigestAppliesToAllRegistries(t *testing.T) { + digest := "sha256:" + strings.Repeat("a", 64) + svc := imagePolicyService(domain.ImageSourcePolicy{ + AllowedRegistries: []string{"registry.example.com"}, + RequireDigest: true, + InstallationRegistry: "gordon.example.com", + }) + assert.ErrorIs(t, svc.validateImageSource("registry.example.com/team/app:1.0", ""), domain.ErrAppImageNotAllowed) + assert.ErrorIs(t, svc.validateImageSource("gordon.example.com/team/app:1.0", ""), domain.ErrAppImageNotAllowed) + require.NoError(t, svc.validateImageSource("registry.example.com/team/app", digest)) + require.NoError(t, svc.validateImageSource("gordon.example.com/team/app", digest)) +} + +func TestPreflightImage_RejectsMalformedResolvedDigestBeforePull(t *testing.T) { + runtime := outmocks.NewMockContainerRuntime(t) + svc := NewService(Deps{ + Runtime: runtime, + Registry: RegistryConfig{Domain: "gordon.example.com"}, + ImagePolicy: domain.ImageSourcePolicy{InstallationRegistry: "gordon.example.com"}, + }, zerowrap.Default()) + + _, err := svc.preflightImage(context.Background(), "gordon.example.com/team/app:1.0", "sha256:short") + + assert.ErrorIs(t, err, domain.ErrAppImageUnresolvable) +} diff --git a/internal/usecase/deployment/internal_readiness_test.go b/internal/usecase/deployment/internal_readiness_test.go new file mode 100644 index 000000000..479cba39d --- /dev/null +++ b/internal/usecase/deployment/internal_readiness_test.go @@ -0,0 +1,363 @@ +package deployment + +import ( + "context" + "errors" + "fmt" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" +) + +// internalSpec is an HTTP service whose only interface is internal. +func internalSpec() domain.AppService { + return domain.AppService{ + Name: "api", + Image: "registry.example.com/blog/api:1.0.0", + HTTP: []domain.AppHTTPInterface{{Port: 8080, Visibility: domain.AppVisibilityInternal}}, + Readiness: domain.AppReadiness{Type: domain.AppReadinessHTTP, Path: "/healthz", Timeout: 2 * time.Second}, + } +} + +// TestBackendPorts_InternalHTTPIsNeverPublished proves internal HTTP ports +// gain no loopback publication, including through readiness metadata, while +// public interfaces keep theirs. +func TestBackendPorts_InternalHTTPIsNeverPublished(t *testing.T) { + internal := internalSpec() + assert.Empty(t, backendPorts(internal), "internal http must not be published") + + public := domain.AppService{ + Name: "web", + HTTP: []domain.AppHTTPInterface{{Host: "blog.example.com", Port: 8080, TLS: "auto"}}, + } + assert.Equal(t, []int{8080}, containerPorts(backendPorts(public))) + + mixed := domain.AppService{ + HTTP: []domain.AppHTTPInterface{ + {Host: "blog.example.com", Port: 8080, TLS: "auto"}, + {Port: 9090, Visibility: domain.AppVisibilityInternal}, + }, + Readiness: domain.AppReadiness{Type: domain.AppReadinessHTTP, Port: 9090}, + } + assert.Equal(t, []int{8080}, containerPorts(backendPorts(mixed)), + "the internal readiness port must not add a second publication") + + tcp := domain.AppService{ + TCP: []domain.AppTCPInterface{{Port: 5432}}, + Readiness: domain.AppReadiness{Type: domain.AppReadinessTCP}, + } + assert.Equal(t, []int{5432}, containerPorts(backendPorts(tcp))) +} + +func containerPorts(ports []domain.ContainerBackendPort) []int { + result := make([]int, 0, len(ports)) + for _, port := range ports { + result = append(result, port.ContainerPort) + } + return result +} + +// TestInternalProbeShape proves readiness type selects the probe transport. +func TestInternalProbeShape(t *testing.T) { + httpProtocol, path := internalProbeShape(internalSpec()) + assert.Equal(t, domain.ProbeProtocolHTTP, httpProtocol) + assert.Equal(t, "/healthz", path) + + tcpSpec := internalSpec() + tcpSpec.Readiness = domain.AppReadiness{Type: domain.AppReadinessTCP} + tcpProtocol, tcpPath := internalProbeShape(tcpSpec) + assert.Equal(t, domain.ProbeProtocolTCP, tcpProtocol) + assert.Empty(t, tcpPath) +} + +// TestWaitInternalReady_RetriesUntilReady proves an unhealthy attempt is +// retried and the request carries the exact identity, network, and shape. +func TestWaitInternalReady_RetriesUntilReady(t *testing.T) { + var requests []domain.ContainerNetworkProbeRequest + attempts := 0 + deps := ProbeDeps{ + containerStart: func(context.Context, string) (time.Time, error) { + return time.Unix(1700000000, 0).UTC(), nil + }, + networkProbe: func(_ context.Context, request domain.ContainerNetworkProbeRequest) (domain.ContainerNetworkProbeResult, error) { + requests = append(requests, request) + attempts++ + if attempts < 3 { + return domain.ContainerNetworkProbeResult{Diagnostic: "no http response"}, nil + } + return domain.ContainerNetworkProbeResult{Ready: true, Status: 200}, nil + }, + } + require.NoError(t, waitInternalReadyWithDeps(context.Background(), deps, "c-api", internalSpec(), "gordon--blog--net", 8080)) + require.Len(t, requests, 3) + assert.Equal(t, "c-api", requests[0].TargetContainerID) + assert.Equal(t, "gordon--blog--net", requests[0].Network) + assert.Equal(t, domain.ProbeProtocolHTTP, requests[0].Protocol) + assert.Equal(t, "/healthz", requests[0].Path) + assert.Equal(t, 8080, requests[0].Port) + assert.Equal(t, time.Unix(1700000000, 0).UTC(), requests[0].ExpectedStartedAt) +} + +// TestWaitInternalReady_InfrastructureErrorFailsImmediately proves a helper +// that cannot run is never retried as an unhealthy attempt. +func TestWaitInternalReady_InfrastructureErrorFailsImmediately(t *testing.T) { + attempts := 0 + deps := ProbeDeps{ + containerStart: func(context.Context, string) (time.Time, error) { + return time.Unix(1700000000, 0).UTC(), nil + }, + networkProbe: func(context.Context, domain.ContainerNetworkProbeRequest) (domain.ContainerNetworkProbeResult, error) { + attempts++ + return domain.ContainerNetworkProbeResult{}, errors.New("helper create failed") + }, + } + err := waitInternalReadyWithDeps(context.Background(), deps, "c-api", internalSpec(), "net", 8080) + require.ErrorContains(t, err, "infrastructure error") + assert.Equal(t, 1, attempts, "an infrastructure failure must not be retried") +} + +func TestRunInternalProbeAttempt_CleanupFailureIsNeverSuppressed(t *testing.T) { + request := domain.ContainerNetworkProbeRequest{Timeout: 10 * time.Millisecond} + deps := ProbeDeps{networkProbe: func(ctx context.Context, _ domain.ContainerNetworkProbeRequest) (domain.ContainerNetworkProbeResult, error) { + <-ctx.Done() + return domain.ContainerNetworkProbeResult{}, fmt.Errorf("%w: remove failed", domain.ErrNetworkProbeCleanup) + }} + + _, err := runInternalProbeAttempt(context.Background(), deps, request) + require.ErrorIs(t, err, domain.ErrNetworkProbeCleanup) +} + +// TestWaitInternalReady_CombinedCleanupAndStaleErrorAbortsAfterOneAttempt +// proves a helper cleanup failure wins over the stale-generation retry: a +// probe error wrapping both a stale sentinel and ErrNetworkProbeCleanup +// aborts the wait after the first attempt instead of polling until the +// readiness deadline. +func TestWaitInternalReady_CombinedCleanupAndStaleErrorAbortsAfterOneAttempt(t *testing.T) { + attempts := 0 + startReads := 0 + deps := ProbeDeps{ + containerStart: func(context.Context, string) (time.Time, error) { + startReads++ + return time.Unix(1700000000, 0).UTC(), nil + }, + networkProbe: func(context.Context, domain.ContainerNetworkProbeRequest) (domain.ContainerNetworkProbeResult, error) { + attempts++ + return domain.ContainerNetworkProbeResult{}, fmt.Errorf("stale candidate: %w (helper not removed: %w)", domain.ErrAppStateConflict, domain.ErrNetworkProbeCleanup) + }, + } + err := waitInternalReadyWithDeps(context.Background(), deps, "c-api", internalSpec(), "net", 8080) + require.ErrorIs(t, err, domain.ErrNetworkProbeCleanup) + assert.NotContains(t, err.Error(), "timeout") + assert.Equal(t, 1, attempts, "a cleanup failure must abort after one attempt") + assert.Equal(t, 1, startReads, "a cleanup failure must not re-read the execution boundary") +} + +// TestWaitInternalReady_Timeout proves a never-ready target fails at the +// readiness deadline. +func TestWaitInternalReady_Timeout(t *testing.T) { + spec := internalSpec() + spec.Readiness.Timeout = 50 * time.Millisecond + deps := ProbeDeps{ + containerStart: func(context.Context, string) (time.Time, error) { + return time.Unix(1700000000, 0).UTC(), nil + }, + networkProbe: func(context.Context, domain.ContainerNetworkProbeRequest) (domain.ContainerNetworkProbeResult, error) { + return domain.ContainerNetworkProbeResult{Diagnostic: "http status 503"}, nil + }, + } + err := waitInternalReadyWithDeps(context.Background(), deps, "c-api", spec, "net", 8080) + require.ErrorContains(t, err, "timeout") + assert.ErrorContains(t, err, "http status 503") +} + +// TestWaitInternalReady_Cancellation proves a canceled deployment context +// stops polling immediately. +func TestWaitInternalReady_Cancellation(t *testing.T) { + ctx, cancel := context.WithCancel(context.Background()) + deps := ProbeDeps{ + containerStart: func(context.Context, string) (time.Time, error) { + return time.Unix(1700000000, 0).UTC(), nil + }, + networkProbe: func(context.Context, domain.ContainerNetworkProbeRequest) (domain.ContainerNetworkProbeResult, error) { + cancel() + return domain.ContainerNetworkProbeResult{}, nil + }, + } + err := waitInternalReadyWithDeps(ctx, deps, "c-api", internalSpec(), "net", 8080) + require.ErrorIs(t, err, context.Canceled) +} + +// TestWaitInternalReady_MissingExecutionStart proves a runtime that reports +// no execution boundary fails closed rather than probing an unidentified +// generation. +func TestWaitInternalReady_MissingExecutionStart(t *testing.T) { + deps := ProbeDeps{ + containerStart: func(context.Context, string) (time.Time, error) { + return time.Time{}, nil + }, + networkProbe: func(context.Context, domain.ContainerNetworkProbeRequest) (domain.ContainerNetworkProbeResult, error) { + t.Fatal("must not probe without an execution start") + return domain.ContainerNetworkProbeResult{}, nil + }, + } + err := waitInternalReadyWithDeps(context.Background(), deps, "c-api", internalSpec(), "net", 8080) + require.ErrorContains(t, err, "execution start") +} + +// TestWaitInternalReady_StaleGenerationIsRetried proves a candidate that +// restarts during the readiness window is re-probed against its new +// execution boundary instead of aborting the wait as an infrastructure +// error. +func TestWaitInternalReady_StaleGenerationIsRetried(t *testing.T) { + first := time.Unix(1700000000, 0).UTC() + second := time.Unix(1700000100, 0).UTC() + starts := []time.Time{first, second, second} + startReads := 0 + attempts := 0 + deps := ProbeDeps{ + containerStart: func(context.Context, string) (time.Time, error) { + start := starts[min(startReads, len(starts)-1)] + startReads++ + return start, nil + }, + networkProbe: func(_ context.Context, request domain.ContainerNetworkProbeRequest) (domain.ContainerNetworkProbeResult, error) { + attempts++ + if attempts == 1 { + return domain.ContainerNetworkProbeResult{}, fmt.Errorf("execution changed: %w", domain.ErrAppStateConflict) + } + assert.Equal(t, second, request.ExpectedStartedAt, "the retry must use the new execution boundary") + return domain.ContainerNetworkProbeResult{Ready: true, Status: 200}, nil + }, + } + require.NoError(t, waitInternalReadyWithDeps(context.Background(), deps, "c-api", internalSpec(), "net", 8080)) + assert.Equal(t, 2, attempts) +} + +// TestWaitInternalReady_StaleGenerationUntilDeadlineTimesOut proves a +// target that keeps restarting never gets reported ready: the wait ends at +// the readiness deadline, not with a false infrastructure error. +func TestWaitInternalReady_StaleGenerationUntilDeadlineTimesOut(t *testing.T) { + spec := internalSpec() + spec.Readiness.Timeout = 50 * time.Millisecond + deps := ProbeDeps{ + containerStart: func(context.Context, string) (time.Time, error) { + return time.Unix(1700000000, 0).UTC(), nil + }, + networkProbe: func(context.Context, domain.ContainerNetworkProbeRequest) (domain.ContainerNetworkProbeResult, error) { + return domain.ContainerNetworkProbeResult{}, fmt.Errorf("gone: %w", domain.ErrContainerNotFound) + }, + } + err := waitInternalReadyWithDeps(context.Background(), deps, "c-api", spec, "net", 8080) + require.Error(t, err) + assert.NotContains(t, err.Error(), "infrastructure error") +} + +// TestWaitInternalReady_VanishedCandidateKeepsPolling proves a candidate +// that is gone on the boundary re-read is still "not ready yet": the wait +// keeps its previous execution boundary and polls to the readiness +// deadline instead of aborting as an infrastructure error. +func TestWaitInternalReady_VanishedCandidateKeepsPolling(t *testing.T) { + spec := internalSpec() + spec.Readiness.Timeout = 600 * time.Millisecond + startReads := 0 + attempts := 0 + deps := ProbeDeps{ + containerStart: func(context.Context, string) (time.Time, error) { + startReads++ + if startReads == 1 { + return time.Unix(1700000000, 0).UTC(), nil + } + return time.Time{}, fmt.Errorf("gone: %w", domain.ErrContainerNotFound) + }, + networkProbe: func(context.Context, domain.ContainerNetworkProbeRequest) (domain.ContainerNetworkProbeResult, error) { + attempts++ + return domain.ContainerNetworkProbeResult{}, fmt.Errorf("gone: %w", domain.ErrContainerNotFound) + }, + } + err := waitInternalReadyWithDeps(context.Background(), deps, "c-api", spec, "net", 8080) + require.ErrorContains(t, err, "timeout") + assert.NotContains(t, err.Error(), "infrastructure error") + assert.GreaterOrEqual(t, attempts, 2, "a vanished candidate must be retried, not abort the wait") + assert.GreaterOrEqual(t, startReads, 2, "a vanished candidate must be re-read, not abort the wait") +} + +// TestReadinessContainerPort_PrefersSinglePublicHTTP proves a public+ +// internal service with no explicit readiness port probes the public HTTP +// backend instead of failing on an unresolved port. +func TestReadinessContainerPort_PrefersSinglePublicHTTP(t *testing.T) { + mixed := domain.AppService{ + HTTP: []domain.AppHTTPInterface{ + {Host: "blog.example.com", Port: 3000, TLS: "auto"}, + {Port: 8080, Visibility: domain.AppVisibilityInternal}, + }, + } + assert.Equal(t, 3000, readinessContainerPort(mixed)) + + single := domain.AppService{HTTP: []domain.AppHTTPInterface{{Host: "blog.example.com", Port: 3000, TLS: "auto"}}} + assert.Equal(t, 3000, readinessContainerPort(single)) + + tcpOnly := domain.AppService{TCP: []domain.AppTCPInterface{{Port: 5432}}} + assert.Equal(t, 5432, readinessContainerPort(tcpOnly)) +} + +// TestWaitServiceReady_InternalUsesNetworkPath proves the dispatcher routes +// an internal readiness port to the bounded network probe and never to the +// loopback path. +func TestWaitServiceReady_InternalUsesNetworkPath(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "inc-1"}, nil) + + networkCalls := 0 + loopbackCalls := 0 + service := NewService(Deps{State: state, Networks: NetworkConfig{Prefix: "gordon"}}, zerowrap.Default()). + WithProbeDeps(ProbeDeps{ + containerStart: func(context.Context, string) (time.Time, error) { + return time.Unix(1700000000, 0).UTC(), nil + }, + networkProbe: func(_ context.Context, request domain.ContainerNetworkProbeRequest) (domain.ContainerNetworkProbeResult, error) { + networkCalls++ + assert.Equal(t, domain.AppPrivateNetworkName("gordon", "inc-1"), request.Network) + return domain.ContainerNetworkProbeResult{Ready: true, Status: 200}, nil + }, + httpGet: func(context.Context, string, string) (int, error) { + loopbackCalls++ + return 200, nil + }, + }) + + require.NoError(t, service.waitServiceReady(ctx, "blog", "c-api", internalSpec(), nil)) + assert.Equal(t, 1, networkCalls) + assert.Zero(t, loopbackCalls, "an internal port must never take the loopback path") +} + +// TestWaitServiceReady_PublicKeepsLoopback proves a public readiness port +// still probes the loopback bind and never the network path. +func TestWaitServiceReady_PublicKeepsLoopback(t *testing.T) { + ctx := context.Background() + networkCalls := 0 + loopbackCalls := 0 + service := NewService(Deps{}, zerowrap.Default()). + WithProbeDeps(ProbeDeps{ + httpGet: func(context.Context, string, string) (int, error) { + loopbackCalls++ + return 200, nil + }, + networkProbe: func(context.Context, domain.ContainerNetworkProbeRequest) (domain.ContainerNetworkProbeResult, error) { + networkCalls++ + return domain.ContainerNetworkProbeResult{}, nil + }, + }) + + spec := readySpec() + require.NoError(t, service.waitServiceReady(ctx, "blog", "c-web", spec, map[int]int{8080: 18080})) + assert.Equal(t, 1, loopbackCalls) + assert.Zero(t, networkCalls, "a public port must never take the network path") +} diff --git a/internal/usecase/deployment/interrupted.go b/internal/usecase/deployment/interrupted.go new file mode 100644 index 000000000..93bc910c2 --- /dev/null +++ b/internal/usecase/deployment/interrupted.go @@ -0,0 +1,244 @@ +package deployment + +import ( + "context" + "fmt" + + "github.com/bnema/zerowrap" + + "github.com/bnema/gordon/internal/domain" +) + +// reconcileInterruptedDeploy converges the journal of a deployment that a +// crash, shutdown, or failed cleanup interrupted. A candidate that was +// created but never published is removed and its loopback claims released, so +// recovery can rebuild the published generation without ever running two +// generations of one service. The interrupted operation is then finalized as +// failed: a repeat of its key replays an explicit failure instead of a +// permanent in-flight claim. +// +// It also retries the candidates of a failed operation whose cleanup did not +// complete, so a leftover stays visible and is eventually removed. The +// published generation recorded in ACTIVE is never touched. +// +// It runs at boot and before every mutation of the app, always under the app +// lock, so a non-terminal journal is definitively interrupted, and a mutation +// that cannot reconcile a predecessor does not proceed. It reads the +// operation journal first and only then ACTIVE, so the common case of a +// successful operation costs one state read. +// +// Reconciliation fails closed on a candidate that cannot be removed: an +// orphan that may still run aborts the caller instead of being journaled as a +// recoverable leftover. Because the journal is never finalized while such a +// leftover exists, it stays the latest operation and a newer operation can +// never be created on top of it and mask it. +func (s *Service) reconcileInterruptedDeploy(ctx context.Context, app string) error { + op, ok, err := s.deps.State.LoadLatestOperation(ctx, app) + if err != nil { + return fmt.Errorf("deployment: load interrupted operation for %q: %w", app, err) + } + if !ok { + return nil + } + if !op.Terminal() && s.operationLive(app, op.Op) { + // The operation is owned by a live goroutine in this process that + // may still be persisting its outcome. Reconcile it only after that + // owner has exited; until then the store record is authoritative and + // must not be finalized here. + return nil + } + terminal := op.Terminal() + needsActive := false + for _, step := range op.Steps { + if reconcileNeedsContainer(step, terminal) { + needsActive = true + break + } + } + var active domain.AppActive + if needsActive { + active, _, err = s.deps.State.LoadActive(ctx, app) + if err != nil { + return fmt.Errorf("deployment: load active for interrupted operation: %w", err) + } + } + log := zerowrap.FromCtx(ctx) + // write is true when the journal itself changes: an interrupted operation + // is always finalized, a terminal one only when a leftover must be + // reported. + changed, err := s.convergeInterruptedSteps(ctx, app, &op, active, terminal, log) + if err != nil { + return err + } + write := !terminal || changed + if !write { + return nil + } + if !terminal { + // The effects ran and only the outcome write was interrupted: that is + // a success, not a failure. + op.Outcome = interruptedOutcome(op) + } + if err := s.deps.State.SaveOperation(ctx, op); err != nil { + return fmt.Errorf("deployment: finalize interrupted operation for %q: %w", app, err) + } + log.Warn().Str("app", app).Str("op", op.Op).Msg("deployment: reconciled an unfinished operation") + return nil +} + +// reconcileNeedsContainer reports whether one step still needs container +// work: a step that recorded a candidate. A terminal operation only retries +// the step that recorded a failed candidate; a pending step exists only in an +// interrupted operation. +func reconcileNeedsContainer(step domain.AppOperationStep, terminal bool) bool { + if step.After == "" { + return false + } + if terminal { + return step.State == domain.AppStepFailed + } + return true +} + +// convergeInterruptedSteps retries the recorded candidate of every step that +// needs container work and accumulates the bounded leftovers in the journal. +// It reports whether the journal changed and fails closed on the first +// candidate that cannot be removed, before any later step or mutation runs. +func (s *Service) convergeInterruptedSteps(ctx context.Context, app string, op *domain.AppOperation, active domain.AppActive, terminal bool, log zerowrap.Logger) (bool, error) { + changed := false + for i := range op.Steps { + step := &op.Steps[i] + if !reconcileNeedsContainer(*step, terminal) { + continue + } + warnings, convErr := s.convergeCandidate(ctx, app, active, step, terminal, log) + if step.After == "" { + // Converging the candidate clears its recorded After, so the + // journal changed even when nothing was left over to report: the + // converged step must be persisted, or every later pass reconciles + // it again. + changed = true + } + if len(warnings) > 0 { + // A candidate that will not go away is operator-visible instead of + // staying untracked. + op.Warnings = append(op.Warnings, journalWarnings(warnings)...) + changed = true + } + if convErr != nil { + // Fail closed: the orphan may still be running, so no later + // mutation may proceed. The journal keeps the leftover (and the + // failed step) for the next pass; a terminal operation stays the + // latest record because no newer claim is opened. + if saveErr := s.deps.State.SaveOperation(ctx, *op); saveErr != nil { + log.Warn().Err(saveErr).Msg("deployment: failed to record reconciliation failure") + } + return changed, fmt.Errorf("deployment: reconcile incomplete operation for %q: %w", app, convErr) + } + } + return changed, nil +} + +// convergeCandidate converges the container one step recorded. The published +// generation is kept; a never-published candidate is removed so recovery can +// rebuild the generation recorded in ACTIVE without two generations running +// at once. An interrupted operation's step is marked failed. +// +// A candidate that cannot be removed is returned as an error, not a +// recoverable warning: the orphan may still run, so every later mutation must +// fail closed instead of proceeding. The returned warnings are the leftovers +// of that removal. +// +// Success clears the step's recorded After and keeps the removed (or kept) +// container id in Detail, so the audit trail survives while the converged step +// is no longer retried on every later pass. Only the failure paths leave After +// in place, keeping a failed convergence eligible for retry. +// +// Once the candidate is converged, the inhibition the replacement wrote for +// the superseded container is cleared. A step carrying both Before and After +// records that the superseded container was confirmed gone before the +// candidate was created (deployService retires it first), so the marker +// protects nothing and would otherwise refuse the boot/start/restart rebuild +// from ACTIVE. +func (s *Service) convergeCandidate(ctx context.Context, app string, active domain.AppActive, step *domain.AppOperationStep, terminal bool, log zerowrap.Logger) ([]CleanupWarning, error) { + candidate := step.After + if publishedContainer(active, candidate) { + // The published generation: keep it. A step that already succeeded is + // left as recorded; a routing apply that never ran is republished by + // the periodic recovery pass. + if !terminal && step.State != domain.AppStepSucceeded { + step.State = domain.AppStepFailed + step.Error = "interrupted before traffic publication; the published container is kept and routing is republished" + } + step.Detail = "converged: published container " + candidate + " kept" + } else { + retired := s.retireContainer(ctx, app, retireOptions{Service: step.Service, Force: true}, candidate) + if !retired.Gone { + if !terminal { + step.State = domain.AppStepFailed + step.Error = "interrupted before publication; the unpublished candidate could not be removed: " + cleanupDetail(retired) + log.Warn().Str("app", app).Str("container", candidate).Msg("deployment: unpublished candidate could not be removed") + } + return retired.Warnings, fmt.Errorf("deployment: candidate %s of service %q could not be removed: %s", candidate, step.Service, cleanupDetail(retired)) + } + if !terminal { + step.State = domain.AppStepFailed + step.Error = "interrupted before publication; the unpublished candidate was removed" + } + step.Detail = "converged: removed unpublished candidate " + candidate + } + // The candidate is gone (or published): the replacement-pending + // inhibition of the superseded container is stale. Dropping it lets + // boot/start/restart rebuild that generation from ACTIVE; the single + // writer is preserved because the superseded container was proven gone + // before the candidate existed. + if step.Service != "" && step.Before != "" && step.Before != candidate { + if err := s.clearRecoveryInhibition(ctx, app, step.Service, step.Before); err != nil { + log.Warn().Err(err).Str("app", app).Str("container", step.Before).Msg("deployment: clear stale replacement inhibition") + return []CleanupWarning{{Service: step.Service, Leftover: step.Before, Detail: "clear recovery inhibition: " + err.Error()}}, nil + } + } + // Only reconciliation removes a recorded candidate: clearing After here is + // what marks the step converged. + step.After = "" + return nil, nil +} + +// interruptedOutcome classifies an interrupted operation: a failure, unless +// every step had already succeeded. +func interruptedOutcome(op domain.AppOperation) string { + if operationSucceeded(op) { + return domain.AppOutcomeSuccess + } + return domain.AppOutcomeFailed +} + +// operationSucceeded reports whether every step of an interrupted operation +// reached success: its effects ran and only the final outcome write was +// interrupted. +func operationSucceeded(op domain.AppOperation) bool { + if len(op.Steps) == 0 { + return false + } + for _, step := range op.Steps { + if step.State != domain.AppStepSucceeded { + return false + } + } + return true +} + +// publishedContainer reports whether any service of the ACTIVE record names +// the container. A container that is the published generation is never a +// leftover, whichever service recorded it. +func publishedContainer(active domain.AppActive, containerID string) bool { + if containerID == "" { + return false + } + for _, service := range active.Services { + if service.Container == containerID { + return true + } + } + return false +} diff --git a/internal/usecase/deployment/lifecycle.go b/internal/usecase/deployment/lifecycle.go new file mode 100644 index 000000000..303b7b3f5 --- /dev/null +++ b/internal/usecase/deployment/lifecycle.go @@ -0,0 +1,837 @@ +package deployment + +import ( + "context" + "errors" + "fmt" + "maps" + "sort" + "strings" + "time" + + "github.com/bnema/zerowrap" + + "github.com/bnema/gordon/internal/domain" +) + +// LifecycleResult carries the terminal outcome of a lifecycle verb. +type LifecycleResult struct { + Op string + App string + Verb string + Outcome string + Services map[string]ServiceResult + // Warnings are operation-level leftovers that belong to no single + // service, such as a private network that could not be verified or + // removed. + Warnings []CleanupWarning +} + +// Stop persists the durable stopped intent first, then stops and removes +// exact active containers by ID. Volumes and secrets are retained. +func (s *Service) Stop(ctx context.Context, app, opID string) (*LifecycleResult, error) { + release, err := s.acquireAppContext(ctx, app) + if err != nil { + return nil, err + } + defer release() + return s.stopLocked(ctx, app, opID) +} + +func (s *Service) stopLocked(ctx context.Context, app, opID string) (*LifecycleResult, error) { + ctx = lifecycleCtx(ctx, "Stop", app) + log := zerowrap.FromCtx(ctx) + if err := s.deps.State.Recover(ctx); err != nil { + return nil, fmt.Errorf("deployment: recover before stop: %w", err) + } + if err := s.reconcileInterruptedDeploy(ctx, app); err != nil { + return nil, err + } + op := domain.AppOperation{ + Kind: "stop", App: app, + StartedAt: time.Now().UTC(), + Request: domain.AppOperationRequestFor("stop", app, "", ""), + } + op, owned, err := s.claimOperation(ctx, opID, op, []domain.AppOperationStep{{ID: "intent.stopped", State: domain.AppStepPending}}) + if err != nil { + return nil, err + } + if !owned { + return replayedLifecycleResult(op), replayError(op) + } + if err := s.deps.State.SaveIntent(ctx, domain.AppStopIntent{ + App: app, Stopped: true, UpdatedBy: op.Op, UpdatedAt: time.Now().UTC(), + }); err != nil { + intentErr := fmt.Errorf("deployment: persist stopped intent: %w", err) + s.failOperation(ctx, &op, intentErr, "error") + return nil, intentErr + } + op.Steps[0].State = domain.AppStepSucceeded + active, _, err := s.deps.State.LoadActive(ctx, app) + if err != nil { + activeErr := fmt.Errorf("deployment: load active: %w", err) + s.failOperation(ctx, &op, activeErr, "error") + return nil, activeErr + } + result := &LifecycleResult{Op: op.Op, App: app, Verb: "stop", Services: map[string]ServiceResult{}} + var failures []string + for _, name := range sortedServiceNames(active) { + container := active.Services[name].Container + step, warnings, err := s.stopService(ctx, app, name, active.Services[name]) + op.Steps = append(op.Steps, step) + if err != nil { + result.Services[name] = ServiceResult{Result: "failed", Before: container, Error: err.Error(), CleanupWarnings: warnings} + failures = append(failures, name) + continue + } + result.Services[name] = ServiceResult{Result: "deployed", Before: container, After: "", CleanupWarnings: warnings} + } + op.Outcome = ComputeOutcome(result.Services) + op.Warnings = journalWarnings(collectCleanupWarnings(result.Services)) + if err := s.deps.State.SaveOperation(ctx, op); err != nil { + log.Warn().Err(err).Msg("deployment: failed to record stop outcome") + } + if err := s.refreshTraffic(ctx, app); err != nil { + s.failTrafficPublication(ctx, &op, err) + return result, err + } + if len(failures) > 0 { + return result, fmt.Errorf("deployment: stop failed for %s: %w", strings.Join(failures, ", "), domain.ErrAppStateConflict) + } + return result, nil +} + +// stopService withdraws one service, then retires its exact container +// with the effective grace. A withdrawal or runtime failure is reported, +// never treated as a successful stop. The confirmed disappearance +// releases the container's backend claims and its recovery inhibition: +// a stopped app is never revived by recovery. +func (s *Service) stopService(ctx context.Context, app, name string, eff domain.AppEffectiveService) (domain.AppOperationStep, []CleanupWarning, error) { + container := eff.Container + step := domain.AppOperationStep{ID: "service." + name + ".stop", State: domain.AppStepPending, Before: container} + fail := func(err error) (domain.AppOperationStep, []CleanupWarning, error) { + step.State = domain.AppStepFailed + step.Error = err.Error() + return step, nil, err + } + if err := s.withdrawForRecovery(ctx, app, name); err != nil { + return fail(err) + } + if container != "" { + retired := s.retireContainer(ctx, app, retireOptions{ + Service: name, Grace: serviceStopGrace(eff), ClearInhibition: true, + }, container) + if !retired.Gone { + return fail(fmt.Errorf("deployment: stop %s/%s container %s: %s", app, name, container, cleanupDetail(retired))) + } + step.After = container + step.State = domain.AppStepSucceeded + return step, retired.Warnings, nil + } + step.State = domain.AppStepSucceeded + return step, nil, nil +} + +// Start clears the stopped intent and ensures running from active records. +// It starts missing/stopped instances without duplicating running ones +// and never activates pending desired revisions. +func (s *Service) Start(ctx context.Context, app, opID string) (*LifecycleResult, error) { + release, err := s.acquireAppContext(ctx, app) + if err != nil { + return nil, err + } + defer release() + return s.startLocked(ctx, app, opID) +} + +func (s *Service) startLocked(ctx context.Context, app, opID string) (*LifecycleResult, error) { + ctx = lifecycleCtx(ctx, "Start", app) + log := zerowrap.FromCtx(ctx) + if err := s.deps.State.Recover(ctx); err != nil { + return nil, fmt.Errorf("deployment: recover before start: %w", err) + } + if err := s.reconcileInterruptedDeploy(ctx, app); err != nil { + return nil, err + } + op := domain.AppOperation{ + Kind: "start", App: app, + StartedAt: time.Now().UTC(), + Request: domain.AppOperationRequestFor("start", app, "", ""), + } + op, owned, err := s.claimOperation(ctx, opID, op, []domain.AppOperationStep{{ID: "intent.running", State: domain.AppStepPending}}) + if err != nil { + return nil, err + } + if !owned { + return replayedLifecycleResult(op), replayError(op) + } + if err := s.deps.State.SaveIntent(ctx, domain.AppStopIntent{ + App: app, Stopped: false, UpdatedBy: op.Op, UpdatedAt: time.Now().UTC(), + }); err != nil { + intentErr := fmt.Errorf("deployment: clear stopped intent: %w", err) + s.failOperation(ctx, &op, intentErr, "error") + return nil, intentErr + } + op.Steps[0].State = domain.AppStepSucceeded + active, _, err := s.deps.State.LoadActive(ctx, app) + if err != nil { + activeErr := fmt.Errorf("deployment: load active: %w", err) + s.failOperation(ctx, &op, activeErr, "error") + return nil, activeErr + } + result := &LifecycleResult{Op: op.Op, App: app, Verb: "start", Services: map[string]ServiceResult{}} + for _, name := range sortedServiceNames(active) { + eff := active.Services[name] + // The step is journaled before the runtime work, so a candidate + // created by a redeploy is durably recorded as soon as it exists. + op.Steps = append(op.Steps, domain.AppOperationStep{ + ID: "service." + name + ".start", State: domain.AppStepPending, Before: eff.Container, Service: name, + }) + s.ensureServiceRunning(ctx, app, opID, name, eff, &op, len(op.Steps)-1, result) + } + op.Outcome = ComputeOutcome(result.Services) + op.Warnings = journalWarnings(collectCleanupWarnings(result.Services)) + if err := s.deps.State.SaveOperation(ctx, op); err != nil { + log.Warn().Err(err).Msg("deployment: failed to record start outcome") + } + if err := s.refreshTraffic(ctx, app); err != nil { + s.failTrafficPublication(ctx, &op, err) + return result, err + } + return result, nil +} + +// Restart restarts from pinned digests without re-resolution. Empty +// service means all services in sorted order. Every service restarts in +// place: traffic is withdrawn, the same pinned container is restarted, its +// readiness probe is checked, and traffic is republished. No second +// container is created and the pinned digest never changes. +func (s *Service) Restart(ctx context.Context, app, service, opID string) (*LifecycleResult, error) { + release, err := s.acquireAppContext(ctx, app) + if err != nil { + return nil, err + } + defer release() + return s.restartLocked(ctx, app, service, opID) +} + +func (s *Service) restartLocked(ctx context.Context, app, service, opID string) (*LifecycleResult, error) { + ctx = lifecycleCtx(ctx, "Restart", app) + if err := s.deps.State.Recover(ctx); err != nil { + return nil, fmt.Errorf("deployment: recover before restart: %w", err) + } + if err := s.reconcileInterruptedDeploy(ctx, app); err != nil { + return nil, err + } + active, ok, err := s.deps.State.LoadActive(ctx, app) + if err != nil { + return nil, fmt.Errorf("deployment: load active: %w", err) + } + if !ok { + return nil, fmt.Errorf("deployment: app %q was never deployed: %w", app, domain.ErrAppStateConflict) + } + names := sortedServiceNames(active) + if service != "" { + if _, ok := active.Services[service]; !ok { + return nil, fmt.Errorf("deployment: service %q not active: %w", service, domain.ErrAppStateConflict) + } + names = []string{service} + } + op := domain.AppOperation{ + Kind: "restart", App: app, + StartedAt: time.Now().UTC(), + Request: domain.AppOperationRequestFor("restart", app, "", service), + } + op, owned, err := s.claimOperation(ctx, opID, op, nil) + if err != nil { + return nil, err + } + if !owned { + return replayedLifecycleResult(op), replayError(op) + } + result := &LifecycleResult{Op: op.Op, App: app, Verb: "restart", Services: map[string]ServiceResult{}} + var failures []string + for _, name := range names { + eff := active.Services[name] + // The step is journaled before the runtime work, so a candidate + // created by a rebuild of a missing generation is durably recorded. + op.Steps = append(op.Steps, domain.AppOperationStep{ + ID: "service." + name + ".restart", State: domain.AppStepPending, Before: eff.Container, Service: name, + }) + svcResult := s.restartOneService(ctx, app, op.Op, name, eff, &op, len(op.Steps)-1) + result.Services[name] = svcResult + if svcResult.Result != "deployed" { + failures = append(failures, name) + } + } + op.Outcome = ComputeOutcome(result.Services) + op.Warnings = journalWarnings(collectCleanupWarnings(result.Services)) + if err := s.deps.State.SaveOperation(ctx, op); err != nil { + return result, fmt.Errorf("deployment: persist restart journal: %w", err) + } + if err := s.refreshTraffic(ctx, app); err != nil { + s.failTrafficPublication(ctx, &op, err) + return result, err + } + if len(failures) > 0 { + return result, fmt.Errorf("deployment: restart failed for %s: %w", strings.Join(failures, ", "), domain.ErrAppStateConflict) + } + return result, nil +} + +// restartOneService withdraws one service, restarts its exact container, +// re-inspects its binds, and verifies readiness before it may be published +// again. Every failure leaves the service withdrawn with its recorded binds +// cleared, so a failed restart is never served. A generation whose recorded +// container no longer exists is rebuilt from the pinned ACTIVE digest, so a +// failed replacement cannot leave restart as a permanent dead end. +func (s *Service) restartOneService(ctx context.Context, app, opID, name string, eff domain.AppEffectiveService, op *domain.AppOperation, index int) ServiceResult { + step := &op.Steps[index] + result := ServiceResult{Result: "failed", Before: eff.Container, After: eff.Container} + fail := func(msg string) ServiceResult { + step.State = domain.AppStepFailed + step.Error = msg + result.Result = "failed" + result.Error = msg + return result + } + if eff.Container == "" { + result.After = "" + return fail("no active container") + } + // Fail closed before any runtime mutation if a bind's policy was + // revoked or became invalid since ACTIVE was published. + if _, err := s.resolveServiceBinds(app, eff.Spec); err != nil { + return fail(err.Error()) + } + // A device grant revoked since ACTIVE was published must fail + // before any runtime mutation, under the same rule as binds. An + // in-place restart never rewrites device configuration. + if _, err := s.resolveServiceDevices(app, eff.Spec); err != nil { + return fail(err.Error()) + } + if err := s.refuseInhibitedRestart(ctx, app, name, eff.Container); err != nil { + _ = s.withdrawForRecovery(ctx, app, name) + return fail(err.Error()) + } + // Withdraw before restarting: a restarting or unverified generation + // must not keep receiving traffic. + if err := s.withdrawForRecovery(ctx, app, name); err != nil { + return fail(err.Error()) + } + if err := s.deps.Runtime.RestartContainer(ctx, eff.Container, serviceStopGrace(eff)); err != nil { + if errors.Is(err, domain.ErrContainerNotFound) { + // The recorded generation is gone: rebuild and publish it from the + // pinned ACTIVE digest instead of leaving the service withdrawn + // until a deploy. + svcResult, rebuildErr := s.redeployPinned(ctx, app, opID, name, eff, s.journalCandidate(op, index)) + if rebuildErr != nil { + svcResult.Error = rebuildErr.Error() + step.State = domain.AppStepFailed + step.Error = rebuildErr.Error() + step.Diagnostics = svcResult.Diagnostics + // The recorded container is proven gone: the inhibition the + // rebuild wrote for it protects nothing, and keeping it would + // refuse every later start, recovery pass, and restart. + if clearErr := s.clearRecoveryInhibition(ctx, app, name, eff.Container); clearErr != nil { + svcResult.CleanupWarnings = append(svcResult.CleanupWarnings, CleanupWarning{ + Service: name, Leftover: eff.Container, Detail: "clear recovery inhibition: " + clearErr.Error(), + }) + } + return svcResult + } + step.State = domain.AppStepSucceeded + step.After = svcResult.After + return svcResult + } + s.clearServiceBinds(ctx, app, name) + return fail(err.Error()) + } + // Ephemeral loopback binds are not guaranteed stable across a runtime + // restart: re-inspect before probing so the proxy never dials a stale + // bind. + binds, udpBinds, err := s.refreshBackendBinds(ctx, app, name, eff, s.deps.Traffic != nil) + if err != nil { + s.clearServiceBinds(ctx, app, name) + return fail(err.Error()) + } + if err := s.waitServiceReady(ctx, app, eff.Container, deploymentReadiness(eff.Spec), binds); err != nil { + s.clearServiceBinds(ctx, app, name) + return fail(err.Error()) + } + step.State = domain.AppStepSucceeded + step.After = eff.Container + return ServiceResult{Result: "deployed", Before: eff.Container, After: eff.Container, BackendBinds: binds, UDPBackendBinds: udpBinds} +} + +// Remove withdraws workloads by exact container ID; volumes, secrets, and +// ownership records are retained under the old UUID. The public name is +// freed for reuse; a new app never implicitly adopts retained resources. +// Stopped intent and a durable recovery inhibition per removed container +// are persisted BEFORE any runtime effect, so native restart policy can +// never revive a removed generation into service. +func (s *Service) Remove(ctx context.Context, app, opID string) (*LifecycleResult, error) { + release, err := s.acquireAppContext(ctx, app) + if err != nil { + return nil, err + } + defer release() + return s.removeLocked(ctx, app, opID) +} + +func (s *Service) removeLocked(ctx context.Context, app, opID string) (*LifecycleResult, error) { + ctx = lifecycleCtx(ctx, "Remove", app) + log := zerowrap.FromCtx(ctx) + if err := s.deps.State.Recover(ctx); err != nil { + return nil, fmt.Errorf("deployment: recover before remove: %w", err) + } + if err := s.reconcileInterruptedDeploy(ctx, app); err != nil { + return nil, err + } + op := domain.AppOperation{ + Kind: "remove", App: app, + StartedAt: time.Now().UTC(), + Request: domain.AppOperationRequestFor("remove", app, "", ""), + } + op, owned, err := s.claimOperation(ctx, opID, op, nil) + if err != nil { + return nil, err + } + if !owned { + return replayedLifecycleResult(op), replayError(op) + } + active, _, err := s.deps.State.LoadActive(ctx, app) + if err != nil { + activeErr := fmt.Errorf("deployment: load active: %w", err) + s.failOperation(ctx, &op, activeErr, "error") + return nil, activeErr + } + result := &LifecycleResult{Op: op.Op, App: app, Verb: "remove", Services: map[string]ServiceResult{}} + // Durable stopped intent and per-container inhibition precede every + // runtime effect: a crash between here and the container stop must + // never leave a generation that recovery could restart. + if err := s.deps.State.SaveIntent(ctx, domain.AppStopIntent{ + App: app, Stopped: true, UpdatedBy: op.Op, UpdatedAt: time.Now().UTC(), + }); err != nil { + intentErr := fmt.Errorf("deployment: persist stopped intent before remove: %w", err) + s.failOperation(ctx, &op, intentErr, "error") + return nil, intentErr + } + steps, err := s.removeServiceContainers(ctx, app, op.Op, active, result) + if err != nil { + op.Steps = steps + s.failOperation(ctx, &op, err, "error") + return nil, err + } + // Every exact container is confirmed gone, so the app's private + // networks are reclaimed immediately, while the ownership record is + // still live and its UUID can be verified against the runtime names + // and labels. Shared networks stay, and retained volumes, secrets, and + // images are never touched. + ownership, err := s.deps.State.LoadOwnership(ctx, app) + if err != nil { + retireErr := fmt.Errorf("deployment: load ownership before reclamation: %w", err) + s.failOperation(ctx, &op, retireErr, "error") + return nil, retireErr + } + networkWarnings, err := s.reclaimPrivateNetworks(ctx, app, ownership) + if err != nil { + s.failOperation(ctx, &op, err, "error") + return nil, err + } + result.Warnings = append(result.Warnings, networkWarnings...) + op.Steps = steps + // End the incarnation atomically: the ownership record is archived + // with its resources retained under the old UUID, the app UUID is + // reset so a name reuse allocates a new incarnation, and desired, + // active, intent, and staged state are cleared. Without this a + // reapply would reuse the UUID and inherit the old secrets and + // volumes. + if err := s.deps.State.RetireApp(ctx, app); err != nil { + retireErr := fmt.Errorf("deployment: retire incarnation: %w", err) + s.failOperation(ctx, &op, retireErr, "error") + return nil, retireErr + } + op.Outcome = domain.AppOutcomeSuccess + op.Warnings = journalWarnings(append(collectCleanupWarnings(result.Services), result.Warnings...)) + if err := s.deps.State.SaveOperation(ctx, op); err != nil { + log.Warn().Err(err).Msg("deployment: failed to record remove outcome") + } + if err := s.refreshTraffic(ctx, app); err != nil { + s.failTrafficPublication(ctx, &op, err) + return result, err + } + return result, nil +} + +// replayedLifecycleResult projects a replayed journal into a lifecycle +// result without touching any workload. +func replayedLifecycleResult(op domain.AppOperation) *LifecycleResult { + return &LifecycleResult{Op: op.Op, App: op.App, Verb: op.Kind, Outcome: op.Outcome} +} + +// refuseInhibitedRestart blocks a restart of a generation whose recovery +// is durably inhibited: a replacement may already have written to the +// volume this generation still owns, so reviving it could corrupt newer +// data. Deploying a new revision supersedes the marker safely. +func (s *Service) refuseInhibitedRestart(ctx context.Context, app, name, containerID string) error { + inhibited, err := s.recoveryInhibited(ctx, app, name, containerID) + if err != nil { + return err + } + if inhibited { + return fmt.Errorf("deployment: restart %q/%q refused: recovery inhibited for container %s", app, name, containerID) + } + return nil +} + +// removeServiceContainers persists one inhibition per removed container +// before stopping and removing it by exact ID. Volumes are never deleted. +// A container that could not be confirmed gone fails the remove: ACTIVE, +// stopped intent, and the inhibition stay in place so the periodic pass +// keeps converging the survivor instead of losing track of it. +func (s *Service) removeServiceContainers(ctx context.Context, app, opID string, active domain.AppActive, result *LifecycleResult) ([]domain.AppOperationStep, error) { + var steps []domain.AppOperationStep + for _, name := range sortedServiceNames(active) { + container := active.Services[name].Container + step := domain.AppOperationStep{ID: "service." + name + ".remove", State: domain.AppStepPending, Before: container} + if container != "" { + if err := s.inhibitRecovery(ctx, app, name, container, "removed", opID); err != nil { + return nil, err + } + retired := s.retireContainer(ctx, app, retireOptions{ + Service: name, Grace: serviceStopGrace(active.Services[name]), ClearInhibition: true, + }, container) + if !retired.Gone { + return nil, fmt.Errorf("deployment: remove %q/%q: container %s not confirmed gone: %s", app, name, container, cleanupDetail(retired)) + } + result.Services[name] = ServiceResult{Result: "deployed", Before: container, After: "", CleanupWarnings: retired.Warnings} + } else { + result.Services[name] = ServiceResult{Result: "deployed", Before: container, After: ""} + } + step.State = domain.AppStepSucceeded + steps = append(steps, step) + } + return steps, nil +} + +// ReconcileBoot reconciles apps intended to run after a daemon boot: +// every non-stopped app with ACTIVE services is verified through Start, +// including all-running apps: ephemeral loopback binds may shift across +// any runtime restart while the daemon is away, so recorded binds are +// re-inspected (never trusted) before the proxy can dial them. +// Recovery first; no duplication of running instances; no activation of +// pending desired revisions. Explicitly stopped apps stay stopped. +// A failing app never blocks the verification of the following apps; +// failures are aggregated and every unverified bind is withdrawn from +// ACTIVE (see failServiceStep), so the post-boot host index rebuild +// projects fail-closed for all of them. +func (s *Service) ReconcileBoot(ctx context.Context) error { + ctx = lifecycleCtx(ctx, "ReconcileBoot", "boot") + if err := s.deps.State.Recover(ctx); err != nil { + return fmt.Errorf("deployment: recover before boot reconcile: %w", err) + } + apps, err := s.deps.State.ListApps(ctx) + if err != nil { + return fmt.Errorf("deployment: list apps: %w", err) + } + var failures []error + for _, app := range apps { + if err := s.reconcileBootApp(ctx, app); err != nil { + failures = append(failures, err) + } + } + return errors.Join(failures...) +} + +// reconcileBootApp verifies one app at boot under the app lock. Explicitly +// stopped apps and apps without ACTIVE services are skipped without error. +// Stopped intent is converged: a container revived by native restart +// policy while the daemon was away is stopped again, never restarted. +func (s *Service) reconcileBootApp(ctx context.Context, app string) error { + release, err := s.acquireAppContext(ctx, app) + if err != nil { + return err + } + defer release() + intent, err := s.deps.State.LoadIntent(ctx, app) + if err != nil { + return fmt.Errorf("deployment: load intent: %w", err) + } + active, ok, err := s.deps.State.LoadActive(ctx, app) + if err != nil { + return fmt.Errorf("deployment: load active: %w", err) + } + if !ok || len(active.Services) == 0 { + // An interrupted operation is reconciled even when ACTIVE carries no + // service: a first deployment can be interrupted between creating its + // candidate and publishing it. No start follows here, so this is the + // only reconciliation of the boot pass. + return s.reconcileInterruptedDeploy(ctx, app) + } + if intent.Stopped { + // A stopped app does not start, so its unfinished operation is + // reconciled here instead. + if err := s.reconcileInterruptedDeploy(ctx, app); err != nil { + return err + } + for _, name := range sortedServiceNames(active) { + if err := s.convergeStoppedService(ctx, app, name, active.Services[name]); err != nil { + return fmt.Errorf("deployment: boot stopped convergence %q/%q: %w", app, name, err) + } + } + return nil + } + // startLocked reconciles the app's unfinished operation before it starts + // anything, so the boot pass reconciles exactly once. + result, err := s.startLocked(ctx, app, "") + if err != nil { + return fmt.Errorf("deployment: boot start %q: %w", app, err) + } + for name, svc := range result.Services { + if svc.Result == "failed" { + return fmt.Errorf("deployment: boot verify %q/%q: %s", app, name, svc.Error) + } + } + return nil +} + +// ensureServiceRunning starts one service when needed. Running +// containers are bind-verified before success: ephemeral loopback binds +// may shift across any runtime restart (stop/start, daemon absence, +// host reboot), so the recorded binds are re-inspected and persisted +// before the proxy can dial them. Verification failure fails the step; +// a stale recorded bind is never served. A recorded container that no +// longer exists is rebuilt from the revision pinned in ACTIVE. +func (s *Service) ensureServiceRunning(ctx context.Context, app, opID, name string, eff domain.AppEffectiveService, op *domain.AppOperation, index int, result *LifecycleResult) { + step := &op.Steps[index] + if eff.Container != "" { + // Fail closed before any runtime mutation if a bind's policy was + // revoked or became invalid since ACTIVE was published. + if _, err := s.resolveServiceBinds(app, eff.Spec); err != nil { + s.failServiceStep(ctx, app, name, eff, step, result, err.Error()) + return + } + // Revoked device grants fail before any runtime mutation. + if _, err := s.resolveServiceDevices(app, eff.Spec); err != nil { + s.failServiceStep(ctx, app, name, eff, step, result, err.Error()) + return + } + // Boot recovery refuses a generation whose recovery is durably + // inhibited: a replacement may already have written to its + // volume, so reviving this ID could corrupt the newer data. + inhibited, err := s.recoveryInhibited(ctx, app, name, eff.Container) + if err != nil { + s.failServiceStep(ctx, app, name, eff, step, result, err.Error()) + return + } + if inhibited { + s.failServiceStep(ctx, app, name, eff, step, result, + fmt.Sprintf("recovery inhibited for container %s", eff.Container)) + return + } + if ok, err := s.deps.Runtime.IsContainerRunning(ctx, eff.Container); err == nil && ok { + s.verifyRunningService(ctx, app, name, eff, step, result) + return + } + // Restart by exact container ID when the runtime still has it. + if err := s.deps.Runtime.StartContainer(ctx, eff.Container); err == nil { + s.verifyRunningService(ctx, app, name, eff, step, result) + return + } + } + // Otherwise rebuild from the revision pinned in ACTIVE. + svcResult, err := s.redeployPinned(ctx, app, opID, name, eff, s.journalCandidate(op, index)) + if err != nil { + step.State = domain.AppStepFailed + step.Error = err.Error() + step.Diagnostics = svcResult.Diagnostics + result.Services[name] = svcResult + return + } + step.State = domain.AppStepSucceeded + step.After = svcResult.After + result.Services[name] = svcResult +} + +// redeployPinned rebuilds one service from the revision pinned in ACTIVE and +// publishes it. It is the shared recovery path for a recorded container that +// no longer exists: boot/start recovery, and a restart of a missing +// generation. The returned result is always populated, so a caller can +// journal the attempt even when err is non-nil. +func (s *Service) redeployPinned(ctx context.Context, app, opID, name string, eff domain.AppEffectiveService, journal candidateJournal) (ServiceResult, error) { + rev, err := s.deps.State.LoadRevision(ctx, app, eff.EffectiveRevision) + if err != nil { + return failedServiceResult(eff.Container, err), err + } + runtimeImage, err := s.preflightImage(ctx, eff.Spec.Image, eff.Digest) + if err != nil { + return failedServiceResult(eff.Container, err), err + } + if _, err := s.resolveServiceBinds(app, eff.Spec); err != nil { + return failedServiceResult(eff.Container, err), err + } + // Revoked device grants fail before any runtime mutation. + if _, err := s.resolveServiceDevices(app, eff.Spec); err != nil { + return failedServiceResult(eff.Container, err), err + } + pinned := pinnedService{ + name: name, spec: eff.Spec, digest: eff.Digest, runtimeImage: runtimeImage, appEnv: maps.Clone(rev.Spec.Env), + appNetworks: append([]domain.AppSharedNetwork(nil), rev.Spec.Networks...), + sharedNetworks: domain.AppServiceSharedNetworks(rev.Spec, name), + } + svcResult := s.deployService(ctx, app, rev.Revision, pinned, opID, eff.Container, serviceStopGrace(eff), journal) + if svcResult.Result == "failed" { + return svcResult, errors.New(svcResult.Error) + } + // The step may only be recorded as succeeded once the effective state is + // published: a journaled success is never ahead of what the proxy reaches. + if err := s.publishService(ctx, app, rev.Revision, pinned, svcResult, opID); err != nil { + svcResult.Result = "failed" + svcResult.Error = err.Error() + return svcResult, err + } + return svcResult, nil +} + +// failedServiceResult builds the terminal result of a service step that +// failed before any container existed. +func failedServiceResult(before string, err error) ServiceResult { + return ServiceResult{Result: "failed", Before: before, Error: err.Error()} +} + +// verifyRunningService withdraws a running or freshly started generation, +// re-inspects its loopback binds, and verifies readiness before the service +// may be published again. Any failure leaves the service withdrawn with its +// recorded binds cleared, so a not-ready backend is never projected. +func (s *Service) verifyRunningService(ctx context.Context, app, name string, eff domain.AppEffectiveService, step *domain.AppOperationStep, result *LifecycleResult) { + if err := s.withdrawForRecovery(ctx, app, name); err != nil { + s.failServiceStep(ctx, app, name, eff, step, result, err.Error()) + return + } + binds, udpBinds, err := s.refreshBackendBinds(ctx, app, name, eff, s.deps.Traffic != nil) + if err != nil { + s.failServiceStep(ctx, app, name, eff, step, result, err.Error()) + return + } + if err := s.waitServiceReady(ctx, app, eff.Container, eff.Spec, binds); err != nil { + s.failServiceStep(ctx, app, name, eff, step, result, err.Error()) + return + } + // The service is running and its re-inspected loopback binds are + // persisted, so the step may succeed: the graph apply for this app + // follows once, after the loop. + step.State = domain.AppStepSucceeded + step.After = eff.Container + result.Services[name] = ServiceResult{Result: "deployed", Before: eff.Container, After: eff.Container, BackendBinds: binds, UDPBackendBinds: udpBinds} +} + +// refreshBackendBinds re-inspects the loopback publishes of a restarted +// container and persists them to the current active record when needed. +// Ephemeral binds may shift across runtime restarts; the proxy must +// never dial the stale recorded bind. Uses the non-destructive +// inspectBackendBinds: an existing ACTIVE container is never stopped +// or removed by re-inspection. +func (s *Service) refreshBackendBinds(ctx context.Context, app, name string, eff domain.AppEffectiveService, forcePersist bool) (map[int]int, map[int]int, error) { + ports := backendPorts(eff.Spec) + if len(ports) == 0 { + return nil, nil, nil + } + binds, udpBinds, err := s.inspectBackendBinds(ctx, app, name, eff.Container, ports) + if err != nil { + return nil, nil, err + } + if !forcePersist && equalBinds(binds, eff.BackendBinds) && equalBinds(udpBinds, eff.UDPBackendBinds) { + return binds, udpBinds, nil + } + active, ok, err := s.deps.State.LoadActive(ctx, app) + if err != nil { + return nil, nil, fmt.Errorf("deployment: reload active for bind refresh: %w", err) + } + if !ok { + return nil, nil, fmt.Errorf("deployment: app %q has no active state: %w", app, domain.ErrAppStateConflict) + } + if err := s.persistBackendBinds(ctx, active, name, eff, binds, udpBinds, forcePersist); err != nil { + return nil, nil, err + } + return binds, udpBinds, nil +} + +func (s *Service) persistBackendBinds(ctx context.Context, active domain.AppActive, name string, eff domain.AppEffectiveService, binds, udpBinds map[int]int, force bool) error { + current, exists := active.Services[name] + if exists && current.Container != eff.Container { + return fmt.Errorf("deployment: active generation changed during bind refresh: %w", domain.ErrAppStateConflict) + } + if !force && exists && equalBinds(binds, current.BackendBinds) && equalBinds(udpBinds, current.UDPBackendBinds) { + return nil + } + if !exists { + current = eff + } + current.BackendBinds = binds + current.UDPBackendBinds = udpBinds + active.Services[name] = current + if err := s.deps.State.SaveActive(ctx, active); err != nil { + return fmt.Errorf("deployment: persist refreshed binds: %w", err) + } + return nil +} + +// failServiceStep records a failed service step and withdraws its +// recorded backend binds from ACTIVE: a bind that failed verification +// (or could not be refreshed) must never be served again — the +// projected backend would otherwise point at a dead or recycled +// loopback port. Invalidation errors are logged, never fatal: the +// step already failed, and the next rebuild projects fail-closed. +func (s *Service) failServiceStep(ctx context.Context, app, name string, eff domain.AppEffectiveService, step *domain.AppOperationStep, result *LifecycleResult, errMsg string) { + step.State = domain.AppStepFailed + step.Error = errMsg + result.Services[name] = ServiceResult{Result: "failed", Before: eff.Container, Error: errMsg} + s.clearServiceBinds(ctx, app, name) +} + +// clearServiceBinds withdraws a service's recorded backend binds from +// ACTIVE through the canonical traffic boundary: a bind that failed +// verification, refresh, or readiness must never be served, because the +// projected backend would point at a dead, unready, or recycled loopback +// port. No graph is applied here: the caller already owns publication +// (or none is due). Errors are logged, never fatal: the step already +// failed and the next rebuild projects fail-closed. +func (s *Service) clearServiceBinds(ctx context.Context, app, name string) { + if s.deps.Traffic == nil { + return + } + if err := s.deps.Traffic.WithdrawServiceState(ctx, app, name); err != nil { + log := zerowrap.FromCtx(ctx) + log.Warn().Err(err).Str("app", app).Str("service", name).Msg("deployment: failed to withdraw unverified binds") + } +} + +func equalBinds(a, b map[int]int) bool { + if len(a) != len(b) { + return false + } + for port, host := range a { + if b[port] != host { + return false + } + } + return true +} +func sortedServiceNames(active domain.AppActive) []string { + names := make([]string, 0, len(active.Services)) + for name := range active.Services { + names = append(names, name) + } + sort.Strings(names) + return names +} + +// lifecycleCtx tags lifecycle use-case logs. +func lifecycleCtx(ctx context.Context, useCase, app string) context.Context { + return zerowrap.CtxWithFields(ctx, map[string]any{ + zerowrap.FieldLayer: "usecase", + zerowrap.FieldUseCase: useCase, + "app": app, + }) +} diff --git a/internal/usecase/deployment/monitor.go b/internal/usecase/deployment/monitor.go new file mode 100644 index 000000000..65df352a8 --- /dev/null +++ b/internal/usecase/deployment/monitor.go @@ -0,0 +1,851 @@ +package deployment + +import ( + "context" + "errors" + "fmt" + "sort" + "sync" + "time" + + "github.com/bnema/zerowrap" + + "github.com/bnema/gordon/internal/domain" +) + +// Recovery pass bounds and policy constants. +const ( + // reconcileServiceDeadline bounds one service's recovery: inspect, + // start/restart, bind inspection, readiness, and publication. A hung + // service times out, releases its app lock, and never starves the + // apps or the services that follow it. + reconcileServiceDeadline = 2 * time.Minute + // stableReadyWindow is how long a generation must be observed + // running, ready, and published before its failure history resets. + stableReadyWindow = 5 * time.Minute + // unhealthyThreshold is the number of consecutive unhealthy + // observations required before recovery restarts a workload. + unhealthyThreshold = 2 +) + +// backoffDelays are the delays applied after the third consecutive +// failure: 1, 2, 4, 8, then a 15-minute cap. +var backoffDelays = []time.Duration{ + 1 * time.Minute, + 2 * time.Minute, + 4 * time.Minute, + 8 * time.Minute, + 15 * time.Minute, +} + +// recoveryKey identifies one workload generation for recovery state. +type recoveryKey struct { + app string + service string + container string +} + +// recoveryBackoff is the in-memory crash-loop budget, keyed by +// (app, service, exact container ID). It is intentionally not durable: +// a daemon restart re-arms one real attempt per generation. +type recoveryBackoff struct { + mu sync.Mutex + now func() time.Time + entries map[recoveryKey]*backoffEntry +} + +type backoffEntry struct { + failures int + nextAttempt time.Time + healthySince time.Time + unhealthy int + // attemptPending marks a recovery intervention (start/restart) that + // has not yet been confirmed by a stable, verified observation. A + // generation found down again while an attempt is still pending is + // a crash loop and is charged, so a workload cannot restart every + // pass forever just because each individual start call succeeded. + attemptPending bool +} + +func newRecoveryBackoff(now func() time.Time) *recoveryBackoff { + return &recoveryBackoff{now: now, entries: map[recoveryKey]*backoffEntry{}} +} + +// allow reports whether one real attempt may run now. +func (b *recoveryBackoff) allow(key recoveryKey) bool { + b.mu.Lock() + defer b.mu.Unlock() + entry, ok := b.entries[key] + if !ok { + return true + } + return !b.now().Before(entry.nextAttempt) +} + +// recordFailure charges one failed start/restart or unhealthy restart. +// The first three consecutive failures are real attempts; the fourth and +// later ones delay the next attempt by 1, 2, 4, 8, then 15 minutes. +// Only five continuously stable minutes (observeStable) or an explicit +// forget reset the escalation, so a crash-looping generation cannot +// escape its budget by waiting. +func (b *recoveryBackoff) recordFailure(key recoveryKey) { + b.mu.Lock() + defer b.mu.Unlock() + now := b.now() + entry := b.ensure(key) + entry.failures++ + entry.healthySince = time.Time{} + if entry.failures >= 3 { + index := entry.failures - 3 + if index >= len(backoffDelays) { + index = len(backoffDelays) - 1 + } + entry.nextAttempt = now.Add(backoffDelays[index]) + } +} + +// observeUnhealthy records one consecutive unhealthy observation and +// returns the current streak length. `starting` and transient health +// errors must not reach this method. +func (b *recoveryBackoff) observeUnhealthy(key recoveryKey) int { + b.mu.Lock() + defer b.mu.Unlock() + entry := b.ensure(key) + entry.unhealthy++ + return entry.unhealthy +} + +// clearUnhealthy drops the unhealthy streak after a non-unhealthy +// observation. +func (b *recoveryBackoff) clearUnhealthy(key recoveryKey) { + b.mu.Lock() + defer b.mu.Unlock() + if entry, ok := b.entries[key]; ok { + entry.unhealthy = 0 + } +} + +// observeStable records a running, ready, published observation and +// resets the generation's history after stableReadyWindow. It never +// creates an entry: a generation with no recorded history is stable by +// definition, so healthy services do not accumulate monitor state. +func (b *recoveryBackoff) observeStable(key recoveryKey) { + b.mu.Lock() + defer b.mu.Unlock() + entry, ok := b.entries[key] + if !ok { + return + } + now := b.now() + entry.unhealthy = 0 + if entry.healthySince.IsZero() { + entry.healthySince = now + return + } + if now.Sub(entry.healthySince) >= stableReadyWindow { + // Five continuously stable minutes: the generation is considered + // recovered, so its failure history and unconfirmed-attempt flag + // are dropped. An intervention is therefore never confirmed by a + // single good pass. + delete(b.entries, key) + } +} + +// invalidateStable drops the stability streak without charging a +// failure: an unhealthy, transitional, or unverifiable observation is +// not five continuously stable minutes. +func (b *recoveryBackoff) invalidateStable(key recoveryKey) { + b.mu.Lock() + defer b.mu.Unlock() + if entry, ok := b.entries[key]; ok { + entry.healthySince = time.Time{} + } +} + +// markAttempt records a recovery intervention that must be confirmed by +// a later stable observation. +func (b *recoveryBackoff) markAttempt(key recoveryKey) { + b.mu.Lock() + defer b.mu.Unlock() + b.ensure(key).attemptPending = true +} + +// attemptPending reports whether a previous recovery intervention of +// this generation was never confirmed stable. +func (b *recoveryBackoff) attemptPending(key recoveryKey) bool { + b.mu.Lock() + defer b.mu.Unlock() + entry, ok := b.entries[key] + return ok && entry.attemptPending +} + +// ensure returns the entry for one key, creating it on first use. The +// caller must hold mu. +func (b *recoveryBackoff) ensure(key recoveryKey) *backoffEntry { + entry, ok := b.entries[key] + if !ok { + entry = &backoffEntry{} + b.entries[key] = entry + } + return entry +} + +// forget drops one generation's history (stopped, removed, or replaced). +func (b *recoveryBackoff) forget(key recoveryKey) { + b.mu.Lock() + defer b.mu.Unlock() + delete(b.entries, key) +} + +// publicationInhibition is the in-memory record of services whose +// fail-closed traffic withdrawal could not be applied. Such a service +// must withdraw successfully again before any later publication. +type publicationInhibition struct { + mu sync.Mutex + entries map[recoveryKey]struct{} +} + +func (p *publicationInhibition) mark(app, service string) { + p.mu.Lock() + defer p.mu.Unlock() + if p.entries == nil { + p.entries = map[recoveryKey]struct{}{} + } + p.entries[recoveryKey{app: app, service: service}] = struct{}{} +} + +func (p *publicationInhibition) clear(app, service string) { + p.mu.Lock() + defer p.mu.Unlock() + delete(p.entries, recoveryKey{app: app, service: service}) +} + +func (p *publicationInhibition) inhibited(app, service string) bool { + p.mu.Lock() + defer p.mu.Unlock() + _, ok := p.entries[recoveryKey{app: app, service: service}] + return ok +} + +// pending returns the services of one app whose withdrawal is still +// unproven, so a caller can retry them before publishing anything. +func (p *publicationInhibition) pending(app string) []string { + p.mu.Lock() + defer p.mu.Unlock() + var services []string + for key := range p.entries { + if key.app == app { + services = append(services, key.service) + } + } + sort.Strings(services) + return services +} + +// retainAppServices drops entries only for services no longer present in +// ACTIVE. Marks for active services survive failed passes until a later +// withdrawal is proven, so another publisher cannot restore stale traffic. +func (p *publicationInhibition) retainAppServices(app string, live map[string]struct{}) { + p.mu.Lock() + defer p.mu.Unlock() + for key := range p.entries { + if key.app != app { + continue + } + if _, ok := live[key.service]; !ok { + delete(p.entries, key) + } + } +} + +// executionTracker remembers the last observed execution start of each +// generation so a native runtime restart is detected without Gordon +// restarting the workload itself. +type executionTracker struct { + mu sync.Mutex + seen map[recoveryKey]time.Time +} + +// isNewExecution reports whether this observation still needs verification. +// An unseen generation is treated as fresh: the daemon may have been absent +// while the runtime restarted it, so accepting it without readiness and bind +// verification could republish stale traffic. The observation is recorded +// only after verification succeeds. +func (t *executionTracker) isNewExecution(key recoveryKey, startedAt time.Time) bool { + t.mu.Lock() + defer t.mu.Unlock() + previous, ok := t.seen[key] + return !ok || !previous.Equal(startedAt) +} + +// record stores the verified execution of one generation. +func (t *executionTracker) record(key recoveryKey, startedAt time.Time) { + t.mu.Lock() + defer t.mu.Unlock() + if t.seen == nil { + t.seen = map[recoveryKey]time.Time{} + } + t.seen[key] = startedAt +} + +// forget drops one generation's observation. +func (t *executionTracker) forget(key recoveryKey) { + t.mu.Lock() + defer t.mu.Unlock() + delete(t.seen, key) +} + +// ReconcileRunning is the daemon-owned periodic recovery pass. It never +// consults DESIRED to select content, never changes intent, never +// pulls, creates, or removes a container, and never starts a +// container whose ID is absent from ACTIVE. Every app is attempted; +// failures are aggregated so one failing app cannot hide the others. +func (s *Service) ReconcileRunning(ctx context.Context) error { + ctx = zerowrap.CtxWithFields(ctx, map[string]any{ + zerowrap.FieldLayer: "usecase", + zerowrap.FieldUseCase: "ReconcileRunning", + }) + if s.deps.State == nil { + return nil + } + apps, err := s.deps.State.ListApps(ctx) + if err != nil { + return fmt.Errorf("deployment: list apps for reconciliation: %w", err) + } + var failures []error + for _, app := range apps { + // Bound the whole app, not each service cumulatively. A multi-service + // app with hung runtime calls must not starve every later app forever. + appCtx, cancel := context.WithTimeout(ctx, reconcileServiceDeadline) + err := s.reconcileAppRunning(appCtx, app) + cancel() + if err != nil { + failures = append(failures, err) + } + } + return errors.Join(failures...) +} + +// reconcileAppRunning recovers one app. A busy app (an in-flight user +// mutation) is skipped, never waited on: the periodic pass must not +// starve later apps or shutdown. +func (s *Service) reconcileAppRunning(ctx context.Context, app string) error { + release, ok := s.tryAcquireAppContext(ctx, app) + if !ok { + s.log.Debug().Str("app", app).Msg("deployment: reconciliation skipped, app is busy") + return nil + } + defer release() + + ctx = zerowrap.CtxWithFields(ctx, map[string]any{ + zerowrap.FieldLayer: "usecase", + zerowrap.FieldUseCase: "ReconcileRunning", + "app": app, + }) + intent, err := s.deps.State.LoadIntent(ctx, app) + if err != nil { + return fmt.Errorf("deployment: reconcile %q: load intent: %w", app, err) + } + active, ok, err := s.deps.State.LoadActive(ctx, app) + if err != nil { + return fmt.Errorf("deployment: reconcile %q: load active: %w", app, err) + } + if !ok || len(active.Services) == 0 { + s.pruneRecoveryHistory(app, active) + return nil + } + inhibitions, err := s.deps.State.LoadRecoveryInhibitions(ctx, app) + if err != nil { + return fmt.Errorf("deployment: reconcile %q: load recovery inhibitions: %w", app, err) + } + inhibited, failures := s.processRecoveryInhibitions(ctx, app, inhibitions) + for _, name := range sortedServiceNames(active) { + eff := active.Services[name] + if intent.Stopped { + if err := s.convergeStoppedService(ctx, app, name, eff); err != nil { + failures = append(failures, err) + } + continue + } + if err := s.reconcileRunningService(ctx, app, name, eff, inhibited); err != nil { + failures = append(failures, err) + } + } + if intent.Stopped { + // A stopped app must not stay published: rebuild the projection + // so no revoked backend remains routable. The publication is + // bounded like any other recovery effect. + publishCtx, cancel := context.WithTimeout(ctx, reconcileServiceDeadline) + if err := s.refreshTraffic(publishCtx, app); err != nil { + failures = append(failures, err) + } + cancel() + } + s.pruneRecoveryHistory(app, active) + return errors.Join(failures...) +} + +func (s *Service) processRecoveryInhibitions(ctx context.Context, app string, inhibitions []domain.AppRecoveryInhibition) (map[recoveryKey]struct{}, []error) { + inhibited := map[recoveryKey]struct{}{} + var failures []error + for _, inhibition := range inhibitions { + if inhibition.Reason != domain.AppInhibitRetirementPending { + inhibited[recoveryKey{app: app, service: inhibition.Service, container: inhibition.ContainerID}] = struct{}{} + continue + } + retired := s.retireContainer(ctx, app, retireOptions{Service: inhibition.Service}, inhibition.ContainerID) + if !retired.Gone { + failures = append(failures, fmt.Errorf("deployment: retirement pending for %s/%s", app, inhibition.ContainerID)) + } + } + return inhibited, failures +} + +// convergeStoppedService enforces durable stopped intent: it never +// starts or publishes, and it stops a verified running exact ACTIVE ID +// without deleting the container or its volumes. +func (s *Service) convergeStoppedService(ctx context.Context, app, name string, eff domain.AppEffectiveService) error { + key := recoveryKey{app: app, service: name, container: eff.Container} + s.backoff.forget(key) + s.executions.forget(key) + if eff.Container == "" || s.deps.Runtime == nil { + return nil + } + serviceCtx, cancel := context.WithTimeout(ctx, reconcileServiceDeadline) + defer cancel() + container, err := s.deps.Runtime.InspectContainer(serviceCtx, eff.Container) + if err != nil { + if errors.Is(err, domain.ErrContainerNotFound) { + return nil + } + return fmt.Errorf("deployment: stopped app %q service %q: inspect %s: %w", app, name, eff.Container, err) + } + if container.Status != string(domain.ContainerStatusRunning) && !isTransitionalStatus(container.Status) { + return nil + } + if err := s.deps.Runtime.StopContainer(serviceCtx, eff.Container, serviceStopGrace(eff)); err != nil && !errors.Is(err, domain.ErrContainerNotFound) { + return fmt.Errorf("deployment: stopped app %q service %q: stop %s: %w", app, name, eff.Container, err) + } + s.log.Info().Str("app", app).Str("service", name).Str("container", eff.Container). + Msg("deployment: stopped intent enforced on a revived container") + return nil +} + +// reconcileRunningService recovers one service to the running intent +// carried by ACTIVE. It only ever acts on the exact ACTIVE container ID. +func (s *Service) reconcileRunningService(ctx context.Context, app, name string, eff domain.AppEffectiveService, inhibited map[recoveryKey]struct{}) error { + if eff.Container == "" { + // Absent ACTIVE: never act, never reconstruct. + return nil + } + if _, blocked := inhibited[recoveryKey{app: app, service: name, container: eff.Container}]; blocked { + return s.refuseInhibitedRecovery(ctx, app, name, eff.Container) + } + if s.deps.Runtime == nil { + return nil + } + serviceCtx, cancel := context.WithTimeout(ctx, reconcileServiceDeadline) + defer cancel() + key := recoveryKey{app: app, service: name, container: eff.Container} + container, fresh, err := s.inspectForRecovery(serviceCtx, app, name, eff, key) + if err != nil { + return err + } + return s.convergeRecoveredService(serviceCtx, app, name, eff, container, key, fresh) +} + +// inspectForRecovery retries a pending withdrawal, classifies the exact +// ACTIVE container, and makes sure nothing unverifiable stays routable +// before the workload is touched. A transient inspection error never +// changes the workload or its traffic. +func (s *Service) inspectForRecovery(ctx context.Context, app, name string, eff domain.AppEffectiveService, key recoveryKey) (*domain.Container, bool, error) { + // A previous withdrawal that could not be applied must be retried + // before anything else: never publish on top of stale forwarding. + withdrawn := false + if s.publication.inhibited(app, name) { + if err := s.withdrawForRecovery(ctx, app, name); err != nil { + return nil, false, err + } + withdrawn = true + } + container, err := s.deps.Runtime.InspectContainer(ctx, eff.Container) + if err != nil { + return nil, false, s.recoveryInspectError(ctx, app, name, eff, err) + } + // Never serve a backend that is not verifiably serving: a native + // restart (fresh execution) or a container that is not running must + // be withdrawn, and that withdrawal must succeed, before Gordon + // touches the workload. A transitional state keeps its traffic: the + // runtime is restarting it right now and it will be back within the + // pass cadence. + fresh := s.executions.isNewExecution(key, container.StartedAt) + unverified := container.Status != string(domain.ContainerStatusRunning) && !isTransitionalStatus(container.Status) + if (fresh || unverified) && !withdrawn { + if err := s.withdrawForRecovery(ctx, app, name); err != nil { + return nil, false, err + } + } + if fresh { + s.backoff.clearUnhealthy(key) + } + return container, fresh, nil +} + +// convergeRecoveredService drives the exact ACTIVE container to a +// verified, published state, or leaves it withdrawn with a reported +// reason. +func (s *Service) convergeRecoveredService(ctx context.Context, app, name string, eff domain.AppEffectiveService, container *domain.Container, key recoveryKey, fresh bool) error { + started, verify, err := s.driveRecoveredContainer(ctx, app, name, eff, container, key) + if err != nil { + return errors.Join(s.withdrawForRecovery(ctx, app, name), err) + } + if !verify { + // Transitional or backoff-delayed: observe now, converge later. + s.backoff.invalidateStable(key) + return nil + } + restarted, stable, err := s.recoverHealth(ctx, app, name, eff, key) + if err != nil { + s.backoff.invalidateStable(key) + return errors.Join(s.withdrawForRecovery(ctx, app, name), err) + } + if err := s.publishRecoveredService(ctx, app, name, eff, container, key, started, restarted, fresh); err != nil { + return err + } + if stable { + s.backoff.observeStable(key) + } else { + s.backoff.invalidateStable(key) + } + return nil +} + +// refuseInhibitedRecovery reports a generation whose recovery is durably +// inhibited and withdraws its backend best-effort: it must not stay +// routable, and it must never be started. +func (s *Service) refuseInhibitedRecovery(ctx context.Context, app, name, containerID string) error { + withdrawCtx, cancel := context.WithTimeout(ctx, reconcileServiceDeadline) + defer cancel() + return errors.Join( + fmt.Errorf( + "deployment: app %q service %q container %s has recovery inhibited; no start attempted", + app, name, containerID, + ), + s.withdrawForRecovery(withdrawCtx, app, name), + ) +} + +// recoveryInspectError classifies one failed inspection: a confirmed +// missing container withdraws traffic and reports without reconstruction, +// while a transient error leaves the workload and its traffic untouched. +func (s *Service) recoveryInspectError(ctx context.Context, app, name string, eff domain.AppEffectiveService, err error) error { + if !errors.Is(err, domain.ErrContainerNotFound) { + return fmt.Errorf("deployment: app %q service %q inspect %s: %w", app, name, eff.Container, err) + } + // The container is confirmed gone: its loopback reservations must not + // outlive it, or a later workload that reuses the port would collide + // with a claim no container holds. + releaseErr := s.deps.State.ReleaseBackendBinds(ctx, app, eff.Container) + if releaseErr != nil { + releaseErr = fmt.Errorf("deployment: release claims of gone container %s: %w", eff.Container, releaseErr) + } + return errors.Join( + s.withdrawForRecovery(ctx, app, name), + releaseErr, + fmt.Errorf("deployment: app %q service %q container %s is gone; traffic withdrawn, no reconstruction: %w", + app, name, eff.Container, domain.ErrContainerNotFound), + ) +} + +// driveRecoveredContainer brings the exact ACTIVE container to a running +// state: an already running container is left alone, a transitional one +// is observed and retried later, an unsafe state errors non-destructively, +// and a stopped one is started under the generation's backoff budget. +// verify reports whether the caller may continue to binds, readiness, and +// publication: a transitional or delayed generation must not be published +// on this pass. +func (s *Service) driveRecoveredContainer(ctx context.Context, app, name string, eff domain.AppEffectiveService, container *domain.Container, key recoveryKey) (started bool, verify bool, err error) { + switch { + case container.Status == string(domain.ContainerStatusRunning): + return false, true, nil + case isTransitionalStatus(container.Status): + // restarting/starting: observe and retry on a later pass. + return false, false, nil + case container.Status == string(domain.ContainerStatusPaused) || container.Status == string(domain.ContainerStatusUnknown) || container.Status == "": + return false, false, fmt.Errorf("deployment: app %q service %q container %s is %q; refusing non-destructive recovery", + app, name, eff.Container, container.Status) + default: + // created/exited/dead/stopped with running intent: start the + // exact ACTIVE ID regardless of exit code. + if !s.backoff.allow(key) { + return false, false, nil + } + // A generation found down again before a previous recovery was + // ever confirmed stable is crash-looping: charge it, so a + // workload that dies after every successful start cannot be + // restarted on every pass forever. + if s.backoff.attemptPending(key) { + s.backoff.recordFailure(key) + if !s.backoff.allow(key) { + return false, false, nil + } + } + started, err := s.startRecovered(ctx, app, name, eff, key) + return started, err == nil, err + } +} + +// publishRecoveredService verifies one generation in the order the plan +// requires: read the container's live binds, probe readiness against +// them, revalidate execution identity and ACTIVE, persist the verified +// binds, and only then publish. Binds are never persisted to ACTIVE +// before readiness, so no other publication can project an unverified +// backend while this pass waits for the workload. The execution is +// recorded only on success, so a failed verification is retried on the +// next pass instead of being silently accepted. +func (s *Service) publishRecoveredService(ctx context.Context, app, name string, eff domain.AppEffectiveService, observed *domain.Container, key recoveryKey, started, restarted, fresh bool) error { + binds, udpBinds, err := s.inspectBackendBinds(ctx, app, name, eff.Container, backendPorts(eff.Spec)) + if err != nil { + s.backoff.recordFailure(key) + return errors.Join(s.withdrawForRecovery(ctx, app, name), err) + } + if started || restarted || fresh { + if err := s.waitServiceReady(ctx, app, eff.Container, eff.Spec, binds); err != nil { + s.backoff.recordFailure(key) + return errors.Join(s.withdrawForRecovery(ctx, app, name), err) + } + } + verifiedBinds, verifiedUDPBinds, err := s.revalidateAndPersistRecoveredBinds(ctx, app, name, eff, observed, binds, udpBinds) + if err != nil { + s.backoff.recordFailure(key) + return errors.Join(s.withdrawForRecovery(ctx, app, name), err) + } + if started || restarted || fresh || !equalBinds(verifiedBinds, eff.BackendBinds) || !equalBinds(verifiedUDPBinds, eff.UDPBackendBinds) { + if err := s.refreshTraffic(ctx, app); err != nil { + return err + } + } + s.executions.record(key, observed.StartedAt) + return nil +} + +// startRecovered starts the exact ACTIVE container. A failed start is +// re-inspected: if native restart policy already revived it, no failure +// is charged and no second restart is issued. +func (s *Service) startRecovered(serviceCtx context.Context, app, name string, eff domain.AppEffectiveService, key recoveryKey) (bool, error) { + // Fail closed before mutating the runtime if a bind's policy was + // revoked or became invalid since ACTIVE was published. + if _, err := s.resolveServiceBinds(app, eff.Spec); err != nil { + return false, err + } + // Revoked device grants fail before any runtime mutation. + if _, err := s.resolveServiceDevices(app, eff.Spec); err != nil { + return false, err + } + if err := s.deps.Runtime.StartContainer(serviceCtx, eff.Container); err != nil { + if errors.Is(err, domain.ErrContainerNotFound) { + s.backoff.forget(key) + return false, errors.Join( + fmt.Errorf("deployment: app %q service %q container %s is gone: %w", app, name, eff.Container, domain.ErrContainerNotFound), + ) + } + recovered, inspectErr := s.deps.Runtime.InspectContainer(serviceCtx, eff.Container) + if inspectErr == nil && (recovered.Status == string(domain.ContainerStatusRunning) || isTransitionalStatus(recovered.Status)) { + s.log.Info().Str("app", app).Str("service", name).Str("container", eff.Container). + Msg("deployment: native restart won the race, no recovery charged") + s.backoff.clearUnhealthy(key) + return true, nil + } + s.backoff.recordFailure(key) + return false, fmt.Errorf("deployment: app %q service %q start %s: %w", app, name, eff.Container, err) + } + s.backoff.markAttempt(key) + s.log.Info().Str("app", app).Str("service", name).Str("container", eff.Container). + Msg("deployment: recovery started the exact active container") + return true, nil +} + +// recoverHealth restarts a workload that reported `unhealthy` on +// consecutive passes. `starting` waits; a missing healthcheck never +// triggers a restart. Restarts share the generation's backoff budget and +// are marked as unconfirmed attempts. The second result reports whether +// the observation counts as stable ("running, ready, published"): an +// unhealthy, still-starting, or unreadable health state must not extend +// the five-minute stability window. +func (s *Service) recoverHealth(serviceCtx context.Context, app, name string, eff domain.AppEffectiveService, key recoveryKey) (restarted bool, stable bool, err error) { + status, hasHealthcheck, statusErr := s.deps.Runtime.GetContainerHealthStatus(serviceCtx, eff.Container) + if statusErr != nil || !hasHealthcheck { + // A transient health error is not an unhealthy workload, but it + // is also not a proven healthy one. + return false, statusErr == nil, nil + } + if status != "unhealthy" { + s.backoff.clearUnhealthy(key) + return false, status != "starting", nil + } + if s.backoff.observeUnhealthy(key) < unhealthyThreshold { + return false, false, nil + } + if !s.backoff.allow(key) { + return false, false, nil + } + // Fail closed before mutating the runtime if a bind's policy was + // revoked or became invalid since ACTIVE was published. + if _, err := s.resolveServiceBinds(app, eff.Spec); err != nil { + return false, false, err + } + // Revoked device grants fail before any runtime mutation. + if _, err := s.resolveServiceDevices(app, eff.Spec); err != nil { + return false, false, err + } + if err := s.deps.Runtime.RestartContainer(serviceCtx, eff.Container, serviceStopGrace(eff)); err != nil { + s.backoff.recordFailure(key) + return false, false, fmt.Errorf("deployment: app %q service %q restart unhealthy %s: %w", app, name, eff.Container, err) + } + s.backoff.markAttempt(key) + s.backoff.clearUnhealthy(key) + s.log.Info().Str("app", app).Str("service", name).Str("container", eff.Container). + Msg("deployment: restarted an unhealthy container") + return true, false, nil +} + +// revalidateAndPersistRecoveredBinds revalidates ACTIVE and the observed +// execution immediately before publication, then persists the verified +// binds (already re-read from the exact container) into the ACTIVE +// record. Stale evidence never publishes: a native restart that happened +// while this pass waited for readiness changes the observed start and +// aborts the pass instead of publishing the previous execution's binds. +func (s *Service) revalidateAndPersistRecoveredBinds(ctx context.Context, app, name string, eff domain.AppEffectiveService, observed *domain.Container, binds, udpBinds map[int]int) (map[int]int, map[int]int, error) { + current, err := s.deps.Runtime.InspectContainer(ctx, eff.Container) + if err != nil { + return nil, nil, fmt.Errorf("deployment: re-inspect %s before publication: %w", eff.Container, err) + } + if observed != nil && !current.StartedAt.Equal(observed.StartedAt) { + return nil, nil, fmt.Errorf("deployment: app %q service %q container %s restarted during recovery: %w", + app, name, eff.Container, domain.ErrAppStateConflict) + } + active, ok, err := s.deps.State.LoadActive(ctx, app) + if err != nil { + return nil, nil, fmt.Errorf("deployment: reload active for recovery: %w", err) + } + if !ok { + return nil, nil, fmt.Errorf("deployment: app %q has no active state: %w", app, domain.ErrAppStateConflict) + } + recorded, ok := active.Services[name] + if !ok || recorded.Container != eff.Container || recorded.EffectiveRevision != eff.EffectiveRevision { + return nil, nil, fmt.Errorf("deployment: app %q service %q generation changed during recovery: %w", app, name, domain.ErrAppStateConflict) + } + if !equalBinds(binds, recorded.BackendBinds) || !equalBinds(udpBinds, recorded.UDPBackendBinds) { + recorded.BackendBinds = binds + recorded.UDPBackendBinds = udpBinds + active.Services[name] = recorded + if err := s.deps.State.SaveActive(ctx, active); err != nil { + return nil, nil, fmt.Errorf("deployment: persist recovered binds: %w", err) + } + } + return binds, udpBinds, nil +} + +// withdrawForRecovery applies one serialized fail-closed withdrawal of +// a service. A failed application retains an in-memory publication +// inhibition and never releases backend claims. +func (s *Service) withdrawForRecovery(ctx context.Context, app, service string) error { + if s.deps.Traffic == nil { + return nil + } + if err := s.deps.Traffic.WithdrawService(ctx, app, service); err != nil { + s.publication.mark(app, service) + return fmt.Errorf("deployment: withdraw %s/%s: %w", app, service, err) + } + s.publication.clear(app, service) + return nil +} + +// recoveryInhibited reports whether one exact generation is durably +// inhibited for recovery. +func (s *Service) recoveryInhibited(ctx context.Context, app, service, containerID string) (bool, error) { + if containerID == "" { + return false, nil + } + inhibitions, err := s.deps.State.LoadRecoveryInhibitions(ctx, app) + if err != nil { + return false, fmt.Errorf("deployment: load recovery inhibitions: %w", err) + } + for _, inhibition := range inhibitions { + if inhibition.Service == service && inhibition.ContainerID == containerID { + return true, nil + } + } + return false, nil +} + +// inhibitRecovery durably records a generation-scoped recovery +// inhibition for one exact container ID. +func (s *Service) inhibitRecovery(ctx context.Context, app, service, containerID, reason, operation string) error { + if containerID == "" { + return nil + } + inhibition := domain.AppRecoveryInhibition{ + App: app, + Service: service, + ContainerID: containerID, + Reason: reason, + Operation: operation, + CreatedAt: time.Now().UTC(), + } + if err := s.deps.State.SaveRecoveryInhibition(ctx, inhibition); err != nil { + return fmt.Errorf("deployment: record recovery inhibition for %s/%s: %w", app, containerID, err) + } + return nil +} + +// clearRecoveryInhibition drops the inhibition of a safely superseded +// generation. +func (s *Service) clearRecoveryInhibition(ctx context.Context, app, service, containerID string) error { + if containerID == "" { + return nil + } + if err := s.deps.State.ClearRecoveryInhibition(ctx, app, service, containerID); err != nil { + return fmt.Errorf("deployment: clear recovery inhibition for %s/%s: %w", app, containerID, err) + } + return nil +} + +// pruneRecoveryHistory drops in-memory state of generations that are no +// longer present in ACTIVE. Durable inhibition records are never +// dropped here: only an explicit supersede or removal clears them. +func (s *Service) pruneRecoveryHistory(app string, active domain.AppActive) { + live := map[recoveryKey]struct{}{} + liveServices := map[string]struct{}{} + for name, svc := range active.Services { + live[recoveryKey{app: app, service: name, container: svc.Container}] = struct{}{} + liveServices[name] = struct{}{} + } + s.backoff.mu.Lock() + for key := range s.backoff.entries { + if key.app != app { + continue + } + if _, ok := live[key]; !ok { + delete(s.backoff.entries, key) + } + } + s.backoff.mu.Unlock() + s.executions.mu.Lock() + for key := range s.executions.seen { + if key.app != app { + continue + } + if _, ok := live[key]; !ok { + delete(s.executions.seen, key) + } + } + s.executions.mu.Unlock() + s.publication.retainAppServices(app, liveServices) +} + +// isTransitionalStatus reports runtime states that must be observed and +// retried rather than acted on. +func isTransitionalStatus(status string) bool { + return status == "restarting" || status == "starting" +} diff --git a/internal/usecase/deployment/network_reclaim_internal_test.go b/internal/usecase/deployment/network_reclaim_internal_test.go new file mode 100644 index 000000000..59a388b8f --- /dev/null +++ b/internal/usecase/deployment/network_reclaim_internal_test.go @@ -0,0 +1,172 @@ +package deployment + +import ( + "context" + "testing" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" +) + +func reclaimService(t *testing.T, state *outmocks.MockAppState, runtime *outmocks.MockContainerRuntime) *Service { + t.Helper() + return NewService(Deps{ + State: state, Runtime: runtime, + Networks: NetworkConfig{Prefix: "gordon"}, + }, zerowrap.Default()) +} + +// TestReclaimPrivateNetworks_RemovesOwnedEmptyIncarnationNetwork proves the +// app's own empty private network is removed immediately, by exact derived +// name and verified ownership. +func TestReclaimPrivateNetworks_RemovesOwnedEmptyIncarnationNetwork(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + + private := domain.AppPrivateNetworkName("gordon", "app-1") + ownership := domain.AppOwnership{ + App: "blog", ID: "app-1", + Networks: []domain.AppOwnedNetwork{{Name: private, Role: domain.AppNetworkRolePrivate}}, + } + runtime.EXPECT().ListNetworks(mock.Anything).Return([]*domain.NetworkInfo{ + {Name: private, Labels: domain.AppPrivateNetworkLabels("blog", "app-1")}, + }, nil).Once() + runtime.EXPECT().RemoveNetwork(mock.Anything, private).Return(nil).Once() + + warnings, err := reclaimService(t, state, runtime).reclaimPrivateNetworks(ctx, "blog", ownership) + require.NoError(t, err) + assert.Empty(t, warnings) +} + +// TestReclaimPrivateNetworks_KeepsSharedNetworks proves a declared shared +// network is never removed by an app removal. +func TestReclaimPrivateNetworks_KeepsSharedNetworks(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + + private := domain.AppPrivateNetworkName("gordon", "app-1") + shared := domain.AppSharedNetworkName("gordon", "shared-backend") + ownership := domain.AppOwnership{ + App: "blog", ID: "app-1", + Networks: []domain.AppOwnedNetwork{ + {Name: private, Role: domain.AppNetworkRolePrivate}, + {Name: shared, Role: domain.AppNetworkRoleShared}, + }, + } + runtime.EXPECT().ListNetworks(mock.Anything).Return([]*domain.NetworkInfo{ + {Name: private, Labels: domain.AppPrivateNetworkLabels("blog", "app-1")}, + {Name: shared, Labels: domain.AppSharedNetworkLabels("shared-backend")}, + }, nil).Once() + runtime.EXPECT().RemoveNetwork(mock.Anything, private).Return(nil).Once() + + warnings, err := reclaimService(t, state, runtime).reclaimPrivateNetworks(ctx, "blog", ownership) + require.NoError(t, err) + assert.Empty(t, warnings) + runtime.AssertNotCalled(t, "RemoveNetwork", mock.Anything, shared) +} + +// TestReclaimPrivateNetworks_RefusesForeignOwnership proves a network whose +// labels do not prove this exact incarnation is left in place and reported, +// never deleted. +func TestReclaimPrivateNetworks_RefusesForeignOwnership(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + + private := domain.AppPrivateNetworkName("gordon", "app-1") + ownership := domain.AppOwnership{ + App: "blog", ID: "app-1", + Networks: []domain.AppOwnedNetwork{{Name: private, Role: domain.AppNetworkRolePrivate}}, + } + // A different incarnation's labels (or a hand-made network) must never + // be adopted, let alone deleted. + runtime.EXPECT().ListNetworks(mock.Anything).Return([]*domain.NetworkInfo{ + {Name: private, Labels: domain.AppPrivateNetworkLabels("blog", "app-99")}, + }, nil).Once() + + warnings, err := reclaimService(t, state, runtime).reclaimPrivateNetworks(ctx, "blog", ownership) + require.NoError(t, err) + require.Len(t, warnings, 1) + assert.Equal(t, private, warnings[0].Leftover) + assert.Contains(t, warnings[0].Detail, "labels") + runtime.AssertNotCalled(t, "RemoveNetwork", mock.Anything, mock.Anything) +} + +// TestReclaimPrivateNetworks_RefusesAttachedNetwork proves a network that +// still has attachments is left in place and reported. +func TestReclaimPrivateNetworks_RefusesAttachedNetwork(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + + private := domain.AppPrivateNetworkName("gordon", "app-1") + ownership := domain.AppOwnership{ + App: "blog", ID: "app-1", + Networks: []domain.AppOwnedNetwork{{Name: private, Role: domain.AppNetworkRolePrivate}}, + } + runtime.EXPECT().ListNetworks(mock.Anything).Return([]*domain.NetworkInfo{ + { + Name: private, + Labels: domain.AppPrivateNetworkLabels("blog", "app-1"), + Containers: []string{"c-still-there"}, + }, + }, nil).Once() + + warnings, err := reclaimService(t, state, runtime).reclaimPrivateNetworks(ctx, "blog", ownership) + require.NoError(t, err) + require.Len(t, warnings, 1) + assert.Contains(t, warnings[0].Detail, "attached containers") + runtime.AssertNotCalled(t, "RemoveNetwork", mock.Anything, mock.Anything) +} + +// TestReclaimPrivateNetworks_NeverTouchesAnotherIncarnationName proves a +// stale ownership entry naming a network this incarnation does not own is +// ignored: a re-used name must never delete the previous incarnation's +// network. +func TestReclaimPrivateNetworks_NeverTouchesAnotherIncarnationName(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + + other := domain.AppPrivateNetworkName("gordon", "app-old") + ownership := domain.AppOwnership{ + App: "blog", ID: "app-1", + Networks: []domain.AppOwnedNetwork{{Name: other, Role: domain.AppNetworkRolePrivate}}, + } + + warnings, err := reclaimService(t, state, runtime).reclaimPrivateNetworks(ctx, "blog", ownership) + require.NoError(t, err) + assert.Empty(t, warnings) + runtime.AssertNotCalled(t, "ListNetworks", mock.Anything) + runtime.AssertNotCalled(t, "RemoveNetwork", mock.Anything, mock.Anything) +} + +// TestReclaimPrivateNetworks_RuntimeFailureIsRetryable proves a failed +// removal is reported so the caller keeps app state and the stopped intent +// for a retry. +func TestReclaimPrivateNetworks_RuntimeFailureIsRetryable(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + + private := domain.AppPrivateNetworkName("gordon", "app-1") + ownership := domain.AppOwnership{ + App: "blog", ID: "app-1", + Networks: []domain.AppOwnedNetwork{{Name: private, Role: domain.AppNetworkRolePrivate}}, + } + runtime.EXPECT().ListNetworks(mock.Anything).Return([]*domain.NetworkInfo{ + {Name: private, Labels: domain.AppPrivateNetworkLabels("blog", "app-1")}, + }, nil).Once() + runtime.EXPECT().RemoveNetwork(mock.Anything, private).Return(assert.AnError).Once() + + _, err := reclaimService(t, state, runtime).reclaimPrivateNetworks(ctx, "blog", ownership) + require.Error(t, err) + assert.Contains(t, err.Error(), "remove private network") +} diff --git a/internal/usecase/deployment/networks.go b/internal/usecase/deployment/networks.go new file mode 100644 index 000000000..0e325efc3 --- /dev/null +++ b/internal/usecase/deployment/networks.go @@ -0,0 +1,226 @@ +package deployment + +import ( + "context" + "fmt" + + "github.com/bnema/gordon/internal/domain" +) + +// ensureIncarnationID returns the app's stable internal UUID, assigning +// and persisting one on first use. Every incarnation-scoped resource +// (network, volumes, secrets) keys off this ID, never the public name, so +// removing and re-adding an app under the same name can never inherit the +// old incarnation's resources. +func (s *Service) ensureIncarnationID(ctx context.Context, app string) (string, error) { + ownership, err := s.deps.State.LoadOwnership(ctx, app) + if err != nil { + return "", fmt.Errorf("deployment: load ownership: %w", err) + } + if ownership.ID != "" { + return ownership.ID, nil + } + ownership.App = app + if err := s.deps.State.SaveOwnership(ctx, ownership); err != nil { + return "", fmt.Errorf("deployment: assign incarnation: %w", err) + } + saved, err := s.deps.State.LoadOwnership(ctx, app) + if err != nil { + return "", fmt.Errorf("deployment: reload ownership: %w", err) + } + if saved.ID == "" { + return "", fmt.Errorf("deployment: app %q has no incarnation id: %w", app, domain.ErrAppStateConflict) + } + return saved.ID, nil +} + +// appNetworks is the runtime name set one service joins. +type appNetworks struct { + // private is the incarnation-owned network every service of the app + // joins; no other app shares it. + private string + // shared are the declared cross-app memberships resolved to runtime + // names. + shared []sharedNetwork +} + +// sharedNetwork ties a declared membership to its derived runtime name. +type sharedNetwork struct { + declared string + runtime string +} + +// resolveAppNetworks derives runtime network names from the app UUID and +// the service's declared shared memberships. Deriving (never accepting a +// caller-supplied runtime name) keeps recovery and deploy in agreement and +// prevents a manifest from naming a foreign network. +func (s *Service) resolveAppNetworks(appID string, memberships []domain.AppSharedNetwork) appNetworks { + prefix := s.deps.Networks.Prefix + resolved := appNetworks{private: domain.AppPrivateNetworkName(prefix, appID)} + for _, membership := range memberships { + resolved.shared = append(resolved.shared, sharedNetwork{ + declared: membership.Network, + runtime: domain.AppSharedNetworkName(prefix, membership.Network), + }) + } + return resolved +} + +// ensureAppNetworks verifies or creates the incarnation network and every +// declared shared network BEFORE any container is created. An existing +// network is reused only when its labels prove Gordon ownership; any other +// network using the derived name is refused fail-closed. Networks are +// never removed here: an empty owned network is inert, and during a +// replacement the superseded generation may still be attached to it. +func (s *Service) ensureAppNetworks(ctx context.Context, app, appID string, nets appNetworks) error { + observed, err := s.deps.Runtime.ListNetworks(ctx) + if err != nil { + return fmt.Errorf("deployment: list networks: %w", err) + } + byName := make(map[string]*domain.NetworkInfo, len(observed)) + for _, network := range observed { + if network == nil { + continue + } + byName[network.Name] = network + } + if err := s.ensureNetwork(ctx, byName, nets.private, domain.AppPrivateNetworkLabels(app, appID)); err != nil { + return err + } + for _, shared := range nets.shared { + if err := s.ensureNetwork(ctx, byName, shared.runtime, domain.AppSharedNetworkLabels(shared.declared)); err != nil { + return err + } + } + return nil +} + +// ensureNetwork reuses an owned network or creates it. +func (s *Service) ensureNetwork(ctx context.Context, byName map[string]*domain.NetworkInfo, name string, labels map[string]string) error { + if existing, ok := byName[name]; ok { + if !domain.NetworkOwnedBy(existing.Labels, labels) { + return fmt.Errorf("deployment: network %q exists without Gordon ownership: %w", name, domain.ErrAppStateConflict) + } + return nil + } + config := domain.NetworkConfig{Driver: "bridge", Internal: s.deps.Networks.Internal, Labels: labels} + if err := s.deps.Runtime.CreateNetwork(ctx, name, config); err != nil { + return fmt.Errorf("deployment: create network %q: %w", name, err) + } + return nil +} + +// connectSharedNetworks attaches a created container to every declared +// shared network. The private network is attached at create through +// NetworkMode; shared memberships are explicit so an app joins only the +// networks its manifest declares. +func (s *Service) connectSharedNetworks(ctx context.Context, containerID string, nets appNetworks) error { + for _, shared := range nets.shared { + if err := s.deps.Runtime.ConnectContainerToNetwork(ctx, containerID, shared.runtime); err != nil { + return fmt.Errorf("deployment: connect %s to network %q: %w", containerID, shared.runtime, err) + } + } + return nil +} + +// recordNetworkOwnership stamps the incarnation network and every declared +// shared membership into the ownership record so prune/remove can see +// exactly which networks this app created or joined. +func recordNetworkOwnership(ownership *domain.AppOwnership, prefix string, shared []domain.AppSharedNetwork) { + if ownership.ID == "" { + return + } + entries := []domain.AppOwnedNetwork{{Name: domain.AppPrivateNetworkName(prefix, ownership.ID), Role: domain.AppNetworkRolePrivate}} + for _, membership := range shared { + entries = append(entries, domain.AppOwnedNetwork{ + Name: domain.AppSharedNetworkName(prefix, membership.Network), + Role: domain.AppNetworkRoleShared, + }) + } + seen := make(map[string]struct{}, len(entries)) + merged := make([]domain.AppOwnedNetwork, 0, len(entries)) + for _, entry := range entries { + if _, ok := seen[entry.Name]; ok { + continue + } + seen[entry.Name] = struct{}{} + merged = append(merged, entry) + } + for _, existing := range ownership.Networks { + if _, ok := seen[existing.Name]; ok { + continue + } + seen[existing.Name] = struct{}{} + merged = append(merged, existing) + } + ownership.Networks = merged +} + +// reclaimPrivateNetworks removes the incarnation-owned private networks of +// one app immediately after every exact container is confirmed gone, while +// the ownership record is still live. The runtime name is re-derived from +// the incarnation UUID and the observed labels must prove that exact +// ownership, so a name reused by another incarnation can never be +// deleted. Shared networks are never removed: they belong to the +// installation, and other apps may still be attached. An unattached, +// ownership-verified network is removed; anything else is left in place +// and reported as a bounded warning, never silently deleted. +func (s *Service) reclaimPrivateNetworks(ctx context.Context, app string, ownership domain.AppOwnership) ([]CleanupWarning, error) { + if ownership.ID == "" { + return nil, nil + } + expectedName := domain.AppPrivateNetworkName(s.deps.Networks.Prefix, ownership.ID) + private := make([]string, 0, len(ownership.Networks)) + for _, network := range ownership.Networks { + if network.Role != domain.AppNetworkRolePrivate { + continue + } + if network.Name != expectedName { + // A record naming anything but this incarnation's derived + // network is never deleted: the name may belong to a + // different incarnation. + continue + } + private = append(private, network.Name) + } + if len(private) == 0 { + return nil, nil + } + observed, err := s.deps.Runtime.ListNetworks(ctx) + if err != nil { + return nil, fmt.Errorf("deployment: list networks for reclamation: %w", err) + } + byName := make(map[string]*domain.NetworkInfo, len(observed)) + for _, network := range observed { + if network == nil { + continue + } + byName[network.Name] = network + } + expected := domain.AppPrivateNetworkLabels(app, ownership.ID) + var warnings []CleanupWarning + for _, name := range private { + network, ok := byName[name] + if !ok { + continue // already gone + } + if !domain.NetworkOwnedBy(network.Labels, expected) { + warnings = append(warnings, CleanupWarning{ + Leftover: name, + Detail: "private network labels do not prove this incarnation; left in place", + }) + continue + } + if len(network.Containers) > 0 { + warnings = append(warnings, CleanupWarning{ + Leftover: name, + Detail: "private network still has attached containers; left in place", + }) + continue + } + if err := s.deps.Runtime.RemoveNetwork(ctx, name); err != nil { + return warnings, fmt.Errorf("deployment: remove private network %q: %w", name, err) + } + } + return warnings, nil +} diff --git a/internal/usecase/deployment/networks_internal_test.go b/internal/usecase/deployment/networks_internal_test.go new file mode 100644 index 000000000..095a900a0 --- /dev/null +++ b/internal/usecase/deployment/networks_internal_test.go @@ -0,0 +1,181 @@ +package deployment + +import ( + "context" + "testing" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" +) + +func networkTestService( + t *testing.T, + state *outmocks.MockAppState, + runtime *outmocks.MockContainerRuntime, + networks NetworkConfig, + limits ResourceLimits, +) *Service { + t.Helper() + return NewService(Deps{State: state, Runtime: runtime, Networks: networks, Limits: limits}, zerowrap.Default()) +} + +// TestCreateAndStart_UsesIncarnationPrivateNetworkAndLimits proves the +// created container joins the app's incarnation-owned network (never the +// runtime default) and carries the configured resource limits. +func TestCreateAndStart_UsesIncarnationPrivateNetworkAndLimits(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + + private := domain.AppPrivateNetworkName("gordon", "app-1") + state.EXPECT().LoadOwnership(mock.Anything, "blog"). + Return(domain.AppOwnership{App: "blog", ID: "app-1"}, nil).Once() + runtime.EXPECT().ListNetworks(mock.Anything).Return([]*domain.NetworkInfo{}, nil).Once() + runtime.EXPECT().CreateNetwork(mock.Anything, private, mock.MatchedBy(func(cfg domain.NetworkConfig) bool { + return cfg.Internal && domain.NetworkOwnedBy(cfg.Labels, domain.AppPrivateNetworkLabels("blog", "app-1")) + })).Return(nil).Once() + runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { + return cfg.NetworkMode == private && + cfg.MemoryLimit == int64(1)<<30 && + cfg.NanoCPUs == 2_000_000_000 && + cfg.PidsLimit == 256 && + assert.Equal(t, []string{"web"}, cfg.Aliases) && + assert.Equal(t, "web", cfg.Hostname) + })).Return(&domain.Container{ID: "c-1", Name: "web"}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-1").Return(nil).Once() + + svc := networkTestService(t, state, runtime, + NetworkConfig{Prefix: "gordon", Internal: true}, + ResourceLimits{MemoryBytes: 1 << 30, NanoCPUs: 2_000_000_000, PidsLimit: 256}, + ) + candidate, err := svc.createAndStart(ctx, "blog", "rev-1", + pinnedService{name: "web", spec: domain.AppService{Name: "web"}}, "op-1", nil) + require.NoError(t, err) + require.NotNil(t, candidate.Container) + assert.Empty(t, candidate.TCPBinds) + assert.Empty(t, candidate.UDPBinds) +} + +// TestCreateAndStart_ReusesOwnedNetwork proves recovery reattaches the +// already-owned incarnation network instead of creating or replacing it. +func TestCreateAndStart_ReusesOwnedNetwork(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + + private := domain.AppPrivateNetworkName("gordon", "app-1") + state.EXPECT().LoadOwnership(mock.Anything, "blog"). + Return(domain.AppOwnership{App: "blog", ID: "app-1"}, nil).Once() + runtime.EXPECT().ListNetworks(mock.Anything).Return([]*domain.NetworkInfo{ + {Name: private, Labels: domain.AppPrivateNetworkLabels("blog", "app-1")}, + }, nil).Once() + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything). + Return(&domain.Container{ID: "c-1", Name: "web"}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-1").Return(nil).Once() + + svc := networkTestService(t, state, runtime, NetworkConfig{Prefix: "gordon"}, ResourceLimits{}) + _, err := svc.createAndStart(ctx, "blog", "rev-1", + pinnedService{name: "web", spec: domain.AppService{Name: "web"}}, "op-1", nil) + require.NoError(t, err) +} + +// TestCreateAndStart_RefusesForeignNetwork proves an unowned network with +// the derived name fails closed before any container is created. +func TestCreateAndStart_RefusesForeignNetwork(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + + private := domain.AppPrivateNetworkName("gordon", "app-1") + state.EXPECT().LoadOwnership(mock.Anything, "blog"). + Return(domain.AppOwnership{App: "blog", ID: "app-1"}, nil).Once() + runtime.EXPECT().ListNetworks(mock.Anything).Return([]*domain.NetworkInfo{ + {Name: private, Labels: map[string]string{domain.LabelManaged: "true"}}, + }, nil).Once() + + svc := networkTestService(t, state, runtime, NetworkConfig{Prefix: "gordon"}, ResourceLimits{}) + _, err := svc.createAndStart(ctx, "blog", "rev-1", + pinnedService{name: "web", spec: domain.AppService{Name: "web"}}, "op-1", nil) + require.ErrorIs(t, err, domain.ErrAppStateConflict) +} + +// TestCreateAndStart_JoinsDeclaredSharedNetworks proves a service is +// connected only to the shared networks its manifest declares. +func TestCreateAndStart_JoinsDeclaredSharedNetworks(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + + private := domain.AppPrivateNetworkName("gordon", "app-1") + shared := domain.AppSharedNetworkName("gordon", "database") + state.EXPECT().LoadOwnership(mock.Anything, "blog"). + Return(domain.AppOwnership{App: "blog", ID: "app-1"}, nil).Once() + runtime.EXPECT().ListNetworks(mock.Anything).Return([]*domain.NetworkInfo{}, nil).Once() + runtime.EXPECT().CreateNetwork(mock.Anything, private, mock.Anything).Return(nil).Once() + runtime.EXPECT().CreateNetwork(mock.Anything, shared, mock.MatchedBy(func(cfg domain.NetworkConfig) bool { + return domain.NetworkOwnedBy(cfg.Labels, domain.AppSharedNetworkLabels("database")) + })).Return(nil).Once() + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything). + Return(&domain.Container{ID: "c-1", Name: "web"}, nil).Once() + runtime.EXPECT().ConnectContainerToNetwork(mock.Anything, "c-1", shared).Return(nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-1").Return(nil).Once() + + svc := networkTestService(t, state, runtime, NetworkConfig{Prefix: "gordon"}, ResourceLimits{}) + _, err := svc.createAndStart(ctx, "blog", "rev-1", pinnedService{ + name: "web", + spec: domain.AppService{Name: "web"}, + sharedNetworks: []domain.AppSharedNetwork{{Network: "database", Services: []string{"web"}}}, + }, "op-1", nil) + require.NoError(t, err) +} + +// TestCreateAndStart_RoutesDeclaredReadOnlyVolumeToReadOnlyMount proves a +// manifest volume declared read-only reaches the runtime config as a +// read-only mount instead of a writable one. +func TestCreateAndStart_RoutesDeclaredReadOnlyVolumeToReadOnlyMount(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + + private := domain.AppPrivateNetworkName("gordon", "app-1") + runtimeName := domain.RuntimeVolumeName("blog", "web", "config") + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-1"}, nil) + state.EXPECT().SaveOwnership(mock.Anything, mock.Anything).Return(nil) + runtime.EXPECT().ListNetworks(mock.Anything).Return([]*domain.NetworkInfo{}, nil).Once() + runtime.EXPECT().CreateNetwork(mock.Anything, private, mock.Anything).Return(nil).Once() + runtime.EXPECT().CreateVolume(mock.Anything, runtimeName, mock.Anything).Return(nil).Once() + runtime.EXPECT().CreateContainer(mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { + _, readOnly := cfg.ReadOnlyVolumes["/config"] + _, writable := cfg.Volumes["/config"] + return readOnly && !writable && cfg.ReadOnlyVolumes["/config"] == runtimeName + })).Return(&domain.Container{ID: "c-1", Name: "web"}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-1").Return(nil).Once() + + svc := networkTestService(t, state, runtime, NetworkConfig{Prefix: "gordon"}, ResourceLimits{}) + _, err := svc.createAndStart(ctx, "blog", "rev-1", pinnedService{ + name: "web", + spec: domain.AppService{ + Name: "web", + Volumes: []domain.AppVolume{{Name: "config", Path: "/config", ReadOnly: true}}, + }, + }, "op-1", nil) + require.NoError(t, err) +} + +// TestCreateAndStart_DistinctAppsNeverShareNetwork proves two apps receive +// different incarnation networks, neither of which is a runtime default. +func TestCreateAndStart_DistinctAppsNeverShareNetwork(t *testing.T) { + networks := map[string]string{} + for _, tc := range []struct{ app, id string }{{"blog", "app-1"}, {"shop", "app-2"}} { + private := domain.AppPrivateNetworkName("gordon", tc.id) + require.NotEqual(t, "default", private) + require.NotEqual(t, "bridge", private) + networks[tc.app] = private + } + assert.NotEqual(t, networks["blog"], networks["shop"]) +} diff --git a/internal/usecase/deployment/operation_claim_integration_test.go b/internal/usecase/deployment/operation_claim_integration_test.go new file mode 100644 index 000000000..cb1c09d85 --- /dev/null +++ b/internal/usecase/deployment/operation_claim_integration_test.go @@ -0,0 +1,97 @@ +package deployment_test + +import ( + "context" + "sync" + "testing" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/deployment" +) + +// TestRemove_ConcurrentSameKeyExecutesOnce proves the guarantee the store +// claim exists for, end to end: two concurrent requests carrying the same +// idempotency key run the removal effects exactly once, and both callers +// observe the same journal. +func TestRemove_ConcurrentSameKeyExecutesOnce(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + runtime := outmocks.NewMockContainerRuntime(t) + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: headlessSpec()}, + }} + require.NoError(t, store.SaveOwnership(ctx, domain.AppOwnership{App: "blog"})) + require.NoError(t, store.SaveActive(ctx, active)) + + // Removal effects must run exactly once for the key. + runtime.EXPECT().StopContainer(mock.Anything, "c-1", mock.Anything).Return(nil).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-1", false).Return(nil).Once() + + svc := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime, + }, zerowrap.Default()) + + const racers = 4 + var ( + wg sync.WaitGroup + mu sync.Mutex + ops []string + errs []error + ) + start := make(chan struct{}) + for range racers { + wg.Add(1) + go func() { + defer wg.Done() + <-start + result, err := svc.Remove(ctx, "blog", "key-1") + mu.Lock() + defer mu.Unlock() + errs = append(errs, err) + if result != nil { + ops = append(ops, result.Op) + } + }() + } + close(start) + wg.Wait() + + require.Len(t, errs, racers) + for i, err := range errs { + require.NoError(t, err, "racer %d", i) + } + require.Len(t, ops, racers) + for i, op := range ops { + assert.Equal(t, "key-1", op, "racer %d must observe the claimed key", i) + } + runtime.AssertNumberOfCalls(t, "StopContainer", 1) + runtime.AssertNumberOfCalls(t, "RemoveContainer", 1) +} + +// TestRemove_UnknownAppCreatesNoState proves a keyed mutation of an +// unknown name fails with ErrAppNotFound and leaves no durable app state +// behind, even with a real store. +func TestRemove_UnknownAppCreatesNoState(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + runtime := outmocks.NewMockContainerRuntime(t) + + svc := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime, + }, zerowrap.Default()) + + _, err := svc.Remove(ctx, "ghost", "key-1") + require.ErrorIs(t, err, domain.ErrAppNotFound) + + apps, err := store.ListApps(ctx) + require.NoError(t, err) + assert.NotContains(t, apps, "ghost") + runtime.AssertNotCalled(t, "StopContainer", mock.Anything, mock.Anything, mock.Anything) +} diff --git a/internal/usecase/deployment/operation_claim_test.go b/internal/usecase/deployment/operation_claim_test.go new file mode 100644 index 000000000..e148152b9 --- /dev/null +++ b/internal/usecase/deployment/operation_claim_test.go @@ -0,0 +1,271 @@ +package deployment_test + +import ( + "context" + "testing" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/deployment" +) + +func keyedService( + t *testing.T, + state *outmocks.MockAppState, + runtime *outmocks.MockContainerRuntime, + images *outmocks.MockImageResolver, + secrets *outmocks.MockSecretProvider, +) *deployment.Service { + t.Helper() + return deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()) +} + +// TestDeploy_ClaimPersistsRequestIdentity proves the first claim writes the +// journal with the immutable request identity of the caller's inputs, so a +// reused key for another request can be rejected later. +func TestDeploy_ClaimPersistsRequestIdentity(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + state.EXPECT().Recover(mock.Anything).Return(nil).Once() + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil).Once() + state.EXPECT().ClaimOperation(mock.Anything, mock.MatchedBy(func(op domain.AppOperation) bool { + return op.Op == "key-1" && + op.Kind == "deploy" && + op.Request == domain.AppOperationRequestFor("deploy", "blog", "", "") && + !op.Terminal() + })).RunAndReturn(func(_ context.Context, op domain.AppOperation) (domain.AppOperation, bool, error) { + return op, true, nil + }).Once() + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-blog"}, nil).Once() + images.EXPECT().ResolveDigest(mock.Anything, rev.Spec.Services[0].Image).Return("", assert.AnError).Once() + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil).Once() + + svc := keyedService(t, state, runtime, images, secrets) + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog", Op: "key-1"}) + + require.ErrorIs(t, err, domain.ErrAppImageUnresolvable) + require.NotNil(t, result) + assert.Equal(t, "key-1", result.Op) +} + +// TestDeploy_TerminalKeyReplayExecutesNoEffect proves a repeat of a +// finished request returns its stored journal without touching a workload. +func TestDeploy_TerminalKeyReplayExecutesNoEffect(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + stored := domain.AppOperation{ + Op: "key-1", Kind: "deploy", App: "blog", InputRevision: "rev-1", + Request: domain.AppOperationRequestFor("deploy", "blog", "", ""), + Outcome: domain.AppOutcomeSuccess, + Steps: []domain.AppOperationStep{{ID: "preflight", State: domain.AppStepSucceeded}}, + } + state.EXPECT().Recover(mock.Anything).Return(nil).Once() + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil).Once() + state.EXPECT().ClaimOperation(mock.Anything, mock.Anything).Return(stored, false, nil).Once() + + svc := keyedService(t, state, runtime, images, secrets) + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog", Op: "key-1"}) + + require.NoError(t, err) + require.NotNil(t, result) + assert.Equal(t, "key-1", result.Op) + assert.Equal(t, domain.AppOutcomeSuccess, result.Outcome) + state.AssertNotCalled(t, "SaveOperation", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) +} + +// TestDeploy_FailedTerminalKeyReplayConflicts proves a failed response does +// not become successful merely because its idempotency key is replayed. +func TestDeploy_FailedTerminalKeyReplayConflicts(t *testing.T) { + for _, outcome := range []string{domain.AppOutcomeFailed, domain.AppOutcomePartial} { + t.Run(outcome, func(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + failed := domain.AppOperation{ + Op: "key-1", Kind: "deploy", App: "blog", InputRevision: "rev-1", + Request: domain.AppOperationRequestFor("deploy", "blog", "", ""), + Outcome: outcome, + Steps: []domain.AppOperationStep{{ID: "preflight", State: domain.AppStepFailed, Error: "image unavailable"}}, + } + state.EXPECT().Recover(mock.Anything).Return(nil).Once() + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil).Once() + state.EXPECT().ClaimOperation(mock.Anything, mock.Anything).Return(failed, false, nil).Once() + + svc := keyedService(t, state, runtime, images, secrets) + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog", Op: "key-1"}) + + require.ErrorIs(t, err, domain.ErrAppStateConflict) + require.NotNil(t, result) + assert.Equal(t, outcome, result.Outcome) + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) + }) + } +} + +// TestDeploy_PendingKeyReplayConflicts proves an in-flight (or interrupted) +// claim is never re-executed: the caller receives the journal and a +// conflict instead of a second deployment. +func TestDeploy_PendingKeyReplayConflicts(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + pending := domain.AppOperation{ + Op: "key-1", Kind: "deploy", App: "blog", InputRevision: "rev-1", + Request: domain.AppOperationRequestFor("deploy", "blog", "", ""), + Steps: []domain.AppOperationStep{{ID: "preflight", State: domain.AppStepPending}}, + } + state.EXPECT().Recover(mock.Anything).Return(nil).Once() + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil).Once() + state.EXPECT().ClaimOperation(mock.Anything, mock.Anything).Return(pending, false, nil).Once() + + svc := keyedService(t, state, runtime, images, secrets) + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog", Op: "key-1"}) + + require.ErrorIs(t, err, domain.ErrAppStateConflict) + require.NotNil(t, result, "the interrupted journal is returned for inspection") + assert.Empty(t, result.Outcome) + state.AssertNotCalled(t, "SaveOperation", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) +} + +// TestDeploy_MismatchedKeyIsRejected proves one key cannot answer a +// different request. +func TestDeploy_MismatchedKeyIsRejected(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + state.EXPECT().Recover(mock.Anything).Return(nil).Once() + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil).Once() + state.EXPECT().ClaimOperation(mock.Anything, mock.Anything). + Return(domain.AppOperation{}, false, domain.ErrAppStateConflict).Once() + + svc := keyedService(t, state, runtime, images, secrets) + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog", Op: "key-1"}) + + require.ErrorIs(t, err, domain.ErrAppStateConflict) + assert.Nil(t, result) + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) +} + +// TestDeploy_UnknownAppIsNotFound proves deploying a name with no app +// identity reports the missing app and claims nothing. +func TestDeploy_UnknownAppIsNotFound(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + + state.EXPECT().Recover(mock.Anything).Return(nil).Once() + state.EXPECT().LoadDesired(mock.Anything, "ghost").Return(domain.AppDesiredRevision{}, false, nil).Once() + state.EXPECT().AppExists(mock.Anything, "ghost").Return(false, nil).Once() + + svc := keyedService(t, state, runtime, images, secrets) + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "ghost", Op: "key-1"}) + + require.ErrorIs(t, err, domain.ErrAppNotFound) + assert.Nil(t, result) + state.AssertNotCalled(t, "ClaimOperation", mock.Anything, mock.Anything) + state.AssertNotCalled(t, "SaveOperation", mock.Anything, mock.Anything) +} + +// TestDeploy_UnknownExplicitRevisionOnKnownApp proves a missing explicit +// revision of a known app stays a revision error, not an app error. +func TestDeploy_UnknownExplicitRevisionOnKnownApp(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + + state.EXPECT().Recover(mock.Anything).Return(nil).Once() + state.EXPECT().LoadRevision(mock.Anything, "blog", "rev-404"). + Return(domain.AppDesiredRevision{}, domain.ErrAppRevisionNotFound).Once() + state.EXPECT().AppExists(mock.Anything, "blog").Return(true, nil).Once() + + svc := keyedService(t, state, runtime, images, secrets) + _, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog", Revision: "rev-404", Op: "key-1"}) + + require.ErrorIs(t, err, domain.ErrAppRevisionNotFound) + state.AssertNotCalled(t, "ClaimOperation", mock.Anything, mock.Anything) +} + +// TestRemove_TerminalKeyReplaySkipsRuntime proves a repeated keyed removal +// replays its stored result without touching the runtime again. +func TestRemove_TerminalKeyReplaySkipsRuntime(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + + stored := domain.AppOperation{ + Op: "key-1", Kind: "remove", App: "blog", + Request: domain.AppOperationRequestFor("remove", "blog", "", ""), + Outcome: domain.AppOutcomeSuccess, + } + state.EXPECT().Recover(mock.Anything).Return(nil).Once() + state.EXPECT().ClaimOperation(mock.Anything, mock.Anything).Return(stored, false, nil).Once() + + svc := keyedService(t, state, runtime, images, secrets) + result, err := svc.Remove(ctx, "blog", "key-1") + + require.NoError(t, err) + require.NotNil(t, result) + assert.Equal(t, "key-1", result.Op) + assert.Equal(t, domain.AppOutcomeSuccess, result.Outcome) + state.AssertNotCalled(t, "SaveIntent", mock.Anything, mock.Anything) + state.AssertNotCalled(t, "RetireApp", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "StopContainer", mock.Anything, mock.Anything, mock.Anything) +} + +// TestStop_PendingKeyReplayConflicts proves a stop replay of an unfinished +// operation never re-stops a workload. +func TestStop_PendingKeyReplayConflicts(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + + pending := domain.AppOperation{ + Op: "key-1", Kind: "stop", App: "blog", + Request: domain.AppOperationRequestFor("stop", "blog", "", ""), + Steps: []domain.AppOperationStep{{ID: "intent.stopped", State: domain.AppStepPending}}, + } + state.EXPECT().Recover(mock.Anything).Return(nil).Once() + state.EXPECT().ClaimOperation(mock.Anything, mock.Anything).Return(pending, false, nil).Once() + + svc := keyedService(t, state, runtime, images, secrets) + result, err := svc.Stop(ctx, "blog", "key-1") + + require.ErrorIs(t, err, domain.ErrAppStateConflict) + require.NotNil(t, result) + state.AssertNotCalled(t, "SaveIntent", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "StopContainer", mock.Anything, mock.Anything, mock.Anything) +} + +// TestRestart_ClaimCarriesServiceIdentity proves the restart claim records +// the targeted service, so a key reused for another service conflicts. +func TestRestart_ClaimCarriesServiceIdentity(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: headlessSpec()}, + }} + state.EXPECT().Recover(mock.Anything).Return(nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil).Once() + state.EXPECT().ClaimOperation(mock.Anything, mock.MatchedBy(func(op domain.AppOperation) bool { + return op.Request == domain.AppOperationRequestFor("restart", "blog", "", "web") + })).Return(domain.AppOperation{}, false, domain.ErrAppStateConflict).Once() + + svc := keyedService(t, state, runtime, images, secrets) + _, err := svc.Restart(ctx, "blog", "web", "key-1") + + require.ErrorIs(t, err, domain.ErrAppStateConflict) + runtime.AssertNotCalled(t, "RestartContainer", mock.Anything, mock.Anything, mock.Anything) +} diff --git a/internal/usecase/deployment/preflight_gate_test.go b/internal/usecase/deployment/preflight_gate_test.go new file mode 100644 index 000000000..7092a9ac6 --- /dev/null +++ b/internal/usecase/deployment/preflight_gate_test.go @@ -0,0 +1,341 @@ +package deployment_test + +import ( + "context" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/deployment" +) + +// recordSavedOperations captures every journal write so a test can assert the +// terminal preflight outcome and that no effect ran before the gate failed. +func recordSavedOperations(state *outmocks.MockAppState) *[]domain.AppOperation { + saved := &[]domain.AppOperation{} + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).RunAndReturn( + func(_ context.Context, op domain.AppOperation) error { + *saved = append(*saved, op) + return nil + }) + return saved +} + +// assertNoDeployMutation proves a gate that failed before execution never +// touched the workload: no container is created, started, stopped, or +// removed, no volume is created, and ACTIVE is never republished. +func assertNoDeployMutation(t *testing.T, runtime *outmocks.MockContainerRuntime, state *outmocks.MockAppState) { + t.Helper() + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "StopContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "CreateVolume", mock.Anything, mock.Anything, mock.Anything) + state.AssertNotCalled(t, "SaveActive", mock.Anything, mock.Anything) +} + +// assertPreflightGateJournal proves a failed preflight gate left a terminal +// failed journal whose preflight step carries the failure. +func assertPreflightGateJournal(t *testing.T, saved []domain.AppOperation) { + t.Helper() + require.NotEmpty(t, saved, "a failed gate must journal the claim and its failure") + last := saved[len(saved)-1] + require.True(t, last.Terminal(), "a failed preflight gate must persist a terminal journal") + assert.Equal(t, domain.AppOutcomeFailed, last.Outcome) + step, ok := opStep(last, "preflight") + require.True(t, ok) + assert.Equal(t, domain.AppStepFailed, step.State) +} + +// TestDeploy_MissingRequiredSecretFailsClosed proves a deploy whose required +// secret cannot be resolved fails closed with ErrAppSecretMissing before any +// workload mutation, and journals the terminal failure. +func TestDeploy_MissingRequiredSecretFailsClosed(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-blog"}, nil) + images.EXPECT().ResolveDigest(mock.Anything, rev.Spec.Services[0].Image).Return(restartTestDigest, nil).Once() + secrets.EXPECT().GetSecret(mock.Anything, "gordon/apps/app-blog/web/database-url").Return("", assert.AnError).Once() + saved := recordSavedOperations(state) + + svc := keyedService(t, state, runtime, images, secrets) + _, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + + require.ErrorIs(t, err, domain.ErrAppSecretMissing) + assertNoDeployMutation(t, runtime, state) + assertPreflightGateJournal(t, *saved) +} + +// TestDeploy_EnvSecretCollisionFailsClosed proves a stored revision whose +// public env collides with one of the service's secret keys is refused +// before any image resolution or workload mutation. +func TestDeploy_EnvSecretCollisionFailsClosed(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + rev.Spec.Env = map[string]string{"DATABASE_URL": "public-value"} + + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-blog"}, nil) + saved := recordSavedOperations(state) + + svc := keyedService(t, state, runtime, images, secrets) + _, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + + require.ErrorIs(t, err, domain.ErrInvalidAppSpec) + images.AssertNotCalled(t, "ResolveDigest", mock.Anything, mock.Anything) + secrets.AssertNotCalled(t, "GetSecret", mock.Anything, mock.Anything) + assertNoDeployMutation(t, runtime, state) + assertPreflightGateJournal(t, *saved) +} + +// TestDeploy_UnmanagedImageVolumeRejected proves an image that declares a +// VOLUME with no operator mapping is refused before any workload mutation, +// and journals the terminal failure. +func TestDeploy_UnmanagedImageVolumeRejected(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-blog"}, nil) + images.EXPECT().ResolveDigest(mock.Anything, rev.Spec.Services[0].Image).Return(restartTestDigest, nil).Once() + secrets.EXPECT().GetSecret(mock.Anything, "gordon/apps/app-blog/web/database-url").Return("x", nil).Once() + runtime.EXPECT().InspectImageVolumes(mock.Anything, rev.Spec.Services[0].Image).Return([]string{"/data"}, nil).Once() + saved := recordSavedOperations(state) + + svc := keyedService(t, state, runtime, images, secrets) + _, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + + require.ErrorIs(t, err, domain.ErrAppUnmanagedImageVolume) + assertNoDeployMutation(t, runtime, state) + assertPreflightGateJournal(t, *saved) +} + +// TestDeploy_RefusesUnownedExistingVolume proves R2: a generated runtime +// volume that already exists but is NOT recorded in the app's ownership is +// refused fail-closed — the deploy must never implicitly adopt (and modify) +// foreign data under a reused name. +func TestDeploy_RefusesUnownedExistingVolume(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + svcSpec := webService() + svcSpec.Volumes = []domain.AppVolume{{Name: "data", Path: "/data"}} + rev := testRevision("blog", svcSpec) + runtimeName := domain.RuntimeVolumeName("blog", "web", "data") + + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + // Ownership records no volume, so the runtime one is foreign. + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-blog"}, nil) + images.EXPECT().ResolveDigest(mock.Anything, svcSpec.Image).Return(restartTestDigest, nil).Once() + secrets.EXPECT().GetSecret(mock.Anything, "gordon/apps/app-blog/web/database-url").Return("x", nil).Once() + runtime.EXPECT().InspectImageVolumes(mock.Anything, svcSpec.Image).Return(nil, nil).Once() + state.EXPECT().LoadCheckpoint(mock.Anything).Return(domain.AppStoreCheckpoint{}, nil).Once() + runtime.EXPECT().VolumeExists(mock.Anything, runtimeName).Return(true, nil).Once() + saved := recordSavedOperations(state) + + svc := keyedService(t, state, runtime, images, secrets) + _, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + + require.ErrorIs(t, err, domain.ErrAppStateConflict) + require.Contains(t, err.Error(), "not owned by app") + assertNoDeployMutation(t, runtime, state) + assertPreflightGateJournal(t, *saved) +} + +// TestStartDeploy_TargetedDiffPolicy proves the targeted-deploy convergence +// rules live in the claim phase: a requested-service image change is allowed +// and claims the operation, while app-wide or other-service divergence is +// refused with ErrAppStateConflict before any claim is written and without +// resolving an image or touching the runtime. +func TestStartDeploy_TargetedDiffPolicy(t *testing.T) { + baseService := webService() + worker := webService() + worker.Name = "worker" + worker.HTTP = []domain.AppHTTPInterface{{Host: "worker.example.com", Port: 8080, TLS: domain.AppTLSAuto}} + base := domain.AppActive{ + App: "blog", ConvergedRevision: "rev-1", Converged: true, + Networks: []domain.AppSharedNetwork{{Network: "shared", Services: []string{"web"}}}, + Services: map[string]domain.AppEffectiveService{ + "web": {Spec: baseService}, + "worker": {Spec: worker}, + }, + } + + tests := []struct { + name string + mutate func(*domain.AppSpec) + wantPath string + }{ + {name: "requested service image", mutate: func(spec *domain.AppSpec) { spec.Services[0].Image = "img:2" }}, + {name: "identical spec", mutate: func(*domain.AppSpec) {}}, + {name: "other service", wantPath: "service/worker/image", mutate: func(spec *domain.AppSpec) { spec.Services[1].Image = "img:2" }}, + {name: "environment", wantPath: "also changes env", mutate: func(spec *domain.AppSpec) { spec.Env = map[string]string{"MODE": "prod"} }}, + {name: "network", wantPath: "network/shared/aliases", mutate: func(spec *domain.AppSpec) { spec.Networks[0].Aliases = []string{"peer"} }}, + {name: "service added", wantPath: "service/job (added)", mutate: func(spec *domain.AppSpec) { + spec.Services = append(spec.Services, domain.AppService{Name: "job", Image: "img:1", Readiness: domain.AppReadiness{Type: "none", Timeout: 30 * time.Second}}) + }}, + {name: "service removed", wantPath: "service/worker (removed)", mutate: func(spec *domain.AppSpec) { + spec.Services = spec.Services[:1] + spec.Networks = nil + }}, + } + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + spec := domain.AppSpec{ + Name: "blog", + Networks: append([]domain.AppSharedNetwork(nil), base.Networks...), + Services: []domain.AppService{baseService, worker}, + } + spec.Networks[0].Aliases = append([]string(nil), base.Networks[0].Aliases...) + tc.mutate(&spec) + rev := domain.AppDesiredRevision{App: "blog", Revision: "rev-2", Spec: spec} + + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadRevision(mock.Anything, "blog", "rev-2").Return(rev, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(base, true, nil).Once() + if tc.wantPath == "" { + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil).Once() + } + + svc := keyedService(t, state, runtime, images, secrets) + started, err := svc.StartDeploy(ctx, deployment.DeployInput{App: "blog", Revision: "rev-2", Service: "web"}) + + if tc.wantPath != "" { + require.ErrorIs(t, err, domain.ErrAppStateConflict) + assert.Contains(t, err.Error(), tc.wantPath) + state.AssertNotCalled(t, "SaveOperation", mock.Anything, mock.Anything) + } else { + require.NoError(t, err) + require.True(t, started.Owned) + } + images.AssertNotCalled(t, "ResolveDigest", mock.Anything, mock.Anything) + assertNoDeployMutation(t, runtime, state) + }) + } +} + +// TestDeploy_HoldsSharedGCBarrierAcrossResourceSelection proves the shared +// GC lease is taken by the execution phase before resource selection and +// released only when the mutation ends. Prune can therefore never observe a +// resource that was selected but whose protection is not yet durable. +func TestDeploy_HoldsSharedGCBarrierAcrossResourceSelection(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + svcSpec := webService() + svcSpec.HTTP = nil + svcSpec.Readiness = domain.AppReadiness{} + rev := testRevision("blog", svcSpec) + + barrier := outmocks.NewMockGCBarrier(t) + lease := outmocks.NewMockGCLease(t) + held := false + barrier.EXPECT().AcquireShared(mock.Anything).Run(func(context.Context) { + held = true + }).Return(lease, nil).Once() + lease.EXPECT().Release().Run(func() { + held = false + }).Once() + + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-blog"}, nil) + // Image resolution and image-volume inspection are the resource-selection + // steps: they must happen inside the shared lease. + images.EXPECT().ResolveDigest(mock.Anything, svcSpec.Image).Run(func(context.Context, string) { + require.True(t, held, "image resolution must run inside the shared GC lease") + }).Return(restartTestDigest, nil).Once() + secrets.EXPECT().GetSecret(mock.Anything, "gordon/apps/app-blog/web/database-url").Return("x", nil) + runtime.EXPECT().InspectImageVolumes(mock.Anything, svcSpec.Image).Run(func(context.Context, string) { + require.True(t, held, "image-volume inspection must run inside the shared GC lease") + }).Return(nil, nil).Once() + state.EXPECT().LoadCheckpoint(mock.Anything).Return(domain.AppStoreCheckpoint{}, nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", Services: map[string]domain.AppEffectiveService{}, + }, true, nil) + expectNetworkProvision(runtime, "app-blog", 1) + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).Return(&domain.Container{ID: "c-new", Name: "web"}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-new").Return(nil).Once() + state.EXPECT().SaveActive(mock.Anything, mock.Anything).Return(nil) + state.EXPECT().SaveOwnership(mock.Anything, mock.Anything).Return(nil) + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) + + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 200, nil }, + func(context.Context, string) error { return nil }, + )).WithGCBarrier(barrier) + + started, err := svc.StartDeploy(ctx, deployment.DeployInput{App: "blog"}) + require.NoError(t, err) + require.True(t, started.Owned) + require.False(t, held, "the claim phase takes no shared GC lease") + + result, err := svc.ExecuteDeploy(ctx, started.Claim) + require.NoError(t, err) + require.NotNil(t, result) + assert.Equal(t, "deployed", result.Services["web"].Result) + assert.False(t, held, "the shared lease must be released when the operation ends") +} + +// deviceRevision returns a device-bearing revision over the standard +// mockRevision shape with device policies authorizing it. +func deviceRevision() domain.AppDesiredRevision { + rev := mockRevision() + rev.Spec.Services[0].Devices = []string{"test_gpu"} + return rev +} + +func devicePolicies() map[string]domain.AppDevicePolicy { + return map[string]domain.AppDevicePolicy{ + "test_gpu": { + Name: "test_gpu", + CDI: []string{"example.com/gpu=GPU-test-uuid"}, + AllowedApps: []string{"blog"}, + AllowedServices: []string{"web"}, + }, + } +} + +// TestDeploy_UnsupportedEngineFailsBeforeMutation proves a device-bearing +// revision on an engine that cannot serve CDI fails in preflight with +// ErrRuntimeUnsupported: the serving generation is never withdrawn and no +// container is created, started, stopped, or removed. +func TestDeploy_UnsupportedEngineFailsBeforeMutation(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := deviceRevision() + + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-blog"}, nil) + images.EXPECT().ResolveDigest(mock.Anything, rev.Spec.Services[0].Image).Return(restartTestDigest, nil).Once() + secrets.EXPECT().GetSecret(mock.Anything, "gordon/apps/app-blog/web/database-url").Return("x", nil) + runtime.EXPECT().InspectImageVolumes(mock.Anything, rev.Spec.Services[0].Image).Return(nil, nil) + runtime.EXPECT().SupportsCDIDevices(mock.Anything).Return(domain.ErrRuntimeUnsupported).Once() + saved := recordSavedOperations(state) + + svc := keyedService(t, state, runtime, images, secrets).WithDevicePolicies(devicePolicies()) + _, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + + require.ErrorIs(t, err, domain.ErrRuntimeUnsupported) + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) + assertNoDeployMutation(t, runtime, state) + assertPreflightGateJournal(t, *saved) +} diff --git a/internal/usecase/deployment/probes.go b/internal/usecase/deployment/probes.go new file mode 100644 index 000000000..82f2499a8 --- /dev/null +++ b/internal/usecase/deployment/probes.go @@ -0,0 +1,69 @@ +package deployment + +import ( + "bufio" + "context" + "fmt" + "io" + "net" + "net/http" + "strings" + "time" +) + +// probeHTTPClient is the dedicated readiness client. It never consults +// environment proxy settings, never follows redirects, and bounds every +// request so a nonresponding workload cannot hold the probe open. +var probeHTTPClient = &http.Client{ + Transport: &http.Transport{Proxy: nil}, + CheckRedirect: func(_ *http.Request, _ []*http.Request) error { + return http.ErrUseLastResponse + }, + Timeout: 5 * time.Second, +} + +// httpGetOnce performs one HTTP GET without following redirects. It dials +// probeURL's authority (the loopback published bind) while sending +// hostAuthority as the Host header, so a workload that validates the Host +// port of its declared container port accepts the probe. +func httpGetOnce(ctx context.Context, probeURL, hostAuthority string) (int, error) { + req, err := http.NewRequestWithContext(ctx, http.MethodGet, probeURL, nil) + if err != nil { + return 0, err + } + if hostAuthority != "" { + req.Host = hostAuthority + } + resp, err := probeHTTPClient.Do(req) + if err != nil { + return 0, err + } + defer func() { _ = resp.Body.Close() }() + return resp.StatusCode, nil +} + +// tcpDialOnce attempts one TCP connection. +func tcpDialOnce(ctx context.Context, addr string) error { + dialer := &net.Dialer{Timeout: 2 * time.Second} + conn, err := dialer.DialContext(ctx, "tcp", addr) + if err != nil { + return err + } + _ = conn.Close() + return nil +} + +// scanForMarker reports whether the stream contains the marker. +func scanForMarker(stream io.ReadCloser, marker string) (bool, error) { + scanner := bufio.NewScanner(stream) + scanner.Buffer(make([]byte, 64*1024), 1024*1024) + for scanner.Scan() { + if strings.Contains(scanner.Text(), marker) { + return true, nil + } + } + if err := scanner.Err(); err != nil { + return false, fmt.Errorf("deployment: read logs: %w", err) + } + return false, nil +} diff --git a/internal/usecase/deployment/publication_internal_test.go b/internal/usecase/deployment/publication_internal_test.go new file mode 100644 index 000000000..8d583017e --- /dev/null +++ b/internal/usecase/deployment/publication_internal_test.go @@ -0,0 +1,160 @@ +package deployment + +import ( + "context" + "errors" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" +) + +// readySpec is an HTTP service whose readiness fails fast. +func readySpec() domain.AppService { + return domain.AppService{ + Name: "web", + Image: "registry.example.com/blog/web:1.4.2", + HTTP: []domain.AppHTTPInterface{{Host: "blog.example.com", Port: 8080, TLS: "auto"}}, + Readiness: domain.AppReadiness{Type: domain.AppReadinessHTTP, Path: "/healthz", Timeout: 50 * time.Millisecond}, + } +} + +// TestVerifyRunningService_NotReadyWithdrawsAndClearsBinds proves a running +// generation that fails readiness is withdrawn, never published, and its +// recorded binds are cleared so the proxy cannot dial an unready port. +func TestVerifyRunningService_NotReadyWithdrawsAndClearsBinds(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + traffic := &stubTraffic{} + eff := domain.AppEffectiveService{ + Container: "c-1", + EffectiveRevision: "rev-1", + Spec: readySpec(), + BackendBinds: map[int]int{8080: 32771}, + } + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-1", mock.Anything).Return( + []domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 32771, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + state.EXPECT().RegisterBackendBinds(mock.Anything, mock.Anything).Return(nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{"web": {Container: "c-1"}}}, true, nil).Once() + state.EXPECT().SaveActive(mock.Anything, mock.Anything).Return(nil).Once() + svc := NewService(Deps{State: state, Runtime: runtime, Traffic: traffic}, zerowrap.Default()). + WithProbeDeps(NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 500, nil }, + func(context.Context, string) error { return nil }, + )) + + step := domain.AppOperationStep{ID: "service.web.start"} + result := &LifecycleResult{Services: map[string]ServiceResult{}} + svc.verifyRunningService(ctx, "blog", "web", eff, &step, result) + + assert.Equal(t, domain.AppStepFailed, step.State) + assert.Equal(t, "failed", result.Services["web"].Result) + // One serialized withdrawal plus one state-only bind clear through + // the canonical boundary; the fail-closed ACTIVE write itself is + // covered by the publisher tests in internal/app. + assert.Equal(t, []string{"blog/web", "blog/web"}, traffic.withdrawn) + state.AssertNumberOfCalls(t, "SaveActive", 1) +} + +// TestVerifyRunningService_ReadyPublishesBinds proves a ready generation is +// published with its freshly inspected binds. +func TestVerifyRunningService_ReadyPublishesBinds(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + traffic := &stubTraffic{} + eff := domain.AppEffectiveService{Container: "c-1", EffectiveRevision: "rev-1", Spec: readySpec()} + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-1", mock.Anything).Return( + []domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 32771, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + state.EXPECT().RegisterBackendBinds(mock.Anything, mock.Anything).Return(nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", Services: map[string]domain.AppEffectiveService{"web": eff}, + }, true, nil).Once() + state.EXPECT().SaveActive(mock.Anything, mock.Anything).Return(nil).Once() + + svc := NewService(Deps{State: state, Runtime: runtime, Traffic: traffic}, zerowrap.Default()). + WithProbeDeps(NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 200, nil }, + func(context.Context, string) error { return nil }, + )) + + step := domain.AppOperationStep{ID: "service.web.start"} + result := &LifecycleResult{Services: map[string]ServiceResult{}} + svc.verifyRunningService(ctx, "blog", "web", eff, &step, result) + + assert.Equal(t, domain.AppStepSucceeded, step.State) + assert.Equal(t, map[int]int{8080: 32771}, result.Services["web"].BackendBinds) + assert.Equal(t, []string{"blog/web"}, traffic.withdrawn, "the service is withdrawn before re-verification") +} + +// TestRedactDiagnostics_RemovesSecretValues proves a known secret value is +// never persisted in failure diagnostics. +func TestRedactDiagnostics_RemovesSecretValues(t *testing.T) { + state := outmocks.NewMockAppState(t) + secrets := outmocks.NewMockSecretProvider(t) + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-1"}, nil).Once() + secrets.EXPECT().GetSecret(mock.Anything, "gordon/apps/app-1/web/database-url").Return("supersecret", nil).Once() + + svc := NewService(Deps{State: state, Secrets: secrets}, zerowrap.Default()) + p := pinnedService{name: "web", spec: domain.AppService{ + Name: "web", Secrets: map[string]string{"DATABASE_URL": "database-url"}, + }} + + got := svc.redactDiagnostics(context.Background(), "blog", p, []string{"connect failed for supersecret", "still running"}) + + assert.Equal(t, []string{"connect failed for [redacted]", "still running"}, got) +} + +// TestRedactDiagnostics_DropsWhenSecretUnreadable proves diagnostics are +// dropped rather than persisted unredacted when a secret cannot be read. +func TestRedactDiagnostics_DropsWhenSecretUnreadable(t *testing.T) { + state := outmocks.NewMockAppState(t) + secrets := outmocks.NewMockSecretProvider(t) + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-1"}, nil).Once() + secrets.EXPECT().GetSecret(mock.Anything, "gordon/apps/app-1/web/database-url").Return("", assert.AnError).Once() + + svc := NewService(Deps{State: state, Secrets: secrets}, zerowrap.Default()) + p := pinnedService{name: "web", spec: domain.AppService{ + Name: "web", Secrets: map[string]string{"DATABASE_URL": "database-url"}, + }} + + assert.Nil(t, svc.redactDiagnostics(context.Background(), "blog", p, []string{"supersecret leaked"})) +} + +// TestStopService_StopFailureSurfaces proves a runtime stop failure is +// reported rather than recorded as a successful stop. +func TestStopService_StopFailureSurfaces(t *testing.T) { + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + runtime.EXPECT().StopContainer(mock.Anything, "c-1", mock.Anything).Return(errors.New("cannot stop")).Once() + + svc := NewService(Deps{State: state, Runtime: runtime, Traffic: &stubTraffic{}}, zerowrap.Default()) + eff := domain.AppEffectiveService{Container: "c-1", Spec: domain.AppService{Name: "web", StopGrace: time.Second}} + step, _, err := svc.stopService(context.Background(), "blog", "web", eff) + + require.Error(t, err) + assert.Equal(t, domain.AppStepFailed, step.State) + assert.Contains(t, step.Error, "cannot stop") +} + +// TestStopService_WithdrawFailureBlocksStop proves withdrawal runs first and +// its failure stops the operation before the container is touched. +func TestStopService_WithdrawFailureBlocksStop(t *testing.T) { + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + + svc := NewService(Deps{State: state, Runtime: runtime, Traffic: &stubTraffic{fail: true}}, zerowrap.Default()) + eff := domain.AppEffectiveService{Container: "c-1", Spec: domain.AppService{Name: "web", StopGrace: time.Second}} + step, _, err := svc.stopService(context.Background(), "blog", "web", eff) + + require.Error(t, err) + assert.Equal(t, domain.AppStepFailed, step.State) + runtime.AssertNotCalled(t, "StopContainer", mock.Anything, mock.Anything, mock.Anything) +} diff --git a/internal/usecase/deployment/readiness.go b/internal/usecase/deployment/readiness.go new file mode 100644 index 000000000..80586765d --- /dev/null +++ b/internal/usecase/deployment/readiness.go @@ -0,0 +1,486 @@ +package deployment + +import ( + "context" + "errors" + "fmt" + "io" + "net" + "net/url" + "strings" + "time" + + "github.com/bnema/gordon/internal/domain" +) + +// ProbeDeps abstracts the runtime hooks readiness needs. The deployment +// service wires these to the ContainerRuntime adapter; tests inject the +// mockery-generated ContainerRuntime mock through NewProbeDeps. +type ProbeDeps struct { + networkInfo func(ctx context.Context, containerID string) (string, int, error) + containerStart func(ctx context.Context, containerID string) (time.Time, error) + logStream func(ctx context.Context, containerID string, since time.Time) (io.ReadCloser, error) + httpGet func(ctx context.Context, url, hostAuthority string) (int, error) + tcpDial func(ctx context.Context, addr string) error + // networkProbe runs one bounded session against an internal port that + // has no host publication. Nil disables the internal path: it must + // never fall back to a loopback bind that does not exist. + networkProbe func(ctx context.Context, request domain.ContainerNetworkProbeRequest) (domain.ContainerNetworkProbeResult, error) +} + +// ProbeRuntime is the runtime subset readiness probing needs. It mirrors +// out.ContainerRuntime signatures so both the live adapter and the +// mockery-generated MockContainerRuntime satisfy it. +type ProbeRuntime interface { + GetContainerNetworkInfo(ctx context.Context, containerID string) (string, int, error) + GetContainerLogsSince(ctx context.Context, containerID string, since time.Time, follow bool) (io.ReadCloser, error) + InspectContainer(ctx context.Context, containerID string) (*domain.Container, error) + ProbeContainerNetwork(ctx context.Context, request domain.ContainerNetworkProbeRequest) (domain.ContainerNetworkProbeResult, error) +} + +// NewProbeDeps wires readiness probes to a runtime implementation. +// Production passes the ContainerRuntime adapter; tests pass the mockery +// MockContainerRuntime which satisfies ProbeRuntime. +func NewProbeDeps(runtime ProbeRuntime) ProbeDeps { + return ProbeDeps{ + networkInfo: func(ctx context.Context, containerID string) (string, int, error) { + return runtime.GetContainerNetworkInfo(ctx, containerID) + }, + containerStart: func(ctx context.Context, containerID string) (time.Time, error) { + ctr, err := runtime.InspectContainer(ctx, containerID) + if err != nil { + return time.Time{}, err + } + return ctr.StartedAt, nil + }, + logStream: func(ctx context.Context, containerID string, since time.Time) (io.ReadCloser, error) { + return runtime.GetContainerLogsSince(ctx, containerID, since, false) + }, + httpGet: httpGetOnce, + tcpDial: tcpDialOnce, + networkProbe: func(ctx context.Context, request domain.ContainerNetworkProbeRequest) (domain.ContainerNetworkProbeResult, error) { + return runtime.ProbeContainerNetwork(ctx, request) + }, + } +} + +// NewTestProbeDeps builds ProbeDeps with injectable HTTP/TCP probers. +// Tests control L4 outcomes without dialing; logStream still reads the +// runtime mock. +func NewTestProbeDeps(runtime ProbeRuntime, httpGet func(ctx context.Context, url, hostAuthority string) (int, error), tcpDial func(ctx context.Context, addr string) error) ProbeDeps { + deps := NewProbeDeps(runtime) + deps.httpGet = httpGet + deps.tcpDial = tcpDial + return deps +} + +// defaultProbeDeps wires readiness to the live runtime adapter. +func (s *Service) defaultProbeDeps() ProbeDeps { + return NewProbeDeps(s.deps.Runtime) +} + +// probeDeps returns the injected test probes or the live adapter wiring. +func (s *Service) probeDeps() ProbeDeps { + if s.probes != nil { + return *s.probes + } + return s.defaultProbeDeps() +} + +// waitServiceReady dispatches readiness for one service. It resolves the +// app private network lazily: only an internal interface needs it. A +// service whose readiness port belongs to an internal HTTP interface +// always probes over the private network; it never falls back to a +// loopback bind, and a public service never takes the network path. +func (s *Service) waitServiceReady(ctx context.Context, app, containerID string, spec domain.AppService, binds map[int]int) error { + switch spec.Readiness.Type { + case domain.AppReadinessHTTP, domain.AppReadinessTCP: + port := readinessContainerPort(spec) + if port > 0 && spec.InternallyOnlyPort(port) { + network, err := s.privateNetworkForApp(ctx, app) + if err != nil { + return err + } + return waitInternalReadyWithDeps(ctx, s.probeDeps(), containerID, spec, network, port) + } + } + return waitServiceReadyWithDeps(ctx, s.probeDeps(), containerID, spec, binds) +} + +// privateNetworkForApp resolves the incarnation-owned private network, the +// only network an internal readiness helper may join. It fails closed when +// the app has no incarnation id. +func (s *Service) privateNetworkForApp(ctx context.Context, app string) (string, error) { + ownership, err := s.deps.State.LoadOwnership(ctx, app) + if err != nil { + return "", fmt.Errorf("deployment: load ownership for private network: %w", err) + } + if ownership.ID == "" { + return "", fmt.Errorf("deployment: app %q has no incarnation id for internal readiness: %w", app, domain.ErrAppStateConflict) + } + return domain.AppPrivateNetworkName(s.deps.Networks.Prefix, ownership.ID), nil +} + +// waitInternalReadyWithDeps polls one bounded private-network session per +// attempt until the target is ready or the readiness timeout expires. An +// infrastructure error fails immediately and is never retried as an +// unhealthy attempt: a helper that cannot run says nothing about the +// target. +func waitInternalReadyWithDeps(ctx context.Context, deps ProbeDeps, containerID string, spec domain.AppService, network string, port int) error { + timeout := readinessTimeout(spec) + protocol, path := internalProbeShape(spec) + if deps.networkProbe == nil || deps.containerStart == nil { + return fmt.Errorf("deployment: internal readiness probe is not wired") + } + startedAt, err := deps.containerStart(ctx, containerID) + if err != nil { + return fmt.Errorf("deployment: internal readiness execution start for %s: %w", containerID, err) + } + if startedAt.IsZero() { + // Without an execution boundary the probe could report readiness + // for a generation that already restarted. Fail closed. + return fmt.Errorf("deployment: internal readiness needs the execution start of %s, runtime reported none", containerID) + } + deadline := time.Now().Add(timeout) + lastAttempt := "no attempt completed" + for { + if err := ctx.Err(); err != nil { + return err + } + remaining := time.Until(deadline) + if remaining <= 0 { + return fmt.Errorf("deployment: internal readiness timeout after %s on container port %d (last attempt: %s)", timeout, port, lastAttempt) + } + attemptTimeout := min(5*time.Second, remaining) + result, probeErr := runInternalProbeAttempt(ctx, deps, domain.ContainerNetworkProbeRequest{ + TargetContainerID: containerID, + ExpectedStartedAt: startedAt, + Network: network, + Protocol: protocol, + Port: port, + Path: path, + Timeout: attemptTimeout, + }) + ready, retryStale, diagnostic, classifyErr := classifyInternalProbe(result, probeErr) + if classifyErr != nil { + return classifyErr + } + if ready { + return nil + } + if diagnostic != "" { + lastAttempt = diagnostic + } + if retryStale { + // The candidate restarted or briefly vanished: re-read its + // execution boundary and keep polling until the deadline. This + // is "not ready yet", not a broken helper, so it must never + // abort the wait as an infrastructure error. + next, startErr := rereadInternalProbeStart(ctx, deps, containerID, startedAt) + if startErr != nil { + return startErr + } + startedAt = next + } + select { + case <-ctx.Done(): + return ctx.Err() + case <-time.After(500 * time.Millisecond): + } + } +} + +func runInternalProbeAttempt(ctx context.Context, deps ProbeDeps, request domain.ContainerNetworkProbeRequest) (domain.ContainerNetworkProbeResult, error) { + attemptCtx, cancel := context.WithTimeout(ctx, request.Timeout) + result, err := deps.networkProbe(attemptCtx, request) + attemptErr := attemptCtx.Err() + cancel() + if err != nil && !errors.Is(err, domain.ErrNetworkProbeCleanup) && errors.Is(attemptErr, context.DeadlineExceeded) && ctx.Err() == nil { + return domain.ContainerNetworkProbeResult{Diagnostic: "probe attempt timed out"}, nil + } + return result, err +} + +// rereadInternalProbeStart refreshes the execution boundary after a stale +// or vanished candidate. A candidate that briefly vanished (not found) +// keeps its previous boundary so the caller keeps polling until the +// deadline; any other runtime failure is infrastructure and aborts. It +// keeps the previous value when the runtime reports none so the next +// attempt still fails closed. +func rereadInternalProbeStart(ctx context.Context, deps ProbeDeps, containerID string, current time.Time) (time.Time, error) { + next, err := deps.containerStart(ctx, containerID) + if err != nil { + // ErrContainerNotFound is the same "not ready yet" condition the + // probe reported: preserve the boundary and keep polling instead of + // aborting the wait as an infrastructure error. + if errors.Is(err, domain.ErrContainerNotFound) { + return current, nil + } + return current, fmt.Errorf("deployment: internal readiness execution start for %s: %w", containerID, err) + } + if next.IsZero() { + return current, nil + } + return next, nil +} + +// classifyInternalProbe maps one attempt's raw outcome to the wait loop's +// next step: ready, not-ready (diagnostic), a stale or vanished candidate +// that should be re-read and retried, or an infrastructure failure that +// must abort the wait immediately. +func classifyInternalProbe(result domain.ContainerNetworkProbeResult, probeErr error) (ready, retryStale bool, diagnostic string, err error) { + switch { + case probeErr == nil: + if result.Ready { + return true, false, "", nil + } + if result.Diagnostic != "" { + return false, false, result.Diagnostic, nil + } + return false, false, "not ready", nil + case errors.Is(probeErr, domain.ErrNetworkProbeCleanup): + // A helper that could not be removed is a broken helper, not a + // stale candidate: an error wrapping both this and a stale or + // vanished sentinel must abort the wait, never be retried. + return false, false, "", fmt.Errorf("deployment: internal readiness probe helper cleanup failed: %w", probeErr) + case errors.Is(probeErr, domain.ErrAppStateConflict), errors.Is(probeErr, domain.ErrContainerNotFound): + return false, true, "target execution changed; retrying", nil + default: + return false, false, "", fmt.Errorf("deployment: internal readiness probe infrastructure error: %w", probeErr) + } +} + +// internalProbeShape selects the probe transport for an internal port: +// http readiness maps to a bounded HTTP probe, tcp readiness to a TCP +// connection probe. +func internalProbeShape(spec domain.AppService) (domain.ProbeProtocol, string) { + if spec.Readiness.Type == domain.AppReadinessHTTP { + path := spec.Readiness.Path + if path == "" { + path = "/" + } + return domain.ProbeProtocolHTTP, path + } + return domain.ProbeProtocolTCP, "" +} + +// waitServiceReadyWithDeps dispatches readiness by manifest type. +// binds carries the container's 127.0.0.1 backend publishes; L4 probes +// dial them rootless-first and never container IPs (plan D3: loopback- +// only generated backends). Restart/start reuse the recorded binds from +// the active record; deploy passes the freshly read binds. +func waitServiceReadyWithDeps(ctx context.Context, deps ProbeDeps, containerID string, spec domain.AppService, binds map[int]int) error { + switch spec.Readiness.Type { + case "", domain.AppReadinessNone: + return nil + case domain.AppReadinessLog: + return waitLogReadyWithDeps(ctx, deps, containerID, spec) + case domain.AppReadinessHTTP: + return waitHTTPReadyWithDeps(ctx, deps, spec, binds) + case domain.AppReadinessTCP: + return waitTCPReadyWithDeps(ctx, deps, spec, binds) + default: + return fmt.Errorf("deployment: unknown readiness type %q", spec.Readiness.Type) + } +} + +// waitHTTPReadyWithDeps polls GET path until 2xx/3xx or timeout. +func waitHTTPReadyWithDeps(ctx context.Context, deps ProbeDeps, spec domain.AppService, binds map[int]int) error { + timeout := readinessTimeout(spec) + containerPort := readinessContainerPort(spec) + port, err := dialPort(spec, binds) + if err != nil { + return err + } + path := spec.Readiness.Path + if path == "" { + path = "/" + } + pathPart, rawQuery, _ := strings.Cut(path, "?") + probeURL := (&url.URL{ + Scheme: "http", + Host: net.JoinHostPort("127.0.0.1", itoa(port)), + Path: pathPart, + RawQuery: rawQuery, + }).String() + // The probe dials the random loopback publication but must address the + // workload at its declared container port: a workload that validates the + // Host port (e.g. qBittorrent) rejects a published-port authority with + // 401 "Invalid Host header, port mismatch". + hostAuthority := net.JoinHostPort("127.0.0.1", itoa(containerPort)) + deadline := time.Now().Add(timeout) + var lastAttempt string + for { + if err := ctx.Err(); err != nil { + return err + } + attemptCtx, cancel := context.WithTimeout(ctx, min(2*time.Second, time.Until(deadline))) + status, err := deps.httpGet(attemptCtx, probeURL, hostAuthority) + cancel() + if err == nil && status >= 200 && status < 400 { + return nil + } + if err != nil { + lastAttempt = err.Error() + } else { + lastAttempt = fmt.Sprintf("HTTP status %d", status) + } + if time.Now().After(deadline) { + return fmt.Errorf("deployment: HTTP readiness timeout after %s: no 2xx/3xx from %s (last attempt: %s)", timeout, probeURL, lastAttempt) + } + select { + case <-ctx.Done(): + return ctx.Err() + case <-time.After(time.Second): + } + } +} + +// waitTCPReadyWithDeps dials the container port until accepted. +func waitTCPReadyWithDeps(ctx context.Context, deps ProbeDeps, spec domain.AppService, binds map[int]int) error { + timeout := readinessTimeout(spec) + port, err := dialPort(spec, binds) + if err != nil { + return err + } + addr := net.JoinHostPort("127.0.0.1", itoa(port)) + deadline := time.Now().Add(timeout) + for { + if err := ctx.Err(); err != nil { + return err + } + if err := deps.tcpDial(ctx, addr); err == nil { + return nil + } + if time.Now().After(deadline) { + return fmt.Errorf("deployment: TCP readiness timeout after %s: %s not reachable", timeout, addr) + } + select { + case <-ctx.Done(): + return ctx.Err() + case <-time.After(500 * time.Millisecond): + } + } +} + +// waitLogReadyWithDeps tails the logs of one exact container ID until the +// marker appears, bounded by the manifest readiness timeout and by the +// caller's deadline context. Logs are read from the observed start of the +// current execution onward, so a marker emitted by a previous execution +// of the same container ID can never satisfy the probe. +func waitLogReadyWithDeps(ctx context.Context, deps ProbeDeps, containerID string, spec domain.AppService) error { + timeout := readinessTimeout(spec) + deadlineCtx, cancel := context.WithTimeout(ctx, timeout) + defer cancel() + marker := spec.Readiness.Contains + if containerID == "" { + return fmt.Errorf("deployment: log readiness requires a container ID") + } + since, err := deps.containerStart(deadlineCtx, containerID) + if err != nil { + return fmt.Errorf("deployment: log readiness execution start for %s: %w", containerID, err) + } + if since.IsZero() { + // Without an execution boundary the probe would scan the + // container's whole history and could match a marker from a + // previous execution. Fail closed instead. + return fmt.Errorf("deployment: log readiness needs the execution start of %s, runtime reported none", containerID) + } + for { + if err := deadlineCtx.Err(); err != nil { + return fmt.Errorf("deployment: log readiness timeout waiting for %q: %w", marker, err) + } + stream, err := deps.logStream(deadlineCtx, containerID, since) + if err != nil { + return fmt.Errorf("deployment: log readiness stream for %s: %w", containerID, err) + } + found, readErr := scanForMarker(stream, marker) + _ = stream.Close() + if readErr != nil { + return readErr + } + if found { + return nil + } + select { + case <-deadlineCtx.Done(): + return fmt.Errorf("deployment: log readiness timeout waiting for %q: %w", marker, deadlineCtx.Err()) + case <-time.After(time.Second): + } + } +} + +// dialPort resolves the 127.0.0.1 host port for the readiness probe: +// the manifest's readiness port when set, else the single TCP-capable +// interface port (validation guarantees unambiguous selection). +func dialPort(spec domain.AppService, binds map[int]int) (int, error) { + containerPort := readinessContainerPort(spec) + hostPort, ok := binds[containerPort] + if !ok || hostPort <= 0 { + return 0, fmt.Errorf("deployment: no 127.0.0.1 backend bind for container port %d", containerPort) + } + return hostPort, nil +} + +// readinessContainerPort resolves the container port a readiness probe +// targets: the manifest's readiness port when set, else the single +// effective-public HTTP port (the backend the proxy dials), else the +// single TCP-capable interface port. Validation guarantees unambiguous +// selection for the latter two; preferring the public HTTP port keeps a +// public+internal service probeable when it declares no explicit +// readiness port. +func readinessContainerPort(spec domain.AppService) int { + if spec.Readiness.Port > 0 { + return spec.Readiness.Port + } + if port := singlePublicHTTPPort(spec); port > 0 { + return port + } + return singleTCPPort(spec) +} + +// singlePublicHTTPPort returns the container port when the service +// declares exactly one effective-public HTTP interface and no TCP +// interface, else 0. +func singlePublicHTTPPort(spec domain.AppService) int { + if len(spec.TCP) > 0 { + return 0 + } + port := 0 + for _, h := range spec.HTTP { + if !h.IsPublic() { + continue + } + if port != 0 && port != h.Port { + return 0 + } + port = h.Port + } + return port +} + +// singleTCPPort returns the container port when exactly one TCP-capable +// interface exists; validation rejects ambiguity (readiness.port +// required) before the engine runs. +func singleTCPPort(spec domain.AppService) int { + port := 0 + count := 0 + for _, h := range spec.HTTP { + port, count = h.Port, count+1 + } + for _, t := range spec.TCP { + port, count = t.Port, count+1 + } + if count == 1 { + return port + } + return 0 +} + +// readinessTimeout applies the manifest default. +func readinessTimeout(spec domain.AppService) time.Duration { + if spec.Readiness.Timeout > 0 { + return spec.Readiness.Timeout + } + return domain.AppDefaultReadinessTimeout +} diff --git a/internal/usecase/deployment/readiness_internal_test.go b/internal/usecase/deployment/readiness_internal_test.go new file mode 100644 index 000000000..4d7e727a3 --- /dev/null +++ b/internal/usecase/deployment/readiness_internal_test.go @@ -0,0 +1,78 @@ +package deployment + +import ( + "context" + "net" + "net/http" + "net/http/httptest" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +// TestWaitHTTPReady_ProbeURLKeepsLoopbackAuthority proves the probe URL is +// built from an immutable loopback host and an escaped path, so no declared +// path can redirect the request to a different authority. The Host header +// must name the declared container port, not the random published port. +func TestWaitHTTPReady_ProbeURLKeepsLoopbackAuthority(t *testing.T) { + var gotURL, gotHost string + deps := ProbeDeps{httpGet: func(_ context.Context, url, hostAuthority string) (int, error) { + gotURL, gotHost = url, hostAuthority + return 200, nil + }} + spec := domain.AppService{ + Name: "web", + HTTP: []domain.AppHTTPInterface{{Host: "web.example.com", Port: 8080}}, + Readiness: domain.AppReadiness{Type: domain.AppReadinessHTTP, Path: "/healthz?probe=1", Timeout: time.Second}, + } + require.NoError(t, waitHTTPReadyWithDeps(context.Background(), deps, spec, map[int]int{8080: 18080})) + assert.Equal(t, "http://127.0.0.1:18080/healthz?probe=1", gotURL) + assert.Equal(t, "127.0.0.1:8080", gotHost, "the Host header names the declared container port, not the published port") +} + +// TestWaitHTTPReady_SendsContainerPortAsHostAuthority is the qBittorrent +// regression: readiness dials the random loopback publication, but the +// workload rejects a Host whose port is not its declared container port +// with 401 "Invalid Host header, port mismatch". The probe must dial the +// published port while sending the container port as the Host authority. +func TestWaitHTTPReady_SendsContainerPortAsHostAuthority(t *testing.T) { + const containerPort = 8080 + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if r.Host != "127.0.0.1:8080" { + http.Error(w, "Invalid Host header, port mismatch", http.StatusUnauthorized) + return + } + w.WriteHeader(http.StatusOK) + })) + defer server.Close() + publishedPort := server.Listener.Addr().(*net.TCPAddr).Port + + spec := domain.AppService{ + Name: "qbittorrent", + HTTP: []domain.AppHTTPInterface{{Host: "qbit.example.com", Port: containerPort}}, + Readiness: domain.AppReadiness{Type: domain.AppReadinessHTTP, Path: "/api/v2/app/version", Timeout: 2 * time.Second}, + } + require.NoError(t, waitHTTPReadyWithDeps(context.Background(), ProbeDeps{httpGet: httpGetOnce}, spec, map[int]int{containerPort: publishedPort})) +} + +// TestWaitHTTPReady_HangingProbeRespectsDeadline proves a workload that +// never answers cannot hold the readiness operation open past its bound. +func TestWaitHTTPReady_HangingProbeRespectsDeadline(t *testing.T) { + deps := ProbeDeps{httpGet: func(ctx context.Context, _, _ string) (int, error) { + <-ctx.Done() + return 0, ctx.Err() + }} + spec := domain.AppService{ + Name: "web", + HTTP: []domain.AppHTTPInterface{{Host: "web.example.com", Port: 8080}}, + Readiness: domain.AppReadiness{Type: domain.AppReadinessHTTP, Path: "/healthz", Timeout: time.Second}, + } + start := time.Now() + err := waitHTTPReadyWithDeps(context.Background(), deps, spec, map[int]int{8080: 18080}) + require.Error(t, err) + assert.Less(t, time.Since(start), 6*time.Second) +} diff --git a/internal/usecase/deployment/reconcile_internal_test.go b/internal/usecase/deployment/reconcile_internal_test.go new file mode 100644 index 000000000..446e9312b --- /dev/null +++ b/internal/usecase/deployment/reconcile_internal_test.go @@ -0,0 +1,156 @@ +package deployment + +import ( + "context" + "testing" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" +) + +type stubTraffic struct { + withdrawn []string + fail bool +} + +func (s *stubTraffic) WithdrawService(_ context.Context, app, service string) error { + if s.fail { + return assert.AnError + } + s.withdrawn = append(s.withdrawn, app+"/"+service) + return nil +} + +// WithdrawServiceState records the state-only withdrawal the canonical +// boundary performs; in-package tests treat it like WithdrawService +// without a graph application. +func (s *stubTraffic) WithdrawServiceState(_ context.Context, app, service string) error { + if s.fail { + return assert.AnError + } + s.withdrawn = append(s.withdrawn, app+"/"+service) + return nil +} + +func (s *stubTraffic) RebuildTraffic(context.Context) error { return nil } + +// TestReconcileRemovedServices_WithdrawsStopsAndClears proves a service the +// new revision drops is withdrawn first, its exact container stopped and +// removed, its claims released, and its ACTIVE entry deleted while a +// healthy sibling is preserved. +func TestReconcileRemovedServices_WithdrawsStopsAndClears(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + traffic := &stubTraffic{} + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-web", EffectiveRevision: "rev-1", Spec: domain.AppService{Name: "web"}}, + "legacy": { + Container: "c-legacy", EffectiveRevision: "rev-1", + Spec: domain.AppService{Name: "legacy", TCP: []domain.AppTCPInterface{{Entrypoint: "tcp", Port: 9000, Publish: "0.0.0.0:9000"}}}, + }, + }} + + state.EXPECT().SaveRecoveryInhibition(mock.Anything, mock.MatchedBy(func(i domain.AppRecoveryInhibition) bool { + return i.App == "blog" && i.Service == "legacy" && i.ContainerID == "c-legacy" && i.Reason == "removed" + })).Return(nil).Once() + runtime.EXPECT().StopContainer(mock.Anything, "c-legacy", mock.Anything).Return(nil).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-legacy", false).Return(nil).Once() + state.EXPECT().ReleaseBackendBinds(mock.Anything, "blog", "c-legacy").Return(nil).Once() + var saved domain.AppActive + var order []string + state.EXPECT().SaveActive(mock.Anything, mock.Anything).RunAndReturn(func(_ context.Context, a domain.AppActive) error { + saved = a + order = append(order, "save-active") + return nil + }).Once() + state.EXPECT().ClearRecoveryInhibition(mock.Anything, "blog", "legacy", "c-legacy").RunAndReturn(func(context.Context, string, string, string) error { + order = append(order, "clear-inhibition") + return nil + }).Once() + + svc := NewService(Deps{State: state, Runtime: runtime, Traffic: traffic}, zerowrap.Default()) + steps, removed, warnings, err := svc.reconcileRemovedServices(ctx, "blog", active, []pinnedService{ + {name: "web", spec: domain.AppService{Name: "web"}}, + }) + + require.NoError(t, err) + assert.Equal(t, []string{"legacy"}, removed) + assert.Empty(t, warnings) + require.Len(t, steps, 1) + assert.Equal(t, domain.AppStepSucceeded, steps[0].State) + assert.Equal(t, []string{"blog/legacy"}, traffic.withdrawn) + _, ok := saved.Services["legacy"] + assert.False(t, ok, "removed service must be absent from ACTIVE") + _, ok = saved.Services["web"] + assert.True(t, ok, "healthy sibling must be preserved") + runtime.AssertNotCalled(t, "RemoveVolume", mock.Anything, mock.Anything, mock.Anything) + assert.Equal(t, []string{"save-active", "clear-inhibition"}, order, "removal must be persisted before its inhibition is cleared") +} + +// TestReconcileRemovedServices_SaveFailureKeepsInhibition proves a removal +// that cannot be persisted keeps its recovery inhibition, so recovery can +// never recreate a service ACTIVE still lists. +func TestReconcileRemovedServices_SaveFailureKeepsInhibition(t *testing.T) { + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "legacy": {Container: "c-legacy", EffectiveRevision: "rev-1", Spec: domain.AppService{Name: "legacy"}}, + }} + + state.EXPECT().SaveRecoveryInhibition(mock.Anything, mock.Anything).Return(nil).Once() + runtime.EXPECT().StopContainer(mock.Anything, "c-legacy", mock.Anything).Return(nil).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-legacy", false).Return(nil).Once() + state.EXPECT().ReleaseBackendBinds(mock.Anything, "blog", "c-legacy").Return(nil).Once() + state.EXPECT().SaveActive(mock.Anything, mock.Anything).Return(assert.AnError).Once() + + svc := NewService(Deps{State: state, Runtime: runtime, Traffic: &stubTraffic{}}, zerowrap.Default()) + _, _, _, err := svc.reconcileRemovedServices(context.Background(), "blog", active, nil) + + require.ErrorIs(t, err, assert.AnError) + state.AssertNotCalled(t, "ClearRecoveryInhibition", mock.Anything, mock.Anything, mock.Anything, mock.Anything) +} + +// TestReconcileRemovedServices_WithdrawFailureAbortsBeforeStop proves a +// failed withdrawal stops the deploy before the container is touched, so +// ACTIVE is never cleared while the service may still be reachable. +func TestReconcileRemovedServices_WithdrawFailureAbortsBeforeStop(t *testing.T) { + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + traffic := &stubTraffic{fail: true} + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "legacy": {Container: "c-legacy", EffectiveRevision: "rev-1", Spec: domain.AppService{Name: "legacy"}}, + }} + + svc := NewService(Deps{State: state, Runtime: runtime, Traffic: traffic}, zerowrap.Default()) + steps, _, _, err := svc.reconcileRemovedServices(context.Background(), "blog", active, nil) + + require.Error(t, err) + require.Len(t, steps, 1) + assert.Equal(t, domain.AppStepFailed, steps[0].State) + runtime.AssertNotCalled(t, "StopContainer", mock.Anything, mock.Anything, mock.Anything) + state.AssertNotCalled(t, "SaveActive", mock.Anything, mock.Anything) +} + +// TestReconcileRemovedServices_NoopWhenAllActiveServicesRemain proves a +// deploy that keeps every active service performs no removal effects. +func TestReconcileRemovedServices_NoopWhenAllActiveServicesRemain(t *testing.T) { + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-web"}, + }} + + svc := NewService(Deps{State: state, Runtime: runtime, Traffic: &stubTraffic{}}, zerowrap.Default()) + steps, removed, warnings, err := svc.reconcileRemovedServices(context.Background(), "blog", active, []pinnedService{{name: "web"}}) + + require.NoError(t, err) + assert.Empty(t, steps) + assert.Empty(t, removed) + assert.Empty(t, warnings) +} diff --git a/internal/usecase/deployment/reconcile_mock_test.go b/internal/usecase/deployment/reconcile_mock_test.go new file mode 100644 index 000000000..6eaa16480 --- /dev/null +++ b/internal/usecase/deployment/reconcile_mock_test.go @@ -0,0 +1,806 @@ +package deployment_test + +import ( + "context" + "fmt" + "io" + "strings" + "sync" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/deployment" +) + +// recordingTraffic is a fail-closed traffic boundary double: it records +// withdrawals and publications and can fail either operation. +type recordingTraffic struct { + mu sync.Mutex + attempts int + withdrawals []string + rebuilds int + failWithdraw bool + failRebuild bool +} + +func (t *recordingTraffic) WithdrawService(_ context.Context, app, service string) error { + t.mu.Lock() + defer t.mu.Unlock() + t.attempts++ + if t.failWithdraw { + return assert.AnError + } + t.withdrawals = append(t.withdrawals, app+"/"+service) + return nil +} + +func (t *recordingTraffic) WithdrawServiceState(_ context.Context, app, service string) error { + t.mu.Lock() + defer t.mu.Unlock() + if t.failWithdraw { + return assert.AnError + } + t.withdrawals = append(t.withdrawals, app+"/"+service) + return nil +} + +func (t *recordingTraffic) withdrawAttempts() int { + t.mu.Lock() + defer t.mu.Unlock() + return t.attempts +} + +func (t *recordingTraffic) RebuildTraffic(context.Context) error { + t.mu.Lock() + defer t.mu.Unlock() + if t.failRebuild { + return assert.AnError + } + t.rebuilds++ + return nil +} + +func (t *recordingTraffic) withdrawn() []string { + t.mu.Lock() + defer t.mu.Unlock() + return append([]string(nil), t.withdrawals...) +} + +func (t *recordingTraffic) rebuildCount() int { + t.mu.Lock() + defer t.mu.Unlock() + return t.rebuilds +} + +// recoveringService builds the engine with a recording traffic boundary +// and injectable L4 probes so recovery passes stay deterministic. +func recoveringService( + t *testing.T, + state *outmocks.MockAppState, + runtime *outmocks.MockContainerRuntime, + traffic *recordingTraffic, + httpGet func(context.Context, string, string) (int, error), +) *deployment.Service { + t.Helper() + return deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Traffic: traffic, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + httpGet, + func(context.Context, string) error { return nil }, + )) +} + +// headlessSpec is a service with no interfaces and no readiness, so a +// recovery pass performs no bind inspection and no probe. +func headlessSpec() domain.AppService { + return domain.AppService{Name: "web", Image: "registry.example.com/blog/web:1.4.2"} +} + +func TestReconcileRunning_EmptyActiveTakesNoRuntimeAction(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{} + + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{}, false, nil).Once() + + svc := recoveringService(t, state, runtime, traffic, nil) + require.NoError(t, svc.ReconcileRunning(ctx)) + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "InspectContainer", mock.Anything, mock.Anything) + assert.Zero(t, traffic.rebuildCount()) +} + +func TestReconcileRunning_StoppedIntentStopsRevivedExactIDWithoutRemoval(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{} + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: headlessSpec()}, + }} + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog", Stopped: true}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + runtime.EXPECT().InspectContainer(mock.Anything, "c-1"). + Return(&domain.Container{ID: "c-1", Status: "running"}, nil).Once() + runtime.EXPECT().StopContainer(mock.Anything, "c-1", mock.Anything).Return(nil).Once() + + svc := recoveringService(t, state, runtime, traffic, nil) + require.NoError(t, svc.ReconcileRunning(ctx)) + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RestartContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, mock.Anything, mock.Anything) + assert.GreaterOrEqual(t, traffic.rebuildCount(), 1, "a stopped app is republished without its backend") +} + +func TestReconcileRunning_AbstainsWhenStoppedContainerAlreadyDown(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{} + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", Spec: headlessSpec()}, + }} + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog", Stopped: true}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + runtime.EXPECT().InspectContainer(mock.Anything, "c-1"). + Return(&domain.Container{ID: "c-1", Status: "exited"}, nil).Once() + + svc := recoveringService(t, state, runtime, traffic, nil) + require.NoError(t, svc.ReconcileRunning(ctx)) + runtime.AssertNotCalled(t, "StopContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) +} + +func TestReconcileRunning_StartsExitedExactIDThenVerifiesAndPublishes(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{} + spec := webService() + spec.Readiness.Timeout = time.Second + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: spec, BackendBinds: map[int]int{8080: 32770}}, + }} + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil).Twice() + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + startedAt := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + runtime.EXPECT().InspectContainer(mock.Anything, "c-1"). + Return(&domain.Container{ID: "c-1", Status: "exited", StartedAt: startedAt}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-1").Return(nil).Once() + runtime.EXPECT().InspectContainer(mock.Anything, "c-1"). + Return(&domain.Container{ID: "c-1", Status: "running", StartedAt: startedAt}, nil).Once() + runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "c-1").Return("", false, nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-1", mock.Anything). + Return([]domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 32771, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + state.EXPECT().RegisterBackendBinds(mock.Anything, mock.MatchedBy(func(claims []domain.AppListenerReservation) bool { + return len(claims) == 1 && claims[0].Port == 32771 && claims[0].ContainerID == "c-1" + })).Return(nil).Once() + state.EXPECT().SaveActive(mock.Anything, mock.MatchedBy(func(a domain.AppActive) bool { + return a.Services["web"].BackendBinds[8080] == 32771 + })).Return(nil).Once() + + var probedURL string + svc := recoveringService(t, state, runtime, traffic, func(_ context.Context, url, _ string) (int, error) { + probedURL = url + return 200, nil + }) + require.NoError(t, svc.ReconcileRunning(ctx)) + + assert.Contains(t, probedURL, "127.0.0.1:32771", "readiness follows the refreshed bind") + assert.Equal(t, []string{"blog/web"}, traffic.withdrawn(), "the stale backend is withdrawn before recovery") + assert.Equal(t, 1, traffic.rebuildCount()) + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "PullImage", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, mock.Anything, mock.Anything) +} + +func TestReconcileRunning_ConfirmedMissingWithdrawsWithoutReconstruction(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{} + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: headlessSpec()}, + }} + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + runtime.EXPECT().InspectContainer(mock.Anything, "c-1"). + Return(nil, fmt.Errorf("inspect: %w", domain.ErrContainerNotFound)).Once() + // A confirmed-missing container must not keep its loopback claims. + state.EXPECT().ReleaseBackendBinds(mock.Anything, "blog", "c-1").Return(nil).Once() + + svc := recoveringService(t, state, runtime, traffic, nil) + err := svc.ReconcileRunning(ctx) + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrContainerNotFound) + assert.Contains(t, err.Error(), "no reconstruction") + assert.Equal(t, []string{"blog/web"}, traffic.withdrawn()) + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "PullImage", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, mock.Anything, mock.Anything) +} + +func TestReconcileRunning_TransientInspectErrorTakesNoAction(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{} + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: headlessSpec()}, + }} + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + runtime.EXPECT().InspectContainer(mock.Anything, "c-1").Return(nil, assert.AnError).Once() + + svc := recoveringService(t, state, runtime, traffic, nil) + err := svc.ReconcileRunning(ctx) + require.Error(t, err) + assert.NotErrorIs(t, err, domain.ErrContainerNotFound) + assert.Empty(t, traffic.withdrawn(), "a transient error is not a dead backend") + assert.Zero(t, traffic.rebuildCount()) + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, mock.Anything, mock.Anything) +} + +func TestReconcileRunning_InhibitedGenerationIsNeverStarted(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{} + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-old", EffectiveRevision: "rev-1", Spec: headlessSpec(), BackendBinds: map[int]int{8080: 32770}}, + }} + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil).Once() + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return([]domain.AppRecoveryInhibition{{ + App: "blog", Service: "web", ContainerID: "c-old", Reason: domain.AppInhibitReplacementPending, + }}, nil).Once() + + svc := recoveringService(t, state, runtime, traffic, nil) + err := svc.ReconcileRunning(ctx) + require.Error(t, err) + assert.Contains(t, err.Error(), "recovery inhibited") + runtime.AssertNotCalled(t, "InspectContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RestartContainer", mock.Anything, mock.Anything, mock.Anything) + assert.Equal(t, []string{"blog/web"}, traffic.withdrawn(), + "an inhibited generation must not stay published") +} + +func TestReconcileRunning_NativeRestartRaceIsNotCharged(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{} + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: headlessSpec()}, + }} + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil).Twice() + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + startedAt := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + runtime.EXPECT().InspectContainer(mock.Anything, "c-1"). + Return(&domain.Container{ID: "c-1", Status: "exited", StartedAt: startedAt}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-1").Return(assert.AnError).Once() + // Native restart policy already revived it: the failed start is not + // charged and no second restart is issued. + runtime.EXPECT().InspectContainer(mock.Anything, "c-1"). + Return(&domain.Container{ID: "c-1", Status: "running", StartedAt: startedAt}, nil).Twice() + runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "c-1").Return("", false, nil).Once() + + svc := recoveringService(t, state, runtime, traffic, nil) + require.NoError(t, svc.ReconcileRunning(ctx)) + runtime.AssertNotCalled(t, "RestartContainer", mock.Anything, mock.Anything, mock.Anything) + assert.Equal(t, 1, traffic.rebuildCount()) +} + +func TestReconcileRunning_NativeRestartRefreshesChangedBinds(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{} + spec := webService() + spec.Readiness = domain.AppReadiness{Type: "none"} + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: spec, BackendBinds: map[int]int{8080: 32770}}, + }} + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil).Twice() + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + runtime.EXPECT().InspectContainer(mock.Anything, "c-1").Return(&domain.Container{ + ID: "c-1", Status: "running", + StartedAt: time.Date(2026, 9, 10, 12, 5, 0, 0, time.UTC), + }, nil).Twice() + runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "c-1").Return("", false, nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-1", mock.Anything). + Return([]domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 32799, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + state.EXPECT().RegisterBackendBinds(mock.Anything, mock.Anything).Return(nil).Once() + state.EXPECT().SaveActive(mock.Anything, mock.MatchedBy(func(a domain.AppActive) bool { + return a.Services["web"].BackendBinds[8080] == 32799 + })).Return(nil).Once() + + svc := recoveringService(t, state, runtime, traffic, nil) + require.NoError(t, svc.ReconcileRunning(ctx)) + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RestartContainer", mock.Anything, mock.Anything, mock.Anything) + assert.Equal(t, 1, traffic.rebuildCount(), "a changed bind is republished once") +} + +func TestReconcileRunning_RestartingObservesAndRetries(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{} + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: headlessSpec()}, + }} + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil).Once() + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + runtime.EXPECT().InspectContainer(mock.Anything, "c-1"). + Return(&domain.Container{ID: "c-1", Status: "restarting"}, nil).Once() + + svc := recoveringService(t, state, runtime, traffic, nil) + require.NoError(t, svc.ReconcileRunning(ctx)) + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RestartContainer", mock.Anything, mock.Anything, mock.Anything) +} + +func TestReconcileRunning_PausedRefusesNonDestructiveRecovery(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{} + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: headlessSpec()}, + }} + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil).Once() + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + runtime.EXPECT().InspectContainer(mock.Anything, "c-1"). + Return(&domain.Container{ID: "c-1", Status: "paused"}, nil).Once() + + svc := recoveringService(t, state, runtime, traffic, nil) + err := svc.ReconcileRunning(ctx) + require.Error(t, err) + assert.Contains(t, err.Error(), "paused") + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RestartContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, mock.Anything, mock.Anything) +} + +func TestReconcileRunning_OneFailureDoesNotBlockLaterApps(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{} + + badActive := domain.AppActive{App: "bad", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-bad", EffectiveRevision: "rev-1", Spec: headlessSpec()}, + }} + bindSpec := webService() + bindSpec.Readiness = domain.AppReadiness{Type: "none"} + goodActive := domain.AppActive{App: "good", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-good", EffectiveRevision: "rev-1", Spec: bindSpec, BackendBinds: map[int]int{8080: 32770}}, + }} + state.EXPECT().ListApps(mock.Anything).Return([]string{"bad", "good"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "bad").Return(domain.AppStopIntent{App: "bad"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "good").Return(domain.AppStopIntent{App: "good"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "bad").Return(badActive, true, nil) + state.EXPECT().LoadActive(mock.Anything, "good").Return(goodActive, true, nil).Twice() + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "bad").Return(nil, nil).Once() + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "good").Return(nil, nil).Once() + runtime.EXPECT().InspectContainer(mock.Anything, "c-bad").Return(nil, assert.AnError).Once() + runtime.EXPECT().InspectContainer(mock.Anything, "c-good").Return(&domain.Container{ + ID: "c-good", Status: "running", + StartedAt: time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC), + }, nil) + runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "c-good").Return("", false, nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-good", mock.Anything). + Return([]domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 32771, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + state.EXPECT().RegisterBackendBinds(mock.Anything, mock.Anything).Return(nil).Once() + state.EXPECT().SaveActive(mock.Anything, mock.Anything).Return(nil).Once() + + svc := recoveringService(t, state, runtime, traffic, nil) + err := svc.ReconcileRunning(ctx) + require.Error(t, err) + assert.Contains(t, err.Error(), `"bad"`) + runtime.AssertCalled(t, "InspectContainer", mock.Anything, "c-good") + assert.Equal(t, 1, traffic.rebuildCount(), "the healthy app is still published") +} + +func TestReconcileRunning_WithdrawalFailureBlocksRuntimeRecovery(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{failWithdraw: true} + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: headlessSpec(), BackendBinds: map[int]int{8080: 32770}}, + }} + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil).Once() + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + runtime.EXPECT().InspectContainer(mock.Anything, "c-1"). + Return(&domain.Container{ID: "c-1", Status: "exited"}, nil).Once() + + svc := recoveringService(t, state, runtime, traffic, nil) + err := svc.ReconcileRunning(ctx) + require.Error(t, err) + assert.Contains(t, err.Error(), "withdraw") + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything, + "runtime recovery requires a proven fail-closed withdrawal") + assert.Zero(t, traffic.rebuildCount()) +} + +func TestReconcileRunning_RetriesWithdrawalBeforeAnyLaterPublication(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{failWithdraw: true} + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: headlessSpec(), BackendBinds: map[int]int{8080: 32770}}, + }} + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil) + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil) + runtime.EXPECT().InspectContainer(mock.Anything, "c-1"). + Return(&domain.Container{ID: "c-1", Status: "exited"}, nil).Times(3) + runtime.EXPECT().StartContainer(mock.Anything, "c-1").Return(nil).Once() + runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "c-1").Return("", false, nil).Once() + state.EXPECT().SaveActive(mock.Anything, mock.Anything).Return(nil).Once() + + svc := recoveringService(t, state, runtime, traffic, nil) + require.Error(t, svc.ReconcileRunning(ctx)) + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) + + // Recovery is retried on a later pass and must withdraw again before + // it may start or publish anything. + traffic.mu.Lock() + traffic.failWithdraw = false + traffic.mu.Unlock() + require.NoError(t, svc.ReconcileRunning(ctx)) + runtime.AssertCalled(t, "StartContainer", mock.Anything, "c-1") + assert.Equal(t, 2, traffic.withdrawAttempts(), "withdrawal is retried, never skipped") +} + +func TestReconcileRunning_RebuildFailureDoesNotRestartHealthyContainer(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{failRebuild: true} + + bindSpec := webService() + bindSpec.Readiness = domain.AppReadiness{Type: "none"} + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: bindSpec, BackendBinds: map[int]int{8080: 32770}}, + }} + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil).Twice() + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + runtime.EXPECT().InspectContainer(mock.Anything, "c-1").Return(&domain.Container{ + ID: "c-1", Status: "running", StartedAt: time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC), + }, nil).Twice() + runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "c-1").Return("", false, nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-1", mock.Anything). + Return([]domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 32771, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + state.EXPECT().RegisterBackendBinds(mock.Anything, mock.Anything).Return(nil).Once() + state.EXPECT().SaveActive(mock.Anything, mock.Anything).Return(nil).Once() + + svc := recoveringService(t, state, runtime, traffic, nil) + err := svc.ReconcileRunning(ctx) + require.Error(t, err) + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RestartContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, mock.Anything, mock.Anything) +} + +func TestReconcileRunning_UnhealthyRestartsOnlyAfterConsecutiveObservations(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{} + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: headlessSpec()}, + }} + startedAt := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil) + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil) + runtime.EXPECT().InspectContainer(mock.Anything, "c-1"). + Return(&domain.Container{ID: "c-1", Status: "running", StartedAt: startedAt}, nil) + runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "c-1").Return("unhealthy", true, nil) + runtime.EXPECT().RestartContainer(mock.Anything, "c-1", mock.Anything).Return(nil).Once() + + svc := recoveringService(t, state, runtime, traffic, nil) + // First observation only arms the streak. + require.NoError(t, svc.ReconcileRunning(ctx)) + runtime.AssertNotCalled(t, "RestartContainer", mock.Anything, mock.Anything, mock.Anything) + + // The second consecutive unhealthy observation restarts the exact ID + // with the effective stop grace. + require.NoError(t, svc.ReconcileRunning(ctx)) + runtime.AssertCalled(t, "RestartContainer", mock.Anything, "c-1", domain.AppDefaultStopGrace) +} + +func TestReconcileRunning_HealthStartingWaits(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{} + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: headlessSpec()}, + }} + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil) + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil) + runtime.EXPECT().InspectContainer(mock.Anything, "c-1").Return(&domain.Container{ + ID: "c-1", Status: "running", StartedAt: time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC), + }, nil) + runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "c-1").Return("starting", true, nil) + + svc := recoveringService(t, state, runtime, traffic, nil) + for range 3 { + require.NoError(t, svc.ReconcileRunning(ctx)) + } + runtime.AssertNotCalled(t, "RestartContainer", mock.Anything, mock.Anything, mock.Anything) +} + +func TestReconcileRunning_NoHealthcheckNeverRestartsOnReadiness(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{} + spec := webService() + spec.Readiness.Timeout = 20 * time.Millisecond + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: spec, BackendBinds: map[int]int{8080: 32770}}, + }} + firstExecution := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil) + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil) + runtime.EXPECT().InspectContainer(mock.Anything, "c-1"). + Return(&domain.Container{ID: "c-1", Status: "running", StartedAt: firstExecution}, nil).Once() + runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "c-1").Return("", false, nil) + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-1", mock.Anything). + Return([]domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 32770, Protocol: domain.NetworkProtocolTCP}}, nil) + state.EXPECT().RegisterBackendBinds(mock.Anything, mock.Anything).Return(nil) + + // An execution unseen by this daemon is verified immediately because it + // may have restarted while Gordon was absent. Readiness failure keeps it + // withdrawn, but never restarts an otherwise running workload. + svc := recoveringService(t, state, runtime, traffic, func(context.Context, string, string) (int, error) { + return 500, nil + }) + require.Error(t, svc.ReconcileRunning(ctx), "the unseen execution fails readiness and stays withdrawn") + runtime.AssertNotCalled(t, "RestartContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) +} + +func TestReconcileRunning_StaleGenerationIsNeverPublished(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{} + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: headlessSpec()}, + }} + // A concurrent mutation replaced the generation while recovery held + // the app lock: the reloaded ACTIVE names a different container. + replaced := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-2", EffectiveRevision: "rev-2", Spec: headlessSpec()}, + }} + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(replaced, true, nil).Once() + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + runtime.EXPECT().InspectContainer(mock.Anything, "c-1").Return(&domain.Container{ + ID: "c-1", Status: "running", StartedAt: time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC), + }, nil).Twice() + runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "c-1").Return("", false, nil).Once() + + svc := recoveringService(t, state, runtime, traffic, nil) + err := svc.ReconcileRunning(ctx) + require.Error(t, err) + assert.Contains(t, err.Error(), "generation changed") + assert.Zero(t, traffic.rebuildCount(), "stale recovery evidence must not publish") +} + +func TestReconcileRunning_LogReadinessUsesTheExactExecutionScope(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{} + spec := headlessSpec() + spec.Readiness = domain.AppReadiness{Type: domain.AppReadinessLog, Contains: "ready", Timeout: time.Second} + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: spec}, + }} + startedAt := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + runtime.EXPECT().InspectContainer(mock.Anything, "c-1"). + Return(&domain.Container{ID: "c-1", Status: "exited", StartedAt: startedAt}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-1").Return(nil).Once() + runtime.EXPECT().InspectContainer(mock.Anything, "c-1"). + Return(&domain.Container{ID: "c-1", Status: "running", StartedAt: startedAt}, nil).Twice() + runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "c-1").Return("", false, nil).Once() + var observedID string + var observedSince time.Time + runtime.EXPECT().GetContainerLogsSince(mock.Anything, "c-1", mock.Anything, false). + RunAndReturn(func(_ context.Context, containerID string, since time.Time, _ bool) (io.ReadCloser, error) { + observedID = containerID + observedSince = since + return io.NopCloser(strings.NewReader("boot\nready\n")), nil + }).Once() + + svc := recoveringService(t, state, runtime, traffic, nil) + require.NoError(t, svc.ReconcileRunning(ctx)) + assert.Equal(t, "c-1", observedID, "readiness reads the exact ACTIVE container") + assert.Equal(t, startedAt, observedSince, "readiness is scoped to the current execution start") + assert.Equal(t, 1, traffic.rebuildCount()) +} + +func TestReconcileRunning_OldExecutionMarkerCannotSatisfyReadiness(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{} + spec := headlessSpec() + spec.Readiness = domain.AppReadiness{Type: domain.AppReadinessLog, Contains: "ready", Timeout: 20 * time.Millisecond} + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: spec}, + }} + startedAt := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + runtime.EXPECT().InspectContainer(mock.Anything, "c-1"). + Return(&domain.Container{ID: "c-1", Status: "exited", StartedAt: startedAt}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-1").Return(nil).Once() + runtime.EXPECT().InspectContainer(mock.Anything, "c-1"). + Return(&domain.Container{ID: "c-1", Status: "running", StartedAt: startedAt}, nil).Once() + runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "c-1").Return("", false, nil).Once() + // The runtime only returns the current execution's logs; the old + // marker is simply not part of the stream. + runtime.EXPECT().GetContainerLogsSince(mock.Anything, "c-1", startedAt, false). + Return(io.NopCloser(strings.NewReader("booting\n")), nil) + + svc := recoveringService(t, state, runtime, traffic, nil) + err := svc.ReconcileRunning(ctx) + require.Error(t, err) + assert.Contains(t, err.Error(), "log readiness timeout") + runtime.AssertNotCalled(t, "RestartContainer", mock.Anything, mock.Anything, mock.Anything) +} + +// TestReconcileRunning_CrashLoopIsChargedEvenWhenStartsSucceed proves the +// crash-loop budget cannot be escaped by starts that succeed and then die +// before the next pass: after three unconfirmed interventions the +// generation is delayed instead of restarted every 15 seconds. +func TestReconcileRunning_CrashLoopIsChargedEvenWhenStartsSucceed(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{} + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: headlessSpec()}, + }} + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil) + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil) + // Every pass finds the exact ACTIVE ID exited again. + started := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + runtime.EXPECT().InspectContainer(mock.Anything, "c-1").Return(&domain.Container{ + ID: "c-1", Status: "exited", StartedAt: started, + }, nil) + runtime.EXPECT().StartContainer(mock.Anything, "c-1").Return(nil).Times(3) + runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "c-1").Return("", false, nil) + + svc := recoveringService(t, state, runtime, traffic, nil) + for pass := 1; pass <= 4; pass++ { + require.NoError(t, svc.ReconcileRunning(ctx), "pass %d", pass) + } + runtime.AssertNumberOfCalls(t, "StartContainer", 3) +} + +// TestReconcileRunning_StaleExecutionAfterReadinessNeverPublishes proves +// the last revalidation: a container that restarted while this pass +// waited for readiness must not republish the previous execution's binds. +func TestReconcileRunning_StaleExecutionAfterReadinessNeverPublishes(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{} + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: headlessSpec()}, + }} + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil).Once() + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + firstExecution := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + runtime.EXPECT().InspectContainer(mock.Anything, "c-1"). + Return(&domain.Container{ID: "c-1", Status: "exited", StartedAt: firstExecution}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-1").Return(nil).Once() + runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "c-1").Return("", false, nil).Once() + // The workload restarted while the pass was verifying it. + runtime.EXPECT().InspectContainer(mock.Anything, "c-1"). + Return(&domain.Container{ID: "c-1", Status: "running", StartedAt: firstExecution.Add(time.Minute)}, nil).Once() + + svc := recoveringService(t, state, runtime, traffic, nil) + err := svc.ReconcileRunning(ctx) + require.Error(t, err) + assert.Contains(t, err.Error(), "restarted during recovery") + assert.Zero(t, traffic.rebuildCount(), "stale evidence must not publish") + state.AssertNotCalled(t, "SaveActive", mock.Anything, mock.Anything) +} + +// TestReconcileRunning_FailedReadinessPersistsNoBinds proves the ordering +// the plan requires: binds reach ACTIVE only after readiness passed, so a +// failed verification cannot be projected by any other publication. +func TestReconcileRunning_FailedReadinessPersistsNoBinds(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + traffic := &recordingTraffic{} + spec := webService() + spec.Readiness.Timeout = 20 * time.Millisecond + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: spec}, + }} + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil).Once() + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + startedAt := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + runtime.EXPECT().InspectContainer(mock.Anything, "c-1"). + Return(&domain.Container{ID: "c-1", Status: "exited", StartedAt: startedAt}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-1").Return(nil).Once() + runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "c-1").Return("", false, nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-1", mock.Anything). + Return([]domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 32771, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + state.EXPECT().RegisterBackendBinds(mock.Anything, mock.Anything).Return(nil).Once() + + svc := recoveringService(t, state, runtime, traffic, func(context.Context, string, string) (int, error) { + return 500, nil + }) + require.Error(t, svc.ReconcileRunning(ctx)) + state.AssertNotCalled(t, "SaveActive", mock.Anything, mock.Anything) + assert.Zero(t, traffic.rebuildCount()) + runtime.AssertNotCalled(t, "RestartContainer", mock.Anything, mock.Anything, mock.Anything) +} diff --git a/internal/usecase/deployment/recovery_integration_test.go b/internal/usecase/deployment/recovery_integration_test.go new file mode 100644 index 000000000..d9c42280d --- /dev/null +++ b/internal/usecase/deployment/recovery_integration_test.go @@ -0,0 +1,428 @@ +package deployment_test + +import ( + "context" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/adapters/out/appstate" + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/deployment" +) + +// newTestStore opens a real app-state store in a temporary directory. +func newTestStore(t *testing.T) *appstate.Store { + t.Helper() + store, err := appstate.NewStore(t.TempDir(), zerowrap.Default()) + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, store.Close()) }) + return store +} + +// TestRecoveryInhibition_RefusesInhibitedGenerationAfterReboot proves the +// durable inhibition contract end to end: an old volume-owning container +// that a replacement may already have superseded is never started again, +// even by a completely reconstructed engine reading a reopened store. +// Unrelated generations of the same app stay recoverable. +func TestRecoveryInhibition_RefusesInhibitedGenerationAfterReboot(t *testing.T) { + ctx := context.Background() + dir := t.TempDir() + + store, err := appstate.NewStore(dir, zerowrap.Default()) + require.NoError(t, err) + require.NoError(t, store.SaveIntent(ctx, domain.AppStopIntent{App: "blog"})) + require.NoError(t, store.SaveActive(ctx, domain.AppActive{ + App: "blog", + Services: map[string]domain.AppEffectiveService{ + "web": { + Container: "c-old", EffectiveRevision: "rev-1", Spec: headlessSpec(), + BackendBinds: map[int]int{8080: 32770}, + }, + "worker": { + Container: "c-worker", EffectiveRevision: "rev-1", Spec: headlessSpec(), + }, + }, + })) + require.NoError(t, store.SaveRecoveryInhibition(ctx, domain.AppRecoveryInhibition{ + App: "blog", Service: "web", ContainerID: "c-old", + Reason: domain.AppInhibitReplacementPending, Operation: "op-1", + })) + require.NoError(t, store.Close()) + + // Reboot: a fresh store handle and a fresh engine, no in-memory state. + reopened, err := appstate.NewStore(dir, zerowrap.Default()) + require.NoError(t, err) + t.Cleanup(func() { require.NoError(t, reopened.Close()) }) + inhibitions, err := reopened.LoadRecoveryInhibitions(ctx, "blog") + require.NoError(t, err) + require.Len(t, inhibitions, 1) + assert.Equal(t, "c-old", inhibitions[0].ContainerID) + + runtime := outmocks.NewMockContainerRuntime(t) + runtime.EXPECT().InspectContainer(mock.Anything, "c-worker"). + Return(&domain.Container{ID: "c-worker", Status: "running", StartedAt: time.Now().UTC()}, nil).Twice() + runtime.EXPECT().GetContainerHealthStatus(mock.Anything, "c-worker").Return("", false, nil).Once() + + svc := deployment.NewService(deployment.Deps{ + State: reopened, Runtime: runtime, Traffic: &recordingTraffic{}, + }, zerowrap.Default()) + + err = svc.ReconcileRunning(ctx) + require.Error(t, err) + assert.Contains(t, err.Error(), "recovery inhibited") + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, "c-old") + runtime.AssertNotCalled(t, "RestartContainer", mock.Anything, "c-old", mock.Anything) + runtime.AssertCalled(t, "InspectContainer", mock.Anything, "c-worker") + runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RemoveVolume", mock.Anything, mock.Anything, mock.Anything) +} + +// seedDesired writes one accepted desired revision through the real +// staged/committed/materialized apply protocol. +func seedDesired(t *testing.T, ctx context.Context, store *appstate.Store, rev domain.AppDesiredRevision) { + t.Helper() + intentID := "intent-1" + require.NoError(t, store.StageApply(ctx, domain.AppApplyIntent{ + Intent: intentID, App: rev.App, Revision: rev.Revision, + SourceSHA256: "0f1e2d", Spec: rev.Spec, CreatedAt: time.Now().UTC(), + })) + require.NoError(t, store.CommitApply(ctx, rev.App, intentID)) + require.NoError(t, store.MaterializeApply(ctx, rev.App, intentID)) +} + +// TestVolumeReplacementFailureLeavesOldGenerationInhibited proves the +// marker is written BEFORE a volume-owning replacement can write and is +// not cleared when that replacement fails: the old generation must never +// be restarted on top of possibly-newer data. +func TestVolumeReplacementFailureLeavesOldGenerationInhibited(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + runtime := outmocks.NewMockContainerRuntime(t) + images := outmocks.NewMockImageResolver(t) + secrets := outmocks.NewMockSecretProvider(t) + + spec := domain.AppService{ + Name: "web", + Image: "registry.example.com/blog/web:1.4.2", + Volumes: []domain.AppVolume{{Name: "data", Path: "/data"}}, + Secrets: map[string]string{}, + Readiness: domain.AppReadiness{ + Type: domain.AppReadinessHTTP, Path: "/healthz", Timeout: 50 * time.Millisecond, + }, + } + rev := testRevision("blog", spec) + rev.Spec.Env = map[string]string{} + + require.NoError(t, store.SaveIntent(ctx, domain.AppStopIntent{App: "blog"})) + require.NoError(t, store.SaveActive(ctx, domain.AppActive{ + App: "blog", + Services: map[string]domain.AppEffectiveService{ + "web": { + Container: "c-old", EffectiveRevision: "rev-0", Spec: headlessSpec(), + BackendBinds: map[int]int{9000: 19000}, + }, + }, + })) + seedDesired(t, ctx, store, rev) + + images.EXPECT().ResolveDigest(mock.Anything, spec.Image).Return("sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", nil).Once() + runtime.EXPECT().InspectImageVolumes(mock.Anything, spec.Image).Return(nil, nil).Once() + runtime.EXPECT().VolumeExists(mock.Anything, mock.Anything).Return(false, nil).Once() + // The superseded writer cannot be removed: the replacement must abort + // rather than risk overlapping writers, and the inhibition written + // before the stop attempt stays in place. + runtime.EXPECT().StopContainer(mock.Anything, "c-old", mock.Anything).Return(nil).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-old", false).Return(assert.AnError).Once() + + svc := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime, Images: images, Secrets: secrets, + ImagePolicy: domain.ImageSourcePolicy{AllowedRegistries: []string{"registry.example.com"}}, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 500, nil }, + func(context.Context, string) error { return assert.AnError }, + )) + + _, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + require.Error(t, err) + + inhibitions, err := store.LoadRecoveryInhibitions(ctx, "blog") + require.NoError(t, err) + require.Len(t, inhibitions, 1) + assert.Equal(t, "c-old", inhibitions[0].ContainerID) + assert.Equal(t, domain.AppInhibitReplacementPending, inhibitions[0].Reason) + + // A later boot must refuse to revive the inhibited generation, and a + // rebuilt engine must reach the same conclusion. + runtime2 := outmocks.NewMockContainerRuntime(t) + rebooted := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime2, Images: images, Secrets: secrets, + }, zerowrap.Default()) + bootErr := rebooted.ReconcileBoot(ctx) + require.Error(t, bootErr) + assert.Contains(t, bootErr.Error(), "recovery inhibited") + runtime2.AssertNotCalled(t, "StartContainer", mock.Anything, "c-old") + runtime2.AssertNotCalled(t, "RestartContainer", mock.Anything, "c-old", mock.Anything) + runtime2.AssertNotCalled(t, "RemoveVolume", mock.Anything, mock.Anything, mock.Anything) +} + +// TestRemove_InhibitsBeforeRuntimeWithdrawal proves remove is durably +// fail-closed: stopped intent and an inhibition per removed container are +// persisted before any runtime effect, and volumes are never deleted. +func TestRemove_InhibitsBeforeRuntimeWithdrawal(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: headlessSpec()}, + }} + state.EXPECT().Recover(mock.Anything).Return(nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil).Once() + + var order []string + state.EXPECT().SaveIntent(mock.Anything, mock.MatchedBy(func(intent domain.AppStopIntent) bool { + return intent.Stopped + })).RunAndReturn(func(context.Context, domain.AppStopIntent) error { + order = append(order, "intent-stopped") + return nil + }).Once() + state.EXPECT().SaveRecoveryInhibition(mock.Anything, mock.MatchedBy(func(inhibition domain.AppRecoveryInhibition) bool { + return inhibition.App == "blog" && inhibition.Service == "web" && inhibition.ContainerID == "c-1" + })).RunAndReturn(func(context.Context, domain.AppRecoveryInhibition) error { + order = append(order, "inhibited") + return nil + }).Once() + runtime.EXPECT().StopContainer(mock.Anything, "c-1", mock.Anything).RunAndReturn(func(context.Context, string, time.Duration) error { + order = append(order, "stopped") + return nil + }).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-1", false).RunAndReturn(func(context.Context, string, bool) error { + order = append(order, "removed") + return nil + }).Once() + // Confirmed disappearance releases the container's backend claims and + // clears its recovery inhibition before the incarnation is retired. + state.EXPECT().ReleaseBackendBinds(mock.Anything, "blog", "c-1").Return(nil).Once() + state.EXPECT().ClearRecoveryInhibition(mock.Anything, "blog", "web", "c-1").Return(nil).Once() + // Ownership is read while it is still live; the name-only record owns + // no network, so nothing is reclaimed. + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog"}, nil).Once() + // The incarnation is retired atomically after the workload is gone. + state.EXPECT().RetireApp(mock.Anything, "blog").RunAndReturn(func(context.Context, string) error { + order = append(order, "retired") + return nil + }).Once() + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) + + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()) + result, err := svc.Remove(ctx, "blog", "") + require.NoError(t, err) + require.NotNil(t, result) + assert.Equal(t, []string{"intent-stopped", "inhibited", "stopped", "removed", "retired"}, order) + runtime.AssertNotCalled(t, "RemoveVolume", mock.Anything, mock.Anything, mock.Anything) +} + +// TestBootStoppedConvergenceStopsRevivedContainer proves durable stopped +// intent wins over native restart policy: a container revived while the +// daemon was away is stopped again, without removal, and is never started. +func TestBootStoppedConvergenceStopsRevivedContainer(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: headlessSpec()}, + }} + state.EXPECT().Recover(mock.Anything).Return(nil).Once() + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog"). + Return(domain.AppStopIntent{App: "blog", Stopped: true}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil).Once() + runtime.EXPECT().InspectContainer(mock.Anything, "c-1"). + Return(&domain.Container{ID: "c-1", Status: "running"}, nil).Once() + runtime.EXPECT().StopContainer(mock.Anything, "c-1", mock.Anything).Return(nil).Once() + + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()) + require.NoError(t, svc.ReconcileBoot(ctx)) + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RestartContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, mock.Anything, mock.Anything) +} + +// TestBootRefusesInhibitedGenerationBeforeAnyRuntimeEffect proves boot +// recovery checks the durable marker before touching the workload. +func TestBootRefusesInhibitedGenerationBeforeAnyRuntimeEffect(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-old", EffectiveRevision: "rev-1", Spec: headlessSpec()}, + }} + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil).Once() + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) + state.EXPECT().SaveIntent(mock.Anything, mock.Anything).Return(nil) + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog"). + Return([]domain.AppRecoveryInhibition{{App: "blog", Service: "web", ContainerID: "c-old"}}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()) + err := svc.ReconcileBoot(ctx) + require.Error(t, err) + assert.Contains(t, err.Error(), "recovery inhibited") + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RestartContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) +} + +// TestWorkloadMutationsSerializePerApp proves a periodic pass cannot +// interleave with an in-flight user mutation of the same app: the pass +// skips the busy app instead of waiting for it. +func TestWorkloadMutationsSerializePerApp(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + + entered := make(chan struct{}) + release := make(chan struct{}) + state.EXPECT().Recover(mock.Anything).RunAndReturn(func(context.Context) error { + close(entered) + <-release + return assert.AnError + }).Once() + // Only the app list may be read: the pass must not touch intent or + // ACTIVE while the mutation holds the app lock. + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Once() + + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()) + + done := make(chan error, 1) + go func() { + _, err := svc.Stop(ctx, "blog", "") + done <- err + }() + <-entered + + // The mutation holds the app lock. The periodic pass must skip the + // app instead of waiting: LoadIntent has no expectation, so reaching + // it would fail the test. + require.NoError(t, svc.ReconcileRunning(ctx)) + state.AssertNotCalled(t, "LoadIntent", mock.Anything, mock.Anything) + + close(release) + require.Error(t, <-done) +} + +// TestRestart_RefusesInhibitedGeneration proves an explicit operator +// restart cannot revive a generation whose recovery is durably inhibited: +// a replacement may already have written to the volume it still owns. +func TestRestart_RefusesInhibitedGeneration(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-old", EffectiveRevision: "rev-1", Spec: headlessSpec()}, + }} + state.EXPECT().Recover(mock.Anything).Return(nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil).Once() + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return([]domain.AppRecoveryInhibition{{ + App: "blog", Service: "web", ContainerID: "c-old", Reason: domain.AppInhibitReplacementPending, + }}, nil).Once() + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) + + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()) + result, err := svc.Restart(ctx, "blog", "", "") + require.ErrorIs(t, err, domain.ErrAppStateConflict) + require.NotNil(t, result) + assert.Equal(t, "failed", result.Services["web"].Result) + assert.Contains(t, result.Services["web"].Error, "recovery inhibited") + runtime.AssertNotCalled(t, "RestartContainer", mock.Anything, mock.Anything, mock.Anything) +} + +// TestRemove_KeepsStateWhenAContainerCannotBeRemoved proves remove never +// loses track of a survivor: ACTIVE, stopped intent, and the inhibition +// stay in place so the periodic pass keeps converging it. +func TestRemove_KeepsStateWhenAContainerCannotBeRemoved(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: headlessSpec()}, + }} + state.EXPECT().Recover(mock.Anything).Return(nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil).Once() + state.EXPECT().SaveIntent(mock.Anything, mock.MatchedBy(func(intent domain.AppStopIntent) bool { + return intent.Stopped + })).Return(nil).Once() + state.EXPECT().SaveRecoveryInhibition(mock.Anything, mock.Anything).Return(nil).Once() + // The claim journal is written before the runtime effects and is + // rewritten terminally when the removal cannot complete: a repeat of + // the same key replays the failure instead of reading a stuck claim. + state.EXPECT().SaveOperation(mock.Anything, mock.MatchedBy(func(op domain.AppOperation) bool { + return op.Op != "" && op.Kind == "remove" && !op.Terminal() + })).Return(nil).Once() + state.EXPECT().SaveOperation(mock.Anything, mock.MatchedBy(func(op domain.AppOperation) bool { + return op.Op != "" && op.Outcome == domain.AppOutcomeFailed && op.Terminal() + })).Return(nil).Once() + runtime.EXPECT().StopContainer(mock.Anything, "c-1", mock.Anything).Return(assert.AnError).Once() + + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()) + _, err := svc.Remove(ctx, "blog", "") + require.Error(t, err) + assert.Contains(t, err.Error(), "not confirmed gone") + // ACTIVE and the stopped intent were never discarded: the monitor can + // still converge the survivor. The survivor's claims and inhibition + // stay in place too: only a confirmed disappearance releases them. + state.AssertNotCalled(t, "SaveActive", mock.Anything, mock.Anything) + state.AssertNotCalled(t, "RetireApp", mock.Anything, mock.Anything) + state.AssertNotCalled(t, "ReleaseBackendBinds", mock.Anything, mock.Anything, mock.Anything) + state.AssertNotCalled(t, "ClearRecoveryInhibition", mock.Anything, mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RemoveVolume", mock.Anything, mock.Anything, mock.Anything) +} + +// TestRemovalOfMissingContainerIsIdempotent proves a container the +// runtime already dropped is a successful removal, not a failure. +func TestRemovalOfMissingContainerIsIdempotent(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: headlessSpec()}, + }} + state.EXPECT().Recover(mock.Anything).Return(nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil).Once() + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) + state.EXPECT().SaveIntent(mock.Anything, mock.Anything).Return(nil) + state.EXPECT().SaveRecoveryInhibition(mock.Anything, mock.Anything).Return(nil).Once() + runtime.EXPECT().StopContainer(mock.Anything, "c-1", mock.Anything).Return(domain.ErrContainerNotFound).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-1", false).Return(domain.ErrContainerNotFound).Once() + // Not-found is confirmation: the stale claims and the inhibition of + // the disappeared container are released. + state.EXPECT().ReleaseBackendBinds(mock.Anything, "blog", "c-1").Return(nil).Once() + state.EXPECT().ClearRecoveryInhibition(mock.Anything, "blog", "web", "c-1").Return(nil).Once() + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog"}, nil).Once() + state.EXPECT().RetireApp(mock.Anything, "blog").Return(nil).Once() + + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()) + result, err := svc.Remove(ctx, "blog", "") + require.NoError(t, err) + require.NotNil(t, result) +} diff --git a/internal/usecase/deployment/removals.go b/internal/usecase/deployment/removals.go new file mode 100644 index 000000000..86e94af23 --- /dev/null +++ b/internal/usecase/deployment/removals.go @@ -0,0 +1,122 @@ +package deployment + +import ( + "context" + "fmt" + "sort" + + "github.com/bnema/zerowrap" + + "github.com/bnema/gordon/internal/domain" +) + +// reconcileRemovalsForDeploy journals and applies the removal of services +// the new revision no longer declares. A failure marks the operation +// failed and aborts before any new state is published. +func (s *Service) reconcileRemovalsForDeploy(ctx context.Context, app string, active domain.AppActive, pinned []pinnedService, op *domain.AppOperation, result *DeployResult, log zerowrap.Logger) error { + removalSteps, removed, warnings, err := s.reconcileRemovedServices(ctx, app, active, pinned) + op.Warnings = append(op.Warnings, journalWarnings(warnings)...) + result.CleanupWarnings = append(result.CleanupWarnings, warnings...) + if err != nil { + op.Steps = append(op.Steps, removalSteps...) + op.Outcome = domain.AppOutcomeFailed + if saveErr := s.deps.State.SaveOperation(ctx, *op); saveErr != nil { + log.Warn().Err(saveErr).Msg("deployment: failed to record removal failure") + } + return err + } + if len(removalSteps) == 0 { + return nil + } + op.Steps = append(op.Steps, removalSteps...) + result.Removed = removed + if saveErr := s.deps.State.SaveOperation(ctx, *op); saveErr != nil { + log.Warn().Err(saveErr).Msg("deployment: failed to record service removals") + } + return nil +} + +// reconcileRemovedServices withdraws and retires every ACTIVE service the +// new revision no longer declares. Traffic is withdrawn first (fail +// closed), then the exact container is inhibited, stopped, and removed +// with its data retained, its backend claims released, and its ACTIVE +// entry deleted and persisted before its recovery inhibition is cleared. A failure leaves the remaining services untouched and +// stops the deploy before any new state is published. +func (s *Service) reconcileRemovedServices(ctx context.Context, app string, active domain.AppActive, pinned []pinnedService) ([]domain.AppOperationStep, []string, []CleanupWarning, error) { + desired := make(map[string]struct{}, len(pinned)) + for _, p := range pinned { + desired[p.name] = struct{}{} + } + var names []string + for name := range active.Services { + if _, ok := desired[name]; !ok { + names = append(names, name) + } + } + sort.Strings(names) + + steps := make([]domain.AppOperationStep, 0, len(names)) + var cleanupWarnings []CleanupWarning + for _, name := range names { + eff := active.Services[name] + step, warnings, err := s.retireRemovedService(ctx, app, name, eff) + steps = append(steps, step) + cleanupWarnings = append(cleanupWarnings, warnings...) + if err != nil { + return steps, names, cleanupWarnings, err + } + // Persist the removal before dropping the recovery inhibition: an + // ACTIVE record that still lists the service without its + // inhibition would let recovery recreate a removed service. + delete(active.Services, name) + if err := s.deps.State.SaveActive(ctx, active); err != nil { + return steps, names, cleanupWarnings, fmt.Errorf("deployment: persist service removal %q: %w", name, err) + } + if err := s.clearRecoveryInhibition(ctx, app, name, eff.Container); err != nil { + cleanupWarnings = append(cleanupWarnings, CleanupWarning{ + Service: name, Leftover: eff.Container, Detail: "clear recovery inhibition: " + err.Error(), + }) + } + } + if len(names) == 0 { + return nil, nil, nil, nil + } + return steps, names, cleanupWarnings, nil +} + +// retireRemovedService withdraws one removed service and removes its exact +// container, releasing its claims only after withdrawal succeeded. +func (s *Service) retireRemovedService(ctx context.Context, app, name string, eff domain.AppEffectiveService) (domain.AppOperationStep, []CleanupWarning, error) { + step := domain.AppOperationStep{ + ID: "service." + name + ".remove", + State: domain.AppStepPending, + Service: name, + Before: eff.Container, + } + fail := func(err error) (domain.AppOperationStep, []CleanupWarning, error) { + step.State = domain.AppStepFailed + step.Error = err.Error() + return step, nil, err + } + if err := s.withdrawForRecovery(ctx, app, name); err != nil { + return fail(err) + } + if eff.Container != "" { + if err := s.inhibitRecovery(ctx, app, name, eff.Container, "removed", ""); err != nil { + return fail(err) + } + // The inhibition stays until the caller persisted the removal. + retired := s.retireContainer(ctx, app, retireOptions{ + Service: name, Grace: serviceStopGrace(eff), + }, eff.Container) + if !retired.Gone { + return fail(fmt.Errorf("deployment: remove %s/%s container %s: %s", app, name, eff.Container, cleanupDetail(retired))) + } + step.State = domain.AppStepSucceeded + step.After = "" + return step, retired.Warnings, nil + } + step.State = domain.AppStepSucceeded + step.After = "" + return step, nil, nil +} diff --git a/internal/usecase/deployment/restart_inplace_test.go b/internal/usecase/deployment/restart_inplace_test.go new file mode 100644 index 000000000..2355a391a --- /dev/null +++ b/internal/usecase/deployment/restart_inplace_test.go @@ -0,0 +1,163 @@ +package deployment_test + +import ( + "context" + "errors" + "sync" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/deployment" +) + +// restartTrafficRecorder records the traffic boundary calls of a restart. +type restartTrafficRecorder struct { + mu sync.Mutex + events *[]string + failRebuild bool +} + +func (r *restartTrafficRecorder) record(event string) { + r.mu.Lock() + defer r.mu.Unlock() + if r.events != nil { + *r.events = append(*r.events, event) + } +} +func (r *restartTrafficRecorder) RebuildTraffic(context.Context) error { + r.record("traffic") + if r.failRebuild { + return errors.New("traffic rebuild failed") + } + return nil +} +func (r *restartTrafficRecorder) WithdrawService(context.Context, string, string) error { + r.record("withdraw") + return nil +} +func (r *restartTrafficRecorder) WithdrawServiceState(context.Context, string, string) error { + r.record("withdraw-state") + return nil +} + +const restartTestDigest = "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + +func restartActive(t *testing.T, spec domain.AppService) domain.AppActive { + t.Helper() + return domain.AppActive{App: "blog", ConvergedRevision: "rev-1", Converged: true, Services: map[string]domain.AppEffectiveService{"web": {EffectiveRevision: "rev-1", ActivatedBy: "op-before", ActivatedAt: time.Now().Add(-time.Hour).UTC(), Image: spec.Image, Digest: restartTestDigest, Container: "c-old", Spec: spec, BackendBinds: map[int]int{8080: 32770}}}} +} + +// TestRestart_InPlaceWithdrawsRestartsVerifiesAndRepublishes proves the +// restart contract: the service is withdrawn from traffic, the SAME pinned +// container is restarted, readiness is verified against its refreshed +// loopback bind, and traffic is republished. No second container is created +// and no container is stopped or removed. +func TestRestart_InPlaceWithdrawsRestartsVerifiesAndRepublishes(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + spec := webService() + spec.Secrets = map[string]string{} + spec.StopGrace = 10 * time.Millisecond + // No declared readiness probe: the implicit HTTP check still applies to a + // stateless public HTTP service on restart, exactly as it does on deploy. + spec.Readiness = domain.AppReadiness{Timeout: time.Second} + active := restartActive(t, spec) + + var order []string + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + runtime.EXPECT().RestartContainer(mock.Anything, "c-old", mock.Anything).RunAndReturn( + func(context.Context, string, time.Duration) error { + recordEvent(&order, "restart") + return nil + }).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-old", mock.Anything).Return( + []domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 32771, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + state.EXPECT().RegisterBackendBinds(mock.Anything, mock.Anything).Return(nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + var saved domain.AppActive + state.EXPECT().SaveActive(mock.Anything, mock.Anything).RunAndReturn( + func(_ context.Context, refreshed domain.AppActive) error { + saved = refreshed + recordEvent(&order, "persist-binds") + return nil + }).Once() + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) + + var probedURL string + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, + Traffic: &restartTrafficRecorder{events: &order}, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(_ context.Context, url, _ string) (int, error) { + probedURL = url + recordEvent(&order, "ready") + return 200, nil + }, + func(context.Context, string) error { return nil }, + )) + + result, err := svc.Restart(ctx, "blog", "web", "") + require.NoError(t, err) + require.NotNil(t, result) + assert.Equal(t, "deployed", result.Services["web"].Result) + assert.Equal(t, "c-old", result.Services["web"].After, "the same pinned container is restarted in place") + assert.Contains(t, probedURL, "127.0.0.1:32771", "readiness dials the refreshed loopback bind") + assert.Equal(t, []string{"withdraw", "restart", "persist-binds", "ready", "traffic"}, order) + assert.Equal(t, map[int]int{8080: 32771}, saved.Services["web"].BackendBinds) + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "StopContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, mock.Anything, mock.Anything) +} + +// TestRestart_ReadinessFailureLeavesTheRestartedContainerWithdrawn proves a +// restart is never served unverified: the failure leaves the same container +// in place but withdraws its recorded binds, and the graph is republished +// fail-closed. Nothing is created, stopped, or removed. +func TestRestart_ReadinessFailureLeavesTheRestartedContainerWithdrawn(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + spec := webService() + spec.Secrets = map[string]string{} + spec.StopGrace = 10 * time.Millisecond + spec.Readiness = domain.AppReadiness{Timeout: 50 * time.Millisecond} + active := restartActive(t, spec) + + var order []string + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + runtime.EXPECT().RestartContainer(mock.Anything, "c-old", mock.Anything).Return(nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-old", mock.Anything).Return( + []domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 32771, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + state.EXPECT().RegisterBackendBinds(mock.Anything, mock.Anything).Return(nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().SaveActive(mock.Anything, mock.Anything).Return(nil).Once() + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) + + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, + Traffic: &restartTrafficRecorder{events: &order}, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 503, nil }, + func(context.Context, string) error { return assert.AnError }, + )) + + result, err := svc.Restart(ctx, "blog", "web", "") + require.Error(t, err) + require.NotNil(t, result) + assert.Equal(t, "failed", result.Services["web"].Result) + assert.Contains(t, result.Services["web"].Error, "readiness") + assert.Contains(t, order, "withdraw-state", "the unverified generation is withdrawn from state") + assert.Contains(t, order, "traffic", "the graph is republished fail-closed") + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "StopContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, mock.Anything, mock.Anything) +} diff --git a/internal/usecase/deployment/retirement.go b/internal/usecase/deployment/retirement.go new file mode 100644 index 000000000..91fc572d9 --- /dev/null +++ b/internal/usecase/deployment/retirement.go @@ -0,0 +1,168 @@ +package deployment + +import ( + "context" + "errors" + "sort" + "time" + + "github.com/bnema/gordon/internal/domain" +) + +// stopGraceOrDefault returns the grace a runtime call must use for one +// container. A zero declared grace means the app default: an app path +// never hands the runtime an immediate kill. +func stopGraceOrDefault(grace time.Duration) time.Duration { + if grace <= 0 { + return domain.AppDefaultStopGrace + } + return grace +} + +// serviceStopGrace returns the effective stop grace of one service. +func serviceStopGrace(eff domain.AppEffectiveService) time.Duration { + return stopGraceOrDefault(eff.Spec.StopGrace) +} + +// activeStopGrace returns the effective stop grace recorded for one +// service of an ACTIVE record, or the app default when the record has no +// such service. Retiring a replaced container uses the grace of the +// generation being stopped, never the grace of its replacement. +func activeStopGrace(active domain.AppActive, service string) time.Duration { + return stopGraceOrDefault(active.Services[service].Spec.StopGrace) +} + +// retireOptions control one exact-container retirement. +type retireOptions struct { + // Service names the app service for warnings. + Service string + // Grace is the effective stop grace of the container being retired. + Grace time.Duration + // Force removes a container that was never published and holds no + // committed state, without waiting for a grace period. + Force bool + // ClearInhibition drops the generation's recovery inhibition once + // the container is confirmed gone. It is only safe when nothing can + // recreate this generation: an operator stop or a removal. A + // replacement keeps its inhibition until the new generation is + // published, because recovery must not recreate the replaced writer + // from the still-active old revision. + ClearInhibition bool +} + +// containerRetirement is the outcome of one exact-container retirement. +type containerRetirement struct { + // ContainerID is the container the retirement targeted. + ContainerID string + // Gone is true only when the runtime confirmed the container no + // longer exists. Not-found counts as confirmation. + Gone bool + // Warnings are bounded, log-free leftovers for operator action. + Warnings []CleanupWarning +} + +// retireContainer is the single retirement path for one exact container: +// stop it with the effective grace (or force-remove a never-published +// candidate), confirm the exact ID is gone, and only then release its +// backend claims and clear its recovery inhibition. A container that is +// not confirmed gone keeps its claims and its inhibition, so no later +// generation can dial a recycled loopback port and no recovery pass can +// revive a half-removed generation. +func (s *Service) retireContainer(ctx context.Context, app string, opts retireOptions, containerID string) containerRetirement { + result := containerRetirement{ContainerID: containerID} + if containerID == "" { + result.Gone = true + return result + } + if err := s.stopForRetirement(ctx, opts, containerID); err != nil { + result.Warnings = append(result.Warnings, CleanupWarning{ + Service: opts.Service, Leftover: containerID, Detail: "stop: " + err.Error(), + }) + return result + } + if err := s.deps.Runtime.RemoveContainer(ctx, containerID, opts.Force); err != nil && !errors.Is(err, domain.ErrContainerNotFound) { + result.Warnings = append(result.Warnings, CleanupWarning{ + Service: opts.Service, Leftover: containerID, Detail: "remove: " + err.Error(), + }) + return result + } + result.Gone = true + if err := s.deps.State.ReleaseBackendBinds(ctx, app, containerID); err != nil { + result.Warnings = append(result.Warnings, CleanupWarning{ + Service: opts.Service, Leftover: containerID, Detail: "release backend claims: " + err.Error(), + }) + } + if opts.ClearInhibition { + // Confirmed gone: the marker has nothing left to protect, and a + // stale record would only confuse the next recreation. + if err := s.clearRecoveryInhibition(ctx, app, opts.Service, containerID); err != nil { + result.Warnings = append(result.Warnings, CleanupWarning{ + Service: opts.Service, Leftover: containerID, Detail: "clear recovery inhibition: " + err.Error(), + }) + } + } + return result +} + +// stopForRetirement stops one container. A forced retirement removes +// without a grace period: only a candidate that was never published and +// holds no committed state may be retired that way. +func (s *Service) stopForRetirement(ctx context.Context, opts retireOptions, containerID string) error { + if opts.Force { + return nil + } + if err := s.deps.Runtime.StopContainer(ctx, containerID, stopGraceOrDefault(opts.Grace)); err != nil && !errors.Is(err, domain.ErrContainerNotFound) { + return err + } + return nil +} + +// retireCandidate removes a candidate that was never published and +// releases its backend claims. The candidate holds no committed state, so +// it is force-removed without a grace period. Cleanup failures are +// returned as bounded warnings: a candidate that could not be removed +// must not hide the failure the caller is already reporting. +func (s *Service) retireCandidate(ctx context.Context, app, service, containerID string) []CleanupWarning { + retired := s.retireContainer(ctx, app, retireOptions{Service: service, Force: true}, containerID) + return retired.Warnings +} + +// cleanupDetail renders the first bounded warning of a failed retirement +// for an error message. Journal and log content never travel here. +func cleanupDetail(retired containerRetirement) string { + if len(retired.Warnings) == 0 { + return "container not confirmed gone" + } + return retired.Warnings[0].Detail +} + +// collectCleanupWarnings gathers one operation's per-service leftovers in +// stable service order. It is the single source of the operation-level +// warning list. +func collectCleanupWarnings(results map[string]ServiceResult) []CleanupWarning { + names := make([]string, 0, len(results)) + for name := range results { + names = append(names, name) + } + sort.Strings(names) + var warnings []CleanupWarning + for _, name := range names { + warnings = append(warnings, results[name].CleanupWarnings...) + } + return warnings +} + +// journalWarnings maps leftovers into the bounded journal shape. The +// service name travels with each warning, so a leftover is always +// attributable without parsing its text. +func journalWarnings(leftovers []CleanupWarning) []domain.AppOperationWarning { + var warnings []domain.AppOperationWarning + for _, warning := range leftovers { + warnings = append(warnings, domain.AppOperationWarning{ + Service: warning.Service, + Leftover: warning.Leftover, + Detail: warning.Detail, + }) + } + return warnings +} diff --git a/internal/usecase/deployment/retirement_internal_test.go b/internal/usecase/deployment/retirement_internal_test.go new file mode 100644 index 000000000..943db30b6 --- /dev/null +++ b/internal/usecase/deployment/retirement_internal_test.go @@ -0,0 +1,70 @@ +package deployment + +import ( + "context" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" +) + +// TestRetireContainer_ForceSkipsTheGracePeriod proves a never-published +// candidate is removed without waiting for a grace period. +func TestRetireContainer_ForceSkipsTheGracePeriod(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + runtime.EXPECT().RemoveContainer(mock.Anything, "c-candidate", true).Return(nil).Once() + state.EXPECT().ReleaseBackendBinds(mock.Anything, "blog", "c-candidate").Return(nil).Once() + + svc := NewService(Deps{State: state, Runtime: runtime}, zerowrap.Default()) + retired := svc.retireContainer(ctx, "blog", retireOptions{Service: "web", Force: true}, "c-candidate") + assert.True(t, retired.Gone) + assert.Empty(t, retired.Warnings) + runtime.AssertNotCalled(t, "StopContainer", mock.Anything, mock.Anything, mock.Anything) +} + +// TestRetireContainer_UnconfirmedRemovalKeepsClaims proves a container the +// runtime would not remove keeps its backend claims and its recovery +// inhibition, and reports a bounded warning instead of a success. +func TestRetireContainer_UnconfirmedRemovalKeepsClaims(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + runtime.EXPECT().StopContainer(mock.Anything, "c-1", 5*time.Second).Return(nil).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-1", false).Return(assert.AnError).Once() + + svc := NewService(Deps{State: state, Runtime: runtime}, zerowrap.Default()) + retired := svc.retireContainer(ctx, "blog", retireOptions{ + Service: "web", Grace: 5 * time.Second, ClearInhibition: true, + }, "c-1") + + assert.False(t, retired.Gone) + require.Len(t, retired.Warnings, 1) + assert.Equal(t, "c-1", retired.Warnings[0].Leftover) + assert.Contains(t, retired.Warnings[0].Detail, "remove") + state.AssertNotCalled(t, "ReleaseBackendBinds", mock.Anything, mock.Anything, mock.Anything) + state.AssertNotCalled(t, "ClearRecoveryInhibition", mock.Anything, mock.Anything, mock.Anything, mock.Anything) +} + +// TestRetireContainer_ZeroGraceUsesTheAppDefault proves a zero grace is +// never an immediate kill on an app path. +func TestRetireContainer_ZeroGraceUsesTheAppDefault(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + runtime.EXPECT().StopContainer(mock.Anything, "c-1", domain.AppDefaultStopGrace).Return(nil).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-1", false).Return(nil).Once() + state.EXPECT().ReleaseBackendBinds(mock.Anything, "blog", "c-1").Return(nil).Once() + + svc := NewService(Deps{State: state, Runtime: runtime}, zerowrap.Default()) + retired := svc.retireContainer(ctx, "blog", retireOptions{Service: "web"}, "c-1") + assert.True(t, retired.Gone) + assert.Empty(t, retired.Warnings) +} diff --git a/internal/usecase/deployment/retirement_test.go b/internal/usecase/deployment/retirement_test.go new file mode 100644 index 000000000..a1e3da5a3 --- /dev/null +++ b/internal/usecase/deployment/retirement_test.go @@ -0,0 +1,204 @@ +package deployment_test + +import ( + "context" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/deployment" +) + +// graceSpec is a headless service carrying one explicit stop grace. +func graceSpec(grace time.Duration) domain.AppService { + spec := headlessSpec() + spec.StopGrace = grace + return spec +} + +func activeWithGrace(container string, grace time.Duration) domain.AppActive { + return domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: container, EffectiveRevision: "rev-1", Spec: graceSpec(grace)}, + }} +} + +// TestLifecycleVerbsPassTheEffectiveStopGrace proves every runtime stop +// uses the effective service grace instead of a fixed adapter timeout. +func TestLifecycleVerbsPassTheEffectiveStopGrace(t *testing.T) { + const grace = 25 * time.Second + + t.Run("stop", func(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + active := activeWithGrace("c-1", grace) + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().SaveIntent(mock.Anything, mock.Anything).Return(nil).Once() + runtime.EXPECT().StopContainer(mock.Anything, "c-1", grace).Return(nil).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-1", false).Return(nil).Once() + state.EXPECT().ReleaseBackendBinds(mock.Anything, "blog", "c-1").Return(nil).Once() + state.EXPECT().ClearRecoveryInhibition(mock.Anything, "blog", "web", "c-1").Return(nil).Once() + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) + + svc := deployment.NewService(deployment.Deps{State: state, Runtime: runtime}, zerowrap.Default()) + _, err := svc.Stop(ctx, "blog", "") + require.NoError(t, err) + }) + + t.Run("restart", func(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + active := activeWithGrace("c-1", grace) + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "blog").Return(nil, nil).Once() + runtime.EXPECT().RestartContainer(mock.Anything, "c-1", grace).Return(nil).Once() + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) + + svc := deployment.NewService(deployment.Deps{State: state, Runtime: runtime}, zerowrap.Default()) + _, err := svc.Restart(ctx, "blog", "web", "") + require.NoError(t, err) + }) + + t.Run("remove", func(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + active := activeWithGrace("c-1", grace) + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().SaveIntent(mock.Anything, mock.Anything).Return(nil).Once() + state.EXPECT().SaveRecoveryInhibition(mock.Anything, mock.Anything).Return(nil).Once() + runtime.EXPECT().StopContainer(mock.Anything, "c-1", grace).Return(nil).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-1", false).Return(nil).Once() + state.EXPECT().ReleaseBackendBinds(mock.Anything, "blog", "c-1").Return(nil).Once() + state.EXPECT().ClearRecoveryInhibition(mock.Anything, "blog", "web", "c-1").Return(nil).Once() + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog"}, nil).Once() + state.EXPECT().RetireApp(mock.Anything, "blog").Return(nil).Once() + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) + + svc := deployment.NewService(deployment.Deps{State: state, Runtime: runtime}, zerowrap.Default()) + _, err := svc.Remove(ctx, "blog", "") + require.NoError(t, err) + }) +} + +// TestStopWithoutDeclaredGraceUsesTheAppDefault proves an effective spec +// with no stop_grace still reaches the runtime as the app default instead +// of an immediate kill. +func TestStopWithoutDeclaredGraceUsesTheAppDefault(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + active := activeWithGrace("c-1", 0) + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().SaveIntent(mock.Anything, mock.Anything).Return(nil).Once() + runtime.EXPECT().StopContainer(mock.Anything, "c-1", domain.AppDefaultStopGrace).Return(nil).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-1", false).Return(nil).Once() + state.EXPECT().ReleaseBackendBinds(mock.Anything, "blog", "c-1").Return(nil).Once() + state.EXPECT().ClearRecoveryInhibition(mock.Anything, "blog", "web", "c-1").Return(nil).Once() + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) + + svc := deployment.NewService(deployment.Deps{State: state, Runtime: runtime}, zerowrap.Default()) + _, err := svc.Stop(ctx, "blog", "") + require.NoError(t, err) +} + +// TestRemove_FailedRetirementKeepsClaimsAndActiveState proves a container +// that is not confirmed gone keeps its backend claims, keeps its ACTIVE +// reference, and fails the operation terminally. +func TestRemove_FailedRetirementKeepsClaimsAndActiveState(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + active := activeWithGrace("c-1", time.Second) + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().SaveIntent(mock.Anything, mock.Anything).Return(nil).Once() + state.EXPECT().SaveRecoveryInhibition(mock.Anything, mock.Anything).Return(nil).Once() + runtime.EXPECT().StopContainer(mock.Anything, "c-1", time.Second).Return(nil).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-1", false).Return(assert.AnError).Once() + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) + + svc := deployment.NewService(deployment.Deps{State: state, Runtime: runtime}, zerowrap.Default()) + _, err := svc.Remove(ctx, "blog", "") + require.Error(t, err) + assert.Contains(t, err.Error(), "not confirmed gone") + state.AssertNotCalled(t, "ReleaseBackendBinds", mock.Anything, mock.Anything, mock.Anything) + state.AssertNotCalled(t, "ClearRecoveryInhibition", mock.Anything, mock.Anything, mock.Anything, mock.Anything) + state.AssertNotCalled(t, "SaveActive", mock.Anything, mock.Anything) + state.AssertNotCalled(t, "RetireApp", mock.Anything, mock.Anything) +} + +// TestDeploy_RetirementFailureBlocksReplacement proves the superseded +// generation must be confirmed gone before a replacement starts: a failed +// retirement aborts the service without creating a candidate, and the +// failure is journaled instead of being downgraded to a warning. +func TestDeploy_RetirementFailureBlocksReplacement(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + oldActive := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {EffectiveRevision: "rev-0", Image: "img:0", Container: "c-old", Spec: graceSpec(time.Second)}, + }} + + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + images.EXPECT().ResolveDigest(mock.Anything, rev.Spec.Services[0].Image).Return( + "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", nil).Once() + secrets.EXPECT().GetSecret(mock.Anything, "gordon/apps/app-blog/web/database-url").Return("x", nil) + runtime.EXPECT().InspectImageVolumes(mock.Anything, rev.Spec.Services[0].Image).Return(nil, nil).Once() + state.EXPECT().LoadCheckpoint(mock.Anything).Return(domain.AppStoreCheckpoint{}, nil).Once() + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-blog"}, nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(oldActive, true, nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + // The superseded container is retired with its own effective grace, and + // its removal fails: the replacement must not start. + runtime.EXPECT().StopContainer(mock.Anything, "c-old", time.Second).Return(nil).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-old", false).Return(assert.AnError).Once() + + var saved []domain.AppOperation + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).RunAndReturn( + func(_ context.Context, op domain.AppOperation) error { + saved = append(saved, op) + return nil + }) + + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 200, nil }, + func(context.Context, string) error { return nil }, + )) + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + require.Error(t, err) + require.NotNil(t, result) + assert.Equal(t, "failed", result.Services["web"].Result) + assert.Contains(t, result.Services["web"].Error, "retire superseded container", + "the unconfirmed retirement is the reported failure") + + // No workload is created, started, or published while the superseded + // generation is not confirmed gone, and no claim or inhibition of that + // generation is released. + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "ListNetworks", mock.Anything) + runtime.AssertNotCalled(t, "RemoveVolume", mock.Anything, mock.Anything, mock.Anything) + state.AssertNotCalled(t, "SaveActive", mock.Anything, mock.Anything) + state.AssertNotCalled(t, "ReleaseBackendBinds", mock.Anything, mock.Anything, mock.Anything) + state.AssertNotCalled(t, "ClearRecoveryInhibition", mock.Anything, mock.Anything, mock.Anything, mock.Anything) + + require.NotEmpty(t, saved) + last := saved[len(saved)-1] + require.Len(t, last.Steps, 2) + assert.Equal(t, "preflight", last.Steps[0].ID) + assert.Equal(t, domain.AppStepSucceeded, last.Steps[0].State) + assert.Equal(t, "service.web.replace", last.Steps[1].ID) + assert.Equal(t, domain.AppStepFailed, last.Steps[1].State) + assert.Equal(t, domain.AppOutcomeFailed, last.Outcome) +} diff --git a/internal/usecase/deployment/sequential_replacement_test.go b/internal/usecase/deployment/sequential_replacement_test.go new file mode 100644 index 000000000..1d2ffa6ea --- /dev/null +++ b/internal/usecase/deployment/sequential_replacement_test.go @@ -0,0 +1,745 @@ +package deployment_test + +import ( + "context" + "strings" + "sync" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/deployment" +) + +// orderedTraffic records traffic boundary calls into a shared event log and +// can fail one withdrawal or one graph apply. +type orderedTraffic struct { + mu sync.Mutex + events *[]string + failWithdraw bool + failRebuild bool +} + +func (t *orderedTraffic) record(event string) { + t.mu.Lock() + defer t.mu.Unlock() + if t.events != nil { + *t.events = append(*t.events, event) + } +} + +func (t *orderedTraffic) RebuildTraffic(context.Context) error { + t.record("traffic") + if t.failRebuild { + return assert.AnError + } + return nil +} + +func (t *orderedTraffic) WithdrawService(context.Context, string, string) error { + t.record("withdraw") + if t.failWithdraw { + return assert.AnError + } + return nil +} + +func (t *orderedTraffic) WithdrawServiceState(context.Context, string, string) error { + t.record("withdraw-state") + if t.failWithdraw { + return assert.AnError + } + return nil +} + +// replaceableActive is one ACTIVE record whose web service is superseded. +func replaceableActive() domain.AppActive { + return domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {EffectiveRevision: "rev-0", Image: "img:0", Container: "c-old"}, + }} +} + +// recordEvent appends one step to a shared event log. Test doubles and mock +// hooks share the log, so the sequence a deployment performs is observable. +func recordEvent(events *[]string, event string) { + if events != nil { + *events = append(*events, event) + } +} + +// expectReplacementPreflight wires the preflight calls of one replacement +// without pre-registering an operation-journal write, so a test can observe +// the terminal journal write itself. +func expectReplacementPreflight( + state *outmocks.MockAppState, + runtime *outmocks.MockContainerRuntime, + images *outmocks.MockImageResolver, + secrets *outmocks.MockSecretProvider, + rev domain.AppDesiredRevision, +) { + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + images.EXPECT().ResolveDigest(mock.Anything, rev.Spec.Services[0].Image).Return( + "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", nil) + secrets.EXPECT().GetSecret(mock.Anything, "gordon/apps/app-blog/web/database-url").Return("x", nil) + runtime.EXPECT().InspectImageVolumes(mock.Anything, rev.Spec.Services[0].Image).Return(nil, nil) + state.EXPECT().LoadCheckpoint(mock.Anything).Return(domain.AppStoreCheckpoint{}, nil) + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-blog"}, nil) +} + +// expectSequentialCreate wires one replacement's candidate creation and loopback +// publication. It records "create-new" and "start-new" in order. +func expectSequentialCreate( + state *outmocks.MockAppState, + runtime *outmocks.MockContainerRuntime, + events *[]string, + containerID string, + hostPort int, +) { + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).RunAndReturn( + func(context.Context, *domain.ContainerConfig) (*domain.Container, error) { + recordEvent(events, "create-new") + return &domain.Container{ID: containerID, Name: "web"}, nil + }).Once() + runtime.EXPECT().StartContainer(mock.Anything, containerID).RunAndReturn( + func(context.Context, string) error { + recordEvent(events, "start-new") + return nil + }).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, containerID, mock.Anything).Return( + []domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: hostPort, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + state.EXPECT().RegisterBackendBinds(mock.Anything, mock.Anything).Return(nil).Once() +} + +// expectRetireOld wires the confirmed retirement of the superseded container. +func expectRetireOld( + state *outmocks.MockAppState, + runtime *outmocks.MockContainerRuntime, + events *[]string, + containerID string, +) { + runtime.EXPECT().StopContainer(mock.Anything, containerID, mock.Anything).RunAndReturn( + func(context.Context, string, time.Duration) error { + recordEvent(events, "stop-old") + return nil + }).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, containerID, false).RunAndReturn( + func(context.Context, string, bool) error { + recordEvent(events, "remove-old") + return nil + }).Once() + state.EXPECT().ReleaseBackendBinds(mock.Anything, "blog", containerID).Return(nil).Once() +} + +// TestDeploy_SequentialReplacementNeverOverlapsGenerations proves the +// replacement order for a public HTTP service without volumes: withdrawal is +// confirmed first, the superseded container is stopped and removed before the +// replacement is created, readiness runs before ACTIVE is published, and the +// graph is applied last. +func TestDeploy_SequentialReplacementNeverOverlapsGenerations(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + expectDeployPreflight(state, runtime, images, secrets, rev) + expectNetworkProvision(runtime, "app-blog", 1) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(replaceableActive(), true, nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + + var order []string + expectRetireOld(state, runtime, &order, "c-old") + expectSequentialCreate(state, runtime, &order, "c-new", 18080) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(replaceableActive(), true, nil) + state.EXPECT().SaveActive(mock.Anything, mock.Anything).RunAndReturn( + func(context.Context, domain.AppActive) error { + recordEvent(&order, "publish-active") + return nil + }).Once() + state.EXPECT().SaveOwnership(mock.Anything, mock.Anything).Return(nil) + state.EXPECT().ClearRecoveryInhibition(mock.Anything, "blog", "web", "c-old").Return(nil).Once() + + var probedURL string + traffic := &orderedTraffic{events: &order} + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, Traffic: traffic, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(_ context.Context, url, _ string) (int, error) { + probedURL = url + recordEvent(&order, "ready") + return 200, nil + }, + func(context.Context, string) error { return nil }, + )) + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + require.NoError(t, err) + require.NotNil(t, result) + assert.Equal(t, "deployed", result.Services["web"].Result) + assert.Contains(t, probedURL, "127.0.0.1:18080", "readiness dials the replacement's fresh loopback bind") + assert.Equal(t, + []string{"withdraw", "stop-old", "remove-old", "create-new", "start-new", "ready", "publish-active", "traffic"}, + order) +} + +// TestDeploy_PreflightFailureLeavesTheWorkloadUntouched proves preflight runs +// to completion before anything is disrupted: an unresolvable image fails the +// deploy without withdrawing traffic or stopping, removing, creating, or +// starting any container. +func TestDeploy_PreflightFailureLeavesTheWorkloadUntouched(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + state.EXPECT().Recover(mock.Anything).Return(nil).Once() + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil).Once() + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-blog"}, nil) + images.EXPECT().ResolveDigest(mock.Anything, mock.Anything).Return("", assert.AnError).Once() + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) + + var events []string + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, + Traffic: &orderedTraffic{events: &events}, + }, zerowrap.Default()) + + _, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + require.ErrorIs(t, err, domain.ErrAppImageUnresolvable) + assert.Empty(t, events, "preflight must not touch traffic before it succeeds") + runtime.AssertNotCalled(t, "StopContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "ListNetworks", mock.Anything) + state.AssertNotCalled(t, "SaveActive", mock.Anything, mock.Anything) +} + +// TestDeploy_WithdrawalFailureBlocksMutation proves a traffic withdrawal that +// cannot be applied leaves the running generation untouched: no container is +// stopped, removed, created, or started, and no ACTIVE state is published. +func TestDeploy_WithdrawalFailureBlocksMutation(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + expectDeployPreflight(state, runtime, images, secrets, rev) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(replaceableActive(), true, nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, + Traffic: &orderedTraffic{failWithdraw: true}, + }, zerowrap.Default()) + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + require.Error(t, err) + require.NotNil(t, result) + assert.Equal(t, "failed", result.Services["web"].Result) + assert.Contains(t, result.Services["web"].Error, "withdraw") + runtime.AssertNotCalled(t, "StopContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "CreateVolume", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "ListNetworks", mock.Anything) + state.AssertNotCalled(t, "SaveActive", mock.Anything, mock.Anything) + state.AssertNotCalled(t, "ReleaseBackendBinds", mock.Anything, mock.Anything, mock.Anything) +} + +// TestDeploy_FirstDeploymentRetiresNothing proves a service with no previous +// ACTIVE generation is created, verified, and published without any +// withdrawal or retirement step. +func TestDeploy_FirstDeploymentRetiresNothing(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + expectDeployPreflight(state, runtime, images, secrets, rev) + expectNetworkProvision(runtime, "app-blog", 1) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", Services: map[string]domain.AppEffectiveService{}, + }, true, nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + + var order []string + expectSequentialCreate(state, runtime, &order, "c-new", 18080) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", Services: map[string]domain.AppEffectiveService{}, + }, true, nil) + state.EXPECT().SaveActive(mock.Anything, mock.Anything).RunAndReturn( + func(context.Context, domain.AppActive) error { + recordEvent(&order, "publish-active") + return nil + }).Once() + state.EXPECT().SaveOwnership(mock.Anything, mock.Anything).Return(nil) + + traffic := &orderedTraffic{events: &order} + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, Traffic: traffic, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { + recordEvent(&order, "ready") + return 200, nil + }, + func(context.Context, string) error { return nil }, + )) + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + require.NoError(t, err) + require.NotNil(t, result) + assert.Equal(t, "deployed", result.Services["web"].Result) + assert.Equal(t, + []string{"create-new", "start-new", "ready", "publish-active", "traffic"}, + order, + "a first deployment withdraws nothing and retires nothing") + runtime.AssertNotCalled(t, "StopContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, mock.Anything, mock.Anything) +} + +// TestDeploy_ReadinessFailureJournalsFailureAndRemovesCandidate proves a +// replacement that is not ready is journaled as failed, that ACTIVE is never +// published for it, and that the unready candidate is removed. The superseded +// container is already gone and is never recreated. +func TestDeploy_ReadinessFailureJournalsFailureAndRemovesCandidate(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + rev.Spec.Services[0].Readiness.Timeout = 50 * time.Millisecond + + expectReplacementPreflight(state, runtime, images, secrets, rev) + expectNetworkProvision(runtime, "app-blog", 1) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(replaceableActive(), true, nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + + var order []string + expectRetireOld(state, runtime, &order, "c-old") + expectSequentialCreate(state, runtime, &order, "c-new", 18080) + runtime.EXPECT().GetContainerLogs(mock.Anything, "c-new", false).Return(nil, assert.AnError).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-new", true).RunAndReturn( + func(context.Context, string, bool) error { + recordEvent(&order, "remove-candidate") + return nil + }).Once() + state.EXPECT().ReleaseBackendBinds(mock.Anything, "blog", "c-new").Return(nil).Once() + + var saved []domain.AppOperation + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).RunAndReturn( + func(_ context.Context, op domain.AppOperation) error { + saved = append(saved, op) + return nil + }) + + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, + Traffic: &orderedTraffic{events: &order}, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 503, nil }, + func(context.Context, string) error { return assert.AnError }, + )) + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + require.Error(t, err) + require.NotNil(t, result) + assert.Equal(t, "failed", result.Services["web"].Result) + assert.Contains(t, result.Services["web"].Error, "readiness", "the failure names the readiness probe") + assert.NotContains(t, result.Services["web"].Error, "logs:", "raw logs must never be embedded in the public error") + assert.Equal(t, "c-new", result.Services["web"].After) + + // ACTIVE was never repointed and the replaced generation's inhibition was + // never cleared: nothing claims the old workload is available again. + state.AssertNotCalled(t, "SaveActive", mock.Anything, mock.Anything) + state.AssertNotCalled(t, "ClearRecoveryInhibition", mock.Anything, mock.Anything, mock.Anything, mock.Anything) + + require.NotEmpty(t, saved) + last := saved[len(saved)-1] + assert.Equal(t, domain.AppOutcomeFailed, last.Outcome, "a failed replacement must not be journaled as success") + step, ok := opStep(last, "service.web.replace") + require.True(t, ok) + assert.Equal(t, domain.AppStepFailed, step.State) + + assert.Equal(t, "remove-candidate", order[len(order)-1], "the unready candidate is the last thing removed") + runtime.AssertNotCalled(t, "StopContainer", mock.Anything, "c-new", mock.Anything) + runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, "c-old", true) +} + +// TestDeploy_CandidateCleanupFailureStaysVisible proves a candidate that +// cannot be removed does not turn a failed replacement into a success: the +// leftover is reported as a cleanup warning on the failed service. +func TestDeploy_CandidateCleanupFailureStaysVisible(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + rev.Spec.Services[0].Readiness.Timeout = 50 * time.Millisecond + + expectDeployPreflight(state, runtime, images, secrets, rev) + expectNetworkProvision(runtime, "app-blog", 1) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(replaceableActive(), true, nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + + expectRetireOld(state, runtime, nil, "c-old") + expectSequentialCreate(state, runtime, nil, "c-new", 18080) + runtime.EXPECT().GetContainerLogs(mock.Anything, "c-new", false).Return(nil, assert.AnError).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-new", true).Return(assert.AnError).Once() + + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 503, nil }, + func(context.Context, string) error { return assert.AnError }, + )) + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + require.Error(t, err) + require.NotNil(t, result) + svcResult := result.Services["web"] + assert.Equal(t, "failed", svcResult.Result) + require.Len(t, svcResult.CleanupWarnings, 1, "an unremovable candidate is reported, never hidden") + assert.Equal(t, "web", svcResult.CleanupWarnings[0].Service) + assert.Equal(t, "c-new", svcResult.CleanupWarnings[0].Leftover) + assert.Contains(t, svcResult.CleanupWarnings[0].Detail, "remove") + state.AssertNotCalled(t, "ReleaseBackendBinds", mock.Anything, "blog", "c-new") + state.AssertNotCalled(t, "SaveActive", mock.Anything, mock.Anything) +} + +// TestDeploy_CancellationDuringReadinessIsNotReportedAsSuccess proves a +// cancellation during the readiness wait never becomes a success: ACTIVE is +// not published, no terminal outcome is written, and the candidate cleanup +// that the canceled context could not complete is reported as a leftover. +// The durable store and the runtime refuse work on a canceled context, so +// this test models both. +func TestDeploy_CancellationDuringReadinessIsNotReportedAsSuccess(t *testing.T) { + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + rev.Spec.Services[0].Readiness.Timeout = 30 * time.Second + + expectReplacementPreflight(state, runtime, images, secrets, rev) + expectNetworkProvision(runtime, "app-blog", 1) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(replaceableActive(), true, nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + + expectRetireOld(state, runtime, nil, "c-old") + expectSequentialCreate(state, runtime, nil, "c-new", 18080) + // The runtime refuses to remove the candidate with a canceled context. + runtime.EXPECT().RemoveContainer(mock.Anything, "c-new", true).Return(context.Canceled).Once() + + var saved []domain.AppOperation + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).RunAndReturn( + func(opCtx context.Context, op domain.AppOperation) error { + if opCtx.Err() != nil { + // The durable store refuses writes on a canceled context, so + // the operation stays in flight instead of becoming terminal. + return opCtx.Err() + } + saved = append(saved, op) + return nil + }) + + probeStarted := make(chan struct{}) + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(probeCtx context.Context, _, _ string) (int, error) { + select { + case <-probeStarted: + default: + close(probeStarted) + } + <-probeCtx.Done() + return 0, probeCtx.Err() + }, + func(context.Context, string) error { return nil }, + )) + + done := make(chan error, 1) + var deployResult *deployment.DeployResult + go func() { + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + deployResult = result + done <- err + }() + select { + case <-probeStarted: + case <-time.After(5 * time.Second): + t.Fatal("timed out waiting for the readiness probe") + } + cancel() + err := <-done + require.Error(t, err) + assert.Contains(t, err.Error(), context.Canceled.Error(), "the canceled probe is reported") + + // No success is claimed anywhere: ACTIVE is untouched, and no journaled + // outcome records the replacement as succeeded. The durable store refuses + // the terminal write on a canceled context, so the operation stays in + // flight rather than becoming a false success. + state.AssertNotCalled(t, "SaveActive", mock.Anything, mock.Anything) + require.NotEmpty(t, saved) + for _, op := range saved { + assert.NotContains(t, []string{domain.AppOutcomeSuccess, domain.AppOutcomePartial}, op.Outcome, + "a canceled replacement is never journaled as a success") + if step, ok := opStep(op, "service.web.replace"); ok { + assert.NotEqual(t, domain.AppStepSucceeded, step.State, + "a canceled replacement step is never recorded as succeeded") + } + } + + // The canceled cleanup is best effort, and what it could not finish is + // reported instead of being hidden. + require.NotNil(t, deployResult) + require.Len(t, deployResult.Services["web"].CleanupWarnings, 1, + "a candidate the canceled cleanup could not remove is reported") + assert.Equal(t, "c-new", deployResult.Services["web"].CleanupWarnings[0].Leftover) + runtime.AssertCalled(t, "RemoveContainer", mock.Anything, "c-new", true) + state.AssertNotCalled(t, "ReleaseBackendBinds", mock.Anything, "blog", "c-new") +} + +// TestDeploy_MultiServiceStopsAtFirstFailure proves services are deployed in +// sorted order and the first failure stops the run: the failing service is +// cleaned up, later services are untouched, and the services already deployed +// are kept without any rollback. +func TestDeploy_MultiServiceStopsAtFirstFailure(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + + httpService := func(name, host, image string, timeout time.Duration) domain.AppService { + return domain.AppService{ + Name: name, Image: image, + Readiness: domain.AppReadiness{Type: domain.AppReadinessHTTP, Path: "/healthz", Timeout: timeout}, + HTTP: []domain.AppHTTPInterface{{Host: host, Port: 8080, TLS: domain.AppTLSAuto}}, + } + } + rev := testRevision("blog", + httpService("alpha", "alpha.example.com", "docker.io/example/alpha:1.4.2", time.Second), + httpService("beta", "beta.example.com", "docker.io/example/beta:1.4.2", 50*time.Millisecond), + httpService("gamma", "gamma.example.com", "docker.io/example/gamma:1.4.2", time.Second), + ) + + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + images.EXPECT().ResolveDigest(mock.Anything, "docker.io/example/alpha:1.4.2").Return("sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", nil).Once() + images.EXPECT().ResolveDigest(mock.Anything, "docker.io/example/beta:1.4.2").Return("sha256:"+strings.Repeat("b", 64), nil).Once() + images.EXPECT().ResolveDigest(mock.Anything, "docker.io/example/gamma:1.4.2").Return("sha256:"+strings.Repeat("c", 64), nil).Once() + runtime.EXPECT().InspectImageVolumes(mock.Anything, "docker.io/example/alpha:1.4.2").Return(nil, nil).Once() + runtime.EXPECT().InspectImageVolumes(mock.Anything, "docker.io/example/beta:1.4.2").Return(nil, nil).Once() + runtime.EXPECT().InspectImageVolumes(mock.Anything, "docker.io/example/gamma:1.4.2").Return(nil, nil).Once() + state.EXPECT().LoadCheckpoint(mock.Anything).Return(domain.AppStoreCheckpoint{}, nil) + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-blog"}, nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", Services: map[string]domain.AppEffectiveService{}, + }, true, nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + expectNetworkProvision(runtime, "app-blog", 2) + + expectSequentialCreate(state, runtime, nil, "c-alpha", 18080) + state.EXPECT().SaveActive(mock.Anything, mock.MatchedBy(func(active domain.AppActive) bool { + return active.Services["alpha"].Container == "c-alpha" + })).Return(nil).Once() + state.EXPECT().SaveOwnership(mock.Anything, mock.Anything).Return(nil) + + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-beta", mock.Anything).Return( + []domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 18081, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + state.EXPECT().RegisterBackendBinds(mock.Anything, mock.Anything).Return(nil).Once() + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).Return(&domain.Container{ID: "c-beta", Name: "beta"}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-beta").Return(nil).Once() + runtime.EXPECT().GetContainerLogs(mock.Anything, "c-beta", false).Return(nil, assert.AnError).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-beta", true).Return(nil).Once() + state.EXPECT().ReleaseBackendBinds(mock.Anything, "blog", "c-beta").Return(nil).Once() + + var saved []domain.AppOperation + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).RunAndReturn( + func(_ context.Context, op domain.AppOperation) error { + saved = append(saved, op) + return nil + }) + + traffic := &orderedTraffic{} + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, Traffic: traffic, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(_ context.Context, url, _ string) (int, error) { + if strings.Contains(url, "18081") { + return 503, nil + } + return 200, nil + }, + func(context.Context, string) error { return nil }, + )) + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + require.Error(t, err) + require.NotNil(t, result) + assert.Equal(t, "deployed", result.Services["alpha"].Result, "an already deployed service is kept") + assert.Equal(t, "failed", result.Services["beta"].Result) + assert.NotContains(t, result.Services, "gamma", "a later service is never attempted after a failure") + + runtime.AssertNumberOfCalls(t, "CreateContainer", 2) + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.MatchedBy(func(cfg *domain.ContainerConfig) bool { + return cfg.Name == "gamma" + })) + // No rollback of the already deployed service. + runtime.AssertNotCalled(t, "StopContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, "c-alpha", mock.Anything) + state.AssertNumberOfCalls(t, "SaveActive", 1) + + require.NotEmpty(t, saved) + last := saved[len(saved)-1] + assert.Equal(t, domain.AppOutcomePartial, last.Outcome, + "the run stops at the first failure; the earlier success is kept, not rolled back") +} + +// TestDeploy_VolumeReplacementPreventsOverlappingWriters proves a stateful +// replacement durably inhibits the superseded generation before it is +// retired, stops and removes it before the candidate starts, and keeps the +// inhibition while the candidate's removal cannot be confirmed. Exclusion +// holds because the previous process is gone before the replacement can write. +func TestDeploy_VolumeReplacementPreventsOverlappingWriters(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + runtime := outmocks.NewMockContainerRuntime(t) + images := outmocks.NewMockImageResolver(t) + secrets := outmocks.NewMockSecretProvider(t) + + image := "registry.example.com/blog/web:1.4.2" + spec := domain.AppService{ + Name: "web", + Image: image, + Secrets: map[string]string{}, + Volumes: []domain.AppVolume{{Name: "data", Path: "/data"}}, + HTTP: []domain.AppHTTPInterface{{Host: "blog.example.com", Port: 8080, TLS: domain.AppTLSAuto}}, + Readiness: domain.AppReadiness{ + Type: domain.AppReadinessHTTP, Path: "/healthz", Timeout: 50 * time.Millisecond, + }, + } + rev := testRevision("blog", spec) + + require.NoError(t, store.SaveIntent(ctx, domain.AppStopIntent{App: "blog"})) + require.NoError(t, store.SaveActive(ctx, domain.AppActive{ + App: "blog", + Services: map[string]domain.AppEffectiveService{ + "web": { + Container: "c-old", EffectiveRevision: "rev-0", Spec: spec, + BackendBinds: map[int]int{8080: 18080}, + }, + }, + })) + seedDesired(t, ctx, store, rev) + // The app keeps one incarnation: the private network every service joins + // is derived from it. + require.NoError(t, store.SaveOwnership(ctx, domain.AppOwnership{App: "blog", ID: "app-blog"})) + expectNetworkProvision(runtime, "app-blog", 1) + + images.EXPECT().ResolveDigest(mock.Anything, image).Return("sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", nil).Once() + runtime.EXPECT().InspectImageVolumes(mock.Anything, image).Return(nil, nil).Once() + runtime.EXPECT().VolumeExists(mock.Anything, mock.Anything).Return(false, nil).Once() + + var order []string + inhibitedBeforeStop := false + runtime.EXPECT().StopContainer(mock.Anything, "c-old", mock.Anything).RunAndReturn( + func(context.Context, string, time.Duration) error { + recordEvent(&order, "stop-old") + // The durable marker must already exist before the superseded + // writer is stopped: recovery must never race the destructive step. + inhibitions, loadErr := store.LoadRecoveryInhibitions(context.Background(), "blog") + require.NoError(t, loadErr) + inhibitedBeforeStop = len(inhibitions) == 1 && + inhibitions[0].ContainerID == "c-old" && + inhibitions[0].Reason == domain.AppInhibitReplacementPending + return nil + }).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-old", false).RunAndReturn( + func(context.Context, string, bool) error { + recordEvent(&order, "remove-old") + return nil + }).Once() + runtime.EXPECT().CreateVolume(mock.Anything, mock.Anything, mock.Anything).RunAndReturn( + func(context.Context, string, map[string]string) error { + recordEvent(&order, "create-volume") + return nil + }).Once() + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).RunAndReturn( + func(context.Context, *domain.ContainerConfig) (*domain.Container, error) { + recordEvent(&order, "create-new") + return &domain.Container{ID: "c-new", Name: "web"}, nil + }).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-new").RunAndReturn( + func(context.Context, string) error { + recordEvent(&order, "start-new") + return nil + }).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-new", mock.Anything).Return( + []domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 18081, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + runtime.EXPECT().GetContainerLogs(mock.Anything, "c-new", false).Return(nil, assert.AnError).Once() + // The failed replacement removes its candidate once. Boot reconciliation + // retries it because a terminal failed operation cannot know the earlier + // removal succeeded without re-checking the runtime; that retry fails, so + // recovery fails closed and never revives the superseded writer whose + // volume the orphan may still touch. + runtime.EXPECT().RemoveContainer(mock.Anything, "c-new", true).Return(nil).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-new", true).Return(assert.AnError).Once() + + svc := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime, Images: images, Secrets: secrets, + ImagePolicy: domain.ImageSourcePolicy{AllowedRegistries: []string{"registry.example.com"}}, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 503, nil }, + func(context.Context, string) error { return assert.AnError }, + )) + + _, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + require.Error(t, err) + + assert.Equal(t, []string{"stop-old", "remove-old", "create-volume", "create-new", "start-new"}, order, + "the superseded writer is gone before the replacement can write") + assert.True(t, inhibitedBeforeStop, + "the durable recovery inhibition exists before the superseded writer is stopped") + + // The failed replacement never cleared the durable inhibition: a later + // boot must not revive the old writer on top of data the candidate may + // already have touched. + inhibitions, loadErr := store.LoadRecoveryInhibitions(ctx, "blog") + require.NoError(t, loadErr) + require.Len(t, inhibitions, 1) + assert.Equal(t, "c-old", inhibitions[0].ContainerID) + assert.Equal(t, domain.AppInhibitReplacementPending, inhibitions[0].Reason) + + // ACTIVE still names the superseded generation: the failure is explicit + // and no availability was restored on its behalf. + active, ok, loadErr := store.LoadActive(ctx, "blog") + require.NoError(t, loadErr) + require.True(t, ok) + assert.Equal(t, "c-old", active.Services["web"].Container) + + // The failed replacement keeps its candidate in the journal: a leftover + // must stay traceable so reconciliation can converge it. + failedOp, hasOp, loadErr := store.LoadLatestOperation(ctx, "blog") + require.NoError(t, loadErr) + require.True(t, hasOp) + failedStep, found := opStep(failedOp, "service.web.replace") + require.True(t, found) + assert.Equal(t, domain.AppStepFailed, failedStep.State) + assert.Equal(t, "c-new", failedStep.After, "the created candidate stays recorded") + + rebooted := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()) + bootErr := rebooted.ReconcileBoot(ctx) + require.Error(t, bootErr) + assert.Contains(t, bootErr.Error(), "could not be removed") + runtime.AssertNotCalled(t, "StartContainer", mock.Anything, "c-old") + runtime.AssertNotCalled(t, "RestartContainer", mock.Anything, "c-old", mock.Anything) + + // The orphan that could not be removed keeps the inhibition: the + // superseded writer is never revived on top of data the candidate may + // already have touched. + keptInhibitions, keptErr := store.LoadRecoveryInhibitions(ctx, "blog") + require.NoError(t, keptErr) + require.Len(t, keptInhibitions, 1) + assert.Equal(t, "c-old", keptInhibitions[0].ContainerID) + assert.Equal(t, domain.AppInhibitReplacementPending, keptInhibitions[0].Reason) +} diff --git a/internal/usecase/deployment/service.go b/internal/usecase/deployment/service.go new file mode 100644 index 000000000..8ebde820a --- /dev/null +++ b/internal/usecase/deployment/service.go @@ -0,0 +1,958 @@ +// Package deployment implements the app deployment engine: preflight, +// journaled execution, and terminal-result computation. +package deployment + +import ( + "context" + "encoding/hex" + "errors" + "fmt" + "maps" + "sort" + "strings" + "sync" + "time" + + "github.com/bnema/zerowrap" + "github.com/google/uuid" + + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/pkg/validation" +) + +// Deps are the driven adapters owned by the deployment engine. +type Deps struct { + State out.AppState + Runtime out.ContainerRuntime + Images out.ImageResolver + Secrets out.SecretProvider + // Registry configures image pulls from the installation registry. + // Domain scopes which refs pull with auth; empty disables pulling. + Registry RegistryConfig + // Networks is the installation network policy applied to every app + // workload: the prefix used to derive Gordon-owned network names and + // whether those networks block external egress. + Networks NetworkConfig + // Limits are the installation container resource limits applied on + // every create and recovery path. Zero means unlimited. + Limits ResourceLimits + // ImagePolicy restricts which registries and references may be + // resolved, pulled, or run. It is enforced on every path, including + // boot, restart, and recovery. + ImagePolicy domain.ImageSourcePolicy + // Traffic is the serialized HTTP/L4 publish boundary. Rebuilt after + // every activation (deploy/start/restart/stop/remove) so the proxy + // host index follows ACTIVE; recovery also withdraws one service + // fail-closed before touching a workload. Nil in tests that assert + // state effects only. Failures are reported to the caller. + Traffic out.AppTrafficRefresher +} + +// RegistryConfig carries the installation registry identity for pulls. +type RegistryConfig struct { + Domain string + PullAddress string + Username string + Password string +} + +// NetworkConfig carries the installation network policy for app workloads. +type NetworkConfig struct { + // Prefix names the Gordon-owned networks. Empty uses the domain default. + Prefix string + // Internal blocks external egress from containers attached to those + // networks (Docker's Internal flag). + Internal bool +} + +// ResourceLimits are the installation container limits applied on every +// create and recovery path. Zero values mean unlimited. +type ResourceLimits struct { + MemoryBytes int64 + NanoCPUs int64 + PidsLimit int64 +} + +// Service orchestrates captured-revision deploys. +type Service struct { + deps Deps + log zerowrap.Logger + probes *ProbeDeps + + // coord serializes workload mutations per app: deploy/start/ + // restart/stop/remove/boot lock blocking, periodic reconciliation + // skips busy apps. + coord *appCoordinator + // barrier is the process-wide GC barrier. A shared lease covers + // every workload mutation from resource selection through durable + // protection publication; prune holds the exclusive lease. + barrier out.GCBarrier + // backoff is the in-memory, generation-scoped recovery budget. + backoff *recoveryBackoff + // publication holds the in-memory publication inhibition: a + // service whose fail-closed withdrawal could not be applied must be + // withdrawn again before any later publication. + publication publicationInhibition + // executions remembers the last observed execution start per + // generation, so a native restart is detected without restarting + // the workload ourselves. + executions executionTracker + // bindMu guards bindPolicies. A config reload replaces the whole + // map atomically; deploy paths read the current map without holding + // the lock. + bindMu sync.RWMutex + bindPolicies map[string]domain.AppBindPolicy + // deviceMu guards devicePolicies. A config reload replaces the whole + // map atomically; deploy paths read the current map without holding + // the lock. + deviceMu sync.RWMutex + devicePolicies map[string]domain.AppDevicePolicy + // liveMu guards liveOps. liveOps holds the operations this process + // owns between the StartDeploy claim and the end of ExecuteDeploy. It + // is in-memory only: a fresh Service (boot) starts empty and may + // reconcile every persisted non-terminal operation, while a + // foreground mutation must never finalize an operation a live + // goroutine still owns. + liveMu sync.Mutex + liveOps map[string]struct{} +} + +// NewService creates the deployment engine. All deps are required; +// Secrets may be nil only in tests that never reach secret preflight. +func NewService(deps Deps, log zerowrap.Logger) *Service { + if deps.ImagePolicy.InstallationRegistry == "" { + deps.ImagePolicy.InstallationRegistry = deps.Registry.Domain + } + return &Service{ + deps: deps, + log: log, + coord: newAppCoordinator(), + backoff: newRecoveryBackoff(time.Now), + } +} + +// markOperationLive records that this process owns one operation from the +// StartDeploy claim until ExecuteDeploy finishes. A live operation is the +// single owner's in-flight work: a foreground reconciliation must not +// finalize it, or the claim would be settled before the owner can execute it. +func (s *Service) markOperationLive(app, op string) { + if app == "" || op == "" { + return + } + s.liveMu.Lock() + defer s.liveMu.Unlock() + if s.liveOps == nil { + s.liveOps = make(map[string]struct{}) + } + s.liveOps[liveOperationKey(app, op)] = struct{}{} +} + +// clearOperationLive drops the live marker. It is called once ExecuteDeploy +// has recorded its outcome or left the claim recoverable, so a later +// reconciliation can converge the operation instead of leaving it in flight. +func (s *Service) clearOperationLive(app, op string) { + s.liveMu.Lock() + defer s.liveMu.Unlock() + delete(s.liveOps, liveOperationKey(app, op)) +} + +// operationLive reports whether this process still owns the operation. +func (s *Service) operationLive(app, op string) bool { + s.liveMu.Lock() + defer s.liveMu.Unlock() + _, ok := s.liveOps[liveOperationKey(app, op)] + return ok +} + +func liveOperationKey(app, op string) string { + return app + "\x00" + op +} + +// WithGCBarrier wires the process-wide GC barrier. Every workload +// mutation holds a shared lease from resource selection through the +// durable publication of the resource's protection, so prune can never +// observe a half-acquired resource. Nil is valid for tests that never +// run prune. +func (s *Service) WithGCBarrier(barrier out.GCBarrier) *Service { + s.barrier = barrier + return s +} + +// WithBindPolicies supplies the initial administrative bind policies. +func (s *Service) WithBindPolicies(policies map[string]domain.AppBindPolicy) *Service { + s.SetBindPolicies(policies) + return s +} + +// SetBindPolicies atomically replaces the administrative bind policies used +// to authorize service binds. The map is copied so later caller mutation +// cannot race an in-flight deploy. +func (s *Service) SetBindPolicies(policies map[string]domain.AppBindPolicy) { + copied := make(map[string]domain.AppBindPolicy, len(policies)) + for name, policy := range policies { + copied[name] = policy + } + s.bindMu.Lock() + s.bindPolicies = copied + s.bindMu.Unlock() +} + +// snapshotBindPolicies returns the current policy map. SetBindPolicies never +// mutates a published map, so the snapshot stays safe after the lock drops. +func (s *Service) snapshotBindPolicies() map[string]domain.AppBindPolicy { + s.bindMu.RLock() + defer s.bindMu.RUnlock() + return s.bindPolicies +} + +// WithDevicePolicies supplies the initial administrative device policies. +func (s *Service) WithDevicePolicies(policies map[string]domain.AppDevicePolicy) *Service { + s.SetDevicePolicies(policies) + return s +} + +// SetDevicePolicies atomically replaces the administrative device policies +// used to authorize service devices. The map and its slices are copied so +// later caller mutation cannot race an in-flight deploy. +func (s *Service) SetDevicePolicies(policies map[string]domain.AppDevicePolicy) { + copied := make(map[string]domain.AppDevicePolicy, len(policies)) + for name, policy := range policies { + policy.CDI = append([]string(nil), policy.CDI...) + policy.AllowedApps = append([]string(nil), policy.AllowedApps...) + policy.AllowedServices = append([]string(nil), policy.AllowedServices...) + copied[name] = policy + } + s.deviceMu.Lock() + s.devicePolicies = copied + s.deviceMu.Unlock() +} + +// snapshotDevicePolicies returns the current policy map. SetDevicePolicies +// never mutates a published map, so the snapshot stays safe after the +// lock drops. +func (s *Service) snapshotDevicePolicies() map[string]domain.AppDevicePolicy { + s.deviceMu.RLock() + defer s.deviceMu.RUnlock() + return s.devicePolicies +} + +// resolveServiceDevices resolves every declared logical device against the +// current administrative policies into runtime CDI IDs. Errors name only +// the app, service, and device: host inventory never appears. +func (s *Service) resolveServiceDevices(app string, svc domain.AppService) ([]string, error) { + if len(svc.Devices) == 0 { + return nil, nil + } + ids, err := domain.ResolveAppDevices(app, svc.Name, svc.Devices, s.snapshotDevicePolicies()) + if err != nil { + return nil, fmt.Errorf("deployment: %w", err) + } + return ids, nil +} + +// resolveServiceBinds resolves every declared bind against the current +// administrative policies into ephemeral runtime binds. Errors name only the +// app, service, and mount: host source paths never appear. +func (s *Service) resolveServiceBinds(app string, svc domain.AppService) ([]domain.ContainerBind, error) { + if len(svc.Binds) == 0 { + return nil, nil + } + policies := s.snapshotBindPolicies() + resolved := make([]domain.ContainerBind, 0, len(svc.Binds)) + for _, bind := range svc.Binds { + policy, ok := policies[bind.Name] + if !ok { + return nil, fmt.Errorf("deployment: app %q service %q mount %q: no administrative mount policy configured: %w", app, svc.Name, bind.Name, domain.ErrBindPolicy) + } + r, err := policy.ResolveAppBind(app, svc.Name, bind) + if err != nil { + return nil, fmt.Errorf("deployment: app %q service %q mount %q: refused by administrative mount policy: %w", app, svc.Name, bind.Name, domain.ErrBindPolicy) + } + resolved = append(resolved, domain.ContainerBind(r)) + } + return resolved, nil +} + +// acquireAppContext takes the GC shared lease first and the per-app +// coordinator second, following the documented lock order +// (GC barrier → per-app coordinator → registry lock → bbolt +// transaction). The returned release drops both, coordinator first. +func (s *Service) acquireAppContext(ctx context.Context, app string) (func(), error) { + lease, err := s.acquireSharedGC(ctx) + if err != nil { + return nil, err + } + release, err := s.coord.acquire(ctx, app) + if err != nil { + lease.Release() + return nil, err + } + return func() { + release() + lease.Release() + }, nil +} + +// tryAcquireAppContext takes the GC shared lease, then tries the per-app +// coordinator without blocking. ok is false when the app is busy; the +// caller skips it. Periodic reconciliation still waits for an in-flight +// prune, because a half-planned snapshot is exactly what must not be +// raced. +func (s *Service) tryAcquireAppContext(ctx context.Context, app string) (func(), bool) { + lease, err := s.acquireSharedGC(ctx) + if err != nil { + return nil, false + } + release, ok := s.coord.tryAcquire(app) + if !ok { + lease.Release() + return nil, false + } + return func() { + release() + lease.Release() + }, true +} + +// acquireSharedGC takes a shared GC lease, or a no-op lease when no +// barrier is wired. +func (s *Service) acquireSharedGC(ctx context.Context) (out.GCLease, error) { + if s.barrier == nil { + return noopGCLease{}, nil + } + return s.barrier.AcquireShared(ctx) +} + +// noopGCLease is the lease used when no barrier is wired. +type noopGCLease struct{} + +func (noopGCLease) Release() {} + +// WithProbeDeps overrides readiness probing (tests inject a mock-backed +// ProbeDeps via NewProbeDeps; production leaves it nil for the live +// runtime adapter). +func (s *Service) WithProbeDeps(probes ProbeDeps) *Service { + s.probes = &probes + return s +} + +// DeployInput selects the revision and service scope. +type DeployInput struct { + App string + Revision string + Service string + Op string +} + +// DeployResult carries terminal per-service results and cleanup warnings. +type DeployResult struct { + Op string + App string + Revision string + Outcome string + Services map[string]ServiceResult + CleanupWarnings []CleanupWarning + // Removed names services retired because the new revision no longer + // declares them. + Removed []string +} + +// ServiceResultUnchanged marks a deploy step that kept the running +// container because its image, spec, and environment were already current. +const ServiceResultUnchanged = domain.AppServiceUnchanged + +// ServiceResult is one service's terminal deployment result. +type ServiceResult struct { + Result string + EffectiveRevision string + Before string + After string + RestartUnsafe bool + Error string + // BackendBinds carries the created container's TCP loopback publishes + // (container port -> 127.0.0.1 host port) for the active record. + BackendBinds map[int]int + // UDPBackendBinds carries the created container's UDP loopback + // publishes for the active record. Nil when the service declares + // no UDP interface. + UDPBackendBinds map[int]int + // CleanupWarnings are bounded leftovers of this service's terminal + // path: a candidate that could not be removed, or a backend claim + // that could not be released. + CleanupWarnings []CleanupWarning + // Diagnostics is bounded, redacted failure output of the failed + // candidate. It is never embedded in Error and is persisted to the + // journal for logs-scoped retrieval only. + Diagnostics []string +} + +// CleanupWarning records post-publication leftovers for operator action. +type CleanupWarning struct { + Service string + Leftover string + Detail string +} + +// pinnedService binds a service spec to its resolved digest plus the +// captured revision's app-wide public env (injected into every service +// alongside that service's own resolved secrets). appEnv is a copy owned +// by this pin: never mutated, never persisted with secret values. +type pinnedService struct { + name string + spec domain.AppService + digest string + runtimeImage string + appEnv map[string]string + appNetworks []domain.AppSharedNetwork + // sharedNetworks are the declared shared-network memberships this + // service joins. The private incarnation network is derived from the + // app UUID at create time so recovery and deploy agree. + sharedNetworks []domain.AppSharedNetwork +} + +// newOpID allocates a time-ordered op- identifier (uuid v7). +func newOpID() string { + id, err := uuid.NewV7() + if err != nil { + id = uuid.New() + } + var buf [32]byte + hex.Encode(buf[:], id[:]) + return "op-" + string(buf[:]) +} + +// claimDeploymentLocked is the first phase of a deploy: it resolves the +// requested revision, enforces the targeted-deploy convergence rules, and +// claims the request key, persisting the non-terminal journal. It performs +// no image pull and no workload mutation, so a crash after it leaves a +// durable claim that reconciliation can converge. owned is true only for the +// single claimer of the key; a replay returns the stored journal instead. +// The caller holds the app lock and has already recovered the store. +func (s *Service) claimDeploymentLocked(ctx context.Context, input DeployInput) (domain.AppDesiredRevision, domain.AppOperation, bool, error) { + rev, err := s.resolveRevision(ctx, input) + if err != nil { + return domain.AppDesiredRevision{}, domain.AppOperation{}, false, err + } + if input.Service != "" { + if err := s.checkConverged(ctx, input.App, input.Service, rev); err != nil { + return domain.AppDesiredRevision{}, domain.AppOperation{}, false, err + } + } + + op := domain.AppOperation{ + Kind: "deploy", + App: input.App, + InputRevision: rev.Revision, + StartedAt: time.Now().UTC(), + Request: domain.AppOperationRequestFor("deploy", input.App, input.Revision, input.Service), + } + op, owned, err := s.claimOperation(ctx, input.Op, op, []domain.AppOperationStep{{ID: "preflight", State: domain.AppStepPending}}) + if err != nil { + return domain.AppDesiredRevision{}, domain.AppOperation{}, false, err + } + return rev, op, owned, nil +} + +// pinPreflightLocked is the second preflight phase: it runs the image +// pull/pin and resource gates of an already claimed operation and records the +// pinned table (or the terminal preflight failure) in the journal. The +// caller holds the app lock and owns the claimed, non-terminal operation. +func (s *Service) pinPreflightLocked(ctx context.Context, app, onlyService string, rev domain.AppDesiredRevision, op *domain.AppOperation) ([]pinnedService, error) { + log := zerowrap.FromCtx(ctx) + + pinned, err := s.preflightServices(ctx, app, rev, onlyService) + if err != nil { + op.Steps[0] = domain.AppOperationStep{ID: "preflight", State: domain.AppStepFailed, Error: err.Error()} + op.Outcome = domain.AppOutcomeFailed + if saveErr := s.deps.State.SaveOperation(ctx, *op); saveErr != nil { + log.Warn().Err(saveErr).Msg("deployment: failed to record preflight failure") + } + return nil, err + } + op.Steps[0] = domain.AppOperationStep{ID: "preflight", State: domain.AppStepSucceeded} + // A resumed claim can carry the service-step plan of an earlier attempt, + // including candidates it recorded. Converge those candidates before + // replacing the plan: an unreconciled leftover may still be running, and + // the replacement this execution creates must never overlap it. + if err := s.convergeResumedCandidates(ctx, app, op, log); err != nil { + return nil, err + } + // Replace the persisted plan instead of appending: the preflight step plus + // one step per pinned service, so runServiceStep's index contract matches a + // normalized layout and no earlier step is overwritten or left pending. + plan := make([]domain.AppOperationStep, 1, len(pinned)+1) + plan[0] = op.Steps[0] + for _, p := range pinned { + plan = append(plan, domain.AppOperationStep{ + ID: "service." + p.name + ".replace", + State: domain.AppStepPending, + Service: p.name, + Digest: p.digest, + Image: p.runtimeImage, + }) + } + op.Steps = plan + if err := s.deps.State.SaveOperation(ctx, *op); err != nil { + return nil, fmt.Errorf("deployment: persist pinned table: %w", err) + } + log.Info().Str("op", op.Op).Str("revision", rev.Revision).Int("services", len(pinned)).Msg("deployment: preflight passed") + return pinned, nil +} + +// convergeResumedCandidates converges any candidate an earlier attempt of a +// resumed claim recorded, so replacing its service-step plan cannot drop a +// leftover that may still be running. A fresh claim has only its preflight +// step and does no work here. +func (s *Service) convergeResumedCandidates(ctx context.Context, app string, op *domain.AppOperation, log zerowrap.Logger) error { + needsActive := false + for _, step := range op.Steps { + if reconcileNeedsContainer(step, false) { + needsActive = true + break + } + } + if !needsActive { + return nil + } + active, _, err := s.deps.State.LoadActive(ctx, app) + if err != nil { + return fmt.Errorf("deployment: load active for resumed claim: %w", err) + } + if _, err := s.convergeInterruptedSteps(ctx, app, op, active, false, log); err != nil { + return err + } + return nil +} + +// claimOperation is the single claim point before any mutation effect. +// A keyed request is claimed atomically: the first claim persists the +// in-flight journal and reports owned=true; a repeat returns the stored +// journal and reports owned=false. A key that already answered a +// different request fails with ErrAppStateConflict, and a key for a name +// with no app identity fails with ErrAppNotFound without writing +// anything. An unkeyed request (internal recovery paths) starts a fresh +// journal under a generated id. +func (s *Service) claimOperation(ctx context.Context, key string, op domain.AppOperation, steps []domain.AppOperationStep) (domain.AppOperation, bool, error) { + op.Steps = steps + // A journal is never opened while a different operation of this app is + // still unfinished: that predecessor must be reconciled first, or this + // claim would mask it and its leftover generation could stay untracked. + latest, found, err := s.deps.State.LoadLatestOperation(ctx, op.App) + if err != nil { + return domain.AppOperation{}, false, fmt.Errorf("deployment: load latest operation: %w", err) + } + if found && !latest.Terminal() && latest.Op != key { + return domain.AppOperation{}, false, fmt.Errorf( + "deployment: app %q has unfinished operation %s: %w", op.App, latest.Op, domain.ErrAppStateConflict) + } + if key == "" { + op.Op = newOpID() + if err := s.deps.State.SaveOperation(ctx, op); err != nil { + return domain.AppOperation{}, false, fmt.Errorf("deployment: persist journal before effects: %w", err) + } + return op, true, nil + } + op.Op = key + existing, claimed, err := s.deps.State.ClaimOperation(ctx, op) + if err != nil { + return domain.AppOperation{}, false, err + } + if !claimed { + return existing, false, nil + } + return op, true, nil +} + +// replayError reports a replayed key whose journal never reached a +// terminal outcome: the operation is still in flight or was interrupted, +// so its effects must not run again. +func replayError(op domain.AppOperation) error { + if !op.Terminal() { + return fmt.Errorf("deployment: operation %s has not reached a terminal outcome: %w", op.Op, domain.ErrAppStateConflict) + } + if op.Outcome != domain.AppOutcomeSuccess { + return fmt.Errorf("deployment: operation %s previously ended with outcome %s: %w", op.Op, op.Outcome, domain.ErrAppStateConflict) + } + return nil +} + +// resolveRevision loads the captured revision (default: current desired). +// A revision that cannot be resolved on a name with no app identity at +// all reports ErrAppNotFound: the failure is the missing app, not the +// missing revision. +func (s *Service) resolveRevision(ctx context.Context, input DeployInput) (domain.AppDesiredRevision, error) { + if input.Revision != "" { + rev, err := s.deps.State.LoadRevision(ctx, input.App, input.Revision) + if err != nil && errors.Is(err, domain.ErrAppRevisionNotFound) { + if knownErr := s.requireKnownApp(ctx, input.App); knownErr != nil { + return domain.AppDesiredRevision{}, knownErr + } + } + return rev, err + } + rev, ok, err := s.deps.State.LoadDesired(ctx, input.App) + if err != nil { + return domain.AppDesiredRevision{}, err + } + if !ok { + if knownErr := s.requireKnownApp(ctx, input.App); knownErr != nil { + return domain.AppDesiredRevision{}, knownErr + } + return domain.AppDesiredRevision{}, fmt.Errorf("deployment: app %q has no desired state: %w", input.App, domain.ErrAppRevisionNotFound) + } + return rev, nil +} + +// requireKnownApp refuses a mutation of a name with no live app identity. +func (s *Service) requireKnownApp(ctx context.Context, app string) error { + exists, err := s.deps.State.AppExists(ctx, app) + if err != nil { + return fmt.Errorf("deployment: app existence: %w", err) + } + if !exists { + return fmt.Errorf("deployment: app %q does not exist: %w", app, domain.ErrAppNotFound) + } + return nil +} + +// checkConverged permits a targeted deploy when every desired/effective +// difference belongs to that service. App-wide and other-service changes +// must be deployed together so ACTIVE never combines incompatible specs. +func (s *Service) checkConverged(ctx context.Context, app, service string, rev domain.AppDesiredRevision) error { + active, ok, err := s.deps.State.LoadActive(ctx, app) + if err != nil { + return err + } + if !ok { + return fmt.Errorf("deployment: app %q was never deployed, targeted deploy refused: %w", app, domain.ErrAppStateConflict) + } + if active.Converged && active.ConvergedRevision == rev.Revision { + return nil + } + diff := domain.DiffAppSpec(rev.Spec, effectiveAppSpec(active)) + prefix := "service/" + service + "/" + if len(diff.Added) > 0 { + return targetedDivergenceError(rev.Revision, diff.Added[0]+" (added)") + } + if len(diff.Removed) > 0 { + return targetedDivergenceError(rev.Revision, diff.Removed[0]+" (removed)") + } + for _, path := range diff.Changed { + if !strings.HasPrefix(path, prefix) { + return targetedDivergenceError(rev.Revision, path) + } + } + return nil +} + +func effectiveAppSpec(active domain.AppActive) domain.AppSpec { + spec := domain.AppSpec{Name: active.App, Networks: append([]domain.AppSharedNetwork(nil), active.Networks...)} + for name, service := range active.Services { + effective := service.Spec + effective.Name = name + spec.Services = append(spec.Services, effective) + } + return spec +} + +func targetedDivergenceError(revision, path string) error { + return fmt.Errorf("%w: service-targeted deploy refused, desired %s also changes %s", domain.ErrAppStateConflict, revision, path) +} + +// selectPreflightServices copies and optionally narrows a revision's services. +func selectPreflightServices(rev domain.AppDesiredRevision, onlyService string) ([]domain.AppService, error) { + services := append([]domain.AppService(nil), rev.Spec.Services...) + if onlyService == "" { + return services, nil + } + for _, svc := range services { + if svc.Name == onlyService { + return []domain.AppService{svc}, nil + } + } + return nil, fmt.Errorf("deployment: service %q not in revision %s: %w", onlyService, rev.Revision, domain.ErrAppStateConflict) +} + +// preflightServices runs the preflight gates in order. No mutation. +func (s *Service) preflightServices(ctx context.Context, app string, rev domain.AppDesiredRevision, onlyService string) ([]pinnedService, error) { + services, err := selectPreflightServices(rev, onlyService) + if err != nil { + return nil, err + } + sort.Slice(services, func(i, j int) bool { return services[i].Name < services[j].Name }) + + var pinned []pinnedService + ownership, err := s.deps.State.LoadOwnership(ctx, app) + if err != nil { + return nil, err + } + // Defensive re-check: stored revisions predate or bypass apply + // validation; an overlap must fail preflight before any mutation. + if err := rev.Spec.CheckEnvSecretCollisions(); err != nil { + return nil, err + } + appEnv := maps.Clone(rev.Spec.Env) + if appEnv == nil { + appEnv = map[string]string{} + } + for _, svc := range services { + digest, err := s.deps.Images.ResolveDigest(ctx, svc.Image) + if err != nil { + return nil, fmt.Errorf("deployment: image %q unresolvable: %w", svc.Image, domain.ErrAppImageUnresolvable) + } + if err := s.checkSecrets(ctx, app, ownership.ID, svc); err != nil { + return nil, err + } + runtimeImage, err := s.preflightImage(ctx, svc.Image, digest) + if err != nil { + return nil, err + } + if err := s.checkImageVolumes(ctx, svc, runtimeImage); err != nil { + return nil, err + } + // Bind resolution stays a preflight gate: an unresolvable bind + // must fail before any workload mutation. The successful values + // are discarded because createAndStart re-resolves against the + // current policy immediately before the runtime mutation. + if _, err := s.resolveServiceBinds(app, svc); err != nil { + return nil, err + } + if err := s.preflightServiceDevices(ctx, app, svc, pinned); err != nil { + return nil, err + } + pinned = append(pinned, pinnedService{ + name: svc.Name, spec: svc, digest: digest, runtimeImage: runtimeImage, appEnv: appEnv, + appNetworks: append([]domain.AppSharedNetwork(nil), rev.Spec.Networks...), + sharedNetworks: domain.AppServiceSharedNetworks(rev.Spec, svc.Name), + }) + } + if err := s.recheckReservations(ctx, app, rev); err != nil { + return nil, err + } + if err := s.checkResources(ctx, app, ownership, services); err != nil { + return nil, err + } + return pinned, nil +} + +// preflightServiceDevices runs the device authorization and engine +// capability gates for one service. Split from preflightServices to keep +// its complexity within budget. +func (s *Service) preflightServiceDevices(ctx context.Context, app string, svc domain.AppService, pinned []pinnedService) error { + if _, err := s.resolveServiceDevices(app, svc); err != nil { + return err + } + // Engine capability is a knowable-before-mutation failure: a + // device-bearing revision on an unsupported engine must fail here, + // before withdrawal retires the serving generation. The check runs + // once per revision, on the first device-bearing service; + // CreateContainer keeps the same gate as defense. + if len(svc.Devices) > 0 && !devicesEngineChecked(pinned) { + if s.deps.Runtime == nil { + return fmt.Errorf("deployment: runtime unavailable: %w", domain.ErrRuntimeUnsupported) + } + // Sanitize the adapter cause: version-probe failures may embed + // the daemon endpoint, which must not reach journals or API + // responses. Cancellation and deadlines are the exception: they + // stay recognizable so aborted deploys are not misread as + // incapable engines. + if err := s.deps.Runtime.SupportsCDIDevices(ctx); err != nil { + return sanitizedEngineProbeError(svc.Name, err) + } + } + return nil +} + +// sanitizedEngineProbeError maps an engine device-capability probe failure +// onto the caller-visible error. Cancellation and deadlines survive so +// shutdown and timeout handling keep working; every other cause collapses +// to ErrRuntimeUnsupported alone, dropping the adapter text that may embed +// the daemon endpoint. +func sanitizedEngineProbeError(service string, err error) error { + switch { + case errors.Is(err, context.Canceled): + return fmt.Errorf("deployment: engine device capability check for service %q: %w", service, context.Canceled) + case errors.Is(err, context.DeadlineExceeded): + return fmt.Errorf("deployment: engine device capability check for service %q: %w", service, context.DeadlineExceeded) + default: + return fmt.Errorf("deployment: engine device capability check for service %q: %w", service, domain.ErrRuntimeUnsupported) + } +} + +// devicesEngineChecked reports whether an earlier pinned service already +// triggered the engine capability probe for this revision. +func devicesEngineChecked(pinned []pinnedService) bool { + for _, p := range pinned { + if len(p.spec.Devices) > 0 { + return true + } + } + return false +} + +// checkSecrets reads every required secret path once. Values stay in +// memory for container creation only; nothing is written anywhere. +// Paths are UUID-keyed so a removed app's secrets are never adopted +// by a new app reusing the name. +func (s *Service) checkSecrets(ctx context.Context, app, appID string, svc domain.AppService) error { + if s.deps.Secrets == nil { + return fmt.Errorf("deployment: secret provider unavailable: %w", domain.ErrAppSecretMissing) + } + for envKey, name := range svc.Secrets { + path := domain.AppSecretPathForID(appID, app, svc.Name, name) + if _, err := s.deps.Secrets.GetSecret(ctx, path); err != nil { + return fmt.Errorf("deployment: secret %s (env %s) missing: %w", path, envKey, domain.ErrAppSecretMissing) + } + } + return nil +} + +// appSecretID returns the stable internal UUID for secret paths. +// Empty means legacy name-keyed paths (records predating UUIDs). +// Used on the execution path where preflight's ownership snapshot is +// not in scope; deploys are infrequent, one extra store read per +// service is acceptable. +func (s *Service) appSecretID(ctx context.Context, app string) (string, error) { + ownership, err := s.deps.State.LoadOwnership(ctx, app) + if err != nil { + return "", err + } + return ownership.ID, nil +} + +func (s *Service) preflightImage(ctx context.Context, image, digest string) (string, error) { + if err := validation.ValidateImageDigest(digest); err != nil { + return "", fmt.Errorf("deployment: image %q resolved to invalid digest: %w", image, domain.ErrAppImageUnresolvable) + } + if err := s.validateImageSource(image, digest); err != nil { + return "", err + } + if s.deps.Registry.Domain == "" { + return image, nil + } + return s.pullImage(ctx, stripImageTag(image)+"@"+digest) +} + +// validateImageSource enforces the installation image policy on a +// reference before any pull. Both the manifest reference and the pinned +// digest are checked, so a digest-pinned ref cannot bypass the hostname +// and port allowlist on deploy or recovery. +func (s *Service) validateImageSource(image, digest string) error { + ref := strings.TrimSpace(image) + if digest != "" && !strings.Contains(ref, "@") { + ref = stripImageTag(image) + "@" + digest + } + if err := s.deps.ImagePolicy.ValidateImageSource(ref); err != nil { + return fmt.Errorf("deployment: image %q: %w", image, err) + } + return nil +} + +// checkImageVolumes rejects images with unmapped VOLUME declarations. +func (s *Service) checkImageVolumes(ctx context.Context, svc domain.AppService, runtimeImage string) error { + declared, err := s.deps.Runtime.InspectImageVolumes(ctx, runtimeImage) + if err != nil { + return fmt.Errorf("deployment: inspect image volumes: %w", err) + } + if len(declared) == 0 { + return nil + } + mapped := map[string]struct{}{} + for _, vol := range svc.Volumes { + mapped[vol.Path] = struct{}{} + } + // Bind destinations also map image-declared volumes: a bind that + // mounts over an image VOLUME is an explicit operator mapping. + for _, bind := range svc.Binds { + mapped[bind.Path] = struct{}{} + } + for _, path := range declared { + if _, ok := mapped[path]; !ok { + return fmt.Errorf( + "deployment: image declares unmanaged volume %q, map it explicitly: %w", + path, domain.ErrAppUnmanagedImageVolume, + ) + } + } + return nil +} + +// recheckReservations re-validates against the live global table. +func (s *Service) recheckReservations(ctx context.Context, app string, rev domain.AppDesiredRevision) error { + checkpoint, err := s.deps.State.LoadCheckpoint(ctx) + if err != nil { + return err + } + reservations := domain.ReservationsFor(rev.Spec) + if err := domain.CheckReservations(checkpoint.Reservations, reservations, app); err != nil { + return fmt.Errorf("deployment: %w", err) + } + return nil +} + +// checkResources verifies volumes exist-or-creatable and ownership is consistent. +// A generated runtime volume that already exists but is NOT recorded in +// this app's ownership is refused fail-closed: mounting it would +// implicitly adopt (and possibly modify) foreign data. Volumes recorded +// in ownership (attached or retained) were created by this app's own +// deploys and are safe to reuse. +func (s *Service) checkResources(ctx context.Context, app string, ownership domain.AppOwnership, services []domain.AppService) error { + owned := map[string]string{} + ownedRuntime := map[string]struct{}{} + for _, vol := range ownership.Volumes { + owned[vol.Name] = vol.Service + ownedRuntime[vol.RuntimeName] = struct{}{} + } + for _, svc := range services { + for _, vol := range svc.Volumes { + if owner, ok := owned[vol.Name]; ok && owner != svc.Name { + return fmt.Errorf( + "deployment: volume %q owned by %q, not %q: %w", + vol.Name, owner, svc.Name, domain.ErrAppStateConflict, + ) + } + runtimeName := domain.RuntimeVolumeName(app, svc.Name, vol.Name) + exists, err := s.deps.Runtime.VolumeExists(ctx, runtimeName) + if err != nil { + return fmt.Errorf("deployment: check volume %q: %w", runtimeName, err) + } + if exists { + if _, ok := ownedRuntime[runtimeName]; !ok { + return fmt.Errorf( + "deployment: runtime volume %q already exists but is not owned by app %q: refusing to adopt foreign data: %w", + runtimeName, app, domain.ErrAppStateConflict, + ) + } + } + } + } + return nil +} + +// ComputeOutcome derives the op outcome over terminal per-service results. +func ComputeOutcome(results map[string]ServiceResult) string { + deployed := 0 + failed := 0 + for _, result := range results { + switch result.Result { + case "deployed", ServiceResultUnchanged: + deployed++ + case "failed": + failed++ + } + } + switch { + case deployed > 0 && failed > 0: + return domain.AppOutcomePartial + case deployed > 0: + return domain.AppOutcomeSuccess + default: + return domain.AppOutcomeFailed + } +} + +// itoa formats integers without extra imports. +func itoa(n int) string { + return fmt.Sprintf("%d", n) +} diff --git a/internal/usecase/deployment/service_test.go b/internal/usecase/deployment/service_test.go new file mode 100644 index 000000000..94b6c3422 --- /dev/null +++ b/internal/usecase/deployment/service_test.go @@ -0,0 +1,123 @@ +package deployment_test + +import ( + "context" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/deployment" +) + +func testRevision(app string, services ...domain.AppService) domain.AppDesiredRevision { + for i := range services { + if services[i].StopGrace == 0 { + services[i].StopGrace = 10 * time.Second + } + if services[i].Readiness.Timeout == 0 { + services[i].Readiness.Timeout = 30 * time.Second + } + } + return domain.AppDesiredRevision{ + Revision: "rev-1", App: app, + Spec: domain.AppSpec{Name: app, Env: map[string]string{}, Services: services}, + } +} + +func webService() domain.AppService { + return domain.AppService{ + Name: "web", Image: "docker.io/example/web:1.4.2", + Readiness: domain.AppReadiness{Type: "http", Path: "/healthz"}, + HTTP: []domain.AppHTTPInterface{{Host: "blog.example.com", Port: 8080, TLS: "auto"}}, + Secrets: map[string]string{"DATABASE_URL": "database-url"}, + } +} + +func preflightService( + t *testing.T, + state *outmocks.MockAppState, + runtime *outmocks.MockContainerRuntime, + images *outmocks.MockImageResolver, + secrets *outmocks.MockSecretProvider, +) *deployment.Service { + t.Helper() + // Every mutation claims a journal only when no other operation of the + // app is unfinished. These tests start clean. + state.EXPECT().LoadLatestOperation(mock.Anything, mock.Anything). + Return(domain.AppOperation{}, false, nil).Maybe() + return deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, + ImagePolicy: domain.ImageSourcePolicy{AllowedRegistries: []string{"registry.example.com"}}, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 200, nil }, + func(context.Context, string) error { return nil }, + )) +} + +func TestComputeOutcome_TerminalResults(t *testing.T) { + deployed := deployment.ServiceResult{Result: "deployed"} + failed := deployment.ServiceResult{Result: "failed", Error: "boom"} + assert.Equal(t, "success", deployment.ComputeOutcome(map[string]deployment.ServiceResult{"a": deployed})) + assert.Equal(t, "failed", deployment.ComputeOutcome(map[string]deployment.ServiceResult{"a": failed})) + assert.Equal(t, "partial", deployment.ComputeOutcome(map[string]deployment.ServiceResult{"a": deployed, "b": failed})) + assert.Equal(t, "failed", deployment.ComputeOutcome(map[string]deployment.ServiceResult{})) +} + +// TestReconcileBoot_ContinuesAfterAppFailure proves R1: a failing app +// never blocks verification of the following apps, and failures are +// aggregated instead of aborting at the first one. +func TestReconcileBoot_ContinuesAfterAppFailure(t *testing.T) { + ctx := context.Background() + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + images := outmocks.NewMockImageResolver(t) + secrets := outmocks.NewMockSecretProvider(t) + svcSvc := preflightService(t, state, runtime, images, secrets) + + badEff := domain.AppEffectiveService{ + Container: "ctr-bad", + Spec: webService(), + BackendBinds: map[int]int{8080: 32768}, + } + badActive := domain.AppActive{App: "bad", Services: map[string]domain.AppEffectiveService{"web": badEff}} + goodEff := domain.AppEffectiveService{ + Container: "ctr-good", + Spec: webService(), + } + goodActive := domain.AppActive{App: "good", Services: map[string]domain.AppEffectiveService{"web": goodEff}} + + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().ListApps(mock.Anything).Return([]string{"bad", "good"}, nil).Once() + // bad app: Start fails bind verification, binds withdrawn. + state.EXPECT().LoadIntent(mock.Anything, "bad").Return(domain.AppStopIntent{App: "bad"}, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "bad").Return(badActive, true, nil) + state.EXPECT().LoadLatestOperation(mock.Anything, "bad").Return(domain.AppOperation{}, false, nil).Maybe() + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).Return(nil) + state.EXPECT().SaveIntent(mock.Anything, mock.Anything).Return(nil) + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "bad").Return(nil, nil).Once() + runtime.EXPECT().IsContainerRunning(mock.Anything, "ctr-bad").Return(true, nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "ctr-bad", mock.Anything).Return(nil, assert.AnError).Once() + // No traffic boundary is wired here, so the fail-closed bind + // withdrawal is a no-op and deployment writes no ACTIVE binds itself. + // good app: still verified after the bad one failed. + state.EXPECT().LoadIntent(mock.Anything, "good").Return(domain.AppStopIntent{App: "good"}, nil).Once() + // Start load + bind-refresh reload. + state.EXPECT().LoadActive(mock.Anything, "good").Return(goodActive, true, nil) + state.EXPECT().LoadLatestOperation(mock.Anything, "good").Return(domain.AppOperation{}, false, nil).Maybe() + state.EXPECT().LoadRecoveryInhibitions(mock.Anything, "good").Return(nil, nil).Once() + runtime.EXPECT().IsContainerRunning(mock.Anything, "ctr-good").Return(true, nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "ctr-good", mock.Anything).Return([]domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 32777, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + state.EXPECT().RegisterBackendBinds(mock.Anything, mock.Anything).Return(nil).Once() + // Bind refresh persists the new bind (differs from recorded nil). + state.EXPECT().SaveActive(mock.Anything, mock.Anything).Return(nil).Once() + + err := svcSvc.ReconcileBoot(ctx) + require.Error(t, err) + assert.Contains(t, err.Error(), `"bad"`) +} diff --git a/internal/usecase/deployment/traffic_failure_test.go b/internal/usecase/deployment/traffic_failure_test.go new file mode 100644 index 000000000..23d97506e --- /dev/null +++ b/internal/usecase/deployment/traffic_failure_test.go @@ -0,0 +1,138 @@ +package deployment_test + +import ( + "context" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/deployment" +) + +// opStep returns one journaled step by id. +func opStep(op domain.AppOperation, id string) (domain.AppOperationStep, bool) { + for _, step := range op.Steps { + if step.ID == id { + return step, true + } + } + return domain.AppOperationStep{}, false +} + +// TestDeploy_TrafficFailureRecordsFailedServiceStep proves the ordering +// invariant: a service step is never journaled as succeeded while the +// routing graph refused its publication. +func TestDeploy_TrafficFailureRecordsFailedServiceStep(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + rev.Spec.Services[0].StopGrace = time.Millisecond + + oldActive := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {EffectiveRevision: "rev-0", Image: "img:0", Container: "c-old"}, + }} + + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + images.EXPECT().ResolveDigest(mock.Anything, rev.Spec.Services[0].Image).Return( + "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", nil).Once() + secrets.EXPECT().GetSecret(mock.Anything, "gordon/apps/app-blog/web/database-url").Return("x", nil) + runtime.EXPECT().InspectImageVolumes(mock.Anything, rev.Spec.Services[0].Image).Return(nil, nil).Once() + state.EXPECT().LoadCheckpoint(mock.Anything).Return(domain.AppStoreCheckpoint{}, nil).Once() + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-blog"}, nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(oldActive, true, nil) + state.EXPECT().SaveOwnership(mock.Anything, mock.Anything).Return(nil) + state.EXPECT().SaveActive(mock.Anything, mock.Anything).Return(nil) + + expectNetworkProvision(runtime, "app-blog", 1) + // The superseded generation is confirmed gone before the replacement is + // created: a deploy never overlaps two generations of one service. + runtime.EXPECT().StopContainer(mock.Anything, "c-old", mock.Anything).Return(nil).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-old", false).Return(nil).Once() + state.EXPECT().ReleaseBackendBinds(mock.Anything, "blog", "c-old").Return(nil).Once() + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).Return(&domain.Container{ID: "c-new", Name: "web"}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-new").Return(nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-new", mock.Anything).Return( + []domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 18080, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + state.EXPECT().RegisterBackendBinds(mock.Anything, mock.Anything).Return(nil).Once() + state.EXPECT().ClearRecoveryInhibition(mock.Anything, "blog", "web", "c-old").Return(nil).Once() + + var saved []domain.AppOperation + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).RunAndReturn( + func(_ context.Context, op domain.AppOperation) error { + saved = append(saved, op) + return nil + }) + + traffic := &recordingTraffic{failRebuild: true} + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, Traffic: traffic, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 200, nil }, + func(context.Context, string) error { return nil }, + )) + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + require.Error(t, err) + require.NotNil(t, result) + assert.Equal(t, "failed", result.Services["web"].Result, "a service without routable publication is not deployed") + + require.NotEmpty(t, saved) + last := saved[len(saved)-1] + assert.Equal(t, domain.AppOutcomeFailed, last.Outcome, "a failed publication must not be journaled as success") + step, ok := opStep(last, "service.web.replace") + require.True(t, ok) + assert.Equal(t, domain.AppStepFailed, step.State, "the service step must not claim success before traffic applied") + // The superseded container is gone: a rejected graph is reported as a + // failure of the new generation, never as a return to the old one. + runtime.AssertCalled(t, "RemoveContainer", mock.Anything, "c-old", false) + runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, "c-new", mock.Anything) +} + +// TestStop_TrafficFailureRecordsTerminalFailure proves a rejected graph +// apply after a stop is recorded as a failure, not as a successful stop +// whose rerun would read success. +func TestStop_TrafficFailureRecordsTerminalFailure(t *testing.T) { + ctx := context.Background() + state, runtime, _, _ := mockDeps(t) + + active := domain.AppActive{App: "blog", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "c-1", EffectiveRevision: "rev-1", Spec: headlessSpec()}, + }} + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(active, true, nil) + state.EXPECT().SaveIntent(mock.Anything, mock.Anything).Return(nil).Once() + runtime.EXPECT().StopContainer(mock.Anything, "c-1", mock.Anything).Return(nil).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-1", false).Return(nil).Once() + state.EXPECT().ReleaseBackendBinds(mock.Anything, "blog", "c-1").Return(nil).Once() + state.EXPECT().ClearRecoveryInhibition(mock.Anything, "blog", "web", "c-1").Return(nil).Once() + + var saved []domain.AppOperation + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).RunAndReturn( + func(_ context.Context, op domain.AppOperation) error { + saved = append(saved, op) + return nil + }) + + traffic := &recordingTraffic{failRebuild: true} + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Traffic: traffic, + }, zerowrap.Default()) + + result, err := svc.Stop(ctx, "blog", "") + require.Error(t, err) + require.NotNil(t, result) + assert.Equal(t, "deployed", result.Services["web"].Result, "the workload really was stopped") + + require.NotEmpty(t, saved) + terminal := saved[len(saved)-1] + assert.Equal(t, domain.AppOutcomeFailed, terminal.Outcome, "the rejected graph apply must not replay as success") + step, ok := opStep(terminal, "traffic.publish") + require.True(t, ok, "the rejected publication is journaled") + assert.Equal(t, domain.AppStepFailed, step.State) +} diff --git a/internal/usecase/deployment/two_phase_deploy_test.go b/internal/usecase/deployment/two_phase_deploy_test.go new file mode 100644 index 000000000..631f8e096 --- /dev/null +++ b/internal/usecase/deployment/two_phase_deploy_test.go @@ -0,0 +1,459 @@ +package deployment_test + +import ( + "context" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/deployment" +) + +// TestStartDeploy_PersistsClaimBeforeAnyExecution proves the first phase +// durably claims the key with a non-terminal journal and touches no runtime +// resource: no image resolution, no pull, no workload mutation. A crash right +// after StartDeploy therefore leaves a claim reconciliation can converge. +func TestStartDeploy_PersistsClaimBeforeAnyExecution(t *testing.T) { + t.Run("keyed", func(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + state.EXPECT().Recover(mock.Anything).Return(nil).Maybe() + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + state.EXPECT().ClaimOperation(mock.Anything, mock.MatchedBy(func(op domain.AppOperation) bool { + return op.Op == "k1" && !op.Terminal() + })).RunAndReturn(func(_ context.Context, op domain.AppOperation) (domain.AppOperation, bool, error) { + return op, true, nil + }).Once() + + svc := keyedService(t, state, runtime, images, secrets) + started, err := svc.StartDeploy(ctx, deployment.DeployInput{App: "blog", Op: "k1"}) + + require.NoError(t, err) + require.True(t, started.Owned) + require.Equal(t, "k1", started.Claim.Op) + require.False(t, started.Claim.Journal.Terminal(), "the claim is non-terminal before execution") + require.Len(t, started.Claim.Journal.Steps, 1) + assert.Equal(t, domain.AppStepPending, started.Claim.Journal.Steps[0].State) + + state.AssertNotCalled(t, "LoadOwnership", mock.Anything, mock.Anything) + images.AssertNotCalled(t, "ResolveDigest", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "StopContainer", mock.Anything, mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "RemoveContainer", mock.Anything, mock.Anything, mock.Anything) + }) + + t.Run("unkeyed", func(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + var saved []domain.AppOperation + state.EXPECT().Recover(mock.Anything).Return(nil).Maybe() + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).RunAndReturn(func(_ context.Context, op domain.AppOperation) error { + saved = append(saved, op) + return nil + }).Maybe() + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + + svc := keyedService(t, state, runtime, images, secrets) + started, err := svc.StartDeploy(ctx, deployment.DeployInput{App: "blog"}) + + require.NoError(t, err) + require.True(t, started.Owned) + require.Len(t, saved, 1, "the claim journal is persisted before any effect") + require.False(t, saved[0].Terminal()) + assert.Equal(t, domain.AppStepPending, saved[0].Steps[0].State) + images.AssertNotCalled(t, "ResolveDigest", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) + }) +} + +// TestStartDeploy_SameKeyHandsOneOwner proves the second phase is only ever +// handed to the single claimer: a repeated key reports Owned false, carries +// the stored journal, and replays with an explicit conflict instead of a +// second execution. +func TestStartDeploy_SameKeyHandsOneOwner(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + stored := domain.AppOperation{ + Op: "k1", Kind: "deploy", App: "blog", InputRevision: "rev-1", + Request: domain.AppOperationRequestFor("deploy", "blog", "", ""), + Steps: []domain.AppOperationStep{{ID: "preflight", State: domain.AppStepPending}}, + } + state.EXPECT().Recover(mock.Anything).Return(nil).Maybe() + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + state.EXPECT().ClaimOperation(mock.Anything, mock.Anything).Return(domain.AppOperation{}, true, nil).Once() + state.EXPECT().ClaimOperation(mock.Anything, mock.Anything).Return(stored, false, nil).Once() + + svc := keyedService(t, state, runtime, images, secrets) + first, err := svc.StartDeploy(ctx, deployment.DeployInput{App: "blog", Op: "k1"}) + require.NoError(t, err) + require.True(t, first.Owned) + + second, err := svc.StartDeploy(ctx, deployment.DeployInput{App: "blog", Op: "k1"}) + require.NoError(t, err) + require.False(t, second.Owned, "the key is never handed to a second owner") + require.Equal(t, "k1", second.Claim.Journal.Op) + require.ErrorIs(t, second.ReplayError(), domain.ErrAppStateConflict) + + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) +} + +// expectHeadlessDeploy wires one full sequential replacement of a headless +// service whose only effect is one container replacement, and records every +// journal write so the caller can assert claim-before-execution ordering. +func expectHeadlessDeploy( + state *outmocks.MockAppState, + runtime *outmocks.MockContainerRuntime, + images *outmocks.MockImageResolver, + secrets *outmocks.MockSecretProvider, +) (domain.AppDesiredRevision, *[]domain.AppOperation) { + svcSpec := webService() + svcSpec.HTTP = nil + svcSpec.Readiness = domain.AppReadiness{} + rev := domain.AppDesiredRevision{ + Revision: "rev-1", App: "blog", + Spec: domain.AppSpec{Name: "blog", Env: map[string]string{}, Services: []domain.AppService{svcSpec}}, + } + + saved := &[]domain.AppOperation{} + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).RunAndReturn(func(_ context.Context, op domain.AppOperation) error { + *saved = append(*saved, op) + return nil + }).Maybe() + state.EXPECT().Recover(mock.Anything).Return(nil) + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + images.EXPECT().ResolveDigest(mock.Anything, svcSpec.Image).Return(restartTestDigest, nil) + secrets.EXPECT().GetSecret(mock.Anything, "gordon/apps/app-blog/web/database-url").Return("x", nil) + runtime.EXPECT().InspectImageVolumes(mock.Anything, svcSpec.Image).Return(nil, nil) + state.EXPECT().LoadCheckpoint(mock.Anything).Return(domain.AppStoreCheckpoint{}, nil) + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-blog"}, nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{ + App: "blog", Services: map[string]domain.AppEffectiveService{}, + }, true, nil) + expectNetworkProvision(runtime, "app-blog", 1) + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).Return(&domain.Container{ID: "c-new", Name: "web"}, nil) + runtime.EXPECT().StartContainer(mock.Anything, "c-new").Return(nil) + state.EXPECT().SaveActive(mock.Anything, mock.Anything).Return(nil) + state.EXPECT().SaveOwnership(mock.Anything, mock.Anything).Return(nil) + return rev, saved +} + +// TestExecuteDeploy_RecordsTerminalOutcome proves the second phase executes +// the already claimed operation and records its terminal success, while the +// claim itself stayed non-terminal up to execution. +func TestExecuteDeploy_RecordsTerminalOutcome(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + _, saved := expectHeadlessDeploy(state, runtime, images, secrets) + + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 200, nil }, + func(context.Context, string) error { return nil }, + )) + + started, err := svc.StartDeploy(ctx, deployment.DeployInput{App: "blog"}) + require.NoError(t, err) + require.True(t, started.Owned) + require.Len(t, *saved, 1) + require.False(t, (*saved)[0].Terminal(), "the claim is non-terminal before execution") + + result, err := svc.ExecuteDeploy(ctx, started.Claim) + + require.NoError(t, err) + require.NotNil(t, result) + assert.Equal(t, "deployed", result.Services["web"].Result) + last := (*saved)[len(*saved)-1] + require.True(t, last.Terminal(), "the second phase records a terminal outcome") + assert.Equal(t, domain.AppOutcomeSuccess, last.Outcome) + runtime.AssertNumberOfCalls(t, "CreateContainer", 1) +} + +// TestExecuteDeploy_RecordsTerminalFailure proves a preflight that fails +// during the second phase records the terminal failure in the claimed +// journal and runs no workload. +func TestExecuteDeploy_RecordsTerminalFailure(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + var saved []domain.AppOperation + state.EXPECT().Recover(mock.Anything).Return(nil).Maybe() + state.EXPECT().SaveOperation(mock.Anything, mock.Anything).RunAndReturn(func(_ context.Context, op domain.AppOperation) error { + saved = append(saved, op) + return nil + }).Maybe() + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + state.EXPECT().LoadOwnership(mock.Anything, "blog").Return(domain.AppOwnership{App: "blog", ID: "app-blog"}, nil) + images.EXPECT().ResolveDigest(mock.Anything, rev.Spec.Services[0].Image).Return("", assert.AnError) + + svc := keyedService(t, state, runtime, images, secrets) + started, err := svc.StartDeploy(ctx, deployment.DeployInput{App: "blog"}) + require.NoError(t, err) + require.True(t, started.Owned) + + result, err := svc.ExecuteDeploy(ctx, started.Claim) + + require.ErrorIs(t, err, domain.ErrAppImageUnresolvable) + require.NotNil(t, result) + require.NotEmpty(t, saved) + last := saved[len(saved)-1] + require.True(t, last.Terminal()) + assert.Equal(t, domain.AppOutcomeFailed, last.Outcome) + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) +} + +// TestExecuteDeploy_TerminalStoreRecordBeatsStaleClaim proves the store is +// authoritative: a claim whose durable record was terminalized by another +// actor between the phases replays that outcome and never executes, even +// though the in-process claim still carries a non-terminal journal. +func TestExecuteDeploy_TerminalStoreRecordBeatsStaleClaim(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + rev := mockRevision() + + state.EXPECT().Recover(mock.Anything).Return(nil).Maybe() + state.EXPECT().LoadDesired(mock.Anything, "blog").Return(rev, true, nil) + state.EXPECT().ClaimOperation(mock.Anything, mock.Anything).RunAndReturn(func(_ context.Context, op domain.AppOperation) (domain.AppOperation, bool, error) { + return op, true, nil + }).Once() + + svc := keyedService(t, state, runtime, images, secrets) + started, err := svc.StartDeploy(ctx, deployment.DeployInput{App: "blog", Op: "k1"}) + require.NoError(t, err) + require.True(t, started.Owned) + require.False(t, started.Claim.Journal.Terminal(), "the in-process claim is still non-terminal") + + // Another actor settled the key in the store: the stale in-memory copy + // must not drive execution. + expectStoredOperation(state, domain.AppOperation{ + Op: "k1", Kind: "deploy", App: "blog", InputRevision: "rev-1", + Outcome: domain.AppOutcomeSuccess, + Steps: []domain.AppOperationStep{{ID: "preflight", State: domain.AppStepSucceeded}}, + }) + + result, err := svc.ExecuteDeploy(ctx, started.Claim) + + require.NoError(t, err) + require.NotNil(t, result) + assert.Equal(t, domain.AppOutcomeSuccess, result.Outcome) + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) + runtime.AssertNotCalled(t, "StopContainer", mock.Anything, mock.Anything, mock.Anything) +} + +// TestExecuteDeploy_CancellationLeavesRecoverableClaim proves cancellation or +// shutdown during the second phase leaves the durable claim non-terminal, and +// the existing reconciliation converges it on the next mutation instead of +// ever re-executing it. +func TestExecuteDeploy_CancellationLeavesRecoverableClaim(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + _, desired := seedReplaceableApp(t, ctx, store) + runtime := outmocks.NewMockContainerRuntime(t) + + barrier := outmocks.NewMockGCBarrier(t) + lease := outmocks.NewMockGCLease(t) + barrier.EXPECT().AcquireShared(mock.Anything).Return(nil, context.Canceled).Once() + barrier.EXPECT().AcquireShared(mock.Anything).Return(lease, nil).Once() + lease.EXPECT().Release().Return().Once() + + svc := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime, + Images: outmocks.NewMockImageResolver(t), Secrets: outmocks.NewMockSecretProvider(t), + }, zerowrap.Default()).WithGCBarrier(barrier) + + started, err := svc.StartDeploy(ctx, deployment.DeployInput{App: "blog", Op: "op-cancel", Revision: desired.Revision}) + require.NoError(t, err) + require.True(t, started.Owned) + + claimed, err := store.LoadOperation(ctx, "blog", "op-cancel") + require.NoError(t, err) + require.False(t, claimed.Terminal(), "the claim is durable and non-terminal before execution") + + cancelled, cancel := context.WithCancel(ctx) + cancel() + _, err = svc.ExecuteDeploy(cancelled, started.Claim) + require.ErrorIs(t, err, context.Canceled) + + interrupted, err := store.LoadOperation(ctx, "blog", "op-cancel") + require.NoError(t, err) + require.False(t, interrupted.Terminal(), "cancellation leaves the claim non-terminal for recovery") + + // The next mutation reconciles the interrupted claim before claiming its + // own key: the operation is finalized and never executes twice. + _, err = svc.StartDeploy(ctx, deployment.DeployInput{App: "blog", Op: "op-next", Revision: desired.Revision}) + require.NoError(t, err) + + reconciled, err := store.LoadOperation(ctx, "blog", "op-cancel") + require.NoError(t, err) + require.True(t, reconciled.Terminal()) + assert.Equal(t, domain.AppOutcomeFailed, reconciled.Outcome) + + _, err = svc.ExecuteDeploy(ctx, deployment.DeployClaim{App: "blog", Op: "op-cancel"}) + require.ErrorIs(t, err, domain.ErrAppStateConflict, "a settled key never re-executes") +} + +// TestDeploy_IsStartDeployThenExecuteDeploy proves the synchronous entry point +// is still the composition of both phases: it journals a non-terminal claim +// before any effect and a terminal outcome after the replacement. +func TestDeploy_IsStartDeployThenExecuteDeploy(t *testing.T) { + ctx := context.Background() + state, runtime, images, secrets := mockDeps(t) + _, saved := expectHeadlessDeploy(state, runtime, images, secrets) + + svc := deployment.NewService(deployment.Deps{ + State: state, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 200, nil }, + func(context.Context, string) error { return nil }, + )) + + result, err := svc.Deploy(ctx, deployment.DeployInput{App: "blog"}) + + require.NoError(t, err) + require.NotNil(t, result) + assert.Equal(t, "deployed", result.Services["web"].Result) + require.NotEmpty(t, *saved) + require.False(t, (*saved)[0].Terminal(), "Deploy journals the claim before execution") + require.True(t, (*saved)[len(*saved)-1].Terminal(), "Deploy records the terminal outcome") +} + +// TestStartDeploy_LiveClaimSurvivesForegroundReconciliation pins the +// StartDeploy/ExecuteDeploy hand-off window: while an owner is between the +// claim and the coordinator (blocked before its GC lease), a same-key and a +// different-key foreground mutation must not reconcile, terminalize, or +// re-claim the live operation. The single owner then executes exactly the +// revision it claimed. +func TestStartDeploy_LiveClaimSurvivesForegroundReconciliation(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + _, desired := seedReplaceableApp(t, ctx, store) + runtime := outmocks.NewMockContainerRuntime(t) + images := outmocks.NewMockImageResolver(t) + secrets := outmocks.NewMockSecretProvider(t) + + // One full headless replacement, the only effects the live owner may run. + images.EXPECT().ResolveDigest(mock.Anything, "docker.io/example/web:1.4.2").Return(restartTestDigest, nil).Once() + runtime.EXPECT().InspectImageVolumes(mock.Anything, "docker.io/example/web:1.4.2").Return(nil, nil).Once() + expectNetworkProvision(runtime, "app-blog", 1) + runtime.EXPECT().StopContainer(mock.Anything, "c-old", mock.Anything).Return(nil).Once() + runtime.EXPECT().RemoveContainer(mock.Anything, "c-old", false).Return(nil).Once() + runtime.EXPECT().CreateContainer(mock.Anything, mock.Anything).Return(&domain.Container{ID: "c-new", Name: "web"}, nil).Once() + runtime.EXPECT().StartContainer(mock.Anything, "c-new").Return(nil).Once() + runtime.EXPECT().GetContainerBackendBinds(mock.Anything, "c-new", mock.Anything).Return( + []domain.ContainerBackendBind{{ContainerPort: 8080, HostPort: 18081, Protocol: domain.NetworkProtocolTCP}}, nil).Once() + + // Block ExecuteDeploy on its GC lease, before it takes the coordinator. + entered := make(chan struct{}) + release := make(chan struct{}) + barrier := outmocks.NewMockGCBarrier(t) + lease := outmocks.NewMockGCLease(t) + barrier.EXPECT().AcquireShared(mock.Anything).Run(func(context.Context) { + close(entered) + <-release + }).Return(lease, nil).Once() + lease.EXPECT().Release().Return().Once() + + svc := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()).WithGCBarrier(barrier).WithProbeDeps(deployment.NewTestProbeDeps(runtime, + func(context.Context, string, string) (int, error) { return 200, nil }, + func(context.Context, string) error { return nil }, + )) + + started, err := svc.StartDeploy(ctx, deployment.DeployInput{App: "blog", Op: "op-live", Revision: desired.Revision}) + require.NoError(t, err) + require.True(t, started.Owned, "the first claim owns the operation") + + done := make(chan error, 1) + go func() { + _, execErr := svc.ExecuteDeploy(ctx, started.Claim) + done <- execErr + }() + select { + case <-entered: + case <-time.After(5 * time.Second): + t.Fatal("timed out waiting for ExecuteDeploy to block before the coordinator") + } + + // Same key while the owner is live: replay, never a second owner. + same, err := svc.StartDeploy(ctx, deployment.DeployInput{App: "blog", Op: "op-live", Revision: desired.Revision}) + require.NoError(t, err) + require.False(t, same.Owned, "a same-key replay never owns the operation twice") + require.ErrorIs(t, same.ReplayError(), domain.ErrAppStateConflict, "the live claim was not terminalized") + + // Different key while the owner is live: the unfinished-operation guard + // refuses, so no second claim is written. + _, err = svc.StartDeploy(ctx, deployment.DeployInput{App: "blog", Op: "op-other", Revision: "rev-0"}) + require.ErrorIs(t, err, domain.ErrAppStateConflict) + + live, err := store.LoadOperation(ctx, "blog", "op-live") + require.NoError(t, err) + require.False(t, live.Terminal(), "a live operation is never terminalized by a foreground mutation") + _, err = store.LoadOperation(ctx, "blog", "op-other") + require.ErrorIs(t, err, domain.ErrAppOperationNotFound, "no second claim was written") + + close(release) + require.NoError(t, <-done) + + final, err := store.LoadOperation(ctx, "blog", "op-live") + require.NoError(t, err) + require.True(t, final.Terminal()) + assert.Equal(t, domain.AppOutcomeSuccess, final.Outcome) + assert.Equal(t, desired.Revision, final.InputRevision, "the owner executed the revision it claimed, not the stale one") + runtime.AssertNumberOfCalls(t, "CreateContainer", 1) +} + +// TestAbandonDeploy_SettlesClaimWithoutStarting proves a claimed but never +// scheduled operation can be settled safely: the journal becomes terminal, the +// in-process live marker is released, and a later claim succeeds instead of +// being refused as a permanently unfinished operation. No runtime effect is +// touched. +func TestAbandonDeploy_SettlesClaimWithoutStarting(t *testing.T) { + ctx := context.Background() + store := newTestStore(t) + _, desired := seedReplaceableApp(t, ctx, store) + runtime := outmocks.NewMockContainerRuntime(t) + images := outmocks.NewMockImageResolver(t) + secrets := outmocks.NewMockSecretProvider(t) + + svc := deployment.NewService(deployment.Deps{ + State: store, Runtime: runtime, Images: images, Secrets: secrets, + }, zerowrap.Default()) + + started, err := svc.StartDeploy(ctx, deployment.DeployInput{App: "blog", Op: "op-abandon", Revision: desired.Revision}) + require.NoError(t, err) + require.True(t, started.Owned) + + // While the claim is live, a different key is refused as unfinished. + _, err = svc.StartDeploy(ctx, deployment.DeployInput{App: "blog", Op: "op-blocked", Revision: desired.Revision}) + require.ErrorIs(t, err, domain.ErrAppStateConflict) + + require.NoError(t, svc.AbandonDeploy(ctx, started.Claim)) + + settled, err := store.LoadOperation(ctx, "blog", "op-abandon") + require.NoError(t, err) + require.True(t, settled.Terminal(), "a settled claim must not remain in flight") + assert.Equal(t, domain.AppOutcomeFailed, settled.Outcome) + + // The live marker is gone and the journal terminal, so the next claim is + // accepted cleanly instead of hanging on a false live operation. + next, err := svc.StartDeploy(ctx, deployment.DeployInput{App: "blog", Op: "op-next", Revision: desired.Revision}) + require.NoError(t, err) + require.True(t, next.Owned, "a settled claim never blocks the next owner") + require.NoError(t, svc.AbandonDeploy(ctx, next.Claim)) + + runtime.AssertNotCalled(t, "CreateContainer", mock.Anything, mock.Anything) +} diff --git a/internal/usecase/health/service.go b/internal/usecase/health/service.go index bf9abb181..125b53cb4 100644 --- a/internal/usecase/health/service.go +++ b/internal/usecase/health/service.go @@ -1,4 +1,4 @@ -// Package health implements the health check use case for routes. +// Package health implements health checking over ACTIVE app services. package health import ( @@ -9,80 +9,147 @@ import ( "github.com/bnema/zerowrap" "github.com/bnema/gordon/internal/boundaries/in" + "github.com/bnema/gordon/internal/boundaries/out" "github.com/bnema/gordon/internal/domain" ) // maxConcurrentProbes limits the number of concurrent health probes to prevent resource exhaustion. const maxConcurrentProbes = 10 -// Service implements the HealthService interface. +// Service implements the HealthService interface over ACTIVE app state. +// Probes dial recorded loopback backends rootless-first, never +// container IPs. type Service struct { - configSvc in.ConfigService - containerSvc in.ContainerService - prober in.HTTPProber - log zerowrap.Logger + state out.AppStateReader + runtime out.ContainerRuntime + prober in.HTTPProber + log zerowrap.Logger } // NewService creates a new health service. func NewService( - configSvc in.ConfigService, - containerSvc in.ContainerService, + state out.AppStateReader, + runtime out.ContainerRuntime, prober in.HTTPProber, log zerowrap.Logger, ) *Service { return &Service{ - configSvc: configSvc, - containerSvc: containerSvc, - prober: prober, - log: log, + state: state, + runtime: runtime, + prober: prober, + log: log, } } -// CheckRoute performs a health check on a single route. -func (s *Service) CheckRoute(ctx context.Context, route domain.Route) *domain.RouteHealth { +// CheckAllRoutes performs health checks on all app-served HTTP hosts. +// The result maps canonical host to health; stopped-intent apps are +// skipped (not routable, not healthy). +func (s *Service) CheckAllRoutes(ctx context.Context) map[string]*domain.RouteHealth { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "CheckRoute", - "domain": route.Domain, + zerowrap.FieldUseCase: "CheckAllRoutes", }) log := zerowrap.FromCtx(ctx) - health := &domain.RouteHealth{ - Domain: route.Domain, - ContainerStatus: "unknown", + targets := s.hostTargets(ctx) + results := make(map[string]*domain.RouteHealth, len(targets)) + + if len(targets) == 0 { + log.Debug().Msg("no app HTTP hosts to check") + return results } - // Validate domain before probing - prevents SSRF via invalid route domains - if !domain.IsValidRouteDomain(route.Domain) { - health.Error = "invalid route domain for health probe" - log.Debug().Msg("health check blocked for invalid domain") - return health + // Check hosts concurrently with semaphore to limit resource usage + var mu sync.Mutex + var wg sync.WaitGroup + sem := make(chan struct{}, maxConcurrentProbes) + + for _, target := range targets { + wg.Add(1) + go func(t hostTarget) { + defer wg.Done() + sem <- struct{}{} // Acquire semaphore + defer func() { <-sem }() // Release semaphore + + health := s.checkHost(ctx, t) + mu.Lock() + results[t.host] = health + mu.Unlock() + }(target) } - // Check container status - container, exists := s.containerSvc.Get(ctx, route.Domain) - if !exists || container == nil { - health.ContainerStatus = "not found" - health.Error = "container not found" - log.Debug().Msg("container not found") - return health + wg.Wait() + + log.Debug().Int("routes_checked", len(results)).Msg("all health checks complete") + return results +} + +// hostTarget is one probed HTTP host. +type hostTarget struct { + host string + backend domain.AppBackend +} + +// hostTargets enumerates probed hosts from ACTIVE state, skipping +// stopped-intent apps. +func (s *Service) hostTargets(ctx context.Context) []hostTarget { + apps, err := s.state.ListApps(ctx) + if err != nil { + return nil + } + var targets []hostTarget + for _, app := range apps { + intent, err := s.state.LoadIntent(ctx, app) + if err != nil || intent.Stopped { + continue + } + active, ok, err := s.state.LoadActive(ctx, app) + if err != nil || !ok { + continue + } + for _, eff := range active.Services { + for _, h := range eff.Spec.HTTP { + if h.Host == "" { + continue + } + targets = append(targets, hostTarget{host: h.Host, backend: eff.BackendFor(h.Port)}) + } + } } + return targets +} + +// checkHost probes one host's recorded loopback backend. +func (s *Service) checkHost(ctx context.Context, target hostTarget) *domain.RouteHealth { + ctx = zerowrap.CtxWithFields(ctx, map[string]any{ + zerowrap.FieldLayer: "usecase", + zerowrap.FieldUseCase: "CheckHost", + "domain": target.host, + }) + log := zerowrap.FromCtx(ctx) - health.ContainerStatus = container.Status + health := &domain.RouteHealth{ + Domain: target.host, + ContainerStatus: "unknown", + } - // Only probe HTTP if container is running - if container.Status != string(domain.ContainerStatusRunning) { - health.Error = fmt.Sprintf("container is %s", container.Status) - log.Debug().Str("status", container.Status).Msg("container not running, skipping HTTP probe") + if !target.backend.Resolved() { + health.ContainerStatus = "not found" + health.Error = "no recorded backend bind for host" + log.Debug().Msg("host has no resolved backend") return health } - // Probe HTTP endpoint - scheme := "http" - if route.HTTPS { - scheme = "https" + running, err := s.runtime.IsContainerRunning(ctx, target.backend.ContainerID) + if err != nil || !running { + health.ContainerStatus = "not running" + health.Error = "container not running" + log.Debug().Msg("container not running, skipping HTTP probe") + return health } - url := fmt.Sprintf("%s://%s/", scheme, route.Domain) + health.ContainerStatus = string(domain.ContainerStatusRunning) + + url := fmt.Sprintf("http://127.0.0.1:%d/", target.backend.Port) statusCode, responseTime, err := s.prober.Probe(ctx, url) if err != nil { health.Error = err.Error() @@ -102,44 +169,3 @@ func (s *Service) CheckRoute(ctx context.Context, route domain.Route) *domain.Ro return health } - -// CheckAllRoutes performs health checks on all configured routes. -func (s *Service) CheckAllRoutes(ctx context.Context) map[string]*domain.RouteHealth { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "CheckAllRoutes", - }) - log := zerowrap.FromCtx(ctx) - - routes := s.configSvc.GetRoutes(ctx) - results := make(map[string]*domain.RouteHealth, len(routes)) - - if len(routes) == 0 { - log.Debug().Msg("no routes configured") - return results - } - - // Check routes concurrently with semaphore to limit resource usage - var mu sync.Mutex - var wg sync.WaitGroup - sem := make(chan struct{}, maxConcurrentProbes) - - for _, route := range routes { - wg.Add(1) - go func(r domain.Route) { - defer wg.Done() - sem <- struct{}{} // Acquire semaphore - defer func() { <-sem }() // Release semaphore - - health := s.CheckRoute(ctx, r) - mu.Lock() - results[r.Domain] = health - mu.Unlock() - }(route) - } - - wg.Wait() - - log.Debug().Int("routes_checked", len(results)).Msg("all health checks complete") - return results -} diff --git a/internal/usecase/health/service_test.go b/internal/usecase/health/service_test.go index 3a468fd2b..610b6c02c 100644 --- a/internal/usecase/health/service_test.go +++ b/internal/usecase/health/service_test.go @@ -9,7 +9,8 @@ import ( "github.com/stretchr/testify/assert" "github.com/stretchr/testify/mock" - "github.com/bnema/gordon/internal/boundaries/in/mocks" + inmocks "github.com/bnema/gordon/internal/boundaries/in/mocks" + outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" "github.com/bnema/gordon/internal/domain" ) @@ -17,91 +18,41 @@ func testLogger() zerowrap.Logger { return zerowrap.Default() } -func TestService_CheckRoute_ContainerNotFound(t *testing.T) { - configSvc := mocks.NewMockConfigService(t) - containerSvc := mocks.NewMockContainerService(t) - prober := mocks.NewMockHTTPProber(t) - - containerSvc.EXPECT().Get(mock.Anything, "app.example.com").Return(nil, false) - - svc := NewService(configSvc, containerSvc, prober, testLogger()) - route := domain.Route{Domain: "app.example.com", Image: "myapp:latest", HTTPS: true} - - health := svc.CheckRoute(context.Background(), route) - - assert.Equal(t, "app.example.com", health.Domain) - assert.Equal(t, "not found", health.ContainerStatus) - assert.Equal(t, 0, health.HTTPStatus) - assert.False(t, health.Healthy) - assert.Equal(t, "container not found", health.Error) -} - -func TestService_CheckRoute_ContainerNotRunning(t *testing.T) { - configSvc := mocks.NewMockConfigService(t) - containerSvc := mocks.NewMockContainerService(t) - prober := mocks.NewMockHTTPProber(t) - - container := &domain.Container{ - ID: "abc123", - Status: "stopped", +func testActive() domain.AppActive { + return domain.AppActive{ + App: "blog", + Services: map[string]domain.AppEffectiveService{ + "web": { + EffectiveRevision: "rev-1", + Image: "img:1", + Container: "c-web", + BackendBinds: map[int]int{8080: 18080}, + Spec: domain.AppService{ + HTTP: []domain.AppHTTPInterface{{Host: "blog.example.com", Port: 8080, TLS: "auto"}}, + }, + }, + }, } - containerSvc.EXPECT().Get(mock.Anything, "app.example.com").Return(container, true) - - svc := NewService(configSvc, containerSvc, prober, testLogger()) - route := domain.Route{Domain: "app.example.com", Image: "myapp:latest", HTTPS: true} - - health := svc.CheckRoute(context.Background(), route) - - assert.Equal(t, "app.example.com", health.Domain) - assert.Equal(t, "stopped", health.ContainerStatus) - assert.Equal(t, 0, health.HTTPStatus) - assert.False(t, health.Healthy) - assert.Contains(t, health.Error, "container is stopped") } -func TestService_CheckRoute_HTTPProbeSuccess(t *testing.T) { - configSvc := mocks.NewMockConfigService(t) - containerSvc := mocks.NewMockContainerService(t) - prober := mocks.NewMockHTTPProber(t) - - container := &domain.Container{ - ID: "abc123", - Status: "running", - } - containerSvc.EXPECT().Get(mock.Anything, "app.example.com").Return(container, true) - prober.EXPECT().Probe(mock.Anything, "https://app.example.com/").Return(200, int64(45), nil) - - svc := NewService(configSvc, containerSvc, prober, testLogger()) - route := domain.Route{Domain: "app.example.com", Image: "myapp:latest", HTTPS: true} +func TestService_CheckAllRoutes_Healthy(t *testing.T) { + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + prober := inmocks.NewMockHTTPProber(t) - health := svc.CheckRoute(context.Background(), route) + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil) + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(testActive(), true, nil) + runtime.EXPECT().IsContainerRunning(mock.Anything, "c-web").Return(true, nil) + prober.EXPECT().Probe(mock.Anything, "http://127.0.0.1:18080/").Return(200, int64(45), nil) - assert.Equal(t, "app.example.com", health.Domain) - assert.Equal(t, "running", health.ContainerStatus) - assert.Equal(t, 200, health.HTTPStatus) - assert.Equal(t, int64(45), health.ResponseTimeMs) - assert.True(t, health.Healthy) - assert.Empty(t, health.Error) -} - -func TestService_CheckRoute_HTTPOnlyRouteUsesHTTP(t *testing.T) { - configSvc := mocks.NewMockConfigService(t) - containerSvc := mocks.NewMockContainerService(t) - prober := mocks.NewMockHTTPProber(t) - - container := &domain.Container{ - ID: "abc123", - Status: "running", - } - containerSvc.EXPECT().Get(mock.Anything, "app.example.com").Return(container, true) - prober.EXPECT().Probe(mock.Anything, "http://app.example.com/").Return(200, int64(45), nil) + svc := NewService(state, runtime, prober, testLogger()) - svc := NewService(configSvc, containerSvc, prober, testLogger()) - route := domain.Route{Domain: "app.example.com", Image: "myapp:latest", HTTPS: false} - - health := svc.CheckRoute(context.Background(), route) + results := svc.CheckAllRoutes(context.Background()) - assert.Equal(t, "app.example.com", health.Domain) + assert.Len(t, results, 1) + health := results["blog.example.com"] + assert.Equal(t, "blog.example.com", health.Domain) assert.Equal(t, "running", health.ContainerStatus) assert.Equal(t, 200, health.HTTPStatus) assert.Equal(t, int64(45), health.ResponseTimeMs) @@ -109,135 +60,71 @@ func TestService_CheckRoute_HTTPOnlyRouteUsesHTTP(t *testing.T) { assert.Empty(t, health.Error) } -func TestService_CheckRoute_HTTPProbeFailure(t *testing.T) { - configSvc := mocks.NewMockConfigService(t) - containerSvc := mocks.NewMockContainerService(t) - prober := mocks.NewMockHTTPProber(t) +func TestService_CheckAllRoutes_ContainerNotRunning(t *testing.T) { + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + prober := inmocks.NewMockHTTPProber(t) - container := &domain.Container{ - ID: "abc123", - Status: "running", - } - containerSvc.EXPECT().Get(mock.Anything, "app.example.com").Return(container, true) - prober.EXPECT().Probe(mock.Anything, "https://app.example.com/").Return(0, int64(0), errors.New("connection refused")) + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil) + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(testActive(), true, nil) + runtime.EXPECT().IsContainerRunning(mock.Anything, "c-web").Return(false, nil) - svc := NewService(configSvc, containerSvc, prober, testLogger()) - route := domain.Route{Domain: "app.example.com", Image: "myapp:latest", HTTPS: true} + svc := NewService(state, runtime, prober, testLogger()) - health := svc.CheckRoute(context.Background(), route) + results := svc.CheckAllRoutes(context.Background()) - assert.Equal(t, "app.example.com", health.Domain) - assert.Equal(t, "running", health.ContainerStatus) - assert.Equal(t, 0, health.HTTPStatus) - assert.False(t, health.Healthy) - assert.Equal(t, "connection refused", health.Error) + assert.Len(t, results, 1) + assert.Equal(t, "not running", results["blog.example.com"].ContainerStatus) + assert.False(t, results["blog.example.com"].Healthy) } -func TestService_CheckRoute_HTTPStatus5xx(t *testing.T) { - configSvc := mocks.NewMockConfigService(t) - containerSvc := mocks.NewMockContainerService(t) - prober := mocks.NewMockHTTPProber(t) +func TestService_CheckAllRoutes_ProbeFailure(t *testing.T) { + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + prober := inmocks.NewMockHTTPProber(t) - container := &domain.Container{ - ID: "abc123", - Status: "running", - } - containerSvc.EXPECT().Get(mock.Anything, "app.example.com").Return(container, true) - prober.EXPECT().Probe(mock.Anything, "https://app.example.com/").Return(502, int64(100), nil) + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil) + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog"}, nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(testActive(), true, nil) + runtime.EXPECT().IsContainerRunning(mock.Anything, "c-web").Return(true, nil) + prober.EXPECT().Probe(mock.Anything, "http://127.0.0.1:18080/").Return(0, int64(0), errors.New("connection refused")) - svc := NewService(configSvc, containerSvc, prober, testLogger()) - route := domain.Route{Domain: "app.example.com", Image: "myapp:latest", HTTPS: true} + svc := NewService(state, runtime, prober, testLogger()) - health := svc.CheckRoute(context.Background(), route) + results := svc.CheckAllRoutes(context.Background()) - assert.Equal(t, "app.example.com", health.Domain) - assert.Equal(t, "running", health.ContainerStatus) - assert.Equal(t, 502, health.HTTPStatus) - assert.Equal(t, int64(100), health.ResponseTimeMs) - assert.False(t, health.Healthy) // 5xx is not healthy - assert.Empty(t, health.Error) + assert.Len(t, results, 1) + assert.Equal(t, "running", results["blog.example.com"].ContainerStatus) + assert.False(t, results["blog.example.com"].Healthy) + assert.Equal(t, "connection refused", results["blog.example.com"].Error) } -func TestService_CheckAllRoutes(t *testing.T) { - configSvc := mocks.NewMockConfigService(t) - containerSvc := mocks.NewMockContainerService(t) - prober := mocks.NewMockHTTPProber(t) - - routes := []domain.Route{ - {Domain: "app1.example.com", Image: "app1:latest", HTTPS: true}, - {Domain: "app2.example.com", Image: "app2:latest", HTTPS: true}, - } - configSvc.EXPECT().GetRoutes(mock.Anything).Return(routes) +func TestService_CheckAllRoutes_StoppedIntentSkipped(t *testing.T) { + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + prober := inmocks.NewMockHTTPProber(t) - container1 := &domain.Container{ID: "abc", Status: "running"} - container2 := &domain.Container{ID: "def", Status: "stopped"} - containerSvc.EXPECT().Get(mock.Anything, "app1.example.com").Return(container1, true) - containerSvc.EXPECT().Get(mock.Anything, "app2.example.com").Return(container2, true) + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil) + state.EXPECT().LoadIntent(mock.Anything, "blog").Return(domain.AppStopIntent{App: "blog", Stopped: true}, nil) - // Only running containers get probed - prober.EXPECT().Probe(mock.Anything, "https://app1.example.com/").Return(200, int64(30), nil) - - svc := NewService(configSvc, containerSvc, prober, testLogger()) + svc := NewService(state, runtime, prober, testLogger()) results := svc.CheckAllRoutes(context.Background()) - assert.Len(t, results, 2) - - assert.NotNil(t, results["app1.example.com"]) - assert.Equal(t, "running", results["app1.example.com"].ContainerStatus) - assert.Equal(t, 200, results["app1.example.com"].HTTPStatus) - assert.True(t, results["app1.example.com"].Healthy) - - assert.NotNil(t, results["app2.example.com"]) - assert.Equal(t, "stopped", results["app2.example.com"].ContainerStatus) - assert.Equal(t, 0, results["app2.example.com"].HTTPStatus) - assert.False(t, results["app2.example.com"].Healthy) + assert.Empty(t, results) } -func TestService_CheckAllRoutes_NoRoutes(t *testing.T) { - configSvc := mocks.NewMockConfigService(t) - containerSvc := mocks.NewMockContainerService(t) - prober := mocks.NewMockHTTPProber(t) +func TestService_CheckAllRoutes_NoApps(t *testing.T) { + state := outmocks.NewMockAppState(t) + runtime := outmocks.NewMockContainerRuntime(t) + prober := inmocks.NewMockHTTPProber(t) - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{}) + state.EXPECT().ListApps(mock.Anything).Return(nil, nil) - svc := NewService(configSvc, containerSvc, prober, testLogger()) + svc := NewService(state, runtime, prober, testLogger()) results := svc.CheckAllRoutes(context.Background()) assert.Empty(t, results) } - -func TestService_CheckRoute_InvalidDomain_Blocked(t *testing.T) { - configSvc := mocks.NewMockConfigService(t) - containerSvc := mocks.NewMockContainerService(t) - prober := mocks.NewMockHTTPProber(t) - - // No container lookup should happen for invalid domains - // No probe should happen either - - svc := NewService(configSvc, containerSvc, prober, testLogger()) - - tests := []struct { - name string - domain string - }{ - {"IP address", "192.168.1.1"}, - {"localhost", "localhost"}, - {"local TLD", "myapp.local"}, - {"internal TLD", "service.internal"}, - {"with port", "example.com:8080"}, - {"IPv6", "::1"}, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - route := domain.Route{Domain: tt.domain, Image: "myapp:latest"} - health := svc.CheckRoute(context.Background(), route) - - assert.Equal(t, tt.domain, health.Domain) - assert.False(t, health.Healthy) - assert.Equal(t, "invalid route domain for health probe", health.Error) - }) - } -} diff --git a/internal/usecase/images/prune.go b/internal/usecase/images/prune.go new file mode 100644 index 000000000..a448f14b8 --- /dev/null +++ b/internal/usecase/images/prune.go @@ -0,0 +1,692 @@ +package images + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "fmt" + "sort" + "time" + + "github.com/bnema/zerowrap" + + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/pruneguard" + "github.com/bnema/gordon/internal/usecase/registrystate" +) + +// Supported OCI/Docker manifest media types. Anything else makes the +// content behind it unknown rather than unreferenced. +const ( + mediaTypeDockerManifest = "application/vnd.docker.distribution.manifest.v2+json" + mediaTypeDockerManifestList = "application/vnd.docker.distribution.manifest.list.v2+json" + mediaTypeOCIManifest = "application/vnd.oci.image.manifest.v1+json" + mediaTypeOCIImageIndex = "application/vnd.oci.image.index.v1+json" +) + +// WithPrunePorts wires the prune ports: the coherent protection store, +// the runtime inventory/deletion port, and the GC barrier. All three +// are required for prune; a missing port fails closed. +func (s *Service) WithPrunePorts(protection out.PruneProtectionStore, pruneRuntime out.PruneRuntime, barrier out.GCBarrier) *Service { + s.protection = protection + s.pruneRuntime = pruneRuntime + s.barrier = barrier + return s +} + +// acquireExclusive takes the GC exclusive lease for the whole prune run, +// or a no-op lease when no barrier is wired. +func (s *Service) acquireExclusive(ctx context.Context) (out.GCLease, error) { + if s.barrier == nil { + return noopGCLease{}, nil + } + return s.barrier.AcquireExclusive(ctx) +} + +// noopGCLease is the lease used when no barrier is wired. +type noopGCLease struct{} + +func (noopGCLease) Release() {} + +// Prune plans and (unless DryRun) executes one image prune run. +// +// The whole plan is computed and validated before anything is deleted. +// A valid plan with zero eligible resources succeeds: protected and +// unknown candidates are normal outcomes, never an operation-wide +// refusal. +func (s *Service) Prune(ctx context.Context, opts domain.ImagePruneOptions) (domain.ImagePruneReport, error) { + ctx = zerowrap.CtxWithFields(ctx, map[string]any{ + zerowrap.FieldLayer: "usecase", + zerowrap.FieldUseCase: "Prune", + "keepLast": opts.KeepLast, + "pruneDangling": opts.PruneDangling, + "pruneRegistry": opts.PruneRegistry, + "dryRun": opts.DryRun, + }) + log := zerowrap.FromCtx(ctx) + + lease, err := s.acquireExclusive(ctx) + if err != nil { + return domain.ImagePruneReport{}, fmt.Errorf("images: acquire prune lease: %w", err) + } + defer lease.Release() + + if s.protection == nil { + return domain.ImagePruneReport{}, fmt.Errorf("images: prune requires a protection store: %w", domain.ErrPruneDisabled) + } + + snapshot, err := s.protection.ProtectionSnapshot(ctx) + if err != nil { + return domain.ImagePruneReport{}, fmt.Errorf("images: read protection snapshot: %w", err) + } + + plan := &domain.PrunePlan{Gaps: append([]domain.InventoryGap(nil), snapshot.Gaps...)} + + if opts.PruneDangling { + verdicts, gaps, err := s.planRuntimeImages(ctx, snapshot) + if err != nil { + return domain.ImagePruneReport{}, err + } + plan.Images = verdicts + plan.Gaps = append(plan.Gaps, gaps...) + } + + // Registry planning and deletion share one registry mutation lock, + // so a concurrent push can never make content look unreferenced + // between the plan and the deletion it authorizes. Runtime deletion + // runs outside that lock: it does not touch registry storage, and + // holding the lock across it would stall every registry push. + if !opts.PruneRegistry { + return s.finishPlan(ctx, plan, opts.DryRun, log) + } + + s.mutationMu.Lock() + tags, blobs, gaps := s.planRegistry(snapshot, opts.KeepLast) + plan.Tags = tags + plan.Blobs = blobs + plan.Gaps = append(plan.Gaps, gaps...) + + if err := plan.Validate(); err != nil { + s.mutationMu.Unlock() + return domain.ImagePruneReport{}, fmt.Errorf("images: refusing to execute an invalid prune plan: %w", err) + } + report := domain.ImagePruneReport{Plan: *domain.NewPruneReport(plan, !opts.DryRun)} + if opts.DryRun { + s.mutationMu.Unlock() + return report, nil + } + if err := s.executeRegistry(ctx, plan, &report, log); err != nil { + s.mutationMu.Unlock() + return report, err + } + s.mutationMu.Unlock() + + s.executeRuntimeImages(ctx, plan, &report, log) + return report, nil +} + +// finishPlan validates the complete plan and executes the runtime part. +// Every deletion happens after the whole plan proved itself well formed. +func (s *Service) finishPlan(ctx context.Context, plan *domain.PrunePlan, dryRun bool, log zerowrap.Logger) (domain.ImagePruneReport, error) { + if err := plan.Validate(); err != nil { + return domain.ImagePruneReport{}, fmt.Errorf("images: refusing to execute an invalid prune plan: %w", err) + } + report := domain.ImagePruneReport{Plan: *domain.NewPruneReport(plan, !dryRun)} + if dryRun { + return report, nil + } + s.executeRuntimeImages(ctx, plan, &report, log) + return report, nil +} + +// planRuntimeImages plans runtime image deletion from one coherent +// inventory. An unwired or unreadable runtime inventory yields gaps +// instead of an empty (apparently safe) runtime. +func (s *Service) planRuntimeImages(ctx context.Context, snapshot *domain.PruneProtectionSnapshot) ([]domain.RuntimeImageVerdict, []domain.InventoryGap, error) { + if s.pruneRuntime == nil { + return nil, []domain.InventoryGap{{ + Source: domain.InventorySourceRuntimeImages, + Reason: domain.PruneReasonUnknownInventory, + Detail: "runtime prune port is not wired", + }}, nil + } + inventory, err := s.pruneRuntime.InventoryRuntime(ctx) + if err != nil { + return nil, nil, fmt.Errorf("images: read runtime inventory: %w", err) + } + verdicts := pruneguard.PlanRuntimeImages(pruneguard.RuntimePlanInput{ + Images: inventory.Images, + Containers: inventory.Containers, + Snapshot: snapshot, + ProtectedContainerIDs: snapshot.RefsOfKind(domain.ProtectionRecoveryInhibition), + Complete: inventory.Complete(), + }) + return verdicts, inventory.Gaps, nil +} + +// executeRuntimeImages deletes exactly the planned runtime image IDs. +// A single deletion failure is isolated: it is reported and the run +// continues with the remaining candidates. +func (s *Service) executeRuntimeImages(ctx context.Context, plan *domain.PrunePlan, report *domain.ImagePruneReport, log zerowrap.Logger) { + for _, ref := range plan.EligibleImages() { + if err := s.pruneRuntime.RemoveImageExact(ctx, ref); err != nil { + log.Warn().Err(err).Str("image", ref.ID).Msg("runtime image prune failed; skipping") + report.Plan.Failures = append(report.Plan.Failures, domain.PruneFailure{ + Kind: domain.PruneResourceRuntimeImage, Ref: ref.ID, Err: err.Error(), + }) + continue + } + report.Plan.Deleted = append(report.Plan.Deleted, domain.PruneCandidateReport{ + Kind: domain.PruneResourceRuntimeImage, Ref: ref.ID, Verdict: domain.PruneVerdictEligible, + }) + report.Runtime.DeletedCount++ + } +} + +// planRegistry plans tag retention and blob garbage collection from the +// registry inventory. It never deletes: the caller executes only after +// the complete plan validates. +// +// The returned gaps record every fact that could not be read. A tag +// whose manifest is unreadable stays unknown, and a blob whose +// reference set cannot be fully proven stays unknown too. +func (s *Service) planRegistry(snapshot *domain.PruneProtectionSnapshot, keepLast int) ([]domain.RegistryTagVerdict, []domain.OCIBlobVerdict, []domain.InventoryGap) { + if keepLast <= 0 { + // Documented behavior: keep_last=0 skips registry tag and blob + // cleanup entirely. + return nil, nil, nil + } + + repositories, err := s.manifestStorage.ListRepositories() + if err != nil { + return nil, nil, []domain.InventoryGap{{ + Source: domain.InventorySourceRegistry, + Reason: domain.PruneReasonUnknownInventory, + Detail: "list repositories failed", + }} + } + sort.Strings(repositories) + + var tags []pruneguard.RegistryTagCandidate + var gaps []domain.InventoryGap + for _, repository := range repositories { + repositoryTags, repositoryGaps := s.inventoryRepository(repository) + tags = append(tags, repositoryTags...) + gaps = append(gaps, repositoryGaps...) + } + + complete := len(gaps) == 0 + verdicts := pruneguard.PlanRegistryTags(pruneguard.RegistryPlanInput{ + Tags: tags, + Snapshot: snapshot, + KeepLast: keepLast, + Complete: complete, + }) + + blobs, blobGaps := s.planRegistryBlobs(snapshot, verdicts, tags, complete, repositories) + gaps = append(gaps, blobGaps...) + return verdicts, blobs, gaps +} + +// inventoryRepository inventories one repository's tags and their +// manifests. A tag whose manifest cannot be read is kept as a candidate +// with ManifestKnown=false, so the planner fails closed on it. +func (s *Service) inventoryRepository(repository string) ([]pruneguard.RegistryTagCandidate, []domain.InventoryGap) { + tagNames, err := s.manifestStorage.ListTags(repository) + if err != nil { + return nil, []domain.InventoryGap{{ + Source: domain.InventorySourceRegistry, + Reason: domain.PruneReasonUnknownInventory, + Detail: "list tags failed for " + repository, + }} + } + + sort.Strings(tagNames) + candidates := make([]pruneguard.RegistryTagCandidate, 0, len(tagNames)) + var gaps []domain.InventoryGap + for _, tagName := range tagNames { + ref := domain.RegistryTagRef{Repository: repository, Tag: tagName} + if !ref.Valid() { + // A tag we cannot name is a tag whose content we cannot + // prove unreferenced, so the whole registry plan fails closed. + gaps = append(gaps, domain.InventoryGap{ + Source: domain.InventorySourceRegistry, + Reason: domain.PruneReasonUnknownIdentity, + Detail: "unnamed or malformed tag in " + repository, + }) + continue + } + candidate := pruneguard.RegistryTagCandidate{Ref: ref} + modTime, err := s.manifestStorage.GetManifestModTime(repository, tagName) + if err != nil { + gaps = append(gaps, domain.InventoryGap{ + Source: domain.InventorySourceRegistry, + Reason: domain.PruneReasonUnknownInventory, + Detail: "manifest modification time unreadable for " + ref.String(), + }) + } + candidate.ModTime = modTime + + data, _, err := s.manifestStorage.GetManifest(repository, tagName) + if err != nil { + gaps = append(gaps, domain.InventoryGap{ + Source: domain.InventorySourceManifests, + Reason: domain.PruneReasonUnknownManifest, + Detail: "manifest unreadable for " + ref.String(), + }) + candidates = append(candidates, candidate) + continue + } + digest := digestOfManifest(data) + _, mediaType, err := parseManifestDocument(data) + if err != nil { + gaps = append(gaps, domain.InventoryGap{ + Source: domain.InventorySourceManifests, + Reason: domain.PruneReasonUnknownManifest, + Detail: "manifest corrupt for " + ref.String(), + }) + candidates = append(candidates, candidate) + continue + } + if !supportedManifestMediaType(mediaType) { + gaps = append(gaps, domain.InventoryGap{ + Source: domain.InventorySourceManifests, + Reason: domain.PruneReasonUnknownMediaType, + Detail: "unsupported manifest media type for " + ref.String(), + }) + candidates = append(candidates, candidate) + continue + } + candidate.Digest = digest + candidate.ManifestKnown = true + candidates = append(candidates, candidate) + } + return candidates, gaps +} + +// planRegistryBlobs computes one verdict per stored blob. +// +// Blobs reachable from the retained (protected or kept) manifest set +// are protected. When any retained closure could not be read or parsed +// completely, unreferenced blobs become unknown instead of eligible: an +// incomplete reference set is never proof of unreachability. +func (s *Service) planRegistryBlobs( + snapshot *domain.PruneProtectionSnapshot, + verdicts []domain.RegistryTagVerdict, + tags []pruneguard.RegistryTagCandidate, + inventoryComplete bool, + repositories []string, +) ([]domain.OCIBlobVerdict, []domain.InventoryGap) { + now := time.Now().UTC() + + // Roots are every tag the plan does not delete: retained tags and + // any tag whose manifest could not be read at all (which has no + // verdict entry yet still must keep protecting its content). + planned := make(map[string]struct{}, len(verdicts)) + var roots []retainedRoot + for _, verdict := range verdicts { + planned[verdict.Ref.Key()] = struct{}{} + if verdict.Verdict == domain.PruneVerdictEligible { + continue + } + roots = append(roots, retainedRoot{Repository: verdict.Ref.Repository, Reference: verdict.Ref.Tag}) + } + for _, tag := range tags { + if _, hasVerdict := planned[tag.Ref.Key()]; hasVerdict { + continue + } + roots = append(roots, retainedRoot{Repository: tag.Ref.Repository, Reference: tag.Ref.Tag}) + } + + closure := newClosure() + // A tag the inventory could not read may reference any blob, so an + // incomplete tag inventory makes the whole reference set unprovable. + closure.complete = inventoryComplete + for _, root := range roots { + s.traverseClosure(root, closure) + } + // Durable roots qualified by repository and digest keep the FULL OCI + // closure of their manifest: a moved tag must not let prune delete the + // config or layers of a digest an active deployment still pins. Roots + // whose repository has no local content are skipped: the image was + // never pushed here, so there is no local closure to preserve. An + // unresolved root in a local repository marks the closure incomplete, + // so unreferenced blobs stay unknown instead of eligible. + localRepositories := make(map[string]struct{}, len(repositories)) + for _, repository := range repositories { + localRepositories[repository] = struct{}{} + } + for _, root := range snapshot.Roots { + if root.Repository == "" || !domain.ValidOCIDigest(root.Ref) { + continue + } + if _, local := localRepositories[root.Repository]; !local { + continue + } + s.traverseClosure(retainedRoot{Repository: root.Repository, Reference: root.Ref}, closure) + } + + // Durable roots claim digests directly as well as through tag + // closure: an app service, apply intent, or unfinished operation + // can pin content by digest alone. Those digests must never be + // judged unreferenced just because no retained tag reaches them. + snapshotComplete := snapshot.Complete() + + pending := s.registryState.PendingDigests(now) + + blobs, err := s.blobStorage.ListBlobs() + if err != nil { + return nil, append(closure.gaps, domain.InventoryGap{ + Source: domain.InventorySourceRegistry, + Reason: domain.PruneReasonUnknownInventory, + Detail: "list blobs failed", + }) + } + sort.Strings(blobs) + + gaps := closure.gaps + decisions := blobDecision{ + snapshot: snapshot, + closure: closure, + pending: pending, + now: now, + snapshotComplete: snapshotComplete, + } + out := make([]domain.OCIBlobVerdict, 0, len(blobs)) + for _, digest := range blobs { + out = append(out, s.decideBlob(digest, decisions)) + } + return out, gaps +} + +// blobDecision carries the shared inputs of one blob verdict. +type blobDecision struct { + snapshot *domain.PruneProtectionSnapshot + closure *closure + pending map[string]struct{} + now time.Time + snapshotComplete bool +} + +// decideBlob resolves one stored blob, strongest protection first. +func (s *Service) decideBlob(digest string, in blobDecision) domain.OCIBlobVerdict { + verdict := domain.OCIBlobVerdict{Ref: domain.OCIRef{Digest: digest}} + protect := func(reason domain.PruneReason) domain.OCIBlobVerdict { + verdict.Verdict = domain.PruneVerdictProtected + verdict.Reasons = domain.CanonicalReasons(reason) + return verdict + } + unknown := func(reason domain.PruneReason) domain.OCIBlobVerdict { + verdict.Verdict = domain.PruneVerdictUnknown + verdict.Reasons = domain.CanonicalReasons(reason) + return verdict + } + + if !verdict.Ref.Valid() { + return unknown(domain.PruneReasonUnknownIdentity) + } + // Durable state can claim a digest directly, without any retained + // tag reaching it. + if matched, reasons := rootReasonsForDigest(in.snapshot, digest); matched { + verdict.Verdict = domain.PruneVerdictProtected + verdict.Reasons = reasons + return verdict + } + if in.closure.references(digest) { + return protect(domain.PruneReasonProtectedSharedContent) + } + if _, isPending := in.pending[digest]; isPending { + return protect(domain.PruneReasonProtectedPendingUpload) + } + if !in.closure.complete || !in.snapshotComplete { + return unknown(domain.PruneReasonUnknownManifest) + } + // A blob that cannot be aged cannot be proven past the upload TTL, + // so it stays unknown instead of becoming eligible. + modTime, err := s.blobStorage.GetBlobModTime(digest) + if err != nil { + return unknown(domain.PruneReasonUnknownInventory) + } + if in.now.Sub(modTime) <= registrystate.PendingBlobTTL { + return protect(domain.PruneReasonProtectedPendingUpload) + } + verdict.Verdict = domain.PruneVerdictEligible + verdict.Reasons = domain.CanonicalReasons(domain.PruneReasonEligibleUnreferencedBlob) + return verdict +} + +// rootReasonsForDigest reports whether the durable snapshot claims the +// digest directly, and with which verdict reasons. +func rootReasonsForDigest(snapshot *domain.PruneProtectionSnapshot, digest string) (bool, []domain.PruneReason) { + if snapshot == nil { + return false, nil + } + matched := false + for _, root := range snapshot.Roots { + if root.Ref == digest { + matched = true + break + } + } + if !matched { + return false, nil + } + reasons := snapshot.RootReasons(digest) + if len(reasons) == 0 { + reasons = domain.CanonicalReasons(domain.PruneReasonProtectedSharedContent) + } + return true, reasons +} + +// retainedRoot is one manifest reference whose closure must survive. +type retainedRoot struct { + Repository string + Reference string +} + +// closure is the transitive reference set of every retained manifest. +type closure struct { + digests map[string]struct{} + visited map[string]struct{} + complete bool + gaps []domain.InventoryGap +} + +func newClosure() *closure { + return &closure{ + digests: make(map[string]struct{}), + visited: make(map[string]struct{}), + complete: true, + } +} + +func (c *closure) references(digest string) bool { + _, ok := c.digests[digest] + return ok +} + +// traverseClosure walks one retained reference transitively. A missing, +// corrupt, or unsupported node marks the closure incomplete: the caller +// then treats unreferenced blobs as unknown instead of eligible. +func (s *Service) traverseClosure(root retainedRoot, c *closure) { + visitKey := root.Repository + "\x00" + root.Reference + if _, seen := c.visited[visitKey]; seen { + return + } + c.visited[visitKey] = struct{}{} + + data, _, err := s.manifestStorage.GetManifest(root.Repository, root.Reference) + if err != nil { + c.complete = false + c.gaps = append(c.gaps, domain.InventoryGap{ + Source: domain.InventorySourceManifests, + Reason: domain.PruneReasonUnknownManifest, + Detail: "retained manifest unreadable for " + root.Repository + "@" + root.Reference, + }) + return + } + references, mediaType, err := parseManifestDocument(data) + if err != nil { + c.complete = false + c.gaps = append(c.gaps, domain.InventoryGap{ + Source: domain.InventorySourceManifests, + Reason: domain.PruneReasonUnknownManifest, + Detail: "retained manifest corrupt for " + root.Repository + "@" + root.Reference, + }) + return + } + if !supportedManifestMediaType(mediaType) { + c.complete = false + c.gaps = append(c.gaps, domain.InventoryGap{ + Source: domain.InventorySourceManifests, + Reason: domain.PruneReasonUnknownMediaType, + Detail: "retained manifest media type unsupported for " + root.Repository + "@" + root.Reference, + }) + return + } + + // The manifest body itself is addressable content: record its digest + // so a stored manifest copy is never judged unreferenced. + c.digests[digestOfManifest(data)] = struct{}{} + + for _, digest := range references.blobs { + c.digests[digest] = struct{}{} + } + for _, child := range references.childManifests { + c.digests[child] = struct{}{} + s.traverseClosure(retainedRoot{Repository: root.Repository, Reference: child}, c) + } +} + +// executeRegistry deletes exactly the planned tag references and then +// exactly the planned unreferenced blobs. It runs after the complete +// plan has been validated, never before. +func (s *Service) executeRegistry(ctx context.Context, plan *domain.PrunePlan, report *domain.ImagePruneReport, log zerowrap.Logger) error { + for _, ref := range plan.EligibleTags() { + if err := s.manifestStorage.DeleteManifest(ref.Repository, ref.Tag); err != nil { + log.Warn().Err(err).Str("tag", ref.String()).Msg("failed to delete manifest, skipping") + report.Plan.Failures = append(report.Plan.Failures, domain.PruneFailure{ + Kind: domain.PruneResourceRegistryTag, Ref: ref.String(), Err: err.Error(), + }) + continue + } + report.Plan.Deleted = append(report.Plan.Deleted, domain.PruneCandidateReport{ + Kind: domain.PruneResourceRegistryTag, Ref: ref.String(), Verdict: domain.PruneVerdictEligible, + }) + report.Registry.TagsRemoved++ + } + + for _, ref := range plan.EligibleBlobs() { + size, err := s.blobStorage.DeleteBlob(ref.Digest) + if err != nil { + log.Warn().Err(err).Str("digest", ref.Digest).Msg("failed to delete blob, skipping") + report.Plan.Failures = append(report.Plan.Failures, domain.PruneFailure{ + Kind: domain.PruneResourceOCIBlob, Ref: ref.String(), Err: err.Error(), + }) + continue + } + report.Plan.Deleted = append(report.Plan.Deleted, domain.PruneCandidateReport{ + Kind: domain.PruneResourceOCIBlob, Ref: ref.String(), Verdict: domain.PruneVerdictEligible, + }) + report.Registry.BlobsRemoved++ + // Each stored blob is deleted once, so its size is exact. + report.Registry.SpaceReclaimed += size + report.Plan.ReclaimedBytes += size + report.Plan.ReclaimedKnown = true + } + + uploadsRemoved, uploadBytes, err := s.blobStorage.CleanupStaleUploads(registrystate.PendingBlobTTL) + if err != nil { + log.Warn().Err(err).Msg("failed to clean up stale uploads, continuing") + return nil + } + report.Registry.UploadsRemoved = uploadsRemoved + report.Registry.UploadSpaceReclaimed = uploadBytes + return nil +} + +// manifestReferences is the parsed reference set of one manifest. +type manifestReferences struct { + blobs []string + childManifests []string +} + +// parseManifestDocument parses one manifest and reports its media type. +func parseManifestDocument(data []byte) (manifestReferences, string, error) { + var document struct { + MediaType string `json:"mediaType"` + SchemaVersion int `json:"schemaVersion"` + Config struct { + Digest string `json:"digest"` + } `json:"config"` + Layers []struct { + Digest string `json:"digest"` + } `json:"layers"` + Manifests []struct { + Digest string `json:"digest"` + } `json:"manifests"` + Subject *struct { + Digest string `json:"digest"` + } `json:"subject"` + } + if err := json.Unmarshal(data, &document); err != nil { + return manifestReferences{}, "", err + } + + refs := manifestReferences{ + blobs: make([]string, 0, len(document.Layers)+1), + childManifests: make([]string, 0, len(document.Manifests)+1), + } + if document.Config.Digest != "" { + refs.blobs = append(refs.blobs, document.Config.Digest) + } + for _, layer := range document.Layers { + if layer.Digest != "" { + refs.blobs = append(refs.blobs, layer.Digest) + } + } + for _, child := range document.Manifests { + if child.Digest != "" { + refs.childManifests = append(refs.childManifests, child.Digest) + } + } + if document.Subject != nil && document.Subject.Digest != "" { + refs.childManifests = append(refs.childManifests, document.Subject.Digest) + } + + mediaType := document.MediaType + if mediaType == "" { + // Older clients omit mediaType. Infer from the shape: a + // document with manifests is an index, one with layers and a + // config is a manifest. Anything else is unsupported. + switch { + case len(document.Manifests) > 0: + mediaType = mediaTypeOCIImageIndex + case document.Config.Digest != "" || len(document.Layers) > 0: + mediaType = mediaTypeOCIManifest + } + } + return refs, mediaType, nil +} + +// supportedManifestMediaType reports whether the closure walker can +// trust the reference set of a manifest with this media type. +func supportedManifestMediaType(mediaType string) bool { + switch mediaType { + case mediaTypeDockerManifest, mediaTypeDockerManifestList, mediaTypeOCIManifest, mediaTypeOCIImageIndex: + return true + default: + // digest-addressed content often carries a content type that + // does not name a media type; treat it as supported only when + // it is one of the known values above. + return false + } +} + +// digestOfManifest returns the sha256 digest of a stored manifest body. +func digestOfManifest(data []byte) string { + sum := sha256.Sum256(data) + return "sha256:" + hex.EncodeToString(sum[:]) +} diff --git a/internal/usecase/images/prune_failclosed_test.go b/internal/usecase/images/prune_failclosed_test.go new file mode 100644 index 000000000..a4dca4e13 --- /dev/null +++ b/internal/usecase/images/prune_failclosed_test.go @@ -0,0 +1,428 @@ +package images + +import ( + "context" + "errors" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/domain" +) + +// TestPrune_SnapshotGapFailsClosed proves an incomplete protection +// snapshot never authorizes deletion: an unreadable app record could be +// hiding a claim on any candidate, so every candidate stays unknown. +func TestPrune_SnapshotGapFailsClosed(t *testing.T) { + now := time.Date(2026, 3, 1, 12, 0, 0, 0, time.UTC) + manifests := newFakeManifestStorage() + manifests.repositories = []string{"app"} + manifests.tagsByRepo["app"] = []string{"latest", "v1", "v2"} + for index, tag := range []string{"latest", "v1", "v2"} { + manifests.modTimes[manifestRefKey("app", tag)] = now.Add(time.Duration(index-3) * time.Hour) + manifests.manifests[manifestRefKey("app", tag)] = mustManifestJSON(t, testDigest(tag), testDigest(tag+"-layer")) + } + + orphan := testDigest("orphan") + blobs := &fakeBlobStorage{ + blobs: []string{orphan}, + blobModTimes: map[string]time.Time{orphan: now.Add(-72 * time.Hour)}, + } + protection := &fakeProtectionStore{snapshot: &domain.PruneProtectionSnapshot{ + Gaps: []domain.InventoryGap{{ + Source: domain.InventorySourceAppState, + Reason: domain.PruneReasonUnknownInventory, + Detail: "app record unreadable", + }}, + }} + runtime := &fakePruneRuntime{inventory: &domain.RuntimeInventory{ + Images: []domain.RuntimeImage{{ID: "sha256:dangling", Labels: map[string]string{domain.LabelApp: "app"}}}, + }} + + svc, _ := newPruneService(t, manifests, blobs, protection, runtime) + report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{ + KeepLast: 1, PruneDangling: true, PruneRegistry: true, + }) + require.NoError(t, err) + + assert.Empty(t, manifests.deletedManifests, "an incomplete snapshot must not authorize tag deletion") + assert.Empty(t, blobs.deletedBlobs, "an incomplete snapshot must not authorize blob deletion") + assert.Empty(t, runtime.removedImages, "an incomplete snapshot must not authorize image deletion") + assert.Zero(t, report.Plan.CountByVerdict(domain.PruneVerdictEligible)) + // Only the explicitly retained window tags stay protected; every + // other candidate is unknown rather than eligible. + assert.Equal(t, 2, report.Plan.CountByVerdict(domain.PruneVerdictProtected)) + assert.Equal(t, len(report.Plan.Candidates)-2, report.Plan.CountByVerdict(domain.PruneVerdictUnknown)) +} + +// TestPrune_MovedTagKeepsActiveDigestClosure proves an ACTIVE digest root +// keeps the full OCI closure of its manifest: moving the mutable tag away +// must not let prune delete the config or layers of the pinned image. +func TestPrune_MovedTagKeepsActiveDigestClosure(t *testing.T) { + now := time.Date(2026, 3, 1, 12, 0, 0, 0, time.UTC) + pinned := testDigest("active-manifest") + configDigest := testDigest("active-config") + layerDigest := testDigest("active-layer") + + manifests := newFakeManifestStorage() + manifests.repositories = []string{"app"} + // The tag now points elsewhere; the pinned manifest survives untagged. + manifests.tagsByRepo["app"] = []string{"latest"} + manifests.modTimes[manifestRefKey("app", "latest")] = now + manifests.manifests[manifestRefKey("app", "latest")] = mustManifestJSON(t, testDigest("latest-config")) + manifests.manifests[manifestRefKey("app", pinned)] = mustManifestJSON(t, configDigest, layerDigest) + + blobs := &fakeBlobStorage{ + blobs: []string{configDigest, layerDigest, testDigest("orphan")}, + blobModTimes: map[string]time.Time{ + configDigest: now.Add(-72 * time.Hour), + layerDigest: now.Add(-72 * time.Hour), + testDigest("orphan"): now.Add(-72 * time.Hour), + }, + } + protection := &fakeProtectionStore{snapshot: &domain.PruneProtectionSnapshot{ + Roots: []domain.ProtectionRoot{{ + Kind: domain.ProtectionActiveService, Ref: pinned, Repository: "app", Owner: "app@active.web", + }}, + }} + + svc, _ := newPruneService(t, manifests, blobs, protection, &fakePruneRuntime{}) + report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{KeepLast: 1, PruneRegistry: true}) + require.NoError(t, err) + + for _, protected := range []string{configDigest, layerDigest} { + protectedVerdicts := report.Plan.CandidatesOfKind(domain.PruneResourceOCIBlob, domain.PruneVerdictProtected) + found := false + for _, verdict := range protectedVerdicts { + if verdict.Ref == protected { + found = true + } + } + assert.True(t, found, "blob %s of the pinned manifest must stay protected", protected) + } + assert.NotContains(t, blobs.deletedBlobs, configDigest) + assert.NotContains(t, blobs.deletedBlobs, layerDigest) +} + +// TestPrune_SubjectClosureProtectsManifestConfigAndLayers proves a retained +// artifact manifest keeps its subject manifest and that subject's blobs. +func TestPrune_SubjectClosureProtectsManifestConfigAndLayers(t *testing.T) { + now := time.Date(2026, 3, 1, 12, 0, 0, 0, time.UTC) + subject := testDigest("subject-manifest") + configDigest := testDigest("subject-config") + layerDigest := testDigest("subject-layer") + + manifests := newFakeManifestStorage() + manifests.repositories = []string{"app"} + manifests.tagsByRepo["app"] = []string{"attestation"} + manifests.modTimes[manifestRefKey("app", "attestation")] = now + manifests.manifests[manifestRefKey("app", "attestation")] = mustSubjectManifestJSON(t, subject) + manifests.manifests[manifestRefKey("app", subject)] = mustManifestJSON(t, configDigest, layerDigest) + + blobs := &fakeBlobStorage{ + blobs: []string{subject, configDigest, layerDigest}, + blobModTimes: map[string]time.Time{ + subject: now.Add(-72 * time.Hour), configDigest: now.Add(-72 * time.Hour), layerDigest: now.Add(-72 * time.Hour), + }, + } + svc, _ := newPruneService(t, manifests, blobs, &fakeProtectionStore{snapshot: &domain.PruneProtectionSnapshot{}}, &fakePruneRuntime{}) + report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{KeepLast: 1, PruneRegistry: true}) + require.NoError(t, err) + + assert.Empty(t, blobs.deletedBlobs) + for _, digest := range []string{subject, configDigest, layerDigest} { + assert.Contains(t, candidateRefs(report.Plan.CandidatesOfKind(domain.PruneResourceOCIBlob, domain.PruneVerdictProtected)), digest) + } +} + +func TestPrune_MissingSubjectFailsClosed(t *testing.T) { + now := time.Date(2026, 3, 1, 12, 0, 0, 0, time.UTC) + orphan := testDigest("orphan") + manifests := newFakeManifestStorage() + manifests.repositories = []string{"app"} + manifests.tagsByRepo["app"] = []string{"attestation"} + manifests.modTimes[manifestRefKey("app", "attestation")] = now + manifests.manifests[manifestRefKey("app", "attestation")] = mustSubjectManifestJSON(t, testDigest("missing-subject")) + blobs := &fakeBlobStorage{blobs: []string{orphan}, blobModTimes: map[string]time.Time{orphan: now.Add(-72 * time.Hour)}} + + svc, _ := newPruneService(t, manifests, blobs, &fakeProtectionStore{snapshot: &domain.PruneProtectionSnapshot{}}, &fakePruneRuntime{}) + report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{KeepLast: 1, PruneRegistry: true}) + require.NoError(t, err) + + assert.Empty(t, blobs.deletedBlobs) + assert.Contains(t, candidateRefs(report.Plan.CandidatesOfKind(domain.PruneResourceOCIBlob, domain.PruneVerdictUnknown)), orphan) +} + +func TestPrune_SubjectCycleTerminatesAndProtectsClosure(t *testing.T) { + now := time.Date(2026, 3, 1, 12, 0, 0, 0, time.UTC) + a := testDigest("subject-a") + b := testDigest("subject-b") + manifests := newFakeManifestStorage() + manifests.repositories = []string{"app"} + manifests.tagsByRepo["app"] = []string{"attestation"} + manifests.modTimes[manifestRefKey("app", "attestation")] = now + manifests.manifests[manifestRefKey("app", "attestation")] = mustSubjectManifestJSON(t, a) + manifests.manifests[manifestRefKey("app", a)] = mustSubjectManifestJSON(t, b) + manifests.manifests[manifestRefKey("app", b)] = mustSubjectManifestJSON(t, a) + blobs := &fakeBlobStorage{blobs: []string{a, b}, blobModTimes: map[string]time.Time{a: now.Add(-72 * time.Hour), b: now.Add(-72 * time.Hour)}} + + svc, _ := newPruneService(t, manifests, blobs, &fakeProtectionStore{snapshot: &domain.PruneProtectionSnapshot{}}, &fakePruneRuntime{}) + report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{KeepLast: 1, PruneRegistry: true}) + require.NoError(t, err) + + assert.Empty(t, blobs.deletedBlobs) + protected := candidateRefs(report.Plan.CandidatesOfKind(domain.PruneResourceOCIBlob, domain.PruneVerdictProtected)) + assert.Contains(t, protected, a) + assert.Contains(t, protected, b) +} + +func mustSubjectManifestJSON(t *testing.T, subject string) []byte { + t.Helper() + return []byte(`{"mediaType":"application/vnd.oci.image.manifest.v1+json","subject":{"digest":"` + subject + `"}}`) +} + +func candidateRefs(candidates []domain.PruneCandidateReport) []string { + refs := make([]string, 0, len(candidates)) + for _, candidate := range candidates { + refs = append(refs, candidate.Ref) + } + return refs +} + +// TestPrune_ExternalRepositoryDigestRootDoesNotDisablePruning proves a +// durable digest root whose repository has no local content does not make +// the registry closure incomplete: external images never introduce local +// content, so unrelated blobs stay eligible. +func TestPrune_ExternalRepositoryDigestRootDoesNotDisablePruning(t *testing.T) { + now := time.Date(2026, 3, 1, 12, 0, 0, 0, time.UTC) + orphan := testDigest("orphan") + + manifests := newFakeManifestStorage() + manifests.repositories = []string{"app"} + manifests.tagsByRepo["app"] = []string{"latest"} + manifests.modTimes[manifestRefKey("app", "latest")] = now + manifests.manifests[manifestRefKey("app", "latest")] = mustManifestJSON(t, testDigest("latest-config")) + + blobs := &fakeBlobStorage{ + blobs: []string{orphan}, + blobModTimes: map[string]time.Time{orphan: now.Add(-72 * time.Hour)}, + } + protection := &fakeProtectionStore{snapshot: &domain.PruneProtectionSnapshot{ + Roots: []domain.ProtectionRoot{{ + Kind: domain.ProtectionActiveService, Ref: testDigest("external-manifest"), + Repository: "docker.io/library/nginx", Owner: "app@active.web", + }}, + }} + + svc, _ := newPruneService(t, manifests, blobs, protection, &fakePruneRuntime{}) + report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{KeepLast: 1, PruneRegistry: true}) + require.NoError(t, err) + + assert.Equal(t, []string{orphan}, blobs.deletedBlobs, "an external root must not block local prune") + assert.Zero(t, report.Plan.CountByVerdict(domain.PruneVerdictUnknown)) +} + +// TestPrune_UnresolvedDigestRootFailsClosed proves a durable digest root +// whose manifest cannot be read makes the registry closure incomplete, so +// unrelated blobs stay unknown instead of being deleted. +func TestPrune_UnresolvedDigestRootFailsClosed(t *testing.T) { + now := time.Date(2026, 3, 1, 12, 0, 0, 0, time.UTC) + missing := testDigest("missing-manifest") + orphan := testDigest("orphan") + + manifests := newFakeManifestStorage() + manifests.repositories = []string{"app"} + manifests.tagsByRepo["app"] = []string{"latest"} + manifests.modTimes[manifestRefKey("app", "latest")] = now + manifests.manifests[manifestRefKey("app", "latest")] = mustManifestJSON(t, testDigest("latest-config")) + + blobs := &fakeBlobStorage{ + blobs: []string{orphan}, + blobModTimes: map[string]time.Time{orphan: now.Add(-72 * time.Hour)}, + } + protection := &fakeProtectionStore{snapshot: &domain.PruneProtectionSnapshot{ + Roots: []domain.ProtectionRoot{{ + Kind: domain.ProtectionActiveService, Ref: missing, Repository: "app", Owner: "app@active.web", + }}, + }} + + svc, _ := newPruneService(t, manifests, blobs, protection, &fakePruneRuntime{}) + report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{KeepLast: 1, PruneRegistry: true}) + require.NoError(t, err) + + assert.Empty(t, blobs.deletedBlobs, "an unresolved durable root must not authorize deletion") + unknown := report.Plan.CandidatesOfKind(domain.PruneResourceOCIBlob, domain.PruneVerdictUnknown) + require.Len(t, unknown, 1) + assert.Equal(t, orphan, unknown[0].Ref) +} + +// TestPrune_BlobClaimedByDigestRootSurvives proves a digest pinned +// directly by durable state (not reachable through any retained tag) +// is protected. +func TestPrune_BlobClaimedByDigestRootSurvives(t *testing.T) { + now := time.Date(2026, 3, 1, 12, 0, 0, 0, time.UTC) + pinned := testDigest("pinned-by-operation") + + manifests := newFakeManifestStorage() + manifests.repositories = []string{"app"} + manifests.tagsByRepo["app"] = []string{"latest", "v1"} + for index, tag := range []string{"latest", "v1"} { + manifests.modTimes[manifestRefKey("app", tag)] = now.Add(time.Duration(index) * time.Hour) + manifests.manifests[manifestRefKey("app", tag)] = mustManifestJSON(t, testDigest(tag)) + } + + blobs := &fakeBlobStorage{ + blobs: []string{pinned}, + blobModTimes: map[string]time.Time{pinned: now.Add(-72 * time.Hour)}, + } + protection := &fakeProtectionStore{snapshot: &domain.PruneProtectionSnapshot{ + Roots: []domain.ProtectionRoot{ + {Kind: domain.ProtectionOperation, Ref: pinned, Owner: "app@op-1"}, + }, + }} + + svc, _ := newPruneService(t, manifests, blobs, protection, &fakePruneRuntime{}) + report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{KeepLast: 1, PruneRegistry: true}) + require.NoError(t, err) + + assert.Empty(t, blobs.deletedBlobs, "a digest pinned by durable state must survive") + protected := report.Plan.CandidatesOfKind(domain.PruneResourceOCIBlob, domain.PruneVerdictProtected) + require.Len(t, protected, 1) + assert.Equal(t, []domain.PruneReason{domain.PruneReasonProtectedOperation}, protected[0].Reasons) +} + +// TestPrune_BlobWithoutModTimeStaysUnknown proves an unreadable blob age +// fails closed instead of falling through to eligibility: a blob that +// cannot be aged cannot be proven past the upload TTL. +func TestPrune_BlobWithoutModTimeStaysUnknown(t *testing.T) { + now := time.Date(2026, 3, 1, 12, 0, 0, 0, time.UTC) + unknownAge := testDigest("unknown-age") + + manifests := newFakeManifestStorage() + manifests.repositories = []string{"app"} + manifests.tagsByRepo["app"] = []string{"latest"} + manifests.modTimes[manifestRefKey("app", "latest")] = now + manifests.manifests[manifestRefKey("app", "latest")] = mustManifestJSON(t, testDigest("cfg")) + + blobs := &fakeBlobStorage{blobs: []string{unknownAge}, blobModTimeErr: errors.New("stat failed")} + svc, _ := newPruneService(t, manifests, blobs, &fakeProtectionStore{}, &fakePruneRuntime{}) + + report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{KeepLast: 1, PruneRegistry: true}) + require.NoError(t, err) + assert.Empty(t, blobs.deletedBlobs) + unknown := report.Plan.CandidatesOfKind(domain.PruneResourceOCIBlob, domain.PruneVerdictUnknown) + require.Len(t, unknown, 1) + assert.Equal(t, []domain.PruneReason{domain.PruneReasonUnknownInventory}, unknown[0].Reasons) +} + +// TestPrune_MalformedTagFailsClosed proves a tag whose identity cannot +// be established makes the registry plan incomplete rather than being +// skipped: skipping it would hide its content from the closure. +func TestPrune_MalformedTagFailsClosed(t *testing.T) { + now := time.Date(2026, 3, 1, 12, 0, 0, 0, time.UTC) + orphan := testDigest("orphan") + + manifests := newFakeManifestStorage() + manifests.repositories = []string{"app"} + manifests.tagsByRepo["app"] = []string{"latest", " "} + manifests.modTimes[manifestRefKey("app", "latest")] = now + manifests.manifests[manifestRefKey("app", "latest")] = mustManifestJSON(t, testDigest("cfg")) + + blobs := &fakeBlobStorage{ + blobs: []string{orphan}, + blobModTimes: map[string]time.Time{orphan: now.Add(-72 * time.Hour)}, + } + svc, _ := newPruneService(t, manifests, blobs, &fakeProtectionStore{}, &fakePruneRuntime{}) + + report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{KeepLast: 1, PruneRegistry: true}) + require.NoError(t, err) + assert.Empty(t, blobs.deletedBlobs) + assert.NotEmpty(t, report.Plan.Gaps) + assert.Zero(t, report.Plan.CountByVerdict(domain.PruneVerdictEligible)) +} + +// TestPrune_SecondExecutionIsIdempotent proves an executed plan is +// complete: re-running it finds nothing left to delete, and the second +// report carries no eligible candidate. +func TestPrune_SecondExecutionIsIdempotent(t *testing.T) { + now := time.Date(2026, 3, 1, 12, 0, 0, 0, time.UTC) + manifests := newFakeManifestStorage() + manifests.repositories = []string{"app"} + manifests.tagsByRepo["app"] = []string{"latest", "v1", "v2", "v3"} + for index, tag := range []string{"latest", "v1", "v2", "v3"} { + manifests.modTimes[manifestRefKey("app", tag)] = now.Add(time.Duration(index-4) * time.Hour) + manifests.manifests[manifestRefKey("app", tag)] = mustManifestJSON(t, testDigest(tag), testDigest(tag+"-layer")) + } + + orphan := testDigest("orphan") + blobs := &fakeBlobStorage{ + blobs: []string{orphan}, + blobModTimes: map[string]time.Time{orphan: now.Add(-72 * time.Hour)}, + } + runtime := &fakePruneRuntime{inventory: &domain.RuntimeInventory{ + Images: []domain.RuntimeImage{{ID: "sha256:dangling", Labels: map[string]string{domain.LabelApp: "app"}}}, + }} + + svc, _ := newPruneService(t, manifests, blobs, releasedImageSnapshot("sha256:dangling"), runtime) + opts := domain.ImagePruneOptions{KeepLast: 1, PruneDangling: true, PruneRegistry: true} + + first, err := svc.Prune(context.Background(), opts) + require.NoError(t, err) + require.NotEmpty(t, first.Plan.Deleted, "the first run must delete something to be worth repeating") + firstDeleted := len(first.Plan.Deleted) + firstTags := len(manifests.deletedManifests) + firstImages := len(runtime.removedImages) + firstBlobs := len(blobs.deletedBlobs) + require.NotZero(t, firstTags+firstImages+firstBlobs) + + second, err := svc.Prune(context.Background(), opts) + require.NoError(t, err) + assert.Empty(t, second.Plan.Deleted, "a second run must delete nothing") + assert.Zero(t, second.Plan.CountByVerdict(domain.PruneVerdictEligible), "nothing may remain eligible") + assert.Zero(t, second.Plan.CountByVerdict(domain.PruneVerdictUnknown), "a completed plan leaves no unknowns behind") + assert.Empty(t, second.Plan.Failures) + assert.Equal(t, firstTags, len(manifests.deletedManifests)) + assert.Equal(t, firstImages, len(runtime.removedImages)) + assert.Equal(t, firstBlobs, len(blobs.deletedBlobs)) + assert.Len(t, first.Plan.Deleted, firstDeleted) +} + +// TestPrune_RuntimeDeletionRunsOutsideRegistryLock proves the registry +// mutation lock is not held while runtime images are deleted, so a slow +// runtime prune cannot stall registry pushes. +func TestPrune_RuntimeDeletionRunsOutsideRegistryLock(t *testing.T) { + now := time.Date(2026, 3, 1, 12, 0, 0, 0, time.UTC) + manifests := newFakeManifestStorage() + manifests.repositories = []string{"app"} + manifests.tagsByRepo["app"] = []string{"latest", "v1"} + manifests.modTimes[manifestRefKey("app", "latest")] = now + manifests.modTimes[manifestRefKey("app", "v1")] = now.Add(-time.Hour) + manifests.manifests[manifestRefKey("app", "latest")] = mustManifestJSON(t, testDigest("l")) + manifests.manifests[manifestRefKey("app", "v1")] = mustManifestJSON(t, testDigest("one")) + + // The runtime port reports whether the registry lock was held when + // the deletion ran. + runtime := &fakePruneRuntime{ + inventory: &domain.RuntimeInventory{ + Images: []domain.RuntimeImage{{ID: "sha256:dangling", Labels: map[string]string{domain.LabelApp: "app"}}}, + }, + } + svc, _ := newPruneService(t, manifests, &fakeBlobStorage{}, releasedImageSnapshot("sha256:dangling"), runtime) + + unlocked := false + runtime.onRemoveImage = func() { + unlocked = svc.mutationMu.TryLock() + if unlocked { + svc.mutationMu.Unlock() + } + } + + _, err := svc.Prune(context.Background(), domain.ImagePruneOptions{ + KeepLast: 1, PruneDangling: true, PruneRegistry: true, + }) + require.NoError(t, err) + assert.Equal(t, []string{"sha256:dangling"}, runtime.removedImages) + assert.True(t, unlocked, "runtime deletion must not run under the registry mutation lock") +} diff --git a/internal/usecase/images/prune_test.go b/internal/usecase/images/prune_test.go new file mode 100644 index 000000000..ba74c9a7f --- /dev/null +++ b/internal/usecase/images/prune_test.go @@ -0,0 +1,483 @@ +package images + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "errors" + "sync" + "testing" + "time" + + "github.com/bnema/zerowrap" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/domain" +) + +// digestOf returns the sha256 digest of a manifest body, matching the +// identity the service derives during inventory. +func manifestBodyDigest(data []byte) string { + sum := sha256.Sum256(data) + return "sha256:" + hex.EncodeToString(sum[:]) +} + +func testDigest(seed string) string { + sum := sha256.Sum256([]byte(seed)) + return "sha256:" + hex.EncodeToString(sum[:]) +} + +// fakeProtectionStore returns one fixed snapshot. +type fakeProtectionStore struct { + snapshot *domain.PruneProtectionSnapshot + err error + calls int +} + +func (f *fakeProtectionStore) ProtectionSnapshot(context.Context) (*domain.PruneProtectionSnapshot, error) { + f.calls++ + if f.err != nil { + return nil, f.err + } + if f.snapshot == nil { + return &domain.PruneProtectionSnapshot{}, nil + } + return f.snapshot, nil +} + +// fakePruneRuntime records every exact deletion request. +type fakePruneRuntime struct { + inventory *domain.RuntimeInventory + inventoryErr error + + removedImages []string + removeImageErr error + + removedVolumes []string + removeVolumeErr error + + // onRemoveImage, when set, runs before a runtime image deletion so a + // test can observe the state it executes under. + onRemoveImage func() +} + +func (f *fakePruneRuntime) InventoryRuntime(context.Context) (*domain.RuntimeInventory, error) { + if f.inventoryErr != nil { + return nil, f.inventoryErr + } + if f.inventory == nil { + return &domain.RuntimeInventory{}, nil + } + return f.inventory, nil +} + +func (f *fakePruneRuntime) RemoveImageExact(_ context.Context, ref domain.RuntimeImageRef) error { + if f.onRemoveImage != nil { + f.onRemoveImage() + } + if f.removeImageErr != nil { + return f.removeImageErr + } + f.removedImages = append(f.removedImages, ref.ID) + // A real adapter no longer reports a removed image, so the double + // must not either: otherwise a second run would look idempotent + // only because the double forgets nothing. + if f.inventory != nil { + kept := f.inventory.Images[:0] + for _, image := range f.inventory.Images { + if image.ID != ref.ID { + kept = append(kept, image) + } + } + f.inventory.Images = kept + } + return nil +} + +func (f *fakePruneRuntime) RemoveVolumeExact(_ context.Context, ref domain.RuntimeVolumeRef) error { + if f.removeVolumeErr != nil { + return f.removeVolumeErr + } + f.removedVolumes = append(f.removedVolumes, ref.Name) + return nil +} + +// fakeBarrier records the exclusive lease lifecycle. +type fakeBarrier struct { + mu sync.Mutex + acquires int + releases int + err error +} + +func (b *fakeBarrier) AcquireShared(context.Context) (out.GCLease, error) { + return noopLease{}, nil +} + +func (b *fakeBarrier) AcquireExclusive(context.Context) (out.GCLease, error) { + if b.err != nil { + return nil, b.err + } + b.mu.Lock() + b.acquires++ + b.mu.Unlock() + return &recordingLease{barrier: b}, nil +} + +type recordingLease struct { + barrier *fakeBarrier + once sync.Once +} + +func (l *recordingLease) Release() { + l.once.Do(func() { + l.barrier.mu.Lock() + l.barrier.releases++ + l.barrier.mu.Unlock() + }) +} + +// noopLease is a lease used by the barrier under test. +type noopLease struct{} + +func (noopLease) Release() {} + +func newPruneService(t *testing.T, manifests *fakeManifestStorage, blobs *fakeBlobStorage, protection *fakeProtectionStore, runtime *fakePruneRuntime) (*Service, *fakeBarrier) { + t.Helper() + barrier := &fakeBarrier{} + svc := NewService(&fakeRuntime{}, manifests, blobs, zerowrap.Default()). + WithPrunePorts(protection, runtime, barrier) + return svc, barrier +} + +// TestPrune_SelectiveMixedBatch is the core acceptance case: app state +// exists, and a mixed batch of safe, protected, and unknown candidates +// produces exactly the safe deletions with no operation-wide refusal. +func TestPrune_SelectiveMixedBatch(t *testing.T) { + now := time.Date(2026, 3, 1, 12, 0, 0, 0, time.UTC) + oldTag := mustManifestJSON(t, testDigest("a"), testDigest("b")) + latestTag := mustManifestJSON(t, testDigest("c"), testDigest("d")) + + manifests := newFakeManifestStorage() + manifests.repositories = []string{"app"} + manifests.tagsByRepo["app"] = []string{"latest", "v1", "v2", "v3", "v4", "v5"} + for index, tag := range []string{"latest", "v1", "v2", "v3", "v4", "v5"} { + manifests.modTimes[manifestRefKey("app", tag)] = now.Add(time.Duration(index-6) * time.Hour) + } + manifests.manifests[manifestRefKey("app", "latest")] = latestTag + manifests.manifests[manifestRefKey("app", "v1")] = oldTag + for _, tag := range []string{"v2", "v3", "v4", "v5"} { + manifests.manifests[manifestRefKey("app", tag)] = mustManifestJSON(t, testDigest(tag), testDigest(tag+"-layer")) + } + + blobs := &fakeBlobStorage{ + blobs: []string{testDigest("b"), testDigest("d"), testDigest("orphan")}, + blobSizes: map[string]int64{testDigest("orphan"): 4096}, + blobModTimes: map[string]time.Time{testDigest("orphan"): now.Add(-48 * time.Hour)}, + } + + activeImage := "sha256:active" + danglingImage := "sha256:dangling" + foreignImage := "sha256:foreign" + protection := &fakeProtectionStore{snapshot: &domain.PruneProtectionSnapshot{ + Roots: []domain.ProtectionRoot{ + {Kind: domain.ProtectionActiveService, Ref: mustTagKey(t, "app", "v1"), Owner: "app@active.web"}, + }, + ImageClaims: []domain.ImageClaim{releasedImageClaim(danglingImage)}, + }} + runtime := &fakePruneRuntime{inventory: &domain.RuntimeInventory{ + Images: []domain.RuntimeImage{ + {ID: activeImage, RepoTags: []string{"app:latest"}, Labels: map[string]string{domain.LabelApp: "app"}}, + {ID: danglingImage, Labels: map[string]string{domain.LabelApp: "app"}}, + {ID: foreignImage}, + }, + Containers: []domain.RuntimeContainerUse{ + {ContainerID: "c1", ImageID: activeImage, Running: true}, + }, + }} + + svc, barrier := newPruneService(t, manifests, blobs, protection, runtime) + report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{ + KeepLast: 3, PruneDangling: true, PruneRegistry: true, + }) + require.NoError(t, err) + + // latest + 3 = latest, v5, v4, v3 retained; v2 and v1 pruned + // only if unprotected. v1 is protected by the active app state. + assert.Equal(t, []manifestRef{{name: "app", reference: "v2"}}, manifests.deletedManifests) + + // The dangling app image is deleted; the foreign and in-use ones + // are not. + assert.Equal(t, []string{danglingImage}, runtime.removedImages) + + // The orphan blob is deleted; the blobs reachable from the retained + // and protected manifests are not. + assert.Equal(t, []string{testDigest("orphan")}, blobs.deletedBlobs) + assert.Equal(t, int64(4096), report.Plan.ReclaimedBytes) + assert.True(t, report.Plan.ReclaimedKnown) + assert.True(t, report.Plan.Applied) + assert.Empty(t, report.Plan.Failures) + + assert.GreaterOrEqual(t, report.Plan.CountByVerdict(domain.PruneVerdictProtected), 1) + assert.GreaterOrEqual(t, report.Plan.CountByVerdict(domain.PruneVerdictEligible), 1) + assert.Equal(t, 1, barrier.acquires) + assert.Equal(t, 1, barrier.releases) + assert.Equal(t, 1, protection.calls) +} + +func TestPrune_DryRunPlansWithoutDeleting(t *testing.T) { + now := time.Date(2026, 3, 1, 12, 0, 0, 0, time.UTC) + manifests := newFakeManifestStorage() + manifests.repositories = []string{"app"} + manifests.tagsByRepo["app"] = []string{"latest", "v1", "v2"} + for index, tag := range []string{"latest", "v1", "v2"} { + manifests.modTimes[manifestRefKey("app", tag)] = now.Add(time.Duration(index-3) * time.Hour) + manifests.manifests[manifestRefKey("app", tag)] = mustManifestJSON(t, testDigest(tag), testDigest(tag+"-l")) + } + blobs := &fakeBlobStorage{blobs: []string{testDigest("x")}, blobModTimes: map[string]time.Time{testDigest("x"): now.Add(-72 * time.Hour)}} + + svc, _ := newPruneService(t, manifests, blobs, &fakeProtectionStore{}, &fakePruneRuntime{}) + report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{ + KeepLast: 1, PruneDangling: true, PruneRegistry: true, DryRun: true, + }) + require.NoError(t, err) + + assert.False(t, report.Plan.Applied) + assert.Empty(t, manifests.deletedManifests) + assert.Empty(t, blobs.deletedBlobs) + assert.NotEmpty(t, report.Plan.Candidates) + + // The dry-run plan must equal the executed plan. + executed, err := svc.Prune(context.Background(), domain.ImagePruneOptions{ + KeepLast: 1, PruneDangling: true, PruneRegistry: true, + }) + require.NoError(t, err) + assert.Equal(t, report.Plan.Candidates, executed.Plan.Candidates) + assert.True(t, executed.Plan.Applied) +} + +func TestPrune_ZeroEligibleSucceeds(t *testing.T) { + now := time.Date(2026, 3, 1, 12, 0, 0, 0, time.UTC) + manifests := newFakeManifestStorage() + manifests.repositories = []string{"app"} + manifests.tagsByRepo["app"] = []string{"latest", "v1"} + for index, tag := range []string{"latest", "v1"} { + manifests.modTimes[manifestRefKey("app", tag)] = now.Add(time.Duration(index) * time.Hour) + manifests.manifests[manifestRefKey("app", tag)] = mustManifestJSON(t, testDigest(tag)) + } + blobs := &fakeBlobStorage{blobs: []string{}} + runtime := &fakePruneRuntime{inventory: &domain.RuntimeInventory{}} + + svc, _ := newPruneService(t, manifests, blobs, &fakeProtectionStore{}, runtime) + report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{ + KeepLast: 3, PruneDangling: true, PruneRegistry: true, + }) + require.NoError(t, err, "a zero-deletion prune must succeed, not fail the operation") + assert.Zero(t, report.Registry.TagsRemoved) + assert.Zero(t, report.Registry.BlobsRemoved) + assert.Zero(t, report.Runtime.DeletedCount) + assert.Equal(t, 0, report.Plan.CountByVerdict(domain.PruneVerdictEligible)) +} + +func TestPrune_MissingChildManifestMakesUnreferencedBlobsUnknown(t *testing.T) { + now := time.Date(2026, 3, 1, 12, 0, 0, 0, time.UTC) + childDigest := testDigest("missing-child") + indexBody := mustManifestIndexJSON(t, childDigest) + + manifests := newFakeManifestStorage() + manifests.repositories = []string{"app"} + manifests.tagsByRepo["app"] = []string{"latest", "v1", "v2"} + manifests.modTimes[manifestRefKey("app", "latest")] = now + manifests.modTimes[manifestRefKey("app", "v1")] = now.Add(-time.Hour) + manifests.modTimes[manifestRefKey("app", "v2")] = now.Add(-2 * time.Hour) + // latest is a multi-arch index whose child manifest is absent. + manifests.manifests[manifestRefKey("app", "latest")] = indexBody + manifests.manifests[manifestRefKey("app", "v1")] = mustManifestJSON(t, testDigest("c1")) + manifests.manifests[manifestRefKey("app", "v2")] = mustManifestJSON(t, testDigest("c2")) + + orphan := testDigest("orphan") + blobs := &fakeBlobStorage{ + blobs: []string{orphan}, + blobModTimes: map[string]time.Time{orphan: now.Add(-72 * time.Hour)}, + } + + svc, _ := newPruneService(t, manifests, blobs, &fakeProtectionStore{}, &fakePruneRuntime{}) + report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{ + KeepLast: 1, PruneRegistry: true, + }) + require.NoError(t, err) + + // The unreadable child makes the retained closure incomplete, so an + // otherwise-unreferenced blob is unknown, not eligible. + assert.Empty(t, blobs.deletedBlobs, "an incomplete closure must never authorize blob deletion") + assert.Contains(t, manifestBodyDigest(indexBody), "sha256:") + var unknown bool + for _, candidate := range report.Plan.Candidates { + if candidate.Kind == domain.PruneResourceOCIBlob && candidate.Verdict == domain.PruneVerdictUnknown { + unknown = true + assert.Equal(t, []domain.PruneReason{domain.PruneReasonUnknownManifest}, candidate.Reasons) + } + } + assert.True(t, unknown, "the orphan blob must be reported unknown") + assert.NotEmpty(t, report.Plan.Gaps) + + // Tags outside the window are still deleted: the failure is scoped + // to the content whose safety depended on the missing manifest. + assert.Len(t, manifests.deletedManifests, 1) + assert.Equal(t, "v2", manifests.deletedManifests[0].reference) +} + +func TestPrune_SharedBlobStaysProtectedThroughClosure(t *testing.T) { + now := time.Date(2026, 3, 1, 12, 0, 0, 0, time.UTC) + sharedLayer := testDigest("shared-layer") + + manifests := newFakeManifestStorage() + manifests.repositories = []string{"app"} + manifests.tagsByRepo["app"] = []string{"latest", "v1", "v2"} + manifests.modTimes[manifestRefKey("app", "latest")] = now + manifests.modTimes[manifestRefKey("app", "v1")] = now.Add(-time.Hour) + manifests.modTimes[manifestRefKey("app", "v2")] = now.Add(-2 * time.Hour) + manifests.manifests[manifestRefKey("app", "latest")] = mustManifestJSON(t, testDigest("cfg-latest"), sharedLayer) + manifests.manifests[manifestRefKey("app", "v1")] = mustManifestJSON(t, testDigest("cfg-v1"), sharedLayer) + manifests.manifests[manifestRefKey("app", "v2")] = mustManifestJSON(t, testDigest("cfg-v2")) + + blobs := &fakeBlobStorage{blobs: []string{sharedLayer}} + svc, _ := newPruneService(t, manifests, blobs, &fakeProtectionStore{}, &fakePruneRuntime{}) + report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{KeepLast: 1, PruneRegistry: true}) + require.NoError(t, err) + + assert.Empty(t, blobs.deletedBlobs, "a layer shared with a retained tag must survive") + protected := report.Plan.CandidatesOfKind(domain.PruneResourceOCIBlob, domain.PruneVerdictProtected) + require.Len(t, protected, 1) + assert.Equal(t, []domain.PruneReason{domain.PruneReasonProtectedSharedContent}, protected[0].Reasons) +} + +func TestPrune_RegistryDeletionFailureIsIsolated(t *testing.T) { + now := time.Date(2026, 3, 1, 12, 0, 0, 0, time.UTC) + manifests := newFakeManifestStorage() + manifests.repositories = []string{"app"} + manifests.tagsByRepo["app"] = []string{"latest", "v1", "v2"} + for index, tag := range []string{"latest", "v1", "v2"} { + manifests.modTimes[manifestRefKey("app", tag)] = now.Add(time.Duration(index-3) * time.Hour) + manifests.manifests[manifestRefKey("app", tag)] = mustManifestJSON(t, testDigest(tag)) + } + manifests.deleteErr = errors.New("storage unavailable") + + blobs := &fakeBlobStorage{} + svc, _ := newPruneService(t, manifests, blobs, &fakeProtectionStore{}, &fakePruneRuntime{}) + report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{KeepLast: 1, PruneRegistry: true}) + require.NoError(t, err, "one deletion failure must not fail the run") + + require.Len(t, report.Plan.Failures, 1) + assert.Equal(t, domain.PruneResourceRegistryTag, report.Plan.Failures[0].Kind) + assert.Zero(t, report.Registry.TagsRemoved) + assert.Empty(t, report.Plan.Deleted) +} + +func TestPrune_RuntimeDeletionFailureIsIsolated(t *testing.T) { + image := "sha256:dangling" + runtime := &fakePruneRuntime{ + inventory: &domain.RuntimeInventory{ + Images: []domain.RuntimeImage{{ID: image, Labels: map[string]string{domain.LabelApp: "app"}}}, + }, + removeImageErr: errors.New("image is in use"), + } + svc, _ := newPruneService(t, newFakeManifestStorage(), &fakeBlobStorage{}, releasedImageSnapshot(image), runtime) + + report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{PruneDangling: true}) + require.NoError(t, err) + require.Len(t, report.Plan.Failures, 1) + assert.Equal(t, domain.PruneResourceRuntimeImage, report.Plan.Failures[0].Kind) + assert.Zero(t, report.Runtime.DeletedCount) +} + +func TestPrune_KeepLastZeroSkipsRegistry(t *testing.T) { + manifests := newFakeManifestStorage() + manifests.repositories = []string{"app"} + manifests.tagsByRepo["app"] = []string{"v1"} + svc, _ := newPruneService(t, manifests, &fakeBlobStorage{blobs: []string{testDigest("x")}}, &fakeProtectionStore{}, &fakePruneRuntime{}) + + report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{KeepLast: 0, PruneRegistry: true}) + require.NoError(t, err) + assert.Empty(t, manifests.deletedManifests) + assert.Empty(t, report.Plan.Candidates) +} + +func TestPrune_RegistryOnlySkipsRuntime(t *testing.T) { + now := time.Date(2026, 3, 1, 12, 0, 0, 0, time.UTC) + manifests := newFakeManifestStorage() + manifests.repositories = []string{"app"} + manifests.tagsByRepo["app"] = []string{"latest", "v1", "v2"} + for index, tag := range []string{"latest", "v1", "v2"} { + manifests.modTimes[manifestRefKey("app", tag)] = now.Add(time.Duration(index-3) * time.Hour) + manifests.manifests[manifestRefKey("app", tag)] = mustManifestJSON(t, testDigest(tag)) + } + runtime := &fakePruneRuntime{ + inventory: &domain.RuntimeInventory{ + Images: []domain.RuntimeImage{{ID: "sha256:dangling", Labels: map[string]string{domain.LabelApp: "app"}}}, + }, + } + svc, _ := newPruneService(t, manifests, &fakeBlobStorage{}, &fakeProtectionStore{}, runtime) + _, err := svc.Prune(context.Background(), domain.ImagePruneOptions{KeepLast: 1, PruneRegistry: true}) + require.NoError(t, err) + assert.Empty(t, runtime.removedImages, "registry-only prune must not touch runtime images") +} + +func TestPrune_RequiresProtectionStore(t *testing.T) { + svc := NewService(&fakeRuntime{}, newFakeManifestStorage(), &fakeBlobStorage{}, zerowrap.Default()) + _, err := svc.Prune(context.Background(), domain.DefaultImagePruneOptions()) + require.Error(t, err) + assert.ErrorIs(t, err, domain.ErrPruneDisabled) +} + +func TestPrune_ProtectionSnapshotFailureFailsClosed(t *testing.T) { + protection := &fakeProtectionStore{err: errors.New("unreadable state")} + svc, _ := newPruneService(t, newFakeManifestStorage(), &fakeBlobStorage{}, protection, &fakePruneRuntime{}) + _, err := svc.Prune(context.Background(), domain.DefaultImagePruneOptions()) + require.Error(t, err) + assert.Contains(t, err.Error(), "protection snapshot") +} + +func TestPrune_RuntimeInventoryGapMakesCandidatesUnknown(t *testing.T) { + runtime := &fakePruneRuntime{inventory: &domain.RuntimeInventory{ + Images: []domain.RuntimeImage{{ID: "sha256:dangling", Labels: map[string]string{domain.LabelApp: "app"}}}, + Gaps: []domain.InventoryGap{{ + Source: domain.InventorySourceRuntimeContainers, + Reason: domain.PruneReasonUnknownContainerUse, + Detail: "list containers failed", + }}, + }} + svc, _ := newPruneService(t, newFakeManifestStorage(), &fakeBlobStorage{}, &fakeProtectionStore{}, runtime) + report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{PruneDangling: true}) + require.NoError(t, err) + assert.Empty(t, runtime.removedImages) + assert.Equal(t, 1, report.Plan.CountByVerdict(domain.PruneVerdictUnknown)) + require.Len(t, report.Plan.Gaps, 1) +} + +// releasedImageClaim marks identity as durably released by an app. Runtime +// image deletion requires this positive ownership; labels alone never +// authorize it. +func releasedImageClaim(identity string) domain.ImageClaim { + return domain.ImageClaim{ + Reference: identity, App: "app", AppID: "app-uuid", + State: domain.VolumeClaimReleased, + } +} + +// releasedImageSnapshot is a complete snapshot containing one released +// runtime image claim. +func releasedImageSnapshot(identity string) *fakeProtectionStore { + return &fakeProtectionStore{snapshot: &domain.PruneProtectionSnapshot{ + ImageClaims: []domain.ImageClaim{releasedImageClaim(identity)}, + }} +} + +func mustTagKey(t *testing.T, repository, tag string) string { + t.Helper() + ref, err := domain.NewRegistryTagRef(repository, tag) + require.NoError(t, err) + return ref.Key() +} diff --git a/internal/usecase/images/service.go b/internal/usecase/images/service.go index 887735459..bbb8da75b 100644 --- a/internal/usecase/images/service.go +++ b/internal/usecase/images/service.go @@ -3,12 +3,9 @@ package images import ( "context" - "encoding/json" - "fmt" "sort" "strings" "sync" - "time" "github.com/bnema/zerowrap" @@ -27,11 +24,15 @@ type Service struct { log zerowrap.Logger mutationMu *sync.RWMutex registryState *registrystate.State + // protection, pruneRuntime, and barrier are the prune ports. All + // three are required for Prune; a missing port fails closed. + protection out.PruneProtectionStore + pruneRuntime out.PruneRuntime + barrier out.GCBarrier } type imageRuntime interface { ListImagesDetailed(ctx context.Context) ([]runtime.ImageDetail, error) - PruneImages(ctx context.Context, danglingOnly bool) (runtime.PruneReport, error) } // NewService creates a new images service. @@ -180,348 +181,6 @@ func (s *Service) appendRegistryImages( return images, nil } -// PruneRuntime prunes dangling images from the runtime. -func (s *Service) PruneRuntime(ctx context.Context) (domain.ImagePruneReport, error) { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "PruneRuntime", - }) - log := zerowrap.FromCtx(ctx) - - pruneReport, err := s.runtime.PruneImages(ctx, true) - if err != nil { - return domain.ImagePruneReport{}, log.WrapErr(err, "failed to prune runtime images") - } - - return domain.ImagePruneReport{ - Runtime: domain.RuntimePruneResult{ - DeletedCount: len(pruneReport.DeletedIDs), - SpaceReclaimed: pruneReport.SpaceReclaimed, - }, - }, nil -} - -// PruneRegistry applies tag retention and blob garbage collection. -// It keeps the "latest" tag when present and keeps keepLast most-recent -// non-latest tags from tagInfos. -func (s *Service) PruneRegistry(ctx context.Context, keepLast int) (domain.ImagePruneReport, error) { - s.mutationMu.Lock() - defer s.mutationMu.Unlock() - - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "PruneRegistry", - "keepLast": keepLast, - }) - log := zerowrap.FromCtx(ctx) - - if keepLast <= 0 { - return domain.ImagePruneReport{}, nil - } - - repositories, err := s.manifestStorage.ListRepositories() - if err != nil { - return domain.ImagePruneReport{}, log.WrapErr(err, "failed to list repositories") - } - - referencedDigests := make(map[string]struct{}) - report := domain.ImagePruneReport{} - - for _, repository := range repositories { - tagInfos, err := s.loadRepositoryTagInfos(repository) - if err != nil { - return domain.ImagePruneReport{}, log.WrapErr(err, "failed to load repository tags") - } - if len(tagInfos) == 0 { - continue - } - - keptTags := buildKeptTagSet(tagInfos, keepLast) - removed, err := s.deleteUnkeptManifests(repository, tagInfos, keptTags) - if err != nil { - return domain.ImagePruneReport{}, log.WrapErr(err, "failed to delete manifest") - } - report.Registry.TagsRemoved += removed - - if err := s.collectKeptTagDigests(log, repository, tagInfos, keptTags, referencedDigests); err != nil { - return domain.ImagePruneReport{}, err - } - } - - now := time.Now().UTC() - for digest := range s.registryState.PendingDigests(now) { - referencedDigests[digest] = struct{}{} - } - - blobsRemoved, spaceReclaimed, err := s.pruneUnreferencedBlobs(log, now, referencedDigests) - if err != nil { - return domain.ImagePruneReport{}, err - } - report.Registry.BlobsRemoved = blobsRemoved - report.Registry.SpaceReclaimed = spaceReclaimed - - // Clean up stale uploads (abandoned push operations) - uploadsRemoved, uploadBytes, err := s.blobStorage.CleanupStaleUploads(24 * time.Hour) - if err != nil { - log.Warn().Err(err).Msg("failed to clean up stale uploads, continuing") - } else { - report.Registry.UploadsRemoved = uploadsRemoved - report.Registry.UploadSpaceReclaimed = uploadBytes - } - - return report, nil -} - -func (s *Service) pruneUnreferencedBlobs(log zerowrap.Logger, now time.Time, referencedDigests map[string]struct{}) (int, int64, error) { - blobs, err := s.blobStorage.ListBlobs() - if err != nil { - return 0, 0, log.WrapErr(err, "failed to list blobs") - } - - var removed int - var spaceReclaimed int64 - for _, digest := range blobs { - if _, referenced := referencedDigests[digest]; referenced { - continue - } - modTime, err := s.blobStorage.GetBlobModTime(digest) - if err != nil { - return 0, 0, log.WrapErr(err, "failed to get blob modification time") - } - if now.Sub(modTime) <= registrystate.PendingBlobTTL { - continue - } - size, err := s.blobStorage.DeleteBlob(digest) - if err != nil { - return 0, 0, log.WrapErr(err, "failed to delete blob") - } - removed++ - spaceReclaimed += size - } - return removed, spaceReclaimed, nil -} - -func (s *Service) loadRepositoryTagInfos(repository string) ([]registryTag, error) { - tags, err := s.manifestStorage.ListTags(repository) - if err != nil { - return nil, err - } - - tagInfos := make([]registryTag, 0, len(tags)) - for _, tag := range tags { - modTime, err := s.manifestStorage.GetManifestModTime(repository, tag) - if err != nil { - return nil, err - } - tagInfos = append(tagInfos, registryTag{name: tag, modTime: modTime}) - } - - sort.Slice(tagInfos, func(i, j int) bool { - if tagInfos[i].modTime.Equal(tagInfos[j].modTime) { - return tagInfos[i].name > tagInfos[j].name - } - return tagInfos[i].modTime.After(tagInfos[j].modTime) - }) - - return tagInfos, nil -} - -func buildKeptTagSet(tagInfos []registryTag, keepLast int) map[string]struct{} { - keptTags := make(map[string]struct{}) - if containsTag(tagInfos, "latest") { - keptTags["latest"] = struct{}{} - } - - kept := 0 - for _, tagInfo := range tagInfos { - if tagInfo.name == "latest" { - continue - } - if kept >= keepLast { - break - } - keptTags[tagInfo.name] = struct{}{} - kept++ - } - - return keptTags -} - -func (s *Service) deleteUnkeptManifests(repository string, tagInfos []registryTag, keptTags map[string]struct{}) (int, error) { - removed := 0 - for _, tag := range tagInfos { - if _, kept := keptTags[tag.name]; kept { - continue - } - if err := s.manifestStorage.DeleteManifest(repository, tag.name); err != nil { - return 0, err - } - removed++ - } - - return removed, nil -} - -func (s *Service) collectKeptTagDigests( - log zerowrap.Logger, - repository string, - tagInfos []registryTag, - keptTags map[string]struct{}, - referencedDigests map[string]struct{}, -) error { - for _, tag := range tagInfos { - if _, kept := keptTags[tag.name]; !kept { - continue - } - if err := s.collectReferencedDigests(log, repository, tag.name, referencedDigests, make(map[string]struct{})); err != nil { - return err - } - } - - return nil -} - -// Prune runs runtime and/or registry prune based on options and aggregates reports. -func (s *Service) Prune(ctx context.Context, opts domain.ImagePruneOptions) (domain.ImagePruneReport, error) { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "Prune", - "keepLast": opts.KeepLast, - "pruneDangling": opts.PruneDangling, - "pruneRegistry": opts.PruneRegistry, - }) - log := zerowrap.FromCtx(ctx) - - var report domain.ImagePruneReport - - if opts.PruneDangling { - runtimeReport, err := s.PruneRuntime(ctx) - if err != nil { - if !opts.PruneRegistry { - // Dangling-only: surface the error to the caller. - return report, fmt.Errorf("runtime prune failed: %w", err) - } - log.Warn().Err(err).Msg("runtime prune failed; continuing with registry prune") - } else { - report.Runtime = runtimeReport.Runtime - } - } - - if opts.PruneRegistry { - registryReport, err := s.PruneRegistry(ctx, opts.KeepLast) - if err != nil { - return report, err - } - report.Registry = registryReport.Registry - } - - return report, nil -} - -type registryTag struct { - name string - modTime time.Time -} - -type manifestReferences struct { - blobs []string - childManifests []string -} - -func containsTag(tags []registryTag, name string) bool { - for _, tag := range tags { - if tag.name == name { - return true - } - } - - return false -} - -func (s *Service) collectReferencedDigests( - log zerowrap.Logger, - repository, reference string, - referencedDigests map[string]struct{}, - visited map[string]struct{}, -) error { - visitKey := repository + "@" + reference - if _, seen := visited[visitKey]; seen { - return nil - } - visited[visitKey] = struct{}{} - - manifestData, _, err := s.manifestStorage.GetManifest(repository, reference) - if err != nil { - log.Warn(). - Str("repository", repository). - Str("reference", reference). - Err(err). - Msg("manifest not found during prune; skipping (orphaned reference)") - return nil - } - - refs, err := parseManifestReferences(manifestData) - if err != nil { - return log.WrapErr(err, "failed to parse manifest") - } - - for _, digest := range refs.blobs { - referencedDigests[digest] = struct{}{} - } - - for _, childRef := range refs.childManifests { - referencedDigests[childRef] = struct{}{} - if err := s.collectReferencedDigests(log, repository, childRef, referencedDigests, visited); err != nil { - return err - } - } - - return nil -} - -func parseManifestReferences(data []byte) (manifestReferences, error) { - var manifest struct { - Config struct { - Digest string `json:"digest"` - } `json:"config"` - Layers []struct { - Digest string `json:"digest"` - } `json:"layers"` - Manifests []struct { - Digest string `json:"digest"` - } `json:"manifests"` - } - - if err := json.Unmarshal(data, &manifest); err != nil { - return manifestReferences{}, err - } - - refs := manifestReferences{ - blobs: make([]string, 0, len(manifest.Layers)+1), - childManifests: make([]string, 0, len(manifest.Manifests)), - } - - if manifest.Config.Digest != "" { - refs.blobs = append(refs.blobs, manifest.Config.Digest) - } - - for _, layer := range manifest.Layers { - if layer.Digest == "" { - continue - } - refs.blobs = append(refs.blobs, layer.Digest) - } - - for _, child := range manifest.Manifests { - if child.Digest == "" { - continue - } - refs.childManifests = append(refs.childManifests, child.Digest) - } - - return refs, nil -} - func isDanglingImage(repoTags []string) bool { if len(repoTags) == 0 { return true diff --git a/internal/usecase/images/service_test.go b/internal/usecase/images/service_test.go index da2126363..a47c6da07 100644 --- a/internal/usecase/images/service_test.go +++ b/internal/usecase/images/service_test.go @@ -13,7 +13,6 @@ import ( "github.com/bnema/gordon/internal/boundaries/out" "github.com/bnema/gordon/internal/domain" - "github.com/bnema/gordon/internal/usecase/registrystate" pkgruntime "github.com/bnema/gordon/pkg/runtime" ) @@ -179,485 +178,8 @@ func TestService_ListImages_IncludesRegistryTagsNotPresentInRuntime(t *testing.T }, images[2]) } -func TestService_PruneRuntime_RemovesDanglingImages(t *testing.T) { - manifestStorage := noopManifestStorage{} - blobStorage := noopBlobStorage{} - rt := &fakeRuntime{ - pruneReport: pkgruntime.PruneReport{ - DeletedIDs: []string{"sha256:a", "sha256:b"}, - SpaceReclaimed: 2048, - }, - } - - svc := NewService(rt, manifestStorage, blobStorage, zerowrap.Default()) - - report, err := svc.PruneRuntime(context.Background()) - - require.NoError(t, err) - assert.Equal(t, 2, report.Runtime.DeletedCount) - assert.Equal(t, int64(2048), report.Runtime.SpaceReclaimed) - assert.True(t, rt.pruneCalled) - assert.True(t, rt.pruneDanglingOnly) -} - -func TestService_PruneRuntime_NoDanglingImages(t *testing.T) { - manifestStorage := noopManifestStorage{} - blobStorage := noopBlobStorage{} - rt := &fakeRuntime{ - pruneReport: pkgruntime.PruneReport{ - DeletedIDs: nil, - SpaceReclaimed: 0, - }, - } - - svc := NewService(rt, manifestStorage, blobStorage, zerowrap.Default()) - - report, err := svc.PruneRuntime(context.Background()) - - require.NoError(t, err) - assert.Equal(t, 0, report.Runtime.DeletedCount) - assert.Equal(t, int64(0), report.Runtime.SpaceReclaimed) - assert.True(t, rt.pruneCalled) - assert.True(t, rt.pruneDanglingOnly) -} - -func TestService_PruneRuntime_ReturnsErrorWhenRuntimeFails(t *testing.T) { - manifestStorage := noopManifestStorage{} - blobStorage := noopBlobStorage{} - rt := &fakeRuntime{pruneErr: errors.New("runtime prune failed")} - - svc := NewService(rt, manifestStorage, blobStorage, zerowrap.Default()) - - _, err := svc.PruneRuntime(context.Background()) - - require.Error(t, err) - assert.Contains(t, err.Error(), "failed to prune runtime images") - assert.True(t, rt.pruneCalled) - assert.True(t, rt.pruneDanglingOnly) -} - -func TestService_PruneRegistry_KeepsLastNTags(t *testing.T) { - manifestStorage := newFakeManifestStorage() - manifestStorage.repositories = []string{"gordon/api"} - manifestStorage.tagsByRepo["gordon/api"] = []string{"latest", "v3", "v2", "v1"} - manifestStorage.modTimes[manifestRefKey("gordon/api", "latest")] = time.Date(2026, 2, 8, 12, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/api", "v3")] = time.Date(2026, 2, 8, 11, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/api", "v2")] = time.Date(2026, 2, 8, 10, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/api", "v1")] = time.Date(2026, 2, 8, 9, 0, 0, 0, time.UTC) - manifestStorage.manifests[manifestRefKey("gordon/api", "latest")] = mustManifestJSON(t, "sha256:cfg-latest", "sha256:layer-latest") - manifestStorage.manifests[manifestRefKey("gordon/api", "v3")] = mustManifestJSON(t, "sha256:cfg-v3", "sha256:layer-v3") - manifestStorage.manifests[manifestRefKey("gordon/api", "v2")] = mustManifestJSON(t, "sha256:cfg-v2", "sha256:layer-v2") - - blobStorage := &fakeBlobStorage{} - svc := NewService(&fakeRuntime{}, manifestStorage, blobStorage, zerowrap.Default()) - - report, err := svc.PruneRegistry(context.Background(), 2) - - require.NoError(t, err) - assert.Equal(t, 1, report.Registry.TagsRemoved) - assert.ElementsMatch(t, []manifestRef{{name: "gordon/api", reference: "v1"}}, manifestStorage.deletedManifests) -} - -func TestService_PruneRegistry_AlwaysKeepsLatestTag(t *testing.T) { - manifestStorage := newFakeManifestStorage() - manifestStorage.repositories = []string{"gordon/api"} - manifestStorage.tagsByRepo["gordon/api"] = []string{"latest", "v3", "v2"} - manifestStorage.modTimes[manifestRefKey("gordon/api", "latest")] = time.Date(2026, 2, 8, 8, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/api", "v3")] = time.Date(2026, 2, 8, 11, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/api", "v2")] = time.Date(2026, 2, 8, 10, 0, 0, 0, time.UTC) - manifestStorage.manifests[manifestRefKey("gordon/api", "latest")] = mustManifestJSON(t, "sha256:cfg-latest", "sha256:layer-shared") - manifestStorage.manifests[manifestRefKey("gordon/api", "v3")] = mustManifestJSON(t, "sha256:cfg-v3", "sha256:layer-shared") - - blobStorage := &fakeBlobStorage{} - svc := NewService(&fakeRuntime{}, manifestStorage, blobStorage, zerowrap.Default()) - - report, err := svc.PruneRegistry(context.Background(), 1) - - require.NoError(t, err) - assert.Equal(t, 1, report.Registry.TagsRemoved) - assert.Equal(t, []manifestRef{{name: "gordon/api", reference: "v2"}}, manifestStorage.deletedManifests) -} - -func TestService_PruneRegistry_SkipsWhenFewerThanKeepLast(t *testing.T) { - manifestStorage := newFakeManifestStorage() - manifestStorage.repositories = []string{"gordon/api"} - manifestStorage.tagsByRepo["gordon/api"] = []string{"latest", "v1"} - manifestStorage.modTimes[manifestRefKey("gordon/api", "latest")] = time.Date(2026, 2, 8, 12, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/api", "v1")] = time.Date(2026, 2, 8, 11, 0, 0, 0, time.UTC) - manifestStorage.manifests[manifestRefKey("gordon/api", "latest")] = mustManifestJSON(t, "sha256:cfg-latest", "sha256:layer-latest") - manifestStorage.manifests[manifestRefKey("gordon/api", "v1")] = mustManifestJSON(t, "sha256:cfg-v1", "sha256:layer-v1") - - blobStorage := &fakeBlobStorage{} - svc := NewService(&fakeRuntime{}, manifestStorage, blobStorage, zerowrap.Default()) - - report, err := svc.PruneRegistry(context.Background(), 5) - - require.NoError(t, err) - assert.Equal(t, 0, report.Registry.TagsRemoved) - assert.Empty(t, manifestStorage.deletedManifests) -} - -func TestService_PruneRegistry_KeepLastZeroSkipsRegistryCleanup(t *testing.T) { - manifestStorage := newFakeManifestStorage() - manifestStorage.repositories = []string{"gordon/api"} - blobStorage := &fakeBlobStorage{blobs: []string{"sha256:orphan"}} - svc := NewService(&fakeRuntime{}, manifestStorage, blobStorage, zerowrap.Default()) - - report, err := svc.PruneRegistry(context.Background(), 0) - - require.NoError(t, err) - assert.Equal(t, domain.RegistryPruneResult{}, report.Registry) - assert.Equal(t, 0, manifestStorage.listRepositoriesCalls) - assert.Empty(t, manifestStorage.deletedManifests) - assert.Empty(t, blobStorage.deletedBlobs) -} - -func TestService_PruneRegistry_GarbageCollectsUnreferencedBlobs(t *testing.T) { - manifestStorage := newFakeManifestStorage() - manifestStorage.repositories = []string{"gordon/api"} - manifestStorage.tagsByRepo["gordon/api"] = []string{"latest"} - manifestStorage.modTimes[manifestRefKey("gordon/api", "latest")] = time.Date(2026, 2, 8, 12, 0, 0, 0, time.UTC) - manifestStorage.manifests[manifestRefKey("gordon/api", "latest")] = mustManifestJSON(t, "sha256:cfg-live", "sha256:layer-live") - - blobStorage := &fakeBlobStorage{ - blobs: []string{"sha256:cfg-live", "sha256:layer-live", "sha256:orphan"}, - blobSizes: map[string]int64{"sha256:orphan": 4096}, - } - svc := NewService(&fakeRuntime{}, manifestStorage, blobStorage, zerowrap.Default()) - - report, err := svc.PruneRegistry(context.Background(), 1) - - require.NoError(t, err) - assert.Equal(t, 1, report.Registry.BlobsRemoved) - assert.Equal(t, int64(4096), report.Registry.SpaceReclaimed) - assert.Equal(t, []string{"sha256:orphan"}, blobStorage.deletedBlobs) -} - -func TestService_PruneRegistry_PreservesRecentlyFinalizedBlobAfterRestart(t *testing.T) { - manifestStorage := newFakeManifestStorage() - blobStorage := &fakeBlobStorage{ - blobs: []string{"sha256:recent"}, - blobModTimes: map[string]time.Time{"sha256:recent": time.Now().UTC()}, - } - svc := NewService(&fakeRuntime{}, manifestStorage, blobStorage, zerowrap.Default(), registrystate.New()) - - report, err := svc.PruneRegistry(context.Background(), 1) - - require.NoError(t, err) - assert.Zero(t, report.Registry.BlobsRemoved) - assert.Empty(t, blobStorage.deletedBlobs) -} - -func TestService_PruneRegistry_PreservesPendingBlob(t *testing.T) { - manifestStorage := newFakeManifestStorage() - manifestStorage.repositories = []string{"gordon/api"} - manifestStorage.tagsByRepo["gordon/api"] = []string{"latest"} - manifestStorage.modTimes[manifestRefKey("gordon/api", "latest")] = time.Now().UTC() - manifestStorage.manifests[manifestRefKey("gordon/api", "latest")] = mustManifestJSON(t, "sha256:cfg-live", "sha256:layer-live") - blobStorage := &fakeBlobStorage{blobs: []string{"sha256:pending"}} - state := registrystate.New() - state.AddPending("sha256:pending", time.Now().UTC()) - svc := NewService(&fakeRuntime{}, manifestStorage, blobStorage, zerowrap.Default(), state) - - report, err := svc.PruneRegistry(context.Background(), 1) - - require.NoError(t, err) - assert.Zero(t, report.Registry.BlobsRemoved) - assert.Empty(t, blobStorage.deletedBlobs) -} - -func TestService_PruneRegistry_PreservesSharedBlobs(t *testing.T) { - manifestStorage := newFakeManifestStorage() - manifestStorage.repositories = []string{"gordon/api"} - manifestStorage.tagsByRepo["gordon/api"] = []string{"latest", "v2", "v1"} - manifestStorage.modTimes[manifestRefKey("gordon/api", "latest")] = time.Date(2026, 2, 8, 12, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/api", "v2")] = time.Date(2026, 2, 8, 11, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/api", "v1")] = time.Date(2026, 2, 8, 10, 0, 0, 0, time.UTC) - manifestStorage.manifests[manifestRefKey("gordon/api", "latest")] = mustManifestJSON(t, "sha256:cfg-latest", "sha256:layer-shared") - manifestStorage.manifests[manifestRefKey("gordon/api", "v2")] = mustManifestJSON(t, "sha256:cfg-v2", "sha256:layer-shared") - manifestStorage.manifests[manifestRefKey("gordon/api", "v1")] = mustManifestJSON(t, "sha256:cfg-v1", "sha256:layer-shared") - - blobStorage := &fakeBlobStorage{blobs: []string{"sha256:cfg-latest", "sha256:cfg-v2", "sha256:layer-shared", "sha256:cfg-v1"}} - svc := NewService(&fakeRuntime{}, manifestStorage, blobStorage, zerowrap.Default()) - - report, err := svc.PruneRegistry(context.Background(), 1) - - require.NoError(t, err) - assert.Equal(t, 1, report.Registry.BlobsRemoved) - assert.Equal(t, []string{"sha256:cfg-v1"}, blobStorage.deletedBlobs) -} - -func TestService_PruneRegistry_PreservesManifestListChildDigests(t *testing.T) { - manifestStorage := newFakeManifestStorage() - manifestStorage.repositories = []string{"gordon/api"} - manifestStorage.tagsByRepo["gordon/api"] = []string{"latest"} - manifestStorage.modTimes[manifestRefKey("gordon/api", "latest")] = time.Date(2026, 2, 8, 12, 0, 0, 0, time.UTC) - manifestStorage.manifests[manifestRefKey("gordon/api", "latest")] = mustManifestIndexJSON(t, "sha256:child-amd64", "sha256:child-arm64") - manifestStorage.manifests[manifestRefKey("gordon/api", "sha256:child-amd64")] = mustManifestJSON(t, "sha256:cfg-amd64", "sha256:layer-amd64") - manifestStorage.manifests[manifestRefKey("gordon/api", "sha256:child-arm64")] = mustManifestJSON(t, "sha256:cfg-arm64", "sha256:layer-arm64") - - blobStorage := &fakeBlobStorage{blobs: []string{ - "sha256:child-amd64", - "sha256:child-arm64", - "sha256:cfg-amd64", - "sha256:layer-amd64", - "sha256:cfg-arm64", - "sha256:layer-arm64", - "sha256:orphan", - }} - svc := NewService(&fakeRuntime{}, manifestStorage, blobStorage, zerowrap.Default()) - - report, err := svc.PruneRegistry(context.Background(), 1) - - require.NoError(t, err) - assert.Equal(t, 1, report.Registry.BlobsRemoved) - assert.Equal(t, []string{"sha256:orphan"}, blobStorage.deletedBlobs) -} - -func TestService_PruneRegistry_TieBreaksByTagName(t *testing.T) { - manifestStorage := newFakeManifestStorage() - manifestStorage.repositories = []string{"gordon/api"} - manifestStorage.tagsByRepo["gordon/api"] = []string{"v1", "v2"} - tie := time.Date(2026, 2, 8, 12, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/api", "v1")] = tie - manifestStorage.modTimes[manifestRefKey("gordon/api", "v2")] = tie - manifestStorage.manifests[manifestRefKey("gordon/api", "v2")] = mustManifestJSON(t, "sha256:cfg-v2", "sha256:layer-v2") - - blobStorage := &fakeBlobStorage{} - svc := NewService(&fakeRuntime{}, manifestStorage, blobStorage, zerowrap.Default()) - - report, err := svc.PruneRegistry(context.Background(), 1) - - require.NoError(t, err) - assert.Equal(t, 1, report.Registry.TagsRemoved) - assert.Equal(t, []manifestRef{{name: "gordon/api", reference: "v1"}}, manifestStorage.deletedManifests) -} - -func TestService_PruneRegistry_LogsWarningAndContinuesWhenChildManifestMissing(t *testing.T) { - manifestStorage := newFakeManifestStorage() - manifestStorage.repositories = []string{"gordon/api"} - manifestStorage.tagsByRepo["gordon/api"] = []string{"latest"} - manifestStorage.modTimes[manifestRefKey("gordon/api", "latest")] = time.Date(2026, 2, 8, 12, 0, 0, 0, time.UTC) - manifestStorage.manifests[manifestRefKey("gordon/api", "latest")] = mustManifestIndexJSON(t, "sha256:child-missing") - - blobStorage := &fakeBlobStorage{blobs: []string{"sha256:child-missing", "sha256:orphan"}} - logger := zerowrap.Default() - svc := NewService(&fakeRuntime{}, manifestStorage, blobStorage, logger) - - report, err := svc.PruneRegistry(context.Background(), 1) - - // Should not error - missing child manifests are logged as warnings and skipped - require.NoError(t, err) - assert.Equal(t, 0, report.Registry.TagsRemoved, "no tags should be removed (latest is kept)") - assert.Equal(t, 1, report.Registry.BlobsRemoved, "orphan blob should be garbage collected") -} - -func TestService_PruneRegistry_MultipleRepositories(t *testing.T) { - manifestStorage := newFakeManifestStorage() - manifestStorage.repositories = []string{"gordon/api", "gordon/web"} - manifestStorage.tagsByRepo["gordon/api"] = []string{"latest", "v2", "v1"} - manifestStorage.tagsByRepo["gordon/web"] = []string{"latest", "v3", "v2"} - manifestStorage.modTimes[manifestRefKey("gordon/api", "latest")] = time.Date(2026, 2, 8, 12, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/api", "v2")] = time.Date(2026, 2, 8, 11, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/api", "v1")] = time.Date(2026, 2, 8, 10, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/web", "latest")] = time.Date(2026, 2, 8, 12, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/web", "v3")] = time.Date(2026, 2, 8, 11, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/web", "v2")] = time.Date(2026, 2, 8, 10, 0, 0, 0, time.UTC) - manifestStorage.manifests[manifestRefKey("gordon/api", "latest")] = mustManifestJSON(t, "sha256:cfg-api-latest", "sha256:layer-api-latest") - manifestStorage.manifests[manifestRefKey("gordon/api", "v2")] = mustManifestJSON(t, "sha256:cfg-api-v2", "sha256:layer-api-v2") - manifestStorage.manifests[manifestRefKey("gordon/web", "latest")] = mustManifestJSON(t, "sha256:cfg-web-latest", "sha256:layer-web-latest") - manifestStorage.manifests[manifestRefKey("gordon/web", "v3")] = mustManifestJSON(t, "sha256:cfg-web-v3", "sha256:layer-web-v3") - - blobStorage := &fakeBlobStorage{} - svc := NewService(&fakeRuntime{}, manifestStorage, blobStorage, zerowrap.Default()) - - report, err := svc.PruneRegistry(context.Background(), 1) - - require.NoError(t, err) - assert.Equal(t, 2, report.Registry.TagsRemoved) - assert.ElementsMatch(t, []manifestRef{ - {name: "gordon/api", reference: "v1"}, - {name: "gordon/web", reference: "v2"}, - }, manifestStorage.deletedManifests) -} - -func TestService_PruneRegistry_EmptyRepository(t *testing.T) { - manifestStorage := newFakeManifestStorage() - manifestStorage.repositories = []string{"gordon/api"} - manifestStorage.tagsByRepo["gordon/api"] = []string{} - blobStorage := &fakeBlobStorage{} - svc := NewService(&fakeRuntime{}, manifestStorage, blobStorage, zerowrap.Default()) - - report, err := svc.PruneRegistry(context.Background(), 2) - - require.NoError(t, err) - assert.Equal(t, domain.RegistryPruneResult{}, report.Registry) - assert.Empty(t, manifestStorage.deletedManifests) - assert.Empty(t, blobStorage.deletedBlobs) -} - -func TestService_Prune_IdempotentMultiplePrunesDoNotAccumulate(t *testing.T) { - manifestStorage := newFakeManifestStorage() - manifestStorage.repositories = []string{"gordon/api"} - manifestStorage.tagsByRepo["gordon/api"] = []string{"latest", "v2", "v1"} - manifestStorage.modTimes[manifestRefKey("gordon/api", "latest")] = time.Date(2026, 2, 8, 12, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/api", "v2")] = time.Date(2026, 2, 8, 11, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/api", "v1")] = time.Date(2026, 2, 8, 10, 0, 0, 0, time.UTC) - manifestStorage.manifests[manifestRefKey("gordon/api", "latest")] = mustManifestJSON(t, "sha256:cfg-live", "sha256:layer-live") - manifestStorage.manifests[manifestRefKey("gordon/api", "v2")] = mustManifestJSON(t, "sha256:cfg-v2", "sha256:layer-v2") - blobStorage := &fakeBlobStorage{blobs: []string{"sha256:cfg-live", "sha256:layer-live", "sha256:orphan"}} - rt := &fakeRuntime{pruneReport: pkgruntime.PruneReport{DeletedIDs: []string{"sha256:img1"}, SpaceReclaimed: 1024}} - - svc := NewService(rt, manifestStorage, blobStorage, zerowrap.Default()) - - // First prune - should remove v1 and orphan - report1, err := svc.Prune(context.Background(), domain.ImagePruneOptions{ - KeepLast: 1, PruneDangling: true, PruneRegistry: true, - }) - require.NoError(t, err) - assert.Equal(t, 1, report1.Registry.TagsRemoved, "first prune should remove v1") - assert.Equal(t, 1, report1.Registry.BlobsRemoved, "first prune should remove orphan") - - // Reset runtime prune report to test idempotency properly (runtime would have nothing to prune second time) - rt.pruneReport = pkgruntime.PruneReport{} - - // Second prune - should be idempotent (nothing else to remove) - report2, err := svc.Prune(context.Background(), domain.ImagePruneOptions{ - KeepLast: 1, PruneDangling: true, PruneRegistry: true, - }) - require.NoError(t, err) - assert.Equal(t, 0, report2.Registry.TagsRemoved, "second prune should remove nothing (idempotent)") - assert.Equal(t, 0, report2.Registry.BlobsRemoved, "second prune should remove nothing (idempotent)") - assert.Equal(t, 0, report2.Runtime.DeletedCount, "second prune should remove no runtime images") - assert.Equal(t, int64(0), report2.Runtime.SpaceReclaimed, "second prune should reclaim no runtime space") -} - -func TestService_Prune_RunsBothRuntimeAndRegistry(t *testing.T) { - manifestStorage := newFakeManifestStorage() - manifestStorage.repositories = []string{"gordon/api"} - manifestStorage.tagsByRepo["gordon/api"] = []string{"latest", "v2", "v1"} - manifestStorage.modTimes[manifestRefKey("gordon/api", "latest")] = time.Date(2026, 2, 8, 12, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/api", "v2")] = time.Date(2026, 2, 8, 11, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/api", "v1")] = time.Date(2026, 2, 8, 10, 0, 0, 0, time.UTC) - manifestStorage.manifests[manifestRefKey("gordon/api", "latest")] = mustManifestJSON(t, "sha256:cfg-live", "sha256:layer-live") - manifestStorage.manifests[manifestRefKey("gordon/api", "v2")] = mustManifestJSON(t, "sha256:cfg-v2", "sha256:layer-v2") - blobStorage := &fakeBlobStorage{blobs: []string{"sha256:cfg-live", "sha256:layer-live", "sha256:orphan"}} - rt := &fakeRuntime{pruneReport: pkgruntime.PruneReport{DeletedIDs: []string{"sha256:img1"}, SpaceReclaimed: 1024}} - - svc := NewService(rt, manifestStorage, blobStorage, zerowrap.Default()) - - report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{ - KeepLast: 1, PruneDangling: true, PruneRegistry: true, - }) - - require.NoError(t, err) - assert.Equal(t, 1, report.Runtime.DeletedCount) - assert.Equal(t, int64(1024), report.Runtime.SpaceReclaimed) - assert.Equal(t, 1, report.Registry.TagsRemoved) - assert.Equal(t, 1, report.Registry.BlobsRemoved) - assert.Equal(t, []manifestRef{{name: "gordon/api", reference: "v1"}}, manifestStorage.deletedManifests) - assert.Equal(t, []string{"sha256:orphan"}, blobStorage.deletedBlobs) -} - -func TestService_Prune_RuntimeFailureDoesNotBlockRegistry(t *testing.T) { - manifestStorage := newFakeManifestStorage() - manifestStorage.repositories = []string{"gordon/api"} - manifestStorage.tagsByRepo["gordon/api"] = []string{"latest", "v2", "v1"} - manifestStorage.modTimes[manifestRefKey("gordon/api", "latest")] = time.Date(2026, 2, 8, 12, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/api", "v2")] = time.Date(2026, 2, 8, 11, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/api", "v1")] = time.Date(2026, 2, 8, 10, 0, 0, 0, time.UTC) - manifestStorage.manifests[manifestRefKey("gordon/api", "latest")] = mustManifestJSON(t, "sha256:cfg-live", "sha256:layer-live") - manifestStorage.manifests[manifestRefKey("gordon/api", "v2")] = mustManifestJSON(t, "sha256:cfg-v2", "sha256:layer-v2") - blobStorage := &fakeBlobStorage{blobs: []string{"sha256:cfg-live", "sha256:layer-live", "sha256:orphan"}} - rt := &fakeRuntime{pruneErr: errors.New("runtime prune failed")} - - svc := NewService(rt, manifestStorage, blobStorage, zerowrap.Default()) - - report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{ - KeepLast: 1, PruneDangling: true, PruneRegistry: true, - }) - - require.NoError(t, err) - assert.Equal(t, 0, report.Runtime.DeletedCount) - assert.Equal(t, int64(0), report.Runtime.SpaceReclaimed) - assert.Equal(t, 1, report.Registry.TagsRemoved) - assert.Equal(t, 1, report.Registry.BlobsRemoved) - assert.Equal(t, []manifestRef{{name: "gordon/api", reference: "v1"}}, manifestStorage.deletedManifests) - assert.Equal(t, []string{"sha256:orphan"}, blobStorage.deletedBlobs) -} - -func TestService_Prune_DanglingOnlySkipsRegistry(t *testing.T) { - manifestStorage := newFakeManifestStorage() - manifestStorage.repositories = []string{"gordon/api"} - manifestStorage.tagsByRepo["gordon/api"] = []string{"latest", "v1"} - blobStorage := &fakeBlobStorage{} - rt := &fakeRuntime{pruneReport: pkgruntime.PruneReport{DeletedIDs: []string{"sha256:img1"}, SpaceReclaimed: 512}} - - svc := NewService(rt, manifestStorage, blobStorage, zerowrap.Default()) - - report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{ - KeepLast: 3, PruneDangling: true, PruneRegistry: false, - }) - - require.NoError(t, err) - assert.True(t, rt.pruneCalled, "runtime prune should have been called") - assert.Equal(t, 1, report.Runtime.DeletedCount) - assert.Equal(t, 0, report.Registry.TagsRemoved, "registry should not be pruned") - assert.Equal(t, 0, manifestStorage.listRepositoriesCalls, "should not list repositories when registry disabled") -} - -func TestService_Prune_DanglingOnlyRuntimeFailureReturnsError(t *testing.T) { - rt := &fakeRuntime{pruneErr: errors.New("docker daemon unavailable")} - svc := NewService(rt, newFakeManifestStorage(), &fakeBlobStorage{}, zerowrap.Default()) - - _, err := svc.Prune(context.Background(), domain.ImagePruneOptions{ - KeepLast: 3, PruneDangling: true, PruneRegistry: false, - }) - - require.Error(t, err) - assert.Contains(t, err.Error(), "runtime prune failed") - assert.Contains(t, err.Error(), "docker daemon unavailable") -} - -func TestService_PruneRegistry_CleansUpStaleUploads(t *testing.T) { - manifestStorage := newFakeManifestStorage() - blobStorage := &fakeCleanupBlobStorage{ - fakeBlobStorage: &fakeBlobStorage{}, - uploadsRemoved: 3, - uploadBytes: 1024 * 1024, - } - svc := NewService(&fakeRuntime{}, manifestStorage, blobStorage, zerowrap.Default()) - - report, err := svc.PruneRegistry(context.Background(), 5) - require.NoError(t, err) - assert.Equal(t, 3, report.Registry.UploadsRemoved) - assert.Equal(t, int64(1024*1024), report.Registry.UploadSpaceReclaimed) - assert.Equal(t, 24*time.Hour, blobStorage.capturedMaxAge) -} - -func TestService_Prune_RegistryOnlySkipsRuntime(t *testing.T) { - manifestStorage := newFakeManifestStorage() - manifestStorage.repositories = []string{"gordon/api"} - manifestStorage.tagsByRepo["gordon/api"] = []string{"latest", "v2", "v1"} - manifestStorage.modTimes[manifestRefKey("gordon/api", "latest")] = time.Date(2026, 2, 8, 12, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/api", "v2")] = time.Date(2026, 2, 8, 11, 0, 0, 0, time.UTC) - manifestStorage.modTimes[manifestRefKey("gordon/api", "v1")] = time.Date(2026, 2, 8, 10, 0, 0, 0, time.UTC) - manifestStorage.manifests[manifestRefKey("gordon/api", "latest")] = mustManifestJSON(t, "sha256:cfg-live", "sha256:layer-live") - manifestStorage.manifests[manifestRefKey("gordon/api", "v2")] = mustManifestJSON(t, "sha256:cfg-v2", "sha256:layer-v2") - blobStorage := &fakeBlobStorage{blobs: []string{"sha256:cfg-live", "sha256:layer-live", "sha256:orphan"}} - rt := &fakeRuntime{} - - svc := NewService(rt, manifestStorage, blobStorage, zerowrap.Default()) - - report, err := svc.Prune(context.Background(), domain.ImagePruneOptions{ - KeepLast: 1, PruneDangling: false, PruneRegistry: true, - }) - - require.NoError(t, err) - assert.False(t, rt.pruneCalled, "runtime prune should not have been called") - assert.Equal(t, 0, report.Runtime.DeletedCount) - assert.Equal(t, 1, report.Registry.TagsRemoved) - assert.Equal(t, 1, report.Registry.BlobsRemoved) +type noopBlobStorage struct { + out.BlobStorage } type fakeRuntime struct { @@ -665,19 +187,6 @@ type fakeRuntime struct { listDetails []pkgruntime.ImageDetail listErr error - - pruneReport pkgruntime.PruneReport - pruneErr error - pruneCalled bool - pruneDanglingOnly bool -} - -type noopManifestStorage struct { - out.ManifestStorage -} - -type noopBlobStorage struct { - out.BlobStorage } type manifestRef struct { @@ -695,39 +204,23 @@ type fakeManifestStorage struct { listRepositoriesCalls int deletedManifests []manifestRef + deleteErr error } type fakeBlobStorage struct { out.BlobStorage - blobs []string - blobSizes map[string]int64 - blobModTimes map[string]time.Time - deletedBlobs []string -} - -type fakeCleanupBlobStorage struct { - *fakeBlobStorage - uploadsRemoved int - uploadBytes int64 - capturedMaxAge time.Duration -} - -func (f *fakeCleanupBlobStorage) CleanupStaleUploads(maxAge time.Duration) (int, int64, error) { - f.capturedMaxAge = maxAge - return f.uploadsRemoved, f.uploadBytes, nil + blobs []string + blobSizes map[string]int64 + blobModTimes map[string]time.Time + blobModTimeErr error + deletedBlobs []string } func (f *fakeRuntime) ListImagesDetailed(context.Context) ([]pkgruntime.ImageDetail, error) { return f.listDetails, f.listErr } -func (f *fakeRuntime) PruneImages(_ context.Context, danglingOnly bool) (pkgruntime.PruneReport, error) { - f.pruneCalled = true - f.pruneDanglingOnly = danglingOnly - return f.pruneReport, f.pruneErr -} - func newFakeManifestStorage() *fakeManifestStorage { return &fakeManifestStorage{ tagsByRepo: make(map[string][]string), @@ -804,6 +297,9 @@ func (f *fakeManifestStorage) GetManifestModTime(name, reference string) (time.T } func (f *fakeManifestStorage) DeleteManifest(name, reference string) error { + if f.deleteErr != nil { + return f.deleteErr + } f.deletedManifests = append(f.deletedManifests, manifestRef{name: name, reference: reference}) tags := f.tagsByRepo[name] filtered := tags[:0] @@ -830,6 +326,9 @@ func (f *fakeBlobStorage) ListBlobs() ([]string, error) { } func (f *fakeBlobStorage) GetBlobModTime(digest string) (time.Time, error) { + if f.blobModTimeErr != nil { + return time.Time{}, f.blobModTimeErr + } return f.blobModTimes[digest], nil } diff --git a/internal/usecase/logexport/access.go b/internal/usecase/logexport/access.go new file mode 100644 index 000000000..af070dc3b --- /dev/null +++ b/internal/usecase/logexport/access.go @@ -0,0 +1,94 @@ +package logexport + +import ( + "context" + "fmt" + "net" + "strconv" + "strings" + + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/domain" +) + +// HostOwnerResolver maps a canonical HTTP host to the app service that +// serves it. apptraffic.HostIndex implements it. +type HostOwnerResolver interface { + HostOwner(host string) (app, service string, ok bool) +} + +// AccessLogExporter exports proxy access entries under the identity of +// the app service that owns the requested host, so one app's access +// and container logs share one source. Unowned hosts (registry, admin, +// unknown) are exported as Gordon's own records. +type AccessLogExporter struct { + hosts HostOwnerResolver + exporter out.LogExporter + next out.AccessLogWriter +} + +var _ out.AccessLogWriter = (*AccessLogExporter)(nil) + +// NewAccessLogExporter creates the exporter. next, when non-nil, also +// receives every entry (the local access log sink). +func NewAccessLogExporter(hosts HostOwnerResolver, exporter out.LogExporter, next out.AccessLogWriter) *AccessLogExporter { + return &AccessLogExporter{hosts: hosts, exporter: exporter, next: next} +} + +// Write implements out.AccessLogWriter. Export never fails the request; +// only the local sink can report an error. +func (a *AccessLogExporter) Write(entry out.AccessLogEntry) error { + a.exporter.Export(context.Background(), a.record(entry)) + if a.next != nil { + return a.next.Write(entry) + } + return nil +} + +func (a *AccessLogExporter) record(entry out.AccessLogEntry) domain.LogRecord { + host := domain.CanonicalHTTPHost(stripPort(entry.Host)) + path := strings.ToValidUTF8(entry.Path, "\uFFFD") + userAgent := strings.ToValidUTF8(entry.UserAgent, "\uFFFD") + var source domain.LogSource + if app, service, ok := a.hosts.HostOwner(host); ok { + source = domain.LogSource{App: app, Service: service} + } + return domain.LogRecord{ + Time: entry.Time, + Source: source, + Type: domain.LogTypeAccess, + Severity: accessSeverity(entry.Status), + Body: fmt.Sprintf("%s %s%s %d", entry.Method, host, path, entry.Status), + Attributes: map[string]string{ + "http.request.method": entry.Method, + "http.response.status_code": strconv.Itoa(entry.Status), + "server.address": host, + "url.path": path, + "client.address": entry.ClientIP, + "user_agent.original": userAgent, + "http.request.header.referer": entry.Referer, + "network.protocol.name": entry.Proto, + "http.response.body.size": strconv.Itoa(entry.BytesSent), + "http.server.duration_ms": strconv.FormatFloat(entry.DurationMS, 'f', 3, 64), + "request.id": entry.RequestID, + }, + } +} + +func stripPort(hostport string) string { + if host, _, err := net.SplitHostPort(hostport); err == nil { + return host + } + return hostport +} + +func accessSeverity(status int) domain.LogSeverity { + switch { + case status >= 500: + return domain.LogSeverityError + case status >= 400: + return domain.LogSeverityWarn + default: + return domain.LogSeverityInfo + } +} diff --git a/internal/usecase/logexport/access_test.go b/internal/usecase/logexport/access_test.go new file mode 100644 index 000000000..b010c5577 --- /dev/null +++ b/internal/usecase/logexport/access_test.go @@ -0,0 +1,104 @@ +package logexport + +import ( + "context" + "errors" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" +) + +type fakeHosts map[string][2]string + +func (f fakeHosts) HostOwner(host string) (string, string, bool) { + owner, ok := f[host] + return owner[0], owner[1], ok +} + +type fakeAccessWriter struct { + entries []out.AccessLogEntry + err error +} + +func (f *fakeAccessWriter) Write(entry out.AccessLogEntry) error { + f.entries = append(f.entries, entry) + return f.err +} + +func captureExport(t *testing.T) (*mocks.MockLogExporter, *[]domain.LogRecord) { + var records []domain.LogRecord + exporter := mocks.NewMockLogExporter(t) + exporter.EXPECT().Export(mock.Anything, mock.Anything).Run(func(_ context.Context, r domain.LogRecord) { + records = append(records, r) + }).Return() + return exporter, &records +} + +func TestAccessLogExporter_AttributesOwnedHostToAppService(t *testing.T) { + exporter, records := captureExport(t) + next := &fakeAccessWriter{} + a := NewAccessLogExporter(fakeHosts{"blog.example.com": {"blog", "web"}}, exporter, next) + ts := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + + err := a.Write(out.AccessLogEntry{Time: ts, Method: "GET", Host: "Blog.Example.com:443", Path: "/x", Status: 502}) + + require.NoError(t, err) + require.Len(t, *records, 1) + record := (*records)[0] + assert.Equal(t, domain.LogSource{App: "blog", Service: "web"}, record.Source) + assert.Equal(t, domain.LogTypeAccess, record.Type) + assert.Equal(t, domain.LogSeverityError, record.Severity) + assert.Equal(t, ts, record.Time) + assert.Equal(t, "GET blog.example.com/x 502", record.Body) + assert.Equal(t, "502", record.Attributes["http.response.status_code"]) + assert.Len(t, next.entries, 1, "local sink still receives the entry") +} + +func TestAccessLogExporter_UnownedHostIsGordon(t *testing.T) { + exporter, records := captureExport(t) + a := NewAccessLogExporter(fakeHosts{}, exporter, nil) + + require.NoError(t, a.Write(out.AccessLogEntry{Host: "registry.example.com", Status: 200})) + + require.Len(t, *records, 1) + assert.True(t, (*records)[0].Source.IsGordon()) + assert.Equal(t, domain.LogSeverityInfo, (*records)[0].Severity) +} + +func TestAccessLogExporter_ReturnsLocalSinkError(t *testing.T) { + exporter, _ := captureExport(t) + sinkErr := errors.New("disk full") + a := NewAccessLogExporter(fakeHosts{}, exporter, &fakeAccessWriter{err: sinkErr}) + + assert.ErrorIs(t, a.Write(out.AccessLogEntry{Status: 404}), sinkErr) +} + +func TestAccessLogExporter_DropsQueryAttribute(t *testing.T) { + exporter, records := captureExport(t) + a := NewAccessLogExporter(fakeHosts{}, exporter, nil) + + require.NoError(t, a.Write(out.AccessLogEntry{Path: "/search", Query: "q=secret", Status: 200})) + + require.Len(t, *records, 1) + _, ok := (*records)[0].Attributes["url.query"] + assert.False(t, ok, "query must not be exported over OTLP") + assert.Equal(t, "/search", (*records)[0].Attributes["url.path"]) +} + +func TestAccessLogExporter_SanitizesInvalidUTF8(t *testing.T) { + exporter, records := captureExport(t) + a := NewAccessLogExporter(fakeHosts{}, exporter, nil) + + require.NoError(t, a.Write(out.AccessLogEntry{Path: "/bad\xff", UserAgent: "agent\xff", Status: 200})) + + require.Len(t, *records, 1) + assert.Equal(t, "/bad\uFFFD", (*records)[0].Attributes["url.path"]) + assert.Equal(t, "agent\uFFFD", (*records)[0].Attributes["user_agent.original"]) +} diff --git a/internal/usecase/logexport/collector.go b/internal/usecase/logexport/collector.go new file mode 100644 index 000000000..492df86c0 --- /dev/null +++ b/internal/usecase/logexport/collector.go @@ -0,0 +1,218 @@ +// Package logexport follows app container output and exports it as +// log records, one source identity per app service. +package logexport + +import ( + "context" + "errors" + "fmt" + "sync" + "time" + + "github.com/bnema/zerowrap" + + "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/domain" +) + +const ( + defaultReconcileInterval = 10 * time.Second + defaultRetryDelay = 2 * time.Second +) + +// Collector keeps one follower per exportable ACTIVE container. +// +// A container is exportable when its app is not stopped and its service +// did not opt out (domain.AppService.LogExportDisabled). Reconciliation +// polls ACTIVE state so deploys, stops, and opt-out changes converge +// without coupling the deployment engine to log export. +type Collector struct { + state out.AppStateReader + streamer out.ContainerLogStreamer + exporter out.LogExporter + + reconcileInterval time.Duration + retryDelay time.Duration + // startedAt bounds replay: a follower exports output emitted since + // the collector was created, so a container deployed later is + // exported from its first line, while restarting Gordon never + // re-exports older output. + startedAt time.Time + + mu sync.Mutex + followers map[string]*follower + cursors map[string]time.Time + wg sync.WaitGroup +} + +type follower struct { + source domain.LogSource + cancel context.CancelFunc +} + +// NewCollector creates a collector. Call Run to start it. +func NewCollector(state out.AppStateReader, streamer out.ContainerLogStreamer, exporter out.LogExporter) *Collector { + return &Collector{ + state: state, + streamer: streamer, + exporter: exporter, + reconcileInterval: defaultReconcileInterval, + retryDelay: defaultRetryDelay, + startedAt: time.Now(), + followers: map[string]*follower{}, + cursors: map[string]time.Time{}, + } +} + +// Run reconciles until ctx is canceled, then stops every follower and +// waits for them to exit. +func (c *Collector) Run(ctx context.Context) { + ctx = zerowrap.CtxWithFields(ctx, map[string]any{ + zerowrap.FieldLayer: "usecase", + zerowrap.FieldUseCase: "LogExport", + }) + defer c.stopAll() + log := zerowrap.FromCtx(ctx) + + ticker := time.NewTicker(c.reconcileInterval) + defer ticker.Stop() + for { + if err := c.Reconcile(ctx); err != nil && ctx.Err() == nil { + log.Warn().Err(err).Msg("log export reconciliation failed") + } + select { + case <-ctx.Done(): + return + case <-ticker.C: + } + } +} + +// Reconcile starts followers for new exportable containers and stops +// followers whose container is gone, stopped, or opted out. +func (c *Collector) Reconcile(ctx context.Context) error { + desired, activeIDs, err := c.desiredContainers(ctx) + if err != nil { + return err + } + + c.mu.Lock() + defer c.mu.Unlock() + for containerID, f := range c.followers { + if source, ok := desired[containerID]; !ok || source != f.source { + f.cancel() + delete(c.followers, containerID) + } + } + for containerID, source := range desired { + if _, ok := c.followers[containerID]; ok { + continue + } + c.startLocked(ctx, containerID, source) + } + for containerID := range c.cursors { + if _, ok := activeIDs[containerID]; !ok { + delete(c.cursors, containerID) + } + } + return nil +} + +// desiredContainers maps exportable ACTIVE container IDs to their source, +// plus the set of every container ID referenced by an ACTIVE service +// (exportable or not) so callers can retire per-container state only when +// the ID is gone from all ACTIVE services. +func (c *Collector) desiredContainers(ctx context.Context) (map[string]domain.LogSource, map[string]struct{}, error) { + apps, err := c.state.ListApps(ctx) + if err != nil { + return nil, nil, fmt.Errorf("logexport: list apps: %w", err) + } + desired := map[string]domain.LogSource{} + activeIDs := map[string]struct{}{} + for _, app := range apps { + active, found, err := c.state.LoadActive(ctx, app) + if err != nil { + return nil, nil, fmt.Errorf("logexport: load active state for app %q: %w", app, err) + } + if !found || active.StopIntent { + continue + } + for name, svc := range active.Services { + if svc.Container == "" { + continue + } + activeIDs[svc.Container] = struct{}{} + if svc.Spec.LogExportDisabled { + continue + } + desired[svc.Container] = domain.LogSource{App: app, Service: name} + } + } + return desired, activeIDs, nil +} + +func (c *Collector) startLocked(parent context.Context, containerID string, source domain.LogSource) { + ctx, cancel := context.WithCancel(parent) + ctx = zerowrap.CtxWithFields(ctx, map[string]any{ + zerowrap.FieldEntityID: containerID, + "app": source.App, + "service": source.Service, + }) + c.followers[containerID] = &follower{source: source, cancel: cancel} + since := c.startedAt + if cursor, ok := c.cursors[containerID]; ok && cursor.After(since) { + since = cursor + } + c.wg.Add(1) + go func() { + defer c.wg.Done() + c.follow(ctx, containerID, source, since) + }() +} + +// follow streams one container until canceled, starting at since. +// A stream that ends (container restarting) resumes after the last +// exported line, so a reconnect neither duplicates nor replays output. +func (c *Collector) follow(ctx context.Context, containerID string, source domain.LogSource, since time.Time) { + log := zerowrap.FromCtx(ctx) + for { + err := c.streamer.StreamContainerLogs(ctx, containerID, since, func(line domain.ContainerLogLine) { + c.exporter.Export(ctx, domain.LogRecord{ + Time: line.Time, + Source: source, + Type: domain.LogTypeContainer, + Stream: line.Stream, + Body: line.Body, + }) + if line.Time.After(since) { + since = line.Time.Add(time.Nanosecond) + c.mu.Lock() + if cursor, ok := c.cursors[containerID]; !ok || since.After(cursor) { + c.cursors[containerID] = since + } + c.mu.Unlock() + } + }) + if ctx.Err() != nil { + return + } + if err != nil && !errors.Is(err, domain.ErrContainerNotFound) { + log.Debug().Err(err).Msg("container log stream interrupted, retrying") + } + select { + case <-ctx.Done(): + return + case <-time.After(c.retryDelay): + } + } +} + +func (c *Collector) stopAll() { + c.mu.Lock() + for containerID, f := range c.followers { + f.cancel() + delete(c.followers, containerID) + } + c.mu.Unlock() + c.wg.Wait() +} diff --git a/internal/usecase/logexport/collector_test.go b/internal/usecase/logexport/collector_test.go new file mode 100644 index 000000000..b261099c3 --- /dev/null +++ b/internal/usecase/logexport/collector_test.go @@ -0,0 +1,298 @@ +package logexport + +import ( + "context" + "errors" + "sync" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + + "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/domain" +) + +func activeApp(app string, stopped bool, services map[string]domain.AppEffectiveService) domain.AppActive { + return domain.AppActive{App: app, StopIntent: stopped, Services: services} +} + +func svc(container string, disabled bool) domain.AppEffectiveService { + return domain.AppEffectiveService{Container: container, Spec: domain.AppService{LogExportDisabled: disabled}} +} + +// blockingStreamer blocks each follower until canceled and records calls. +type blockingStreamer struct { + mu sync.Mutex + started map[string]int + stopped map[string]int +} + +func newBlockingStreamer() *blockingStreamer { + return &blockingStreamer{started: map[string]int{}, stopped: map[string]int{}} +} + +func (s *blockingStreamer) StreamContainerLogs(ctx context.Context, containerID string, _ time.Time, _ func(domain.ContainerLogLine)) error { + s.mu.Lock() + s.started[containerID]++ + s.mu.Unlock() + <-ctx.Done() + s.mu.Lock() + s.stopped[containerID]++ + s.mu.Unlock() + return ctx.Err() +} + +func (s *blockingStreamer) counts(containerID string) (int, int) { + s.mu.Lock() + defer s.mu.Unlock() + return s.started[containerID], s.stopped[containerID] +} + +func TestCollector_FollowsOnlyExportableContainers(t *testing.T) { + state := mocks.NewMockAppStateReader(t) + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog", "paused"}, nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(activeApp("blog", false, map[string]domain.AppEffectiveService{ + "web": svc("c-web", false), + "db": svc("c-db", true), // service opted out + "new": svc("", false), // no container yet + }), true, nil) + state.EXPECT().LoadActive(mock.Anything, "paused").Return(activeApp("paused", true, map[string]domain.AppEffectiveService{ + "web": svc("c-paused", false), + }), true, nil) + streamer := newBlockingStreamer() + c := NewCollector(state, streamer, mocks.NewMockLogExporter(t)) + + ctx, cancel := context.WithCancel(context.Background()) + require.NoError(t, c.Reconcile(ctx)) + require.Eventually(t, func() bool { started, _ := streamer.counts("c-web"); return started == 1 }, time.Second, time.Millisecond) + cancel() + c.stopAll() + + for _, id := range []string{"c-db", "c-paused"} { + started, _ := streamer.counts(id) + assert.Zero(t, started, id) + } + _, stopped := streamer.counts("c-web") + assert.Equal(t, 1, stopped, "stopAll waits for followers") +} + +func TestCollector_StopsFollowerWhenServiceOptsOut(t *testing.T) { + state := mocks.NewMockAppStateReader(t) + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(activeApp("blog", false, map[string]domain.AppEffectiveService{ + "web": svc("c-web", false), + }), true, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(activeApp("blog", false, map[string]domain.AppEffectiveService{ + "web": svc("c-web", true), + }), true, nil).Once() + streamer := newBlockingStreamer() + c := NewCollector(state, streamer, mocks.NewMockLogExporter(t)) + defer c.stopAll() + + require.NoError(t, c.Reconcile(context.Background())) + require.NoError(t, c.Reconcile(context.Background())) + + require.Eventually(t, func() bool { _, stopped := streamer.counts("c-web"); return stopped == 1 }, time.Second, time.Millisecond) +} + +func TestCollector_ExportsLinesWithSourceIdentity(t *testing.T) { + state := mocks.NewMockAppStateReader(t) + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(activeApp("blog", false, map[string]domain.AppEffectiveService{ + "web": svc("c-web", false), + }), true, nil) + + ts := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + streamer := mocks.NewMockContainerLogStreamer(t) + streamer.EXPECT().StreamContainerLogs(mock.Anything, "c-web", mock.Anything, mock.Anything). + RunAndReturn(func(ctx context.Context, _ string, _ time.Time, emit func(domain.ContainerLogLine)) error { + emit(domain.ContainerLogLine{Time: ts, Stream: domain.LogStreamStderr, Body: "boom"}) + <-ctx.Done() + return ctx.Err() + }) + + exported := make(chan domain.LogRecord, 1) + exporter := mocks.NewMockLogExporter(t) + exporter.EXPECT().Export(mock.Anything, mock.Anything).Run(func(_ context.Context, r domain.LogRecord) { + exported <- r + }).Return() + + c := NewCollector(state, streamer, exporter) + defer c.stopAll() + require.NoError(t, c.Reconcile(context.Background())) + + select { + case record := <-exported: + assert.Equal(t, domain.LogRecord{ + Time: ts, + Source: domain.LogSource{App: "blog", Service: "web"}, + Type: domain.LogTypeContainer, + Stream: domain.LogStreamStderr, + Body: "boom", + }, record) + assert.Equal(t, "blog.web", record.Source.ServiceName()) + case <-time.After(time.Second): + t.Fatal("no record exported") + } +} + +func TestCollector_ResumesAfterLastLineOnReconnect(t *testing.T) { + state := mocks.NewMockAppStateReader(t) + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(activeApp("blog", false, map[string]domain.AppEffectiveService{ + "web": svc("c-web", false), + }), true, nil) + + start := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + last := start.Add(time.Minute) + streamer := mocks.NewMockContainerLogStreamer(t) + streamer.EXPECT().StreamContainerLogs(mock.Anything, "c-web", start, mock.Anything). + RunAndReturn(func(_ context.Context, _ string, _ time.Time, emit func(domain.ContainerLogLine)) error { + emit(domain.ContainerLogLine{Time: last, Body: "line"}) + return nil // stream ended: container restarting + }).Once() + resumed := make(chan time.Time, 1) + streamer.EXPECT().StreamContainerLogs(mock.Anything, "c-web", mock.Anything, mock.Anything). + RunAndReturn(func(ctx context.Context, _ string, since time.Time, _ func(domain.ContainerLogLine)) error { + resumed <- since + <-ctx.Done() + return ctx.Err() + }).Once() + exporter := mocks.NewMockLogExporter(t) + exporter.EXPECT().Export(mock.Anything, mock.Anything).Return() + + c := NewCollector(state, streamer, exporter) + c.startedAt = start + c.retryDelay = time.Millisecond + defer c.stopAll() + require.NoError(t, c.Reconcile(context.Background())) + + select { + case since := <-resumed: + assert.Equal(t, last.Add(time.Nanosecond), since) + case <-time.After(time.Second): + t.Fatal("follower did not reconnect") + } +} + +func TestCollector_ReconcileWrapsStateErrors(t *testing.T) { + listErr := assert.AnError + state := mocks.NewMockAppStateReader(t) + state.EXPECT().ListApps(mock.Anything).Return(nil, listErr) + c := NewCollector(state, newBlockingStreamer(), mocks.NewMockLogExporter(t)) + defer c.stopAll() + + err := c.Reconcile(context.Background()) + require.Error(t, err) + assert.ErrorIs(t, err, listErr) + + loadErr := assert.AnError + state2 := mocks.NewMockAppStateReader(t) + state2.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil) + state2.EXPECT().LoadActive(mock.Anything, "blog").Return(domain.AppActive{}, false, loadErr) + c2 := NewCollector(state2, newBlockingStreamer(), mocks.NewMockLogExporter(t)) + defer c2.stopAll() + + err = c2.Reconcile(context.Background()) + require.Error(t, err) + assert.ErrorIs(t, err, loadErr) +} + +func TestCollector_RecreatedFollowResumesFromSavedCursor(t *testing.T) { + state := mocks.NewMockAppStateReader(t) + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(activeApp("blog", false, map[string]domain.AppEffectiveService{ + "web": svc("c-web", false), + }), true, nil) + + start := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + last := start.Add(time.Minute) + streamed := make(chan time.Time, 4) + streamer := mocks.NewMockContainerLogStreamer(t) + first := make(chan struct{}) + // The first stream emits one line then ends (container restart); + // later streams only record their resume point. + streamer.EXPECT().StreamContainerLogs(mock.Anything, "c-web", mock.Anything, mock.Anything). + RunAndReturn(func(_ context.Context, _ string, since time.Time, emit func(domain.ContainerLogLine)) error { + select { + case streamed <- since: + default: + } + select { + case <-first: + return errors.New("boom") + default: + close(first) + emit(domain.ContainerLogLine{Time: last, Body: "line"}) + return nil + } + }).Maybe() + exporter := mocks.NewMockLogExporter(t) + exporter.EXPECT().Export(mock.Anything, mock.Anything).Return().Maybe() + + c := NewCollector(state, streamer, exporter) + c.startedAt = start + c.retryDelay = 50 * time.Millisecond + defer c.stopAll() + require.NoError(t, c.Reconcile(context.Background())) + + select { + case since := <-streamed: + assert.Equal(t, start, since, "first follow starts at collector start") + case <-time.After(2 * time.Second): + t.Fatal("first follow did not start") + } + + // The emitted line must advance the saved cursor past startedAt. + require.Eventually(t, func() bool { + c.mu.Lock() + defer c.mu.Unlock() + cursor, ok := c.cursors["c-web"] + return ok && cursor.Equal(last.Add(time.Nanosecond)) + }, 2*time.Second, 10*time.Millisecond, "cursor must advance past the exported line") + + // Simulate the supervision loop recreating the follow (source + // reassignment): the new follow must resume from the saved cursor. + c.mu.Lock() + if f, ok := c.followers["c-web"]; ok { + f.cancel() + delete(c.followers, "c-web") + } + c.mu.Unlock() + require.NoError(t, c.Reconcile(context.Background())) + + select { + case since := <-streamed: + assert.Equal(t, last.Add(time.Nanosecond), since, "recreated follow resumes from the saved cursor") + case <-time.After(2 * time.Second): + t.Fatal("recreated follow did not start") + } +} + +func TestCollector_RetiresCursorWhenIDLeavesActive(t *testing.T) { + state := mocks.NewMockAppStateReader(t) + state.EXPECT().ListApps(mock.Anything).Return([]string{"blog"}, nil).Times(2) + state.EXPECT().LoadActive(mock.Anything, "blog").Return(activeApp("blog", false, map[string]domain.AppEffectiveService{ + "web": svc("c-web", false), + }), true, nil).Once() + state.EXPECT().LoadActive(mock.Anything, "blog").Return(activeApp("blog", false, map[string]domain.AppEffectiveService{ + "web": svc("c-new", false), + }), true, nil).Once() + streamer := newBlockingStreamer() + c := NewCollector(state, streamer, mocks.NewMockLogExporter(t)) + defer c.stopAll() + + require.NoError(t, c.Reconcile(context.Background())) + c.mu.Lock() + c.cursors["c-web"] = time.Now() + c.mu.Unlock() + require.NoError(t, c.Reconcile(context.Background())) + + c.mu.Lock() + _, ok := c.cursors["c-web"] + c.mu.Unlock() + assert.False(t, ok, "cursor is dropped once the ID is gone from ACTIVE services") +} diff --git a/internal/usecase/logs/service.go b/internal/usecase/logs/service.go index c2663e165..f08f9a618 100644 --- a/internal/usecase/logs/service.go +++ b/internal/usecase/logs/service.go @@ -16,38 +16,68 @@ import ( "github.com/bnema/zerowrap" - "github.com/bnema/gordon/internal/boundaries/in" "github.com/bnema/gordon/internal/boundaries/out" + "github.com/bnema/gordon/internal/domain" ) +// appStateProvider resolves app services from the durable ACTIVE record. +type appStateProvider interface { + LoadActive(ctx context.Context, app string) (domain.AppActive, bool, error) +} + // Service implements the LogService interface. type Service struct { logFilePath string fileLoggingEnabled bool - containerSvc in.ContainerService runtime out.ContainerRuntime log zerowrap.Logger + // appState resolves log references to containers recorded in ACTIVE. + appState appStateProvider } var execCommandContext = exec.CommandContext -// NewService creates a new log service. +// NewService creates a new log service. App state is wired after the store opens. func NewService( logFilePath string, fileLoggingEnabled bool, - containerSvc in.ContainerService, runtime out.ContainerRuntime, log zerowrap.Logger, ) *Service { return &Service{ logFilePath: logFilePath, fileLoggingEnabled: fileLoggingEnabled, - containerSvc: containerSvc, runtime: runtime, log: log, } } +// WithAppState wires the durable ACTIVE app-state reader used for log resolution. +func (s *Service) WithAppState(provider appStateProvider) *Service { + s.appState = provider + return s +} + +// containerForService resolves an app/service reference exclusively through ACTIVE. +func (s *Service) containerForService(ctx context.Context, ref string) (string, error) { + app, service, ok := strings.Cut(ref, "/") + if !ok || app == "" || service == "" || strings.Contains(service, "/") || s.appState == nil { + return "", fmt.Errorf("invalid app/service log reference: %s", ref) + } + active, found, err := s.appState.LoadActive(ctx, app) + if err != nil { + return "", fmt.Errorf("failed to load active app %s: %w", app, err) + } + if !found { + return "", fmt.Errorf("active app/service not found: %s", ref) + } + target, found := active.Services[service] + if !found || target.Container == "" { + return "", fmt.Errorf("active app/service not found: %s", ref) + } + return target.Container, nil +} + // GetProcessLogs returns the last N lines of Gordon process logs. func (s *Service) GetProcessLogs(ctx context.Context, lines int) ([]string, error) { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ @@ -357,14 +387,14 @@ func (s *Service) GetContainerLogs(ctx context.Context, domain string, lines int }) log := zerowrap.FromCtx(ctx) - // Get container by domain - container, ok := s.containerSvc.Get(ctx, domain) - if !ok || container == nil { - return nil, fmt.Errorf("container not found for domain: %s", domain) + // Resolve the app/service to its ACTIVE container. + containerID, err := s.containerForService(ctx, domain) + if err != nil { + return nil, err } // Get logs from container runtime (non-follow mode) - reader, err := s.runtime.GetContainerLogs(ctx, container.ID, false) + reader, err := s.runtime.GetContainerLogs(ctx, containerID, false) if err != nil { return nil, log.WrapErr(err, "failed to get container logs") } @@ -398,14 +428,14 @@ func (s *Service) FollowContainerLogs(ctx context.Context, domain string, initia }) log := zerowrap.FromCtx(ctx) - // Get container by domain - container, ok := s.containerSvc.Get(ctx, domain) - if !ok || container == nil { - return nil, fmt.Errorf("container not found for domain: %s", domain) + // Resolve the app/service to its ACTIVE container. + containerID, err := s.containerForService(ctx, domain) + if err != nil { + return nil, err } // Get logs from container runtime (follow mode) - reader, err := s.runtime.GetContainerLogs(ctx, container.ID, true) + reader, err := s.runtime.GetContainerLogs(ctx, containerID, true) if err != nil { return nil, log.WrapErr(err, "failed to get container logs") } @@ -426,11 +456,17 @@ func (s *Service) FollowContainerLogs(ctx context.Context, domain string, initia return ch, nil } +// maxTailLines bounds the ring buffer allocated by tailLines. +const maxTailLines = 10000 + // tailLines reads the last N lines from a file using a ring buffer. func tailLines(file *os.File, n int) ([]string, error) { if n <= 0 { return []string{}, nil } + if n > maxTailLines { + n = maxTailLines + } // Seek to beginning if _, err := file.Seek(0, io.SeekStart); err != nil { diff --git a/internal/usecase/logs/service_test.go b/internal/usecase/logs/service_test.go index 13a5757f7..531138b75 100644 --- a/internal/usecase/logs/service_test.go +++ b/internal/usecase/logs/service_test.go @@ -14,8 +14,7 @@ import ( "github.com/stretchr/testify/mock" "github.com/stretchr/testify/require" - "github.com/bnema/gordon/internal/boundaries/in/mocks" - outMocks "github.com/bnema/gordon/internal/boundaries/out/mocks" + "github.com/bnema/gordon/internal/boundaries/out/mocks" "github.com/bnema/gordon/internal/domain" ) @@ -30,10 +29,9 @@ func TestService_GetProcessLogs(t *testing.T) { err := os.WriteFile(logPath, []byte(content), 0644) require.NoError(t, err) - containerSvc := mocks.NewMockContainerService(t) - runtime := outMocks.NewMockContainerRuntime(t) + runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(logPath, true, containerSvc, runtime, log) + svc := NewService(logPath, true, runtime, log) lines, err := svc.GetProcessLogs(context.Background(), 3) require.NoError(t, err) @@ -42,10 +40,9 @@ func TestService_GetProcessLogs(t *testing.T) { }) t.Run("returns empty slice for non-existent file", func(t *testing.T) { - containerSvc := mocks.NewMockContainerService(t) - runtime := outMocks.NewMockContainerRuntime(t) + runtime := mocks.NewMockContainerRuntime(t) - svc := NewService("/nonexistent/file.log", true, containerSvc, runtime, log) + svc := NewService("/nonexistent/file.log", true, runtime, log) lines, err := svc.GetProcessLogs(context.Background(), 10) require.NoError(t, err) @@ -53,10 +50,9 @@ func TestService_GetProcessLogs(t *testing.T) { }) t.Run("returns error when log file path not configured", func(t *testing.T) { - containerSvc := mocks.NewMockContainerService(t) - runtime := outMocks.NewMockContainerRuntime(t) + runtime := mocks.NewMockContainerRuntime(t) - svc := NewService("", true, containerSvc, runtime, log) + svc := NewService("", true, runtime, log) _, err := svc.GetProcessLogs(context.Background(), 10) assert.Error(t, err) @@ -70,10 +66,9 @@ func TestService_GetProcessLogs(t *testing.T) { err := os.WriteFile(logPath, []byte(content), 0644) require.NoError(t, err) - containerSvc := mocks.NewMockContainerService(t) - runtime := outMocks.NewMockContainerRuntime(t) + runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(logPath, true, containerSvc, runtime, log) + svc := NewService(logPath, true, runtime, log) lines, err := svc.GetProcessLogs(context.Background(), 10) require.NoError(t, err) @@ -89,10 +84,9 @@ func TestService_GetProcessLogs(t *testing.T) { execCommandContext = origExec }() - containerSvc := mocks.NewMockContainerService(t) - runtime := outMocks.NewMockContainerRuntime(t) + runtime := mocks.NewMockContainerRuntime(t) - svc := NewService("/tmp/unused.log", false, containerSvc, runtime, log) + svc := NewService("/tmp/unused.log", false, runtime, log) lines, err := svc.GetProcessLogs(context.Background(), 10) require.NoError(t, err) @@ -110,10 +104,9 @@ func TestService_FollowProcessLogs(t *testing.T) { err := os.WriteFile(logPath, []byte(content), 0644) require.NoError(t, err) - containerSvc := mocks.NewMockContainerService(t) - runtime := outMocks.NewMockContainerRuntime(t) + runtime := mocks.NewMockContainerRuntime(t) - svc := NewService(logPath, true, containerSvc, runtime, log) + svc := NewService(logPath, true, runtime, log) ctx, cancel := context.WithCancel(context.Background()) defer cancel() @@ -150,35 +143,46 @@ func TestService_GetContainerLogs(t *testing.T) { log := zerowrap.New(zerowrap.Config{Level: "warn"}) t.Run("returns error when container not found", func(t *testing.T) { - containerSvc := mocks.NewMockContainerService(t) - runtime := outMocks.NewMockContainerRuntime(t) + runtime := mocks.NewMockContainerRuntime(t) - containerSvc.EXPECT().Get(mock.Anything, "unknown.local").Return(nil, false) + svc := NewService("/tmp/test.log", true, runtime, log) + svc.WithAppState(stubActiveApps{}) - svc := NewService("/tmp/test.log", true, containerSvc, runtime, log) - - _, err := svc.GetContainerLogs(context.Background(), "unknown.local", 10) + _, err := svc.GetContainerLogs(context.Background(), "unknown/web", 10) assert.Error(t, err) - assert.Contains(t, err.Error(), "container not found") + assert.Contains(t, err.Error(), "active app/service not found") }) - t.Run("calls runtime with correct container ID", func(t *testing.T) { - containerSvc := mocks.NewMockContainerService(t) - runtime := outMocks.NewMockContainerRuntime(t) + for _, tc := range []struct { + name string + ref string + }{ + {name: "raw Docker ID", ref: "abc123def456"}, + {name: "foreign container ID", ref: "fedcba654321"}, + {name: "stale container ID", ref: "012345abcdef"}, + } { + t.Run("rejects "+tc.name+" without calling the runtime", func(t *testing.T) { + runtime := mocks.NewMockContainerRuntime(t) + + svc := NewService("/tmp/test.log", true, runtime, log) + _, err := svc.GetContainerLogs(context.Background(), tc.ref, 10) + require.Error(t, err) + assert.Contains(t, err.Error(), "app/service") + }) + } - container := &domain.Container{ - ID: "abc123", - Name: "app.local", - Status: "running", - } - containerSvc.EXPECT().Get(mock.Anything, "app.local").Return(container, true) + t.Run("calls runtime with correct container ID", func(t *testing.T) { + runtime := mocks.NewMockContainerRuntime(t) // Create a mock reader that returns empty content runtime.EXPECT().GetContainerLogs(mock.Anything, "abc123", false).Return(&mockReader{}, nil) - svc := NewService("/tmp/test.log", true, containerSvc, runtime, log) + svc := NewService("/tmp/test.log", true, runtime, log) + svc.WithAppState(stubActiveApps{"app": {App: "app", Services: map[string]domain.AppEffectiveService{ + "web": {Container: "abc123"}, + }}}) - lines, err := svc.GetContainerLogs(context.Background(), "app.local", 10) + lines, err := svc.GetContainerLogs(context.Background(), "app/web", 10) require.NoError(t, err) assert.Empty(t, lines) }) @@ -188,17 +192,31 @@ func TestService_FollowContainerLogs(t *testing.T) { log := zerowrap.New(zerowrap.Config{Level: "warn"}) t.Run("returns error when container not found", func(t *testing.T) { - containerSvc := mocks.NewMockContainerService(t) - runtime := outMocks.NewMockContainerRuntime(t) - - containerSvc.EXPECT().Get(mock.Anything, "unknown.local").Return(nil, false) + runtime := mocks.NewMockContainerRuntime(t) - svc := NewService("/tmp/test.log", true, containerSvc, runtime, log) + svc := NewService("/tmp/test.log", true, runtime, log) + svc.WithAppState(stubActiveApps{}) - _, err := svc.FollowContainerLogs(context.Background(), "unknown.local", 10) + _, err := svc.FollowContainerLogs(context.Background(), "unknown/web", 10) assert.Error(t, err) - assert.Contains(t, err.Error(), "container not found") + assert.Contains(t, err.Error(), "active app/service not found") }) + + t.Run("rejects raw container ID without calling the runtime", func(t *testing.T) { + runtime := mocks.NewMockContainerRuntime(t) + + svc := NewService("/tmp/test.log", true, runtime, log) + _, err := svc.FollowContainerLogs(context.Background(), "abc123def456", 10) + require.Error(t, err) + assert.Contains(t, err.Error(), "app/service") + }) +} + +type stubActiveApps map[string]domain.AppActive + +func (s stubActiveApps) LoadActive(_ context.Context, app string) (domain.AppActive, bool, error) { + active, ok := s[app] + return active, ok, nil } func TestTailLines(t *testing.T) { diff --git a/internal/usecase/pki/service.go b/internal/usecase/pki/service.go index e4863dbc3..39097e29a 100644 --- a/internal/usecase/pki/service.go +++ b/internal/usecase/pki/service.go @@ -33,7 +33,7 @@ type cachedCert struct { // especially given the 10-year root CA lifetime. type Service struct { ca out.CertificateAuthority - routes out.RouteChecker + routes out.AppRoutes allowedMu sync.RWMutex allowedDomains map[string]struct{} log zerowrap.Logger @@ -47,7 +47,7 @@ type Service struct { // NewService creates a PKI service and starts background maintenance goroutines. // It performs an initial intermediate renewal check synchronously so the first // TLS handshakes never use a nearly-expired intermediate. -func NewService(ctx context.Context, ca out.CertificateAuthority, routes out.RouteChecker, allowedDomains []string, log zerowrap.Logger) *Service { +func NewService(ctx context.Context, ca out.CertificateAuthority, routes out.AppRoutes, allowedDomains []string, log zerowrap.Logger) *Service { ctx, cancel := context.WithCancel(ctx) allowed := canonicalDomainSet(allowedDomains) svc := &Service{ @@ -195,8 +195,8 @@ func (s *Service) isDomainAllowed(ctx context.Context, domainName string) bool { if additionallyAllowed { return true } - for _, r := range s.routes.GetRoutes(ctx) { - if r.Domain == domainName { + for _, h := range s.routes.AppHosts() { + if h.Host == domainName && h.TLSMode != domain.AppTLSNever { return true } } @@ -253,18 +253,18 @@ func (s *Service) renewIntermediateIfNeeded() { func (s *Service) sweepExpiredCerts(ctx context.Context) { now := time.Now() - // Fetch routes once for all cache entries. - routes := s.routes.GetRoutes(ctx) + // Fetch app hosts once for all cache entries. + hosts := s.routes.AppHosts() extRoutes := s.routes.GetExternalRoutes() s.allowedMu.RLock() - allowed := make(map[string]struct{}, len(routes)+len(extRoutes)+len(s.allowedDomains)) + allowed := make(map[string]struct{}, len(hosts)+len(extRoutes)+len(s.allowedDomains)) for domain := range s.allowedDomains { allowed[domain] = struct{}{} } s.allowedMu.RUnlock() - for _, r := range routes { - allowed[r.Domain] = struct{}{} + for _, h := range hosts { + allowed[h.Host] = struct{}{} } for d := range extRoutes { allowed[d] = struct{}{} diff --git a/internal/usecase/pki/service_test.go b/internal/usecase/pki/service_test.go index 87d98f282..1aab898bb 100644 --- a/internal/usecase/pki/service_test.go +++ b/internal/usecase/pki/service_test.go @@ -9,13 +9,11 @@ import ( "github.com/bnema/zerowrap" "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/mock" "github.com/stretchr/testify/require" pkiadapter "github.com/bnema/gordon/internal/adapters/out/pki" "github.com/bnema/gordon/internal/boundaries/out" "github.com/bnema/gordon/internal/boundaries/out/mocks" - "github.com/bnema/gordon/internal/domain" pkiusecase "github.com/bnema/gordon/internal/usecase/pki" ) @@ -33,15 +31,21 @@ func (b *blockingCertificateAuthority) IssueCertificate(domain string) (*tls.Cer return b.CertificateAuthority.IssueCertificate(domain) } -func newRouteCheckerMock(t *testing.T, domains ...string) *mocks.MockRouteChecker { - m := mocks.NewMockRouteChecker(t) - routes := make([]domain.Route, len(domains)) - for i, d := range domains { - routes[i] = domain.Route{Domain: d} +// stubAppRoutes is an ACTIVE-derived host source for PKI tests. +type stubAppRoutes struct { + hosts []out.AppHost +} + +func (s *stubAppRoutes) AppHosts() []out.AppHost { return s.hosts } + +func (s *stubAppRoutes) GetExternalRoutes() map[string]string { return nil } + +func newRouteCheckerMock(_ *testing.T, domains ...string) *stubAppRoutes { + hosts := make([]out.AppHost, 0, len(domains)) + for _, d := range domains { + hosts = append(hosts, out.AppHost{Host: d}) } - m.EXPECT().GetRoutes(mock.Anything).Return(routes).Maybe() - m.EXPECT().GetExternalRoutes().Return(nil).Maybe() - return m + return &stubAppRoutes{hosts: hosts} } func TestService_GetCertificate_KnownDomain(t *testing.T) { @@ -190,6 +194,31 @@ func TestService_GetCertificate_WrapsIssuanceError(t *testing.T) { assert.Equal(t, `issue leaf certificate for "broken.example.com": issuer unavailable`, err.Error()) } +func TestService_GetCertificate_NeverTLSHostDenied(t *testing.T) { + dir := t.TempDir() + ca, err := pkiadapter.NewCA(dir, testLogger()) + require.NoError(t, err) + + cfg := &stubAppRoutes{hosts: []out.AppHost{ + {Host: "plain.example.com", TLSMode: "never"}, + {Host: "auto.example.com", TLSMode: "auto"}, + }} + + ctx, cancel := context.WithCancel(t.Context()) + defer cancel() + + svc := pkiusecase.NewService(ctx, ca, cfg, nil, testLogger()) + defer svc.Stop() + + cert, err := svc.GetCertificate(&tls.ClientHelloInfo{ServerName: "plain.example.com"}) + assert.NoError(t, err) + assert.Nil(t, cert, "tls=never hosts stay plain HTTP: no internal certificate") + + cert, err = svc.GetCertificate(&tls.ClientHelloInfo{ServerName: "auto.example.com"}) + require.NoError(t, err) + require.NotNil(t, cert) +} + func TestService_GetCertificate_UnknownDomain(t *testing.T) { dir := t.TempDir() ca, err := pkiadapter.NewCA(dir, testLogger()) diff --git a/internal/usecase/proxy/events.go b/internal/usecase/proxy/events.go index bd01fe6d7..47a13db5b 100644 --- a/internal/usecase/proxy/events.go +++ b/internal/usecase/proxy/events.go @@ -8,60 +8,6 @@ import ( "github.com/bnema/gordon/internal/domain" ) -// TargetInvalidator defines the interface for invalidating proxy targets. -type TargetInvalidator interface { - InvalidateTarget(ctx context.Context, domainName string) -} - -// ContainerDeployedHandler handles container.deployed events to invalidate proxy cache. -type ContainerDeployedHandler struct { - invalidator TargetInvalidator - ctx context.Context -} - -// NewContainerDeployedHandler creates a new ContainerDeployedHandler. -func NewContainerDeployedHandler(ctx context.Context, invalidator TargetInvalidator) *ContainerDeployedHandler { - return &ContainerDeployedHandler{ - invalidator: invalidator, - ctx: ctx, - } -} - -// Handle handles a container.deployed event by invalidating the proxy cache. -func (h *ContainerDeployedHandler) Handle(ctx context.Context, event domain.Event) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldHandler: "ContainerDeployedHandler", - zerowrap.FieldEvent: string(event.Type), - "event_id": event.ID, - }) - log := zerowrap.FromCtx(ctx) - - // Get domain from event - domainName := event.Route - if domainName == "" { - // Try to get from payload - if payload, ok := event.Data.(*domain.ContainerEventPayload); ok { - domainName = payload.Domain - } - } - - if domainName == "" { - log.Debug().Msg("no domain in container deployed event, skipping cache invalidation") - return nil - } - - log.Debug().Str("domain", domainName).Msg("invalidating proxy cache for deployed container") - h.invalidator.InvalidateTarget(ctx, domainName) - - return nil -} - -// CanHandle returns whether this handler can handle the given event type. -func (h *ContainerDeployedHandler) CanHandle(eventType domain.EventType) bool { - return eventType == domain.EventContainerDeployed -} - // TargetRefresher defines the interface for refreshing all proxy targets. type TargetRefresher interface { RefreshTargets(ctx context.Context) error diff --git a/internal/usecase/proxy/service.go b/internal/usecase/proxy/service.go index a59030a40..d39fb80b2 100644 --- a/internal/usecase/proxy/service.go +++ b/internal/usecase/proxy/service.go @@ -5,9 +5,7 @@ import ( "context" "fmt" "net" - "os" "strconv" - "strings" "sync" "sync/atomic" "time" @@ -19,7 +17,6 @@ import ( "go.opentelemetry.io/otel/trace" "github.com/bnema/gordon/internal/boundaries/in" - "github.com/bnema/gordon/internal/boundaries/out" "github.com/bnema/gordon/internal/domain" ) @@ -34,33 +31,67 @@ type Config struct { MaxConcurrentConns int // Maximum concurrent proxy connections (0 = no limit) } +// TargetProvider resolves a canonical HTTP host to its app backend. +// Implemented by the apptraffic host index (derived from ACTIVE state); +// the proxy never interprets app state itself. +type TargetProvider interface { + // LookupHost returns the projected backend for a canonical host. + LookupHost(host string) (domain.AppBackend, bool) +} + // Service implements the ProxyService interface. +// App backends resolve through appTargets (ACTIVE-derived projection); +// the pre-v3 container-service + image-label resolution is removed. type Service struct { - runtime out.ContainerRuntime - containerSvc in.ContainerService - configSvc in.ConfigService - config Config - targets map[string]*domain.ProxyTarget + configSvc in.ConfigService + config Config + // appTargets is the ACTIVE-derived host index, wired after the app + // store opens (WithAppTargets). Nil means no app serves this host. + // App resolutions bypass the target cache entirely (see + // resolveAppTarget); only external routes and explicit RegisterTarget + // entries live in targets. + appTargets TargetProvider + targets map[string]cachedTarget + // gen is the LOCAL cache invalidation epoch for external/explicit + // targets only. It is unrelated to the apptraffic HostIndex version: + // app resolutions never consult it. + gen uint64 mu sync.RWMutex inFlight map[string]int inFlightMu sync.Mutex registryInFlight atomic.Int64 // active registry proxy requests, for graceful drain } -// NewService creates a new proxy service. +// cachedTarget carries the local cache epoch its resolution was built from. +// Applies to external/explicit targets only; app targets are never cached. +type cachedTarget struct { + target *domain.ProxyTarget + gen uint64 +} + +// WithAppTargets wires the ACTIVE-derived host index. The proxy is +// built before the app store opens, so wiring is deferred like the +// backup WithAppSources hook. +func (s *Service) WithAppTargets(provider TargetProvider) *Service { + s.mu.Lock() + defer s.mu.Unlock() + s.appTargets = provider + s.gen++ + s.targets = make(map[string]cachedTarget) + return s +} + +// NewService creates a new proxy service. The app target provider is +// wired later via WithAppTargets once the app store opens. func NewService( - runtime out.ContainerRuntime, - containerSvc in.ContainerService, configSvc in.ConfigService, config Config, ) *Service { return &Service{ - runtime: runtime, - containerSvc: containerSvc, - configSvc: configSvc, - config: config, - targets: make(map[string]*domain.ProxyTarget), - inFlight: make(map[string]int), + configSvc: configSvc, + config: config, + targets: make(map[string]cachedTarget), + inFlight: make(map[string]int), } } @@ -89,16 +120,17 @@ func (s *Service) GetTarget(ctx context.Context, domainName string) (target *dom }) log := zerowrap.FromCtx(ctx) - // Check cache first + // Check cache first (generation-guarded: a resolution built before + // the latest invalidation is dropped, never served). s.mu.RLock() - if target, exists := s.targets[domainName]; exists { + if cached, exists := s.targets[domainName]; exists { s.mu.RUnlock() log.Debug(). - Str("host", target.Host). - Int("port", target.Port). - Str("container_id", target.ContainerID). + Str("host", cached.target.Host). + Int("port", cached.target.Port). + Str("container_id", cached.target.ContainerID). Msg("using cached proxy target") - return target, nil + return cached.target, nil } s.mu.RUnlock() @@ -108,75 +140,36 @@ func (s *Service) GetTarget(ctx context.Context, domainName string) (target *dom return s.resolveExternalRoute(ctx, domainName, targetAddr, log) } - // Get container for this domain - container, exists := s.containerSvc.Get(ctx, domainName) - if !exists { - log.Debug().Msg("container not found for domain") + // App backends resolve from the ACTIVE-derived host index. + return s.resolveAppTarget(domainName, log) +} + +// resolveAppTarget maps a canonical host to its recorded loopback +// backend. App targets are NEVER cached: every resolution reads the +// ACTIVE-derived host index (an in-memory RWMutex lookup), so a freshly +// rebuilt index is observed on the very next request with no invalidation +// wiring. Unbound or unknown hosts fail closed (no container-IP +// fallback, no image-label inference, no stale-target fallback). +func (s *Service) resolveAppTarget(domainName string, log zerowrap.Logger) (*domain.ProxyTarget, error) { + s.mu.RLock() + provider := s.appTargets + s.mu.RUnlock() + if provider == nil { + log.Debug().Msg("no app target provider wired") return nil, domain.ErrNoTargetAvailable } - log.Debug().Str("container_id", container.ID).Str("image", container.Image).Msg("found container for domain") - - // Build target based on runtime mode - - if s.isRunningInContainer() { - // Gordon is in a container - use container network - meta, err := s.resolveTargetMetadata(ctx, container.Image) - if err != nil { - return nil, log.WrapErr(err, "failed to resolve target metadata") - } - containerIP, _, err := s.runtime.GetContainerNetworkInfo(ctx, container.ID) - if err != nil { - return nil, log.WrapErrWithFields(err, "failed to get container network info", map[string]any{zerowrap.FieldEntityID: container.ID}) - } - target = &domain.ProxyTarget{ - Host: containerIP, - Port: meta.Port, - ContainerID: container.ID, - Scheme: "http", - Protocol: meta.Protocol, - RouteHost: domainName, - } - } else { - // Gordon is on the host - use host port mapping - routes := s.configSvc.GetRoutes(ctx) - var route *domain.Route - for _, r := range routes { - if r.Domain == domainName { - route = &r - break - } - } - - if route == nil { - return nil, domain.ErrRouteNotFound - } - - meta, err := s.resolveTargetMetadata(ctx, container.Image) - if err != nil { - return nil, log.WrapErr(err, "failed to resolve target metadata") - } - - hostPort, err := s.runtime.GetContainerPort(ctx, container.ID, meta.Port) - if err != nil { - return nil, log.WrapErrWithFields(err, "failed to get host port mapping", map[string]any{"internal_port": meta.Port}) - } - - target = &domain.ProxyTarget{ - Host: "localhost", - Port: hostPort, - ContainerID: container.ID, - Scheme: "http", - Protocol: meta.Protocol, - RouteHost: domainName, - } + backend, ok := provider.LookupHost(domainName) + if !ok || !backend.Resolved() { + log.Debug().Msg("no resolved app backend for host") + return nil, domain.ErrNoTargetAvailable } - - // Cache the target - s.mu.Lock() - s.targets[domainName] = target - s.mu.Unlock() - - return target, nil + return &domain.ProxyTarget{ + Host: backend.Host, + Port: backend.Port, + ContainerID: backend.ContainerID, + Scheme: "http", + RouteHost: domainName, + }, nil } // resolveExternalRoute resolves an external route target address into a ProxyTarget, @@ -225,7 +218,7 @@ func (s *Service) resolveExternalRoute(_ context.Context, domainName, targetAddr // Cache external route target s.mu.Lock() - s.targets[domainName] = t + s.targets[domainName] = cachedTarget{target: t, gen: s.gen} s.mu.Unlock() log.Debug(). @@ -245,7 +238,7 @@ func (s *Service) RegisterTarget(_ context.Context, domainName string, target *d s.mu.Lock() defer s.mu.Unlock() - s.targets[canonicalDomain] = target + s.targets[canonicalDomain] = cachedTarget{target: target, gen: s.gen} return nil } @@ -265,6 +258,8 @@ func (s *Service) UnregisterTarget(_ context.Context, domainName string) error { // InvalidateTarget removes a cached proxy target, forcing re-lookup on next request. // This is used during zero-downtime deployments to switch traffic to a new container. +// Every invalidation bumps the cache generation so concurrent stale +// resolutions are dropped instead of cached. func (s *Service) InvalidateTarget(_ context.Context, domainName string) { canonicalDomain, ok := domain.CanonicalRouteDomain(domainName) if !ok { @@ -274,6 +269,7 @@ func (s *Service) InvalidateTarget(_ context.Context, domainName string) { s.mu.Lock() defer s.mu.Unlock() + s.gen++ delete(s.targets, canonicalDomain) } @@ -308,10 +304,11 @@ func (s *Service) WaitForNoInFlight(ctx context.Context, containerID string, tim } } -// RefreshTargets refreshes all proxy targets from container state. +// RefreshTargets drops all cached proxy targets, forcing re-lookup. func (s *Service) RefreshTargets(ctx context.Context) error { s.mu.Lock() - s.targets = make(map[string]*domain.ProxyTarget) + s.gen++ + s.targets = make(map[string]cachedTarget) s.mu.Unlock() log := zerowrap.FromCtx(ctx) @@ -341,8 +338,9 @@ func (s *Service) IsRegistryDomain(host string) bool { return ok && canonicalHost == registryDomain } -// IsKnownHost returns true if host is configured as registry, route, or external route. -func (s *Service) IsKnownHost(ctx context.Context, host string) bool { +// IsKnownHost returns true for the registry domain, external routes, +// and app-served hosts from the ACTIVE-derived index. +func (s *Service) IsKnownHost(_ context.Context, host string) bool { canonicalHost, ok := domain.CanonicalRouteDomain(host) if !ok { return false @@ -350,10 +348,16 @@ func (s *Service) IsKnownHost(ctx context.Context, host string) bool { if s.IsRegistryDomain(canonicalHost) { return true } - if _, err := s.configSvc.GetRoute(ctx, canonicalHost); err == nil { + if _, ok := s.configSvc.GetExternalRoutes()[canonicalHost]; ok { return true } - _, ok = s.configSvc.GetExternalRoutes()[canonicalHost] + s.mu.RLock() + provider := s.appTargets + s.mu.RUnlock() + if provider == nil { + return false + } + _, ok = provider.LookupHost(canonicalHost) return ok } @@ -421,33 +425,3 @@ func (s *Service) DrainRegistryInFlight(timeout time.Duration) bool { } return false } - -func (s *Service) isRunningInContainer() bool { - // Check for /.dockerenv - if _, err := os.Stat("/.dockerenv"); err == nil { - return true - } - - // Check cgroup for container indicators - if data, err := os.ReadFile("/proc/1/cgroup"); err == nil { - content := string(data) - if strings.Contains(content, "docker") || - strings.Contains(content, "containerd") || - strings.Contains(content, "podman") { - return true - } - } - - // NOTE: Hostname length check (12 or 64 chars) was removed because it produced - // false positives on hosts with short hostnames (e.g., "web-server-1" = 12 chars), - // which would cause the proxy to use container network IPs instead of host port mappings. - - // Check environment variables - if os.Getenv("KUBERNETES_SERVICE_HOST") != "" || - os.Getenv("DOCKER_CONTAINER") != "" || - os.Getenv("container") != "" { - return true - } - - return false -} diff --git a/internal/usecase/proxy/service_test.go b/internal/usecase/proxy/service_test.go index b40e681e9..c6984cce9 100644 --- a/internal/usecase/proxy/service_test.go +++ b/internal/usecase/proxy/service_test.go @@ -7,10 +7,8 @@ import ( "github.com/bnema/zerowrap" "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/mock" inmocks "github.com/bnema/gordon/internal/boundaries/in/mocks" - outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" "github.com/bnema/gordon/internal/domain" ) @@ -18,44 +16,61 @@ func testContext() context.Context { return zerowrap.WithCtx(context.Background(), zerowrap.Default()) } -func TestService_GetTarget_FromCache(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) +// stubProvider is a fixed TargetProvider for proxy tests. +type stubProvider struct { + backends map[string]domain.AppBackend +} - config := Config{ - RegistryDomain: "registry.example.com", - RegistryPort: 5000, +func (s *stubProvider) LookupHost(host string) (domain.AppBackend, bool) { + backend, ok := s.backends[host] + return backend, ok +} + +func testService(t *testing.T, configSvc *inmocks.MockConfigService, provider TargetProvider) *Service { + if configSvc == nil { + configSvc = inmocks.NewMockConfigService(t) } - svc := NewService(runtime, containerSvc, configSvc, config) + svc := NewService(configSvc, Config{}) + if provider != nil { + svc.WithAppTargets(provider) + } + return svc +} + +func appBackend(host string, port int) domain.AppBackend { + return domain.AppBackend{ + Host: "127.0.0.1", + Port: port, + ContainerPort: 8080, + ContainerID: "c-app", + } +} + +func TestService_GetTarget_FromCache(t *testing.T) { + svc := testService(t, nil, nil) ctx := testContext() // Pre-populate cache - cachedTarget := &domain.ProxyTarget{ - Host: "192.168.1.100", - Port: 8080, + cached := &domain.ProxyTarget{ + Host: "127.0.0.1", + Port: 18080, ContainerID: "container-123", Scheme: "http", } - svc.targets["app.example.com"] = cachedTarget - - // No mock calls expected - should return from cache + svc.targets["app.example.com"] = cachedTarget{target: cached} + // No provider needed - should return from cache result, err := svc.GetTarget(ctx, "app.example.com") assert.NoError(t, err) - assert.Equal(t, cachedTarget, result) + assert.Equal(t, cached, result) } func TestService_GetTarget_CanonicalizesHostForLookup(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) configSvc := inmocks.NewMockConfigService(t) - svc := NewService(runtime, containerSvc, configSvc, Config{}) - ctx := testContext() - configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{}) - containerSvc.EXPECT().Get(mock.Anything, "app.example.com").Return(nil, false) + svc := testService(t, configSvc, &stubProvider{backends: map[string]domain.AppBackend{}}) + ctx := testContext() result, err := svc.GetTarget(ctx, "App.Example.com") assert.ErrorIs(t, err, domain.ErrNoTargetAvailable) @@ -63,37 +78,92 @@ func TestService_GetTarget_CanonicalizesHostForLookup(t *testing.T) { } func TestService_GetTarget_RejectsInvalidHostAuthority(t *testing.T) { - svc := NewService(outmocks.NewMockContainerRuntime(t), inmocks.NewMockContainerService(t), inmocks.NewMockConfigService(t), Config{}) + svc := testService(t, nil, nil) result, err := svc.GetTarget(testContext(), "app.example.com:8080") assert.ErrorIs(t, err, domain.ErrNoTargetAvailable) assert.Nil(t, result) } -func TestService_GetTarget_ContainerNotFound(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) +func TestService_GetTarget_AppBackend(t *testing.T) { + configSvc := inmocks.NewMockConfigService(t) + configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{}) + svc := testService(t, configSvc, &stubProvider{backends: map[string]domain.AppBackend{ + "app.example.com": appBackend("127.0.0.1", 18080), + }}) + ctx := testContext() + + result, err := svc.GetTarget(ctx, "app.example.com") + + assert.NoError(t, err) + assert.Equal(t, "127.0.0.1", result.Host) + assert.Equal(t, 18080, result.Port) + assert.Equal(t, "c-app", result.ContainerID) + assert.Equal(t, "http", result.Scheme) + assert.Equal(t, "app.example.com", result.RouteHost) +} + +func TestService_GetTarget_AppBackendUnresolved(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) + configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{}) + // Zero backend: recorded but unbound (fail closed). + svc := testService(t, configSvc, &stubProvider{backends: map[string]domain.AppBackend{ + "app.example.com": {ContainerPort: 8080}, + }}) + ctx := testContext() - config := Config{} - svc := NewService(runtime, containerSvc, configSvc, config) + result, err := svc.GetTarget(ctx, "app.example.com") + assert.ErrorIs(t, err, domain.ErrNoTargetAvailable) + assert.Nil(t, result) +} + +func TestService_GetTarget_NoProvider(t *testing.T) { + configSvc := inmocks.NewMockConfigService(t) + configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{}) + svc := testService(t, configSvc, nil) ctx := testContext() - // Mock external routes (empty - no match) + result, err := svc.GetTarget(ctx, "app.example.com") + assert.ErrorIs(t, err, domain.ErrNoTargetAvailable) + assert.Nil(t, result) +} + +func TestService_GetTarget_NoStaleAppBackendAfterReplacement(t *testing.T) { + configSvc := inmocks.NewMockConfigService(t) configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{}) - containerSvc.EXPECT().Get(mock.Anything, "app.example.com").Return(nil, false) + provider := &stubProvider{backends: map[string]domain.AppBackend{ + "app.example.com": appBackend("127.0.0.1", 32768), + }} + svc := testService(t, configSvc, provider) + ctx := testContext() + // First resolution observes the v1 bind. result, err := svc.GetTarget(ctx, "app.example.com") + assert.NoError(t, err) + assert.Equal(t, 32768, result.Port) + + // Deploy replacement: the index now records the v2 bind. No + // invalidation call happens; the next lookup must observe v2. + provider.backends["app.example.com"] = appBackend("127.0.0.1", 32769) + result, err = svc.GetTarget(ctx, "app.example.com") + assert.NoError(t, err) + assert.Equal(t, 32769, result.Port) + + // App resolutions never populate the target cache. + svc.mu.RLock() + _, exists := svc.targets["app.example.com"] + svc.mu.RUnlock() + assert.False(t, exists) + // Withdrawal fails closed: no stale-target fallback. + delete(provider.backends, "app.example.com") + result, err = svc.GetTarget(ctx, "app.example.com") assert.ErrorIs(t, err, domain.ErrNoTargetAvailable) assert.Nil(t, result) } func TestService_GetTarget_ExternalRoute(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{}) + svc := testService(t, configSvc, nil) ctx := testContext() // Mock external routes - use public IP to pass SSRF check @@ -111,11 +181,8 @@ func TestService_GetTarget_ExternalRoute(t *testing.T) { } func TestService_GetTarget_ExternalRoute_Cached(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{}) + svc := testService(t, configSvc, nil) ctx := testContext() // First call - should resolve external route (use public IP) @@ -134,11 +201,8 @@ func TestService_GetTarget_ExternalRoute_Cached(t *testing.T) { } func TestService_GetTarget_ExternalRoute_SSRFBlocked(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{}) + svc := testService(t, configSvc, nil) ctx := testContext() tests := []struct { @@ -156,7 +220,7 @@ func TestService_GetTarget_ExternalRoute_SSRFBlocked(t *testing.T) { for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { // Clear cache - svc.targets = make(map[string]*domain.ProxyTarget) + svc.targets = make(map[string]cachedTarget) configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{ "ssrf.example.com": tt.target, @@ -172,11 +236,8 @@ func TestService_GetTarget_ExternalRoute_SSRFBlocked(t *testing.T) { } func TestService_GetTarget_ExternalRoute_InvalidTarget(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{}) + svc := testService(t, configSvc, nil) ctx := testContext() // Mock external routes with invalid format (missing port) @@ -202,11 +263,8 @@ func TestService_GetTarget_ExternalRoute_InvalidPort(t *testing.T) { for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{}) + svc := testService(t, configSvc, nil) ctx := testContext() configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{ @@ -222,16 +280,12 @@ func TestService_GetTarget_ExternalRoute_InvalidPort(t *testing.T) { } func TestService_RegisterTarget(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{}) + svc := testService(t, nil, nil) ctx := testContext() target := &domain.ProxyTarget{ - Host: "192.168.1.100", - Port: 8080, + Host: "127.0.0.1", + Port: 18080, ContainerID: "container-123", Scheme: "http", } @@ -245,22 +299,18 @@ func TestService_RegisterTarget(t *testing.T) { cached := svc.targets["app.example.com"] svc.mu.RUnlock() - assert.Equal(t, target, cached) + assert.Equal(t, target, cached.target) } func TestService_UnregisterTarget(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{}) + svc := testService(t, nil, nil) ctx := testContext() // Pre-populate - svc.targets["app.example.com"] = &domain.ProxyTarget{ - Host: "192.168.1.100", - Port: 8080, - } + svc.targets["app.example.com"] = cachedTarget{target: &domain.ProxyTarget{ + Host: "127.0.0.1", + Port: 18080, + }} err := svc.UnregisterTarget(ctx, "App.Example.com") @@ -275,16 +325,12 @@ func TestService_UnregisterTarget(t *testing.T) { } func TestService_RefreshTargets(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{}) + svc := testService(t, nil, nil) ctx := testContext() // Pre-populate with some targets - svc.targets["app1.example.com"] = &domain.ProxyTarget{Host: "192.168.1.100"} - svc.targets["app2.example.com"] = &domain.ProxyTarget{Host: "192.168.1.101"} + svc.targets["app1.example.com"] = cachedTarget{target: &domain.ProxyTarget{Host: "127.0.0.1"}} + svc.targets["app2.example.com"] = cachedTarget{target: &domain.ProxyTarget{Host: "127.0.0.1"}} err := svc.RefreshTargets(ctx) @@ -299,11 +345,8 @@ func TestService_RefreshTargets(t *testing.T) { } func TestService_UpdateConfig(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{ + svc := testService(t, nil, nil) + svc.UpdateConfig(Config{ RegistryDomain: "old.registry.com", RegistryPort: 5000, }) @@ -319,33 +362,16 @@ func TestService_UpdateConfig(t *testing.T) { assert.Equal(t, 5001, svc.config.RegistryPort) } -func TestService_isRunningInContainer(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{}) - - // This test just verifies the method doesn't panic - // The actual result depends on the environment - result := svc.isRunningInContainer() - assert.IsType(t, true, result) // Just verify it returns a bool -} - func TestService_InvalidateTarget(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{}) + svc := testService(t, nil, nil) ctx := testContext() // Pre-populate cache with target - svc.targets["app.example.com"] = &domain.ProxyTarget{ - Host: "192.168.1.100", - Port: 8080, + svc.targets["app.example.com"] = cachedTarget{target: &domain.ProxyTarget{ + Host: "127.0.0.1", + Port: 18080, ContainerID: "old-container", - } + }} // Invalidate the target using mixed-case input. svc.InvalidateTarget(ctx, "App.Example.com") @@ -358,77 +384,16 @@ func TestService_InvalidateTarget(t *testing.T) { assert.False(t, exists, "target should be removed from cache after invalidation") } -func TestContainerDeployedHandler_CanHandle(t *testing.T) { - handler := NewContainerDeployedHandler(testContext(), nil) - - assert.True(t, handler.CanHandle(domain.EventContainerDeployed)) - assert.False(t, handler.CanHandle(domain.EventImagePushed)) - assert.False(t, handler.CanHandle(domain.EventConfigReload)) -} - -func TestContainerDeployedHandler_Handle_InvalidatesCache(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) +func TestService_IsKnownHost_AppBackend(t *testing.T) { configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{}) - ctx := testContext() - - // Pre-populate cache - svc.targets["app.example.com"] = &domain.ProxyTarget{ - Host: "192.168.1.100", - Port: 8080, - ContainerID: "old-container", - } - - // Create handler with service as invalidator - handler := NewContainerDeployedHandler(ctx, svc) - - // Simulate container deployed event - event := domain.Event{ - ID: "event-123", - Type: domain.EventContainerDeployed, - Route: "app.example.com", - Data: &domain.ContainerEventPayload{ - ContainerID: "new-container", - Domain: "app.example.com", - }, - } - - err := handler.Handle(context.Background(), event) - - assert.NoError(t, err) - - // Verify target was invalidated - svc.mu.RLock() - _, exists := svc.targets["app.example.com"] - svc.mu.RUnlock() - - assert.False(t, exists, "cache should be invalidated after container deployed event") -} - -func TestContainerDeployedHandler_Handle_NoDomain(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{}) + configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{}).Maybe() + svc := testService(t, configSvc, &stubProvider{backends: map[string]domain.AppBackend{ + "app.example.com": appBackend("127.0.0.1", 18080), + }}) ctx := testContext() - handler := NewContainerDeployedHandler(ctx, svc) - - // Event with no domain - event := domain.Event{ - ID: "event-123", - Type: domain.EventContainerDeployed, - Data: &domain.ContainerEventPayload{ - ContainerID: "new-container", - }, - } - - // Should not error, just skip - err := handler.Handle(context.Background(), event) - assert.NoError(t, err) + assert.True(t, svc.IsKnownHost(ctx, "app.example.com")) + assert.False(t, svc.IsKnownHost(ctx, "unknown.example.com")) } func TestRegistryInFlightTracking(t *testing.T) { @@ -492,11 +457,8 @@ func TestDrainRegistryInFlightTimeout(t *testing.T) { } func TestService_ProxyConfig_ReflectsUpdates(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{ + svc := testService(t, nil, nil) + svc.UpdateConfig(Config{ RegistryDomain: "old.registry.com", RegistryPort: 5000, MaxBodySize: 1024, @@ -528,11 +490,8 @@ func TestService_ProxyConfig_ReflectsUpdates(t *testing.T) { } func TestService_IsRegistryDomain(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{ + svc := testService(t, nil, nil) + svc.UpdateConfig(Config{ RegistryDomain: "registry.example.com", }) @@ -542,11 +501,8 @@ func TestService_IsRegistryDomain(t *testing.T) { } func TestService_IsRegistryDomain_EmptyConfig(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{}) + svc := testService(t, nil, nil) + svc.UpdateConfig(Config{}) assert.False(t, svc.IsRegistryDomain("registry.example.com")) assert.False(t, svc.IsRegistryDomain("")) @@ -606,209 +562,3 @@ func TestService_TrackRegistryRequest(t *testing.T) { svc.ReleaseRegistryRequest() assert.Equal(t, int64(0), svc.RegistryInFlight()) } - -func TestService_GetTarget_HostMode_UsesProxyPortLabel(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{}) - ctx := testContext() - - // No external routes - configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{}) - - // Container exists for this domain - container := &domain.Container{ - ID: "c-gitea", - Image: "gitea/gitea:latest", - } - containerSvc.EXPECT().Get(mock.Anything, "git.example.com").Return(container, true) - - // Route exists - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{ - {Domain: "git.example.com", Image: "gitea/gitea:latest"}, - }) - - // Image has gordon.proxy.port=3000 label - runtime.EXPECT().GetImageLabels(mock.Anything, "gitea/gitea:latest").Return(map[string]string{ - domain.LabelProxyPort: "3000", - }, nil) - - // Host port mapping for internal port 3000 - runtime.EXPECT().GetContainerPort(mock.Anything, "c-gitea", 3000).Return(32000, nil) - - result, err := svc.GetTarget(ctx, "git.example.com") - - assert.NoError(t, err) - assert.Equal(t, "localhost", result.Host) - assert.Equal(t, 32000, result.Port) - assert.Equal(t, "c-gitea", result.ContainerID) - assert.Equal(t, "git.example.com", result.RouteHost) -} - -func TestService_GetTarget_HostMode_UsesDeprecatedPortLabel(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{}) - ctx := testContext() - - configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{}) - - container := &domain.Container{ - ID: "c-app", - Image: "myapp:latest", - } - containerSvc.EXPECT().Get(mock.Anything, "app.example.com").Return(container, true) - - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{ - {Domain: "app.example.com", Image: "myapp:latest"}, - }) - - // Image has deprecated gordon.port label — should still work - runtime.EXPECT().GetImageLabels(mock.Anything, "myapp:latest").Return(map[string]string{ - domain.LabelPort: "8080", - }, nil) - - runtime.EXPECT().GetContainerPort(mock.Anything, "c-app", 8080).Return(33000, nil) - - result, err := svc.GetTarget(ctx, "app.example.com") - - assert.NoError(t, err) - assert.Equal(t, 33000, result.Port) -} - -func TestService_GetTarget_HostMode_ProxyPortWinsOverPort(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{}) - ctx := testContext() - - configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{}) - - container := &domain.Container{ - ID: "c-dual", - Image: "dualapp:latest", - } - containerSvc.EXPECT().Get(mock.Anything, "dual.example.com").Return(container, true) - - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{ - {Domain: "dual.example.com", Image: "dualapp:latest"}, - }) - - // Both labels set — gordon.proxy.port=9000 should win over gordon.port=3000 - runtime.EXPECT().GetImageLabels(mock.Anything, "dualapp:latest").Return(map[string]string{ - domain.LabelProxyPort: "9000", - domain.LabelPort: "3000", - }, nil) - - // Should use 9000 (gordon.proxy.port), NOT 3000 (gordon.port) - runtime.EXPECT().GetContainerPort(mock.Anything, "c-dual", 9000).Return(34000, nil) - - result, err := svc.GetTarget(ctx, "dual.example.com") - - assert.NoError(t, err) - assert.Equal(t, 34000, result.Port) -} - -func TestService_GetTarget_HostMode_FallsBackToExposedPort(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{}) - ctx := testContext() - - configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{}) - - container := &domain.Container{ - ID: "c-plain", - Image: "plain:latest", - } - containerSvc.EXPECT().Get(mock.Anything, "plain.example.com").Return(container, true) - - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{ - {Domain: "plain.example.com", Image: "plain:latest"}, - }) - - // No port labels — should fall back to first exposed port - runtime.EXPECT().GetImageLabels(mock.Anything, "plain:latest").Return(map[string]string{}, nil) - runtime.EXPECT().GetImageExposedPorts(mock.Anything, "plain:latest").Return([]int{8080}, nil) - - runtime.EXPECT().GetContainerPort(mock.Anything, "c-plain", 8080).Return(35000, nil) - - result, err := svc.GetTarget(ctx, "plain.example.com") - - assert.NoError(t, err) - assert.Equal(t, 35000, result.Port) -} - -func TestService_GetTarget_HostMode_H2CProtocolPropagated(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{}) - ctx := testContext() - - configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{}) - - container := &domain.Container{ - ID: "c-grpc", - Image: "grpc-app:latest", - } - containerSvc.EXPECT().Get(mock.Anything, "grpc.example.com").Return(container, true) - - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{ - {Domain: "grpc.example.com", Image: "grpc-app:latest"}, - }) - - runtime.EXPECT().GetImageLabels(mock.Anything, "grpc-app:latest").Return(map[string]string{ - domain.LabelProxyPort: "50051", - domain.LabelProxyProtocol: "h2c", - }, nil) - - runtime.EXPECT().GetContainerPort(mock.Anything, "c-grpc", 50051).Return(50051, nil) - - result, err := svc.GetTarget(ctx, "grpc.example.com") - - assert.NoError(t, err) - assert.Equal(t, "h2c", result.Protocol) - assert.Equal(t, 50051, result.Port) -} - -func TestService_GetTarget_HostMode_DefaultProtocol(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - containerSvc := inmocks.NewMockContainerService(t) - configSvc := inmocks.NewMockConfigService(t) - - svc := NewService(runtime, containerSvc, configSvc, Config{}) - ctx := testContext() - - configSvc.EXPECT().GetExternalRoutes().Return(map[string]string{}) - - container := &domain.Container{ - ID: "c-web", - Image: "web:latest", - } - containerSvc.EXPECT().Get(mock.Anything, "web.example.com").Return(container, true) - - configSvc.EXPECT().GetRoutes(mock.Anything).Return([]domain.Route{ - {Domain: "web.example.com", Image: "web:latest"}, - }) - - runtime.EXPECT().GetImageLabels(mock.Anything, "web:latest").Return(map[string]string{ - domain.LabelProxyPort: "8080", - }, nil) - - runtime.EXPECT().GetContainerPort(mock.Anything, "c-web", 8080).Return(8080, nil) - - result, err := svc.GetTarget(ctx, "web.example.com") - - assert.NoError(t, err) - assert.Equal(t, "", result.Protocol) -} diff --git a/internal/usecase/proxy/target_metadata.go b/internal/usecase/proxy/target_metadata.go deleted file mode 100644 index b680b28cd..000000000 --- a/internal/usecase/proxy/target_metadata.go +++ /dev/null @@ -1,74 +0,0 @@ -package proxy - -import ( - "context" - "fmt" - "strconv" - - "github.com/bnema/zerowrap" - - "github.com/bnema/gordon/internal/domain" -) - -// TargetMetadata holds resolved backend connection metadata for a container image. -type TargetMetadata struct { - Port int // The port to proxy traffic to - Protocol string // "" for HTTP/1.1, "h2c" for cleartext HTTP/2 -} - -// resolveTargetMetadata determines the port and protocol for an image by reading -// image labels. It checks gordon.proxy.port first, then the deprecated gordon.port -// alias, then falls back to the first exposed port. Protocol is read from -// gordon.proxy.protocol (only "h2c" is recognized; unknown values are ignored). -func (s *Service) resolveTargetMetadata(ctx context.Context, imageRef string) (TargetMetadata, error) { - log := zerowrap.FromCtx(ctx) - log.Debug().Str("image_ref", imageRef).Msg("resolving target metadata for image") - - labels, err := s.runtime.GetImageLabels(ctx, imageRef) - if err != nil { - log.Debug().Err(err).Msg("failed to get image labels, falling back to exposed ports") - labels = nil - } - - port := 0 - if labels != nil { - for _, key := range []string{domain.LabelProxyPort, domain.LabelPort} { - if portStr, ok := labels[key]; ok && portStr != "" { - p, convErr := strconv.Atoi(portStr) - if convErr == nil && p > 0 && p <= 65535 { - log.Debug().Int("port", p).Str("label", key).Msg("using proxy port from image label") - port = p - break - } - log.Warn().Str("label", key).Str("port_value", portStr).Msg("invalid port label value") - } - } - } - - if port == 0 { - exposedPorts, portsErr := s.runtime.GetImageExposedPorts(ctx, imageRef) - if portsErr != nil { - return TargetMetadata{}, fmt.Errorf("failed to get exposed ports for image %s: %w", imageRef, portsErr) - } - if len(exposedPorts) == 0 { - return TargetMetadata{}, fmt.Errorf("no exposed ports found for image %s", imageRef) - } - port = exposedPorts[0] - log.Debug().Int("port", port).Msg("using first exposed port") - } - - protocol := "" - if labels != nil { - proto := labels[domain.LabelProxyProtocol] - switch proto { - case "h2c": - protocol = "h2c" - log.Debug().Str("protocol", proto).Msg("container opts into h2c") - case "": - default: - log.Warn().Str("protocol", proto).Msg("unknown proxy protocol label value, ignoring") - } - } - - return TargetMetadata{Port: port, Protocol: protocol}, nil -} diff --git a/internal/usecase/proxy/target_metadata_test.go b/internal/usecase/proxy/target_metadata_test.go deleted file mode 100644 index 599da3890..000000000 --- a/internal/usecase/proxy/target_metadata_test.go +++ /dev/null @@ -1,137 +0,0 @@ -package proxy - -import ( - "fmt" - "testing" - - "github.com/stretchr/testify/assert" - - outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" - "github.com/bnema/gordon/internal/domain" -) - -func TestResolveTargetMetadata_PortAndProtocolFromLabels(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - svc := &Service{runtime: runtime} - ctx := testContext() - - runtime.EXPECT().GetImageLabels(ctx, "grpc-app:latest").Return(map[string]string{ - domain.LabelProxyPort: "50051", - domain.LabelProxyProtocol: "h2c", - }, nil) - - meta, err := svc.resolveTargetMetadata(ctx, "grpc-app:latest") - - assert.NoError(t, err) - assert.Equal(t, 50051, meta.Port) - assert.Equal(t, "h2c", meta.Protocol) -} - -func TestResolveTargetMetadata_DefaultProtocolWhenAbsent(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - svc := &Service{runtime: runtime} - ctx := testContext() - - runtime.EXPECT().GetImageLabels(ctx, "web:latest").Return(map[string]string{ - domain.LabelProxyPort: "8080", - }, nil) - - meta, err := svc.resolveTargetMetadata(ctx, "web:latest") - - assert.NoError(t, err) - assert.Equal(t, 8080, meta.Port) - assert.Equal(t, "", meta.Protocol) -} - -func TestResolveTargetMetadata_DeprecatedPortLabel(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - svc := &Service{runtime: runtime} - ctx := testContext() - - runtime.EXPECT().GetImageLabels(ctx, "old:latest").Return(map[string]string{ - domain.LabelPort: "3000", - }, nil) - - meta, err := svc.resolveTargetMetadata(ctx, "old:latest") - - assert.NoError(t, err) - assert.Equal(t, 3000, meta.Port) - assert.Equal(t, "", meta.Protocol) -} - -func TestResolveTargetMetadata_ProxyPortWinsOverDeprecated(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - svc := &Service{runtime: runtime} - ctx := testContext() - - runtime.EXPECT().GetImageLabels(ctx, "dual:latest").Return(map[string]string{ - domain.LabelProxyPort: "9000", - domain.LabelPort: "3000", - }, nil) - - meta, err := svc.resolveTargetMetadata(ctx, "dual:latest") - - assert.NoError(t, err) - assert.Equal(t, 9000, meta.Port) -} - -func TestResolveTargetMetadata_FallsBackToExposedPort(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - svc := &Service{runtime: runtime} - ctx := testContext() - - runtime.EXPECT().GetImageLabels(ctx, "plain:latest").Return(map[string]string{}, nil) - runtime.EXPECT().GetImageExposedPorts(ctx, "plain:latest").Return([]int{8080}, nil) - - meta, err := svc.resolveTargetMetadata(ctx, "plain:latest") - - assert.NoError(t, err) - assert.Equal(t, 8080, meta.Port) - assert.Equal(t, "", meta.Protocol) -} - -func TestResolveTargetMetadata_UnknownProtocolIgnored(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - svc := &Service{runtime: runtime} - ctx := testContext() - - runtime.EXPECT().GetImageLabels(ctx, "exotic:latest").Return(map[string]string{ - domain.LabelProxyPort: "8080", - domain.LabelProxyProtocol: "quic", - }, nil) - - meta, err := svc.resolveTargetMetadata(ctx, "exotic:latest") - - assert.NoError(t, err) - assert.Equal(t, 8080, meta.Port) - assert.Equal(t, "", meta.Protocol) -} - -func TestResolveTargetMetadata_NoPortsAvailable(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - svc := &Service{runtime: runtime} - ctx := testContext() - - runtime.EXPECT().GetImageLabels(ctx, "empty:latest").Return(map[string]string{}, nil) - runtime.EXPECT().GetImageExposedPorts(ctx, "empty:latest").Return([]int{}, nil) - - _, err := svc.resolveTargetMetadata(ctx, "empty:latest") - - assert.Error(t, err) - assert.Contains(t, err.Error(), "no exposed ports") -} - -func TestResolveTargetMetadata_LabelErrorFallsBackToExposedPort(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - svc := &Service{runtime: runtime} - ctx := testContext() - - runtime.EXPECT().GetImageLabels(ctx, "broken:latest").Return(nil, fmt.Errorf("image inspect failed")) - runtime.EXPECT().GetImageExposedPorts(ctx, "broken:latest").Return([]int{3000}, nil) - - meta, err := svc.resolveTargetMetadata(ctx, "broken:latest") - - assert.NoError(t, err) - assert.Equal(t, 3000, meta.Port) - assert.Equal(t, "", meta.Protocol) -} diff --git a/internal/usecase/pruneguard/guard.go b/internal/usecase/pruneguard/guard.go new file mode 100644 index 000000000..dfa48849a --- /dev/null +++ b/internal/usecase/pruneguard/guard.go @@ -0,0 +1,103 @@ +// Package pruneguard holds the ownership-aware prune policy: pure +// planning from complete inputs to per-candidate verdicts. Every +// exported function here is pure — inputs in, verdicts out, no store +// reads, no runtime calls, no deletion. +// +// The policy is fail-closed per candidate, never per operation: +// +// - A resource is eligible only when every fact needed to prove it +// safe was read completely and no durable root claims it. +// - A durable root (desired/active/recovery/apply/operation/ownership +// /container use/pending upload/shared closure/pin) makes the +// candidate protected. +// - Anything else — unknown identity, unknown provenance, incomplete +// inventory, unreadable manifest — makes the candidate unknown. +// Unknown candidates are never deleted; they do not disable the +// whole operation. +// +// Ownership is never inferred from names, and labels never override a +// contradictory or missing durable ownership record. +package pruneguard + +import ( + "github.com/bnema/gordon/internal/domain" +) + +// Provenance classifies a runtime resource by the labels it carries. +type Provenance int + +const ( + // ProvenanceUnmanaged carries no Gordon labels: never adopted, + // never deleted. + ProvenanceUnmanaged Provenance = iota + // ProvenanceLegacyManaged carries the managed label but no app + // ownership label: the legacy/unknown class. + ProvenanceLegacyManaged + // ProvenanceServiceManaged carries standalone service labels. + ProvenanceServiceManaged + // ProvenanceAppOwned carries app ownership labels. + ProvenanceAppOwned +) + +// ClassifyResource maps a label set to its provenance. App ownership +// wins over service labels, and the managed marker alone is never +// treated as app ownership. +func ClassifyResource(labels map[string]string) Provenance { + if len(labels) == 0 { + return ProvenanceUnmanaged + } + if labels[domain.LabelApp] != "" { + return ProvenanceAppOwned + } + if labels[domain.LabelService] != "" || labels[domain.LabelServiceName] != "" { + return ProvenanceServiceManaged + } + if labels[domain.LabelManaged] == "true" { + return ProvenanceLegacyManaged + } + return ProvenanceUnmanaged +} + +// protected builds a protected verdict with canonical reasons, falling +// back to a reason that always fits when no root kind mapped. +func protected(reasons []domain.PruneReason, fallback domain.PruneReason) (domain.PruneVerdict, []domain.PruneReason) { + canonical := domain.CanonicalReasons(reasons...) + if len(canonical) == 0 { + canonical = domain.CanonicalReasons(fallback) + } + return domain.PruneVerdictProtected, canonical +} + +// eligibleResult builds an eligible verdict with its single reason. +func eligibleResult(reason domain.PruneReason) (domain.PruneVerdict, []domain.PruneReason) { + return domain.PruneVerdictEligible, domain.CanonicalReasons(reason) +} + +// unknownResult builds an unknown verdict with its single reason. +func unknownResult(reason domain.PruneReason) (domain.PruneVerdict, []domain.PruneReason) { + return domain.PruneVerdictUnknown, domain.CanonicalReasons(reason) +} + +// rootReasonsOrFallback collects the mapped reasons of every root +// claiming refs, falling back to a shared-content reason when a root +// matched but its kind cannot justify a verdict reason. +func rootReasonsOrFallback(snapshot *domain.PruneProtectionSnapshot, refs ...string) (bool, []domain.PruneReason) { + if snapshot == nil { + return false, nil + } + matched := false + for _, ref := range refs { + if ref != "" && snapshot.ProtectsAnyRef(ref) { + matched = true + break + } + } + if !matched { + return false, nil + } + reasons := snapshot.RootReasons(refs...) + if len(reasons) == 0 { + reasons = domain.CanonicalReasons(domain.PruneReasonProtectedSharedContent) + } + return true, reasons +} diff --git a/internal/usecase/pruneguard/plan_test.go b/internal/usecase/pruneguard/plan_test.go new file mode 100644 index 000000000..c687292f1 --- /dev/null +++ b/internal/usecase/pruneguard/plan_test.go @@ -0,0 +1,617 @@ +package pruneguard + +import ( + "strings" + "testing" + "time" + + "github.com/bnema/gordon/internal/domain" +) + +func digest(seed string) string { + repeated := strings.Repeat(seed, 64) + return "sha256:" + repeated[:64] +} + +func tagKey(repository, tag string) string { + return domain.RegistryTagRef{Repository: repository, Tag: tag}.Key() +} + +func TestClassifyResource(t *testing.T) { + tests := []struct { + name string + labels map[string]string + want Provenance + }{ + {"nil", nil, ProvenanceUnmanaged}, + {"foreign", map[string]string{"com.example.role": "db"}, ProvenanceUnmanaged}, + {"managed only", map[string]string{domain.LabelManaged: "true"}, ProvenanceLegacyManaged}, + {"managed false", map[string]string{domain.LabelManaged: "false"}, ProvenanceUnmanaged}, + {"service", map[string]string{domain.LabelService: "srv"}, ProvenanceServiceManaged}, + {"service name", map[string]string{domain.LabelServiceName: "srv"}, ProvenanceServiceManaged}, + { + "app wins over service", + map[string]string{domain.LabelApp: "shop", domain.LabelService: "srv", domain.LabelManaged: "true"}, + ProvenanceAppOwned, + }, + } + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + if got := ClassifyResource(tc.labels); got != tc.want { + t.Fatalf("ClassifyResource(%v) = %v, want %v", tc.labels, got, tc.want) + } + }) + } +} + +func TestPlanRegistryTagsRetentionWindow(t *testing.T) { + base := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + tags := make([]RegistryTagCandidate, 0, 6) + tags = append(tags, candidate("app", "latest", base.Add(6*time.Hour), digest("a"))) + for index := 0; index < 5; index++ { + tags = append(tags, candidate("app", "v"+string(rune('1'+index)), base.Add(time.Duration(index)*time.Hour), digest(string(rune('b'+index))))) + } + + verdicts := PlanRegistryTags(RegistryPlanInput{Tags: tags, Snapshot: emptySnapshot(), KeepLast: 3, Complete: true}) + byTag := indexVerdicts(t, verdicts) + + if byTag["latest"] != domain.PruneVerdictProtected { + t.Fatalf("latest = %v, want protected", byTag["latest"]) + } + for _, tag := range []string{"v5", "v4", "v3"} { + if byTag[tag] != domain.PruneVerdictProtected { + t.Fatalf("%s = %v, want protected (inside latest + 3)", tag, byTag[tag]) + } + } + for _, tag := range []string{"v1", "v2"} { + if byTag[tag] != domain.PruneVerdictEligible { + t.Fatalf("%s = %v, want eligible (outside latest + 3)", tag, byTag[tag]) + } + } + + // Output order is deterministic regardless of input order. + shuffled := []RegistryTagCandidate{tags[3], tags[0], tags[5], tags[1], tags[4], tags[2]} + again := PlanRegistryTags(RegistryPlanInput{Tags: shuffled, Snapshot: emptySnapshot(), KeepLast: 3, Complete: true}) + for index := range verdicts { + if verdicts[index].Ref != again[index].Ref { + t.Fatalf("output order changed with input order at %d: %v vs %v", index, verdicts[index].Ref, again[index].Ref) + } + } +} + +func TestPlanRegistryTagsLatestCountsSeparately(t *testing.T) { + base := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + tags := []RegistryTagCandidate{ + candidate("app", "latest", base, digest("a")), + candidate("app", "v1", base.Add(time.Hour), digest("b")), + } + verdicts := PlanRegistryTags(RegistryPlanInput{Tags: tags, Snapshot: emptySnapshot(), KeepLast: 1, Complete: true}) + byTag := indexVerdicts(t, verdicts) + if byTag["latest"] != domain.PruneVerdictProtected || byTag["v1"] != domain.PruneVerdictProtected { + t.Fatalf("latest + 1 must retain both tags, got %v", byTag) + } +} + +func TestPlanRegistryTagsEqualTimestampsAreDeterministic(t *testing.T) { + base := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + tags := []RegistryTagCandidate{ + candidate("app", "b", base, digest("a")), + candidate("app", "a", base, digest("b")), + } + verdicts := PlanRegistryTags(RegistryPlanInput{Tags: tags, Snapshot: emptySnapshot(), KeepLast: 1, Complete: true}) + byTag := indexVerdicts(t, verdicts) + // Tag ordering descending resolves the tie: "b" is kept. + if byTag["b"] != domain.PruneVerdictProtected || byTag["a"] != domain.PruneVerdictEligible { + t.Fatalf("equal timestamps must resolve deterministically, got %v", byTag) + } +} + +func TestPlanRegistryTagsProtectedRootsSurviveRetention(t *testing.T) { + base := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + oldDigest := digest("a") + tags := []RegistryTagCandidate{ + candidate("app", "latest", base.Add(10*time.Hour), digest("f")), + candidate("app", "v9", base.Add(9*time.Hour), digest("e")), + candidate("app", "v8", base.Add(8*time.Hour), digest("d")), + candidate("app", "v7", base.Add(7*time.Hour), digest("c")), + candidate("app", "v1", base.Add(time.Hour), oldDigest), + } + + snapshot := &domain.PruneProtectionSnapshot{Roots: []domain.ProtectionRoot{ + {Kind: domain.ProtectionDesiredRevision, Ref: tagKey("app", "v1"), Owner: "app"}, + }} + verdicts := PlanRegistryTags(RegistryPlanInput{Tags: tags, Snapshot: snapshot, KeepLast: 3, Complete: true}) + + var v1 domain.RegistryTagVerdict + for _, verdict := range verdicts { + if verdict.Ref.Tag == "v1" { + v1 = verdict + } + } + if v1.Verdict != domain.PruneVerdictProtected { + t.Fatalf("protected v1 = %v, want protected", v1.Verdict) + } + if len(v1.Reasons) != 1 || v1.Reasons[0] != domain.PruneReasonProtectedDesiredRevision { + t.Fatalf("v1 reasons = %v, want the desired-revision root", v1.Reasons) + } + + // The same protection reached through the digest must also hold. + snapshot = &domain.PruneProtectionSnapshot{Roots: []domain.ProtectionRoot{ + {Kind: domain.ProtectionActiveService, Ref: oldDigest, Owner: "app/web"}, + }} + verdicts = PlanRegistryTags(RegistryPlanInput{Tags: tags, Snapshot: snapshot, KeepLast: 3, Complete: true}) + for _, verdict := range verdicts { + if verdict.Ref.Tag == "v1" && verdict.Verdict != domain.PruneVerdictProtected { + t.Fatalf("digest-protected v1 = %v, want protected", verdict.Verdict) + } + } +} + +func TestPlanRegistryTagsFailClosed(t *testing.T) { + base := time.Date(2026, 1, 1, 0, 0, 0, 0, time.UTC) + tags := []RegistryTagCandidate{ + candidate("app", "latest", base.Add(10*time.Hour), digest("f")), + candidate("app", "v9", base.Add(9*time.Hour), digest("e")), + candidate("app", "v1", base.Add(time.Hour), digest("a")), + candidate("app", "v2", base.Add(2*time.Hour), digest("b")), + } + tags[3].ManifestKnown = false + + verdicts := PlanRegistryTags(RegistryPlanInput{Tags: tags, Snapshot: emptySnapshot(), KeepLast: 1, Complete: true}) + byTag := indexVerdicts(t, verdicts) + if byTag["v2"] != domain.PruneVerdictUnknown { + t.Fatalf("unreadable manifest tag = %v, want unknown", byTag["v2"]) + } + if byTag["v1"] != domain.PruneVerdictEligible { + t.Fatalf("readable out-of-window tag = %v, want eligible", byTag["v1"]) + } + + verdicts = PlanRegistryTags(RegistryPlanInput{Tags: tags, KeepLast: 1, Complete: false}) + byTag = indexVerdicts(t, verdicts) + if byTag["v1"] != domain.PruneVerdictUnknown { + t.Fatalf("incomplete registry inventory = %v, want unknown", byTag["v1"]) + } + // Protection still wins over incompleteness. + snapshot := &domain.PruneProtectionSnapshot{Roots: []domain.ProtectionRoot{ + {Kind: domain.ProtectionPinned, Ref: tagKey("app", "v1")}, + }} + verdicts = PlanRegistryTags(RegistryPlanInput{Tags: tags, Snapshot: snapshot, KeepLast: 1, Complete: false}) + byTag = indexVerdicts(t, verdicts) + if byTag["v1"] != domain.PruneVerdictProtected { + t.Fatalf("pinned tag under incomplete inventory = %v, want protected", byTag["v1"]) + } +} + +func TestPlanRegistryTagsKeepLastZeroSkips(t *testing.T) { + tags := []RegistryTagCandidate{candidate("app", "v1", time.Now(), digest("a"))} + for _, keepLast := range []int{0, -1} { + if verdicts := PlanRegistryTags(RegistryPlanInput{Tags: tags, KeepLast: keepLast, Complete: true}); verdicts != nil { + t.Fatalf("KeepLast=%d produced %d verdicts, want none", keepLast, len(verdicts)) + } + } +} + +func TestPlanRegistryTagsRejectsInvalidIdentity(t *testing.T) { + verdicts := PlanRegistryTags(RegistryPlanInput{ + Tags: []RegistryTagCandidate{{Ref: domain.RegistryTagRef{}, ManifestKnown: true}}, + Snapshot: emptySnapshot(), + KeepLast: 1, + Complete: true, + }) + if len(verdicts) != 1 || verdicts[0].Verdict != domain.PruneVerdictUnknown { + t.Fatalf("invalid tag identity = %+v, want unknown", verdicts) + } +} + +func TestPlanRuntimeImagesMixedBatch(t *testing.T) { + activeID := "sha256:active" + stoppedID := "sha256:stopped" + danglingID := "sha256:dangling" + foreignID := "sha256:foreign" + legacyID := "sha256:legacy" + serviceID := "sha256:service" + taggedID := "sha256:tagged" + + images := []domain.RuntimeImage{ + {ID: activeID, Labels: appLabels("shop", "web")}, + {ID: stoppedID, Labels: appLabels("shop", "web")}, + {ID: danglingID, Labels: appLabels("shop", "web")}, + {ID: foreignID}, + {ID: legacyID, Labels: map[string]string{domain.LabelManaged: "true"}}, + {ID: serviceID, Labels: map[string]string{domain.LabelService: "edge"}}, + {ID: taggedID, RepoTags: []string{"app:latest"}, Labels: appLabels("shop", "web")}, + } + containers := []domain.RuntimeContainerUse{ + {ContainerID: "c1", ImageID: activeID, Running: true}, + {ContainerID: "c2", ImageID: stoppedID}, + } + + verdicts := PlanRuntimeImages(RuntimePlanInput{Images: images, Containers: containers, Snapshot: &domain.PruneProtectionSnapshot{ + ImageClaims: []domain.ImageClaim{{Reference: danglingID, App: "shop", State: domain.VolumeClaimReleased}}, + }, Complete: true}) + byID := indexImageVerdicts(t, verdicts) + + expect := map[string]domain.PruneVerdict{ + activeID: domain.PruneVerdictProtected, + stoppedID: domain.PruneVerdictProtected, + danglingID: domain.PruneVerdictEligible, + foreignID: domain.PruneVerdictProtected, + legacyID: domain.PruneVerdictProtected, + serviceID: domain.PruneVerdictProtected, + taggedID: domain.PruneVerdictProtected, + } + for id, want := range expect { + if byID[id] != want { + t.Fatalf("%s = %v, want %v", id, byID[id], want) + } + } + if byID[stoppedID] != domain.PruneVerdictProtected { + t.Fatal("stopped container must protect its image") + } + if got := byID[foreignID]; got != domain.PruneVerdictProtected { + t.Fatalf("unmanaged dangling image = %v, want protected", got) + } +} + +// TestPlanRuntimeImagesForgedLabelsNeverAuthorizeDeletion proves image +// labels are hints only: a dangling image that carries app ownership labels +// but has no durable ownership record stays unknown, never eligible. +func TestPlanRuntimeImagesForgedLabelsNeverAuthorizeDeletion(t *testing.T) { + forged := "sha256:forged" + verdicts := PlanRuntimeImages(RuntimePlanInput{ + Images: []domain.RuntimeImage{{ + ID: forged, + Labels: map[string]string{domain.LabelApp: "shop", domain.LabelAppID: "shop-uuid", domain.LabelManaged: "true"}, + }}, + Snapshot: emptySnapshot(), + Complete: true, + }) + + require := verdicts[0] + if require.Verdict != domain.PruneVerdictUnknown { + t.Fatalf("forged app labels = %v, want unknown", require.Verdict) + } + if require.Reasons[0] != domain.PruneReasonUnknownProvenance { + t.Fatalf("forged app label reason = %v", require.Reasons) + } +} + +// TestPlanRuntimeImagesAnyMatchingClaimProtects proves an attached claim +// protects an image even when a released claim for the same identity also +// exists, in either order: one live owner is enough to forbid deletion. +func TestPlanRuntimeImagesAnyMatchingClaimProtects(t *testing.T) { + const identity = "sha256:shared" + images := []domain.RuntimeImage{{ID: identity, Labels: map[string]string{domain.LabelApp: "shop"}}} + orders := [][]domain.ImageClaim{ + { + {Reference: identity, App: "shop", State: domain.VolumeClaimReleased}, + {Reference: identity, App: "shop", State: domain.VolumeClaimAttached}, + }, + { + {Reference: identity, App: "shop", State: domain.VolumeClaimAttached}, + {Reference: identity, App: "shop", State: domain.VolumeClaimReleased}, + }, + } + for index, claims := range orders { + verdicts := PlanRuntimeImages(RuntimePlanInput{ + Images: images, + Snapshot: &domain.PruneProtectionSnapshot{ImageClaims: claims}, + Complete: true, + }) + if verdicts[0].Verdict != domain.PruneVerdictProtected { + t.Fatalf("order %d: image = %v, want protected while an attached claim exists", index, verdicts[0].Verdict) + } + } +} + +// TestPlanRuntimeImagesReleasedClaimIsEligible proves a durable released +// claim with no container use and a complete inventory is eligible. +func TestPlanRuntimeImagesReleasedClaimIsEligible(t *testing.T) { + released := "sha256:released" + verdicts := PlanRuntimeImages(RuntimePlanInput{ + Images: []domain.RuntimeImage{{ID: released, Labels: map[string]string{domain.LabelApp: "shop"}}}, + Snapshot: &domain.PruneProtectionSnapshot{ImageClaims: []domain.ImageClaim{{Reference: released, App: "shop", State: domain.VolumeClaimReleased}}}, + Complete: true, + }) + if verdicts[0].Verdict != domain.PruneVerdictEligible { + t.Fatalf("released image = %v, want eligible", verdicts[0].Verdict) + } +} + +// TestPlanRuntimeImagesAttachedClaimProtects proves an attached durable +// claim protects an otherwise dangling image. +func TestPlanRuntimeImagesAttachedClaimProtects(t *testing.T) { + attached := "sha256:attached" + verdicts := PlanRuntimeImages(RuntimePlanInput{ + Images: []domain.RuntimeImage{{ID: attached, Labels: map[string]string{domain.LabelApp: "shop"}}}, + Snapshot: &domain.PruneProtectionSnapshot{ImageClaims: []domain.ImageClaim{{Reference: attached, App: "shop", State: domain.VolumeClaimAttached}}}, + Complete: true, + }) + if verdicts[0].Verdict != domain.PruneVerdictProtected { + t.Fatalf("attached image = %v, want protected", verdicts[0].Verdict) + } +} + +func TestPlanRuntimeImagesProtectionRootsAndUnknowns(t *testing.T) { + pinnedID := "sha256:pinned" + danglingID := "sha256:dangling" + + images := []domain.RuntimeImage{ + {ID: pinnedID, Labels: appLabels("shop", "web")}, + {ID: danglingID, Labels: appLabels("shop", "web")}, + } + snapshot := &domain.PruneProtectionSnapshot{Roots: []domain.ProtectionRoot{ + {Kind: domain.ProtectionRecoveryInhibition, Ref: pinnedID, Owner: "shop"}, + }} + + verdicts := PlanRuntimeImages(RuntimePlanInput{Images: images, Snapshot: snapshot, Complete: true}) + byID := indexImageVerdicts(t, verdicts) + if byID[pinnedID] != domain.PruneVerdictProtected { + t.Fatalf("recovery-inhibited image = %v, want protected", byID[pinnedID]) + } + if len(verdicts) != 2 || verdicts[1].Reasons[0] != domain.PruneReasonProtectedRecoveryInhibition { + t.Fatalf("reasons = %v, want the recovery-inhibition root", verdicts[0].Reasons) + } + + // An unresolvable container protects every candidate: its image may + // be any of them. + unknown := PlanRuntimeImages(RuntimePlanInput{ + Images: images[1:], + Containers: []domain.RuntimeContainerUse{{ContainerID: "c9", ImageUnknown: true}}, + Complete: true, + }) + if unknown[0].Verdict != domain.PruneVerdictUnknown { + t.Fatalf("unknown container image = %v, want unknown", unknown[0].Verdict) + } + if unknown[0].Reasons[0] != domain.PruneReasonUnknownContainerUse { + t.Fatalf("unknown container reason = %v", unknown[0].Reasons) + } + + incomplete := PlanRuntimeImages(RuntimePlanInput{Images: images[1:], Complete: false}) + if incomplete[0].Verdict != domain.PruneVerdictUnknown { + t.Fatalf("incomplete runtime inventory = %v, want unknown", incomplete[0].Verdict) + } +} + +func TestPlanRuntimeImagesProtectsUnknownIdentity(t *testing.T) { + verdicts := PlanRuntimeImages(RuntimePlanInput{ + Images: []domain.RuntimeImage{{ID: ""}}, + Complete: true, + }) + if len(verdicts) != 1 || verdicts[0].Verdict != domain.PruneVerdictUnknown { + t.Fatalf("placeholder image identity = %+v, want unknown", verdicts) + } +} + +func TestPlanVolumesProtectsEverythingExceptReleased(t *testing.T) { + released := "gordon-shop-web-data" + attached := "gordon-shop-api-data" + retained := "gordon-old-data" + legacy := "gordon-legacy-data" + serviceVolume := "gordon-edge-data" + foreign := "pgdata" + labelled := "gordon-mystery-data" + inUse := "gordon-shop-cache" + + volumes := []*domain.VolumeInfo{ + {Name: released, Labels: appLabels("shop", "web")}, + {Name: attached, Labels: appLabels("shop", "api")}, + {Name: retained, Labels: appLabels("old", "web")}, + {Name: legacy, Labels: map[string]string{domain.LabelManaged: "true"}}, + {Name: serviceVolume, Labels: map[string]string{domain.LabelService: "edge"}}, + {Name: foreign}, + {Name: labelled, Labels: appLabels("ghost", "web")}, + {Name: inUse, Labels: appLabels("shop", "web"), InUse: true, Containers: []string{"c1"}}, + } + + snapshot := &domain.PruneProtectionSnapshot{VolumeClaims: []domain.VolumeClaim{ + {Name: released, App: "shop", AppID: "shop-uuid", Service: "web", State: domain.VolumeClaimReleased}, + {Name: attached, App: "shop", AppID: "shop-uuid", Service: "api", State: domain.VolumeClaimAttached}, + {Name: retained, App: "old", AppID: "old-uuid", Service: "web", State: domain.VolumeClaimRetained}, + {Name: inUse, App: "shop", AppID: "shop-uuid", Service: "web", State: domain.VolumeClaimAttached}, + }} + + verdicts := PlanVolumes(VolumePlanInput{Volumes: volumes, Snapshot: snapshot, Complete: true}) + byName := indexVolumeVerdicts(t, verdicts) + + expect := map[string]domain.PruneVerdict{ + released: domain.PruneVerdictEligible, + attached: domain.PruneVerdictProtected, + retained: domain.PruneVerdictProtected, + legacy: domain.PruneVerdictProtected, + serviceVolume: domain.PruneVerdictProtected, + foreign: domain.PruneVerdictProtected, + labelled: domain.PruneVerdictUnknown, + inUse: domain.PruneVerdictProtected, + } + for name, want := range expect { + if byName[name] != want { + t.Fatalf("%s = %v, want %v", name, byName[name], want) + } + } + + // A retained record is never reinterpreted as released, even when + // the labels look released-compatible. + if got := byName[retained]; got != domain.PruneVerdictProtected { + t.Fatalf("retained volume = %v, want protected", got) + } +} + +func TestPlanVolumesReleasedRequiresAgreeingLabels(t *testing.T) { + name := "gordon-shop-web-data" + claim := domain.VolumeClaim{Name: name, App: "shop", AppID: "shop-uuid", Service: "web", State: domain.VolumeClaimReleased} + + tests := []struct { + name string + labels map[string]string + want domain.PruneVerdict + }{ + {"labels agree by name", map[string]string{domain.LabelManaged: "true", domain.LabelApp: "shop", domain.LabelAppID: "shop-uuid", domain.LabelAppService: "web"}, domain.PruneVerdictEligible}, + {"labels agree by uuid", map[string]string{domain.LabelManaged: "true", domain.LabelApp: "shop-uuid", domain.LabelAppID: "shop-uuid"}, domain.PruneVerdictEligible}, + {"missing managed marker", map[string]string{domain.LabelApp: "shop", domain.LabelAppID: "shop-uuid"}, domain.PruneVerdictUnknown}, + {"missing app label", map[string]string{domain.LabelManaged: "true", domain.LabelAppID: "shop-uuid"}, domain.PruneVerdictUnknown}, + {"contradicting app", map[string]string{domain.LabelManaged: "true", domain.LabelApp: "other", domain.LabelAppID: "shop-uuid"}, domain.PruneVerdictUnknown}, + {"contradicting service", map[string]string{domain.LabelManaged: "true", domain.LabelApp: "shop", domain.LabelAppID: "shop-uuid", domain.LabelAppService: "api"}, domain.PruneVerdictUnknown}, + {"no labels", nil, domain.PruneVerdictUnknown}, + {"missing incarnation stamp", map[string]string{domain.LabelManaged: "true", domain.LabelApp: "shop", domain.LabelAppService: "web"}, domain.PruneVerdictUnknown}, + } + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + verdicts := PlanVolumes(VolumePlanInput{ + Volumes: []*domain.VolumeInfo{{Name: name, Labels: tc.labels}}, + Snapshot: &domain.PruneProtectionSnapshot{VolumeClaims: []domain.VolumeClaim{claim}}, + Complete: true, + }) + if verdicts[0].Verdict != tc.want { + t.Fatalf("verdict = %v (%v), want %v", verdicts[0].Verdict, verdicts[0].Reasons, tc.want) + } + }) + } +} + +func TestPlanVolumesFailClosedOnIncompleteOwnership(t *testing.T) { + name := "gordon-shop-web-data" + volume := &domain.VolumeInfo{Name: name, Labels: appLabels("shop", "web")} + claim := domain.VolumeClaim{Name: name, App: "shop", AppID: "shop-uuid", Service: "web", State: domain.VolumeClaimReleased} + + snapshot := &domain.PruneProtectionSnapshot{ + VolumeClaims: []domain.VolumeClaim{claim}, + Gaps: []domain.InventoryGap{{ + Source: domain.InventorySourceOwnership, + Reason: domain.PruneReasonUnknownInventory, + Detail: "unreadable ownership record", + }}, + } + verdicts := PlanVolumes(VolumePlanInput{Volumes: []*domain.VolumeInfo{volume}, Snapshot: snapshot, Complete: true}) + if verdicts[0].Verdict != domain.PruneVerdictUnknown { + t.Fatalf("released volume under incomplete ownership = %v, want unknown", verdicts[0].Verdict) + } + + // A nil snapshot proves nothing: it is incomplete by definition. + verdicts = PlanVolumes(VolumePlanInput{Volumes: []*domain.VolumeInfo{volume}, Complete: true}) + if verdicts[0].Verdict == domain.PruneVerdictEligible { + t.Fatalf("nil snapshot made a volume eligible: %+v", verdicts[0]) + } + + verdicts = PlanVolumes(VolumePlanInput{ + Volumes: []*domain.VolumeInfo{volume}, + Snapshot: &domain.PruneProtectionSnapshot{VolumeClaims: []domain.VolumeClaim{claim}}, + Complete: false, + }) + if verdicts[0].Verdict != domain.PruneVerdictUnknown { + t.Fatalf("incomplete runtime inventory = %v, want unknown", verdicts[0].Verdict) + } +} + +func TestPlanVolumesZeroDeletionIsSuccess(t *testing.T) { + volumes := []*domain.VolumeInfo{ + {Name: "pgdata"}, + {Name: "gordon-shop-data", Labels: appLabels("shop", "web"), InUse: true}, + } + verdicts := PlanVolumes(VolumePlanInput{Volumes: volumes, Snapshot: &domain.PruneProtectionSnapshot{}, Complete: true}) + if len(verdicts) != 2 { + t.Fatalf("expected one verdict per volume, got %d", len(verdicts)) + } + for _, verdict := range verdicts { + if verdict.Verdict == domain.PruneVerdictEligible { + t.Fatalf("volume %q unexpectedly eligible", verdict.Ref.Name) + } + if len(verdict.Reasons) == 0 { + t.Fatalf("volume %q carries no reason", verdict.Ref.Name) + } + } +} + +func TestPlanVolumesOrderIsDeterministic(t *testing.T) { + volumes := []*domain.VolumeInfo{{Name: "c"}, {Name: "a"}, {Name: "b"}} + first := PlanVolumes(VolumePlanInput{Volumes: volumes, Complete: true}) + second := PlanVolumes(VolumePlanInput{Volumes: []*domain.VolumeInfo{volumes[1], volumes[2], volumes[0]}, Complete: true}) + for index := range first { + if first[index].Ref != second[index].Ref { + t.Fatalf("order changed at %d: %v vs %v", index, first[index].Ref, second[index].Ref) + } + } +} + +func TestPlansProduceValidPlanEntries(t *testing.T) { + // Every verdict the planners emit must survive plan validation. + plan := &domain.PrunePlan{} + plan.Tags = PlanRegistryTags(RegistryPlanInput{ + Tags: []RegistryTagCandidate{ + candidate("app", "latest", time.Now(), digest("a")), + candidate("app", "v1", time.Now().Add(-time.Hour), digest("b")), + }, + Snapshot: emptySnapshot(), + KeepLast: 1, + Complete: true, + }) + plan.Images = PlanRuntimeImages(RuntimePlanInput{ + Images: []domain.RuntimeImage{ + {ID: "sha256:one", Labels: appLabels("app", "web")}, + {ID: "sha256:two"}, + }, + Snapshot: emptySnapshot(), + Complete: true, + }) + plan.Volumes = PlanVolumes(VolumePlanInput{ + Volumes: []*domain.VolumeInfo{{Name: "pgdata"}, {Name: "gordon-app-data", Labels: appLabels("app", "web")}}, + Snapshot: &domain.PruneProtectionSnapshot{}, + Complete: true, + }) + if err := plan.Validate(); err != nil { + t.Fatalf("planner output failed plan validation: %v", err) + } + if len(plan.EligibleTags()) != 0 { + t.Fatalf("latest + 1 must retain every tag, got %v", plan.EligibleTags()) + } +} + +func candidate(repository, tag string, modTime time.Time, manifestDigest string) RegistryTagCandidate { + return RegistryTagCandidate{ + Ref: domain.RegistryTagRef{Repository: repository, Tag: tag}, + Digest: manifestDigest, + ModTime: modTime, + ManifestKnown: true, + } +} + +// emptySnapshot is a complete snapshot that protects nothing. +func emptySnapshot() *domain.PruneProtectionSnapshot { + return &domain.PruneProtectionSnapshot{} +} + +func appLabels(app, service string) map[string]string { + return map[string]string{ + domain.LabelManaged: "true", + domain.LabelApp: app, + domain.LabelAppID: app + "-uuid", + domain.LabelAppService: service, + } +} + +func indexVerdicts(t *testing.T, verdicts []domain.RegistryTagVerdict) map[string]domain.PruneVerdict { + t.Helper() + out := make(map[string]domain.PruneVerdict, len(verdicts)) + for _, verdict := range verdicts { + out[verdict.Ref.Tag] = verdict.Verdict + } + return out +} + +func indexImageVerdicts(t *testing.T, verdicts []domain.RuntimeImageVerdict) map[string]domain.PruneVerdict { + t.Helper() + out := make(map[string]domain.PruneVerdict, len(verdicts)) + for _, verdict := range verdicts { + out[verdict.Ref.ID] = verdict.Verdict + } + return out +} + +func indexVolumeVerdicts(t *testing.T, verdicts []domain.VolumeVerdict) map[string]domain.PruneVerdict { + t.Helper() + out := make(map[string]domain.PruneVerdict, len(verdicts)) + for _, verdict := range verdicts { + out[verdict.Ref.Name] = verdict.Verdict + } + return out +} diff --git a/internal/usecase/pruneguard/registry.go b/internal/usecase/pruneguard/registry.go new file mode 100644 index 000000000..440a11b35 --- /dev/null +++ b/internal/usecase/pruneguard/registry.go @@ -0,0 +1,144 @@ +package pruneguard + +import ( + "sort" + "time" + + "github.com/bnema/gordon/internal/domain" +) + +// RegistryTagCandidate is one registry tag as inventoried. +type RegistryTagCandidate struct { + Ref domain.RegistryTagRef + // Digest is the manifest digest the tag resolved to; empty when + // unknown. + Digest string + // ModTime is the tag's last modification time. + ModTime time.Time + // ManifestKnown is false when the tag's manifest could not be read, + // so the content behind it cannot be proven unreferenced. + ManifestKnown bool +} + +// RegistryPlanInput is the complete input of one registry tag plan. +// KeepLast semantics match the product contract: latest plus the +// keep_last newest non-latest tags per repository. KeepLast <= 0 skips +// registry tag cleanup entirely, so the function returns no verdicts. +type RegistryPlanInput struct { + Tags []RegistryTagCandidate + Snapshot *domain.PruneProtectionSnapshot + KeepLast int + // Complete is false when the registry inventory could not be read + // in full. Candidates that would otherwise be eligible become + // unknown instead. + Complete bool +} + +// PlanRegistryTags computes one verdict per candidate tag. Output is +// ordered by repository then tag, so callers and reports are +// deterministic regardless of input order. +func PlanRegistryTags(input RegistryPlanInput) []domain.RegistryTagVerdict { + if input.KeepLast <= 0 { + return nil + } + + kept := retentionWindow(input.Tags, input.KeepLast) + + verdicts := make([]domain.RegistryTagVerdict, 0, len(input.Tags)) + for _, candidate := range input.Tags { + verdicts = append(verdicts, planRegistryTag(candidate, kept, input)) + } + sortRegistryTagVerdicts(verdicts) + return verdicts +} + +func planRegistryTag(candidate RegistryTagCandidate, kept map[string]struct{}, input RegistryPlanInput) domain.RegistryTagVerdict { + verdict := domain.RegistryTagVerdict{Ref: candidate.Ref, Digest: candidate.Digest} + + if !candidate.Ref.Valid() { + verdict.Verdict, verdict.Reasons = unknownResult(domain.PruneReasonUnknownIdentity) + return verdict + } + + // A durable root wins over every other consideration: protected + // candidates stay protected even when the inventory is incomplete. + refs := []string{candidate.Ref.Key(), candidate.Ref.String()} + if candidate.Digest != "" { + refs = append(refs, candidate.Digest) + } + if matched, reasons := rootReasonsOrFallback(input.Snapshot, refs...); matched { + verdict.Verdict, verdict.Reasons = protected(reasons, domain.PruneReasonProtectedSharedContent) + return verdict + } + + if candidate.Ref.Tag == "latest" { + verdict.Verdict, verdict.Reasons = protected(nil, domain.PruneReasonProtectedLatest) + return verdict + } + if _, inWindow := kept[candidate.Ref.Key()]; inWindow { + verdict.Verdict, verdict.Reasons = protected(nil, domain.PruneReasonProtectedRetentionWindow) + return verdict + } + + if !candidate.ManifestKnown { + verdict.Verdict, verdict.Reasons = unknownResult(domain.PruneReasonUnknownManifest) + return verdict + } + // Safety is only provable against a complete view of both the + // registry and the durable facts that claim it. An incomplete app + // snapshot could be hiding a root for this very tag. + if !input.Complete || !input.Snapshot.Complete() { + verdict.Verdict, verdict.Reasons = unknownResult(domain.PruneReasonUnknownInventory) + return verdict + } + + verdict.Verdict, verdict.Reasons = eligibleResult(domain.PruneReasonEligibleRetention) + return verdict +} + +// retentionWindow returns the canonical keys of the latest tag plus the +// keep_last newest non-latest tags of every repository. Ordering is by +// modification time descending, then tag descending, so equal +// timestamps still produce one deterministic window. +func retentionWindow(tags []RegistryTagCandidate, keepLast int) map[string]struct{} { + byRepository := make(map[string][]RegistryTagCandidate) + for _, tag := range tags { + if !tag.Ref.Valid() { + continue + } + byRepository[tag.Ref.Repository] = append(byRepository[tag.Ref.Repository], tag) + } + + kept := make(map[string]struct{}) + for _, repositoryTags := range byRepository { + sort.SliceStable(repositoryTags, func(i, j int) bool { + if repositoryTags[i].ModTime.Equal(repositoryTags[j].ModTime) { + return repositoryTags[i].Ref.Tag > repositoryTags[j].Ref.Tag + } + return repositoryTags[i].ModTime.After(repositoryTags[j].ModTime) + }) + + count := 0 + for _, tag := range repositoryTags { + if tag.Ref.Tag == "latest" { + kept[tag.Ref.Key()] = struct{}{} + continue + } + if count >= keepLast { + break + } + kept[tag.Ref.Key()] = struct{}{} + count++ + } + } + return kept +} + +func sortRegistryTagVerdicts(verdicts []domain.RegistryTagVerdict) { + sort.SliceStable(verdicts, func(i, j int) bool { + if verdicts[i].Ref.Repository != verdicts[j].Ref.Repository { + return verdicts[i].Ref.Repository < verdicts[j].Ref.Repository + } + return verdicts[i].Ref.Tag < verdicts[j].Ref.Tag + }) +} diff --git a/internal/usecase/pruneguard/runtime.go b/internal/usecase/pruneguard/runtime.go new file mode 100644 index 000000000..05f20e2d3 --- /dev/null +++ b/internal/usecase/pruneguard/runtime.go @@ -0,0 +1,263 @@ +package pruneguard + +import ( + "sort" + "strings" + + "github.com/bnema/gordon/internal/domain" +) + +// RuntimePlanInput is the complete input of one runtime image plan. +type RuntimePlanInput struct { + Images []domain.RuntimeImage + Containers []domain.RuntimeContainerUse + Snapshot *domain.PruneProtectionSnapshot + // ProtectedContainerIDs are container IDs that must retain their + // images whatever their running state: durable recovery + // inhibitions name the generation that a replacement may already + // have superseded. A protected ID with no matching container makes + // every candidate unknown, because its image cannot be identified. + ProtectedContainerIDs []string + // Complete is false when the runtime inventory could not be read in + // full. Candidates that would otherwise be eligible become unknown. + Complete bool +} + +// containerUseIndex indexes container image usage for one planning run. +type containerUseIndex struct { + imageIDs map[string]struct{} + refs map[string]struct{} + // protectedImageIDs are the images held by protected containers. + protectedImageIDs map[string]struct{} + // unknown marks a container whose image identity could not be + // resolved. It makes every image candidate unknown: an image used + // by any container, including an unknown one, must survive. + unknown bool +} + +func buildContainerUseIndex(containers []domain.RuntimeContainerUse, protectedIDs []string) containerUseIndex { + index := containerUseIndex{ + imageIDs: make(map[string]struct{}, len(containers)), + refs: make(map[string]struct{}, len(containers)), + protectedImageIDs: make(map[string]struct{}), + } + byContainerID := make(map[string]domain.RuntimeContainerUse, len(containers)) + for _, use := range containers { + if use.ContainerID != "" { + byContainerID[use.ContainerID] = use + } + if use.ImageID != "" { + index.imageIDs[use.ImageID] = struct{}{} + } + if use.ImageRef != "" { + index.refs[use.ImageRef] = struct{}{} + } + if use.ImageUnknown || (use.ImageID == "" && use.ImageRef == "") { + index.unknown = true + } + } + + for _, containerID := range protectedIDs { + use, ok := byContainerID[containerID] + if !ok || use.ImageID == "" { + // The protected generation's image cannot be identified, so + // no candidate can be proven unused. + index.unknown = true + continue + } + index.protectedImageIDs[use.ImageID] = struct{}{} + } + return index +} + +func (i containerUseIndex) usesImage(image domain.RuntimeImage) bool { + if _, ok := i.imageIDs[image.ID]; ok { + return true + } + for _, tag := range image.RepoTags { + if _, ok := i.refs[tag]; ok { + return true + } + } + for _, digest := range image.RepoDigests { + if _, ok := i.refs[digest]; ok { + return true + } + } + return false +} + +// PlanRuntimeImages computes one verdict per runtime image. Output is +// ordered by image ID so callers and reports are deterministic. +func PlanRuntimeImages(input RuntimePlanInput) []domain.RuntimeImageVerdict { + uses := buildContainerUseIndex(input.Containers, input.ProtectedContainerIDs) + + verdicts := make([]domain.RuntimeImageVerdict, 0, len(input.Images)) + for _, image := range input.Images { + verdicts = append(verdicts, planRuntimeImage(image, uses, input)) + } + sort.SliceStable(verdicts, func(i, j int) bool { + return verdicts[i].Ref.ID < verdicts[j].Ref.ID + }) + return verdicts +} + +func planRuntimeImage(image domain.RuntimeImage, uses containerUseIndex, input RuntimePlanInput) domain.RuntimeImageVerdict { + verdict := domain.RuntimeImageVerdict{Ref: domain.RuntimeImageRef{ID: image.ID}} + + if !verdict.Ref.Valid() { + verdict.Verdict, verdict.Reasons = unknownResult(domain.PruneReasonUnknownIdentity) + return verdict + } + + // A durable root wins over provenance: a still-referenced image + // stays protected even if its labels are gone. Repo tags are the + // runtime-side identity of app-pinned content, so they are matched + // against tag-keyed roots as well as digests. + refs := append([]string{image.ID}, image.RepoDigests...) + for _, repoTag := range image.RepoTags { + if repoTag == "" || repoTag == ":" || repoTag == "" { + continue + } + refs = append(refs, repoTag) + if tagRef, ok := domain.ParseRegistryTagRef(repoTag); ok { + refs = append(refs, tagRef.Key()) + } + } + if matched, reasons := rootReasonsOrFallback(input.Snapshot, refs...); matched { + verdict.Verdict, verdict.Reasons = protected(reasons, domain.PruneReasonProtectedSharedContent) + return verdict + } + + // Live or ambiguous container use protects before any ownership claim + // is consulted: an image a container could be running must survive. + if useVerdict, reasons, decided := containerUseVerdict(image, uses); decided { + return domain.RuntimeImageVerdict{Ref: domain.RuntimeImageRef{ID: image.ID}, Verdict: useVerdict, Reasons: reasons} + } + + // Labels are hints only. Deletion requires positive durable ownership: + // a matching claim from an app's ownership record. Every matching + // claim is considered: an attached or retained record protects even + // when a newer released record for the same image also exists. + if claimVerdict, reasons, decided := imageClaimVerdict(image, input.Snapshot); decided { + return domain.RuntimeImageVerdict{Ref: domain.RuntimeImageRef{ID: image.ID}, Verdict: claimVerdict, Reasons: reasons} + } + + // Safety is only provable against a complete view of both the + // runtime and the durable facts that claim the image. + if !input.Complete || !input.Snapshot.Complete() { + verdict.Verdict, verdict.Reasons = unknownResult(domain.PruneReasonUnknownInventory) + return verdict + } + + verdict.Verdict, verdict.Reasons = eligibleResult(domain.PruneReasonEligibleDanglingRuntimeImage) + return verdict +} + +// containerUseVerdict decides a candidate from container/tag state alone. +func containerUseVerdict(image domain.RuntimeImage, uses containerUseIndex) (domain.PruneVerdict, []domain.PruneReason, bool) { + if hasRepoTags(image.RepoTags) { + verdict, reasons := protected(nil, domain.PruneReasonProtectedTagged) + return verdict, reasons, true + } + if uses.usesImage(image) { + verdict, reasons := protected(nil, domain.PruneReasonProtectedContainerUse) + return verdict, reasons, true + } + if _, held := uses.protectedImageIDs[image.ID]; held { + verdict, reasons := protected(nil, domain.PruneReasonProtectedRecoveryInhibition) + return verdict, reasons, true + } + if uses.unknown { + verdict, reasons := unknownResult(domain.PruneReasonUnknownContainerUse) + return verdict, reasons, true + } + return "", nil, false +} + +// noClaimVerdict classifies a runtime image with no durable ownership +// record. App labels without a record are never eligible. +func noClaimVerdict(image domain.RuntimeImage) (domain.PruneVerdict, []domain.PruneReason) { + switch ClassifyResource(image.Labels) { + case ProvenanceUnmanaged: + return protected(nil, domain.PruneReasonProtectedUnmanaged) + case ProvenanceLegacyManaged: + return protected(nil, domain.PruneReasonProtectedManagedOnly) + case ProvenanceServiceManaged: + return protected(nil, domain.PruneReasonProtectedServiceManaged) + default: + return unknownResult(domain.PruneReasonUnknownProvenance) + } +} + +// imageClaimVerdict classifies an image from every matching ownership claim. +// It reports whether the claims decided the verdict; false means all matching +// claims are released and the ordinary eligibility gates still apply. +func imageClaimVerdict(image domain.RuntimeImage, snapshot *domain.PruneProtectionSnapshot) (domain.PruneVerdict, []domain.PruneReason, bool) { + claims := snapshot.ImageClaimsFor(imageIdentities(image)...) + if len(claims) == 0 { + verdict, reasons := noClaimVerdict(image) + return verdict, reasons, true + } + protecting, released, invalid := false, false, false + for _, claim := range claims { + if claim.State.Protects() { + protecting = true + break + } + if !claim.Valid() { + invalid = true + continue + } + released = true + } + switch { + case protecting: + verdict, reasons := protected(nil, domain.PruneReasonProtectedOwnership) + return verdict, reasons, true + case invalid || !released: + verdict, reasons := unknownResult(domain.PruneReasonUnknownProvenance) + return verdict, reasons, true + default: + return "", nil, false + } +} + +// imageIdentities returns every identity a durable image claim may match: +// the runtime image ID, full repo tags and digests, the digest part of a +// repo digest, and the registry tag key of a repo tag. +func imageIdentities(image domain.RuntimeImage) []string { + identities := make([]string, 0, 1+len(image.RepoTags)+2*len(image.RepoDigests)) + identities = append(identities, image.ID) + for _, repoTag := range image.RepoTags { + if repoTag == "" || repoTag == ":" || repoTag == "" { + continue + } + identities = append(identities, repoTag) + if tagRef, ok := domain.ParseRegistryTagRef(repoTag); ok { + identities = append(identities, tagRef.Key()) + } + } + for _, repoDigest := range image.RepoDigests { + if repoDigest == "" { + continue + } + identities = append(identities, repoDigest) + if _, digest, ok := strings.Cut(repoDigest, "@"); ok && digest != "" { + identities = append(identities, digest) + } + } + return identities +} + +// hasRepoTags reports whether the runtime still has a usable tag for the +// image. Placeholder tags never count. +func hasRepoTags(repoTags []string) bool { + for _, tag := range repoTags { + if tag == "" || tag == ":" || tag == "" { + continue + } + return true + } + return false +} diff --git a/internal/usecase/pruneguard/volume.go b/internal/usecase/pruneguard/volume.go new file mode 100644 index 000000000..bf1371f3d --- /dev/null +++ b/internal/usecase/pruneguard/volume.go @@ -0,0 +1,111 @@ +package pruneguard + +import ( + "sort" + + "github.com/bnema/gordon/internal/domain" +) + +// VolumePlanInput is the complete input of one volume plan. +type VolumePlanInput struct { + Volumes []*domain.VolumeInfo + Snapshot *domain.PruneProtectionSnapshot + // Complete is false when the runtime volume inventory could not be + // read in full. Candidates that would otherwise be eligible become + // unknown. + Complete bool +} + +// PlanVolumes computes one verdict per runtime volume. +// +// A volume is eligible only when a durable ownership record explicitly +// records it as released, the runtime labels agree with that record, and +// no container uses it. Labels alone never make a volume deletable, and +// a retained record is never reinterpreted as released. A legitimately +// empty result is a success, not an error. +// +// Output is ordered by volume name so callers are deterministic. +func PlanVolumes(input VolumePlanInput) []domain.VolumeVerdict { + ownershipIncomplete := input.Snapshot.HasGapFor(domain.InventorySourceOwnership) + + verdicts := make([]domain.VolumeVerdict, 0, len(input.Volumes)) + for _, volume := range input.Volumes { + verdicts = append(verdicts, planVolume(volume, input, ownershipIncomplete)) + } + sort.SliceStable(verdicts, func(i, j int) bool { + return verdicts[i].Ref.Name < verdicts[j].Ref.Name + }) + return verdicts +} + +func planVolume(volume *domain.VolumeInfo, input VolumePlanInput, ownershipIncomplete bool) domain.VolumeVerdict { + verdict := domain.VolumeVerdict{} + if volume == nil { + verdict.Verdict, verdict.Reasons = unknownResult(domain.PruneReasonUnknownIdentity) + return verdict + } + verdict.Ref = domain.RuntimeVolumeRef{Name: volume.Name} + + if !verdict.Ref.Valid() { + verdict.Verdict, verdict.Reasons = unknownResult(domain.PruneReasonUnknownIdentity) + return verdict + } + + if volume.InUse { + verdict.Verdict, verdict.Reasons = protected(nil, domain.PruneReasonProtectedContainerUse) + return verdict + } + + claim, hasClaim := input.Snapshot.VolumeClaimFor(volume.Name) + + // Positive protection first: any attached or retained record wins, + // even when the rest of the inventory is incomplete, and a newer + // incarnation never un-protects an older claim. + if input.Snapshot.VolumeProtects(volume.Name) { + verdict.Verdict, verdict.Reasons = protected(nil, domain.PruneReasonProtectedOwnership) + return verdict + } + + if !hasClaim { + switch ClassifyResource(volume.Labels) { + case ProvenanceUnmanaged: + verdict.Verdict, verdict.Reasons = protected(nil, domain.PruneReasonProtectedUnmanaged) + case ProvenanceLegacyManaged: + verdict.Verdict, verdict.Reasons = protected(nil, domain.PruneReasonProtectedManagedOnly) + case ProvenanceServiceManaged: + verdict.Verdict, verdict.Reasons = protected(nil, domain.PruneReasonProtectedServiceManaged) + default: + // App labels with no durable record: ownership history is + // missing, and labels never override that. + verdict.Verdict, verdict.Reasons = unknownResult(domain.PruneReasonUnknownProvenance) + } + return verdict + } + + // Released: the durable record and the runtime labels must agree on + // the owning app incarnation before the volume can be deleted. + if !claim.MatchesLabels(volume.Labels) { + verdict.Verdict, verdict.Reasons = unknownResult(domain.PruneReasonUnknownProvenance) + return verdict + } + + if ownershipIncomplete || !input.Complete { + verdict.Verdict, verdict.Reasons = unknownResult(domain.PruneReasonUnknownInventory) + return verdict + } + if !claim.Valid() { + verdict.Verdict, verdict.Reasons = unknownResult(domain.PruneReasonUnknownProvenance) + return verdict + } + // Eligibility needs both halves of the proof: a durable released + // record and the incarnation UUID that ties the runtime volume to + // it. A record without an incarnation cannot prove which volume it + // claims, so it never authorizes deletion. + if claim.AppID == "" { + verdict.Verdict, verdict.Reasons = unknownResult(domain.PruneReasonUnknownProvenance) + return verdict + } + + verdict.Verdict, verdict.Reasons = eligibleResult(domain.PruneReasonEligibleReleasedVolume) + return verdict +} diff --git a/internal/usecase/publictls/integration_test.go b/internal/usecase/publictls/integration_test.go index a1f7a0f4e..81c40455b 100644 --- a/internal/usecase/publictls/integration_test.go +++ b/internal/usecase/publictls/integration_test.go @@ -22,10 +22,10 @@ func TestPublicTLSIntegration_DNS01WildcardStatusAndCertificateLookup(t *testing // Routes: app.example.com, api.prod.example.com, example.com routes := &fakeRoutes{ - routes: []domain.Route{ - {Domain: "app.example.com"}, - {Domain: "api.prod.example.com"}, - {Domain: "example.com"}, + hosts: []out.AppHost{ + {Host: "app.example.com"}, + {Host: "api.prod.example.com"}, + {Host: "example.com"}, }, } @@ -168,8 +168,8 @@ func TestPublicTLSIntegration_HTTP01PerRouteChallengeFlow(t *testing.T) { ctx := context.Background() routes := &fakeRoutes{ - routes: []domain.Route{ - {Domain: "app.example.com"}, + hosts: []out.AppHost{ + {Host: "app.example.com"}, }, } @@ -250,8 +250,8 @@ func TestPublicTLSIntegration_MissingRequiredCertReportsCoverageError(t *testing ctx := context.Background() routes := &fakeRoutes{ - routes: []domain.Route{ - {Domain: "missing.example.com"}, + hosts: []out.AppHost{ + {Host: "missing.example.com"}, }, } diff --git a/internal/usecase/publictls/renewal_test.go b/internal/usecase/publictls/renewal_test.go index 6581c2384..98b035dce 100644 --- a/internal/usecase/publictls/renewal_test.go +++ b/internal/usecase/publictls/renewal_test.go @@ -127,7 +127,7 @@ func TestRenewalLoop_ReconcilesMissingCertificatesOnTick(t *testing.T) { TLSPort: 8443, } - routes := &fakeRoutes{routes: []domain.Route{{Domain: "app.example.com"}}} + routes := &fakeRoutes{hosts: []out.AppHost{{Host: "app.example.com"}}} issuer, recorder := newMockPublicCertificateIssuer(t, nil, nil) store, _ := newMockCertificateStore(t) diff --git a/internal/usecase/publictls/service.go b/internal/usecase/publictls/service.go index 244628842..ecf5496c4 100644 --- a/internal/usecase/publictls/service.go +++ b/internal/usecase/publictls/service.go @@ -15,9 +15,10 @@ import ( "github.com/bnema/gordon/internal/domain" ) -// RouteSource provides routes from which certificate targets are derived. +// RouteSource provides ACTIVE-derived hosts from which certificate +// targets are derived, plus installation external routes. type RouteSource interface { - GetRoutes(ctx context.Context) []domain.Route + out.AppHostSource GetExternalRoutes() map[string]string } @@ -104,9 +105,8 @@ func (s *Service) Load(ctx context.Context) error { required := make(map[string]struct{}) if s.deps.Routes != nil { - routes := s.deps.Routes.GetRoutes(ctx) external := s.deps.Routes.GetExternalRoutes() - required = canonicalHostSet(routeHosts(routes, external, s.additionalHosts)) + required = canonicalHostSet(routeHosts(s.deps.Routes.AppHosts(), external, s.additionalHosts)) } s.mu.Lock() @@ -143,9 +143,8 @@ func (s *Service) SetAdditionalHosts(ctx context.Context, hosts []string) { s.additionalHosts = append([]string(nil), hosts...) required := canonicalHostSet(s.additionalHosts) if s.deps.Routes != nil { - routes := s.deps.Routes.GetRoutes(ctx) external := s.deps.Routes.GetExternalRoutes() - required = canonicalHostSet(routeHosts(routes, external, s.additionalHosts)) + required = canonicalHostSet(routeHosts(s.deps.Routes.AppHosts(), external, s.additionalHosts)) } s.mu.Lock() @@ -199,9 +198,8 @@ func (s *Service) Reconcile(ctx context.Context) error { // Get route hosts early to build required hosts set before target derivation. // This ensures GetCertificate returns ErrTLSRouteNotCovered even if // DeriveCertificateTargets fails (e.g. broken DNS-01 zone resolver). - routes := s.deps.Routes.GetRoutes(ctx) external := s.deps.Routes.GetExternalRoutes() - hosts := routeHosts(routes, external, s.additionalHosts) + hosts := routeHosts(s.deps.Routes.AppHosts(), external, s.additionalHosts) // Build required hosts set from route hosts (before target derivation). required := canonicalHostSet(hosts) @@ -215,7 +213,7 @@ func (s *Service) Reconcile(ctx context.Context) error { // Derive desired targets. targets, err := DeriveCertificateTargets(ctx, effective.Mode, - routes, external, + s.deps.Routes.AppHosts(), external, s.additionalHosts, s.deps.ZoneResolver, ) diff --git a/internal/usecase/publictls/service_test.go b/internal/usecase/publictls/service_test.go index 409e9d889..b558a42dc 100644 --- a/internal/usecase/publictls/service_test.go +++ b/internal/usecase/publictls/service_test.go @@ -32,19 +32,19 @@ import ( // copy-on-read semantics. Generated mocks cannot easily replicate this // pattern because the interface methods return slices/maps by value and // callers may modify the returned data; a hand-rolled fake ensures each -// GetRoutes/GetExternalRoutes call returns a defensive copy. +// AppHosts/GetExternalRoutes call returns a defensive copy. type fakeRoutes struct { mu sync.Mutex - routes []domain.Route + hosts []out.AppHost external map[string]string } -func (f *fakeRoutes) GetRoutes(_ context.Context) []domain.Route { +func (f *fakeRoutes) AppHosts() []out.AppHost { f.mu.Lock() defer f.mu.Unlock() - routesCopy := make([]domain.Route, len(f.routes)) - copy(routesCopy, f.routes) - return routesCopy + hostsCopy := make([]out.AppHost, len(f.hosts)) + copy(hostsCopy, f.hosts) + return hostsCopy } func (f *fakeRoutes) GetExternalRoutes() map[string]string { @@ -303,10 +303,10 @@ func TestServiceReconcileLimitsMissingObtainsPerRun(t *testing.T) { ctx := context.Background() routes := &fakeRoutes{ - routes: []domain.Route{ - {Domain: "one.example.com"}, - {Domain: "two.example.com"}, - {Domain: "three.example.com"}, + hosts: []out.AppHost{ + {Host: "one.example.com"}, + {Host: "two.example.com"}, + {Host: "three.example.com"}, }, } issuer, recorder := newMockPublicCertificateIssuer(t, nil, nil) @@ -348,11 +348,18 @@ func TestServiceReconcileLimitsMissingObtainsPerRun(t *testing.T) { assert.Len(t, storeState.All(), 3) } +// stubAppRoutes is an ACTIVE-derived host source for public-TLS tests. +type stubAppRoutes struct { + hosts []out.AppHost +} + +func (s *stubAppRoutes) AppHosts() []out.AppHost { return s.hosts } + +func (s *stubAppRoutes) GetExternalRoutes() map[string]string { return nil } + func TestServiceSetAdditionalHosts_ImmediatelyRevokesRemovedManagementDomain(t *testing.T) { ctx := t.Context() - routes := outmocks.NewMockRouteChecker(t) - routes.EXPECT().GetRoutes(mock.Anything).Return(nil).Maybe() - routes.EXPECT().GetExternalRoutes().Return(map[string]string{}).Maybe() + routes := &stubAppRoutes{} issuer, _ := newMockPublicCertificateIssuer(t, nil, nil) store, _ := newMockCertificateStore(t) @@ -417,9 +424,9 @@ func TestServiceReconcileRotatesBatchAfterFailedObtain(t *testing.T) { ctx := context.Background() routes := &fakeRoutes{ - routes: []domain.Route{ - {Domain: "b.example.com"}, - {Domain: "a.example.com"}, + hosts: []out.AppHost{ + {Host: "b.example.com"}, + {Host: "a.example.com"}, }, } issuer, recorder := newMockPublicCertificateIssuer(t, func(_ context.Context, order out.CertificateOrder) (*out.StoredCertificate, error) { @@ -471,9 +478,9 @@ func TestServiceLoadUsesZeroCursorWhenLoadStateFails(t *testing.T) { ctx := context.Background() routes := &fakeRoutes{ - routes: []domain.Route{ - {Domain: "b.example.com"}, - {Domain: "a.example.com"}, + hosts: []out.AppHost{ + {Host: "b.example.com"}, + {Host: "a.example.com"}, }, } issuer, recorder := newMockPublicCertificateIssuer(t, nil, nil) @@ -515,9 +522,9 @@ func TestServiceReconcileUsesInMemoryCursorWhenPersistFails(t *testing.T) { ctx := context.Background() routes := &fakeRoutes{ - routes: []domain.Route{ - {Domain: "b.example.com"}, - {Domain: "a.example.com"}, + hosts: []out.AppHost{ + {Host: "b.example.com"}, + {Host: "a.example.com"}, }, } issuer, recorder := newMockPublicCertificateIssuer(t, func(_ context.Context, order out.CertificateOrder) (*out.StoredCertificate, error) { @@ -570,9 +577,9 @@ func TestServiceReconcilePersistsBatchCursorAcrossRestart(t *testing.T) { ctx := context.Background() routes := &fakeRoutes{ - routes: []domain.Route{ - {Domain: "b.example.com"}, - {Domain: "a.example.com"}, + hosts: []out.AppHost{ + {Host: "b.example.com"}, + {Host: "a.example.com"}, }, } issuer, recorder := newMockPublicCertificateIssuer(t, func(_ context.Context, order out.CertificateOrder) (*out.StoredCertificate, error) { @@ -628,8 +635,8 @@ func TestServiceReconcileObtainsMissingHTTP01Cert(t *testing.T) { ctx := context.Background() routes := &fakeRoutes{ - routes: []domain.Route{ - {Domain: "app.example.com"}, + hosts: []out.AppHost{ + {Host: "app.example.com"}, }, } issuer, recorder := newMockPublicCertificateIssuer(t, nil, nil) @@ -684,7 +691,7 @@ func TestServiceReconcileObtainsMissingHTTP01Cert(t *testing.T) { func TestServiceReconcile_SerializesConcurrentRuns(t *testing.T) { ctx := context.Background() - routes := &fakeRoutes{routes: []domain.Route{{Domain: "app.example.com"}}} + routes := &fakeRoutes{hosts: []out.AppHost{{Host: "app.example.com"}}} obtainStarted := make(chan struct{}, 2) releaseObtain := make(chan struct{}) @@ -743,7 +750,7 @@ func TestServiceReconcile_SerializesConcurrentRuns(t *testing.T) { func TestServiceReconcileResolvesMissingEffectiveChallenge(t *testing.T) { ctx := context.Background() - routes := &fakeRoutes{routes: []domain.Route{{Domain: "app.example.com"}}} + routes := &fakeRoutes{hosts: []out.AppHost{{Host: "app.example.com"}}} issuer, _ := newMockPublicCertificateIssuer(t, nil, nil) store, _ := newMockCertificateStore(t) cfg := Config{ @@ -787,8 +794,8 @@ func TestServiceStatusReportsCoverage(t *testing.T) { }) routes := &fakeRoutes{ - routes: []domain.Route{ - {Domain: "app.example.com"}, + hosts: []out.AppHost{ + {Host: "app.example.com"}, }, } cfg := Config{ @@ -837,8 +844,8 @@ func TestServiceGetCertificateReturnsErrTLSRouteNotCovered(t *testing.T) { ctx := context.Background() routes := &fakeRoutes{ - routes: []domain.Route{ - {Domain: "app.example.com"}, + hosts: []out.AppHost{ + {Host: "app.example.com"}, }, } // Make Obtain fail so no cert is cached. @@ -888,8 +895,8 @@ func TestServiceReconcileDNS01BrokenResolverReturnsErrTLSRouteNotCovered(t *test ctx := context.Background() routes := &fakeRoutes{ - routes: []domain.Route{ - {Domain: "app.example.com"}, + hosts: []out.AppHost{ + {Host: "app.example.com"}, }, } issuer, _ := newMockPublicCertificateIssuer(t, nil, nil) @@ -960,7 +967,7 @@ func TestServiceLoadParsesPEMIntoTLSCertificate(t *testing.T) { // Certificate field is zero — will be populated by Load. }) - routes := &fakeRoutes{routes: []domain.Route{{Domain: "test.example.com"}}} + routes := &fakeRoutes{hosts: []out.AppHost{{Host: "test.example.com"}}} cfg := Config{Enabled: true} svc := NewService(cfg, ServiceDeps{ @@ -991,7 +998,7 @@ func TestServiceLoadParsesPEMIntoTLSCertificate(t *testing.T) { func TestServiceStatusReportsRouteErrorAfterObtainFailure(t *testing.T) { ctx := context.Background() - routes := &fakeRoutes{routes: []domain.Route{{Domain: "app.example.com"}}} + routes := &fakeRoutes{hosts: []out.AppHost{{Host: "app.example.com"}}} issuer, _ := newMockPublicCertificateIssuer(t, func(_ context.Context, _ out.CertificateOrder) (*out.StoredCertificate, error) { return nil, fmt.Errorf("acme unavailable token=sk-secret") }, nil) @@ -1042,7 +1049,7 @@ func TestServiceStatusCoverageUsesServingCertDespiteTransientError(t *testing.T) cfg := Config{Enabled: true} svc := NewService(cfg, ServiceDeps{ Config: cfg, - Routes: &fakeRoutes{routes: []domain.Route{{Domain: "app.example.com"}}}, + Routes: &fakeRoutes{hosts: []out.AppHost{{Host: "app.example.com"}}}, Store: store, }) require.NoError(t, svc.Load(ctx)) @@ -1075,8 +1082,8 @@ func TestServiceStatusRedactsSensitiveStrings(t *testing.T) { }) routes := &fakeRoutes{ - routes: []domain.Route{ - {Domain: "app.example.com"}, + hosts: []out.AppHost{ + {Host: "app.example.com"}, }, } cfg := Config{ diff --git a/internal/usecase/publictls/status.go b/internal/usecase/publictls/status.go index 6708a327b..34392b1be 100644 --- a/internal/usecase/publictls/status.go +++ b/internal/usecase/publictls/status.go @@ -111,8 +111,12 @@ func collectRouteDomains(ctx context.Context, routes RouteSource) []string { routeSet := make(map[string]struct{}) var domains []string - for _, r := range routes.GetRoutes(ctx) { - canonical, ok := domain.CanonicalRouteDomain(r.Domain) + for _, h := range routes.AppHosts() { + // tls=never interfaces stay plain HTTP: not ACME coverage candidates. + if h.TLSMode == domain.AppTLSNever { + continue + } + canonical, ok := domain.CanonicalRouteDomain(h.Host) if ok { if _, exists := routeSet[canonical]; !exists { routeSet[canonical] = struct{}{} diff --git a/internal/usecase/publictls/targets.go b/internal/usecase/publictls/targets.go index b3186f585..d47288eee 100644 --- a/internal/usecase/publictls/targets.go +++ b/internal/usecase/publictls/targets.go @@ -22,12 +22,12 @@ type CertificateTarget struct { func DeriveCertificateTargets( ctx context.Context, mode domain.ACMEChallengeMode, - routes []domain.Route, + appHosts []out.AppHost, external map[string]string, additionalHosts []string, resolver out.CloudflareZoneResolver, ) ([]CertificateTarget, error) { - hosts := routeHosts(routes, external, additionalHosts) + hosts := routeHosts(appHosts, external, additionalHosts) switch mode { case domain.ACMEChallengeHTTP01: @@ -39,9 +39,9 @@ func DeriveCertificateTargets( } } -// routeHosts collects all unique canonical hosts from routes and external keys, +// routeHosts collects all unique canonical hosts from app hosts and external keys, // sorted alphabetically. Trailing dots are stripped before canonicalization. -func routeHosts(routes []domain.Route, external map[string]string, additionalHosts []string) []string { +func routeHosts(appHosts []out.AppHost, external map[string]string, additionalHosts []string) []string { seen := make(map[string]struct{}) var hosts []string @@ -57,8 +57,12 @@ func routeHosts(routes []domain.Route, external map[string]string, additionalHos } } - for _, r := range routes { - addHost(r.Domain) + for _, h := range appHosts { + // tls=never interfaces stay plain HTTP: no certificate target. + if h.TLSMode == domain.AppTLSNever { + continue + } + addHost(h.Host) } for h := range external { addHost(h) diff --git a/internal/usecase/publictls/targets_test.go b/internal/usecase/publictls/targets_test.go index 61e5ba0b6..733f9f347 100644 --- a/internal/usecase/publictls/targets_test.go +++ b/internal/usecase/publictls/targets_test.go @@ -14,8 +14,8 @@ import ( ) func TestDeriveTargets_HTTP01PerRoute(t *testing.T) { - routes := []domain.Route{{Domain: "app.example.com"}, {Domain: "registry.example.com"}} - targets, err := DeriveCertificateTargets(context.Background(), domain.ACMEChallengeHTTP01, routes, nil, nil, nil) + hosts := []out.AppHost{{Host: "app.example.com"}, {Host: "registry.example.com"}} + targets, err := DeriveCertificateTargets(context.Background(), domain.ACMEChallengeHTTP01, hosts, nil, nil, nil) require.NoError(t, err) require.Len(t, targets, 2) @@ -28,10 +28,10 @@ func TestDeriveTargets_HTTP01PerRoute(t *testing.T) { } func TestDeriveTargets_DNS01WildcardBases(t *testing.T) { - routes := []domain.Route{{Domain: "app.example.com"}, {Domain: "api.prod.example.com"}, {Domain: "example.com"}} + hosts := []out.AppHost{{Host: "app.example.com"}, {Host: "api.prod.example.com"}, {Host: "example.com"}} resolver := outmocks.NewMockCloudflareZoneResolver(t) resolver.EXPECT().FindZone(mock.Anything, mock.Anything).Return(out.CloudflareZone{Name: "example.com"}, nil).Times(3) - targets, err := DeriveCertificateTargets(context.Background(), domain.ACMEChallengeCloudflareDNS01, routes, nil, nil, resolver) + targets, err := DeriveCertificateTargets(context.Background(), domain.ACMEChallengeCloudflareDNS01, hosts, nil, nil, resolver) require.NoError(t, err) require.Len(t, targets, 2) @@ -59,9 +59,9 @@ func TestDeriveTargets_IncludesAdditionalHosts(t *testing.T) { } func TestDeriveTargets_IncludesExternalRoutes(t *testing.T) { - routes := []domain.Route{{Domain: "app.example.com"}} + hosts := []out.AppHost{{Host: "app.example.com"}} external := map[string]string{"external.example.com": "127.0.0.1:8080"} - targets, err := DeriveCertificateTargets(context.Background(), domain.ACMEChallengeHTTP01, routes, external, nil, nil) + targets, err := DeriveCertificateTargets(context.Background(), domain.ACMEChallengeHTTP01, hosts, external, nil, nil) require.NoError(t, err) require.Len(t, targets, 2) @@ -70,25 +70,37 @@ func TestDeriveTargets_IncludesExternalRoutes(t *testing.T) { assert.Equal(t, []string{"external.example.com"}, targets[1].Names) } +func TestDeriveTargets_SkipsNeverTLSHosts(t *testing.T) { + hosts := []out.AppHost{ + {Host: "plain.example.com", TLSMode: "never"}, + {Host: "app.example.com", TLSMode: "auto"}, + } + targets, err := DeriveCertificateTargets(context.Background(), domain.ACMEChallengeHTTP01, hosts, nil, nil, nil) + require.NoError(t, err) + require.Len(t, targets, 1, "tls=never hosts stay plain HTTP: no certificate target") + assert.Equal(t, "http01-app.example.com", targets[0].ID) + assert.Equal(t, []string{"app.example.com"}, targets[0].Names) +} + func TestDeriveTargets_DNS01NilResolver_ReturnsError(t *testing.T) { - routes := []domain.Route{{Domain: "app.example.com"}} - _, err := DeriveCertificateTargets(context.Background(), domain.ACMEChallengeCloudflareDNS01, routes, nil, nil, nil) + hosts := []out.AppHost{{Host: "app.example.com"}} + _, err := DeriveCertificateTargets(context.Background(), domain.ACMEChallengeCloudflareDNS01, hosts, nil, nil, nil) require.Error(t, err) assert.Contains(t, err.Error(), "resolver is nil") } func TestDeriveTargets_DNS01MismatchedZone_ReturnsError(t *testing.T) { - routes := []domain.Route{{Domain: "app.example.com"}} + hosts := []out.AppHost{{Host: "app.example.com"}} resolver := outmocks.NewMockCloudflareZoneResolver(t) resolver.EXPECT().FindZone(mock.Anything, mock.Anything).Return(out.CloudflareZone{Name: "other.test"}, nil).Once() - _, err := DeriveCertificateTargets(context.Background(), domain.ACMEChallengeCloudflareDNS01, routes, nil, nil, resolver) + _, err := DeriveCertificateTargets(context.Background(), domain.ACMEChallengeCloudflareDNS01, hosts, nil, nil, resolver) require.Error(t, err) assert.Contains(t, err.Error(), "does not match host") } func TestDeriveTargets_DuplicateCanonicalization(t *testing.T) { - routes := []domain.Route{{Domain: "App.Example.Com"}, {Domain: "app.example.com"}} - targets, err := DeriveCertificateTargets(context.Background(), domain.ACMEChallengeHTTP01, routes, nil, nil, nil) + hosts := []out.AppHost{{Host: "App.Example.Com"}, {Host: "app.example.com"}} + targets, err := DeriveCertificateTargets(context.Background(), domain.ACMEChallengeHTTP01, hosts, nil, nil, nil) require.NoError(t, err) require.Len(t, targets, 1, "mixed-case duplicates should collapse to one target") assert.Equal(t, "http01-app.example.com", targets[0].ID) diff --git a/internal/usecase/registry/service.go b/internal/usecase/registry/service.go index 311f79cfd..f9979deb6 100644 --- a/internal/usecase/registry/service.go +++ b/internal/usecase/registry/service.go @@ -6,7 +6,6 @@ import ( "crypto/sha256" "crypto/sha512" "encoding/json" - "errors" "fmt" "io" "strings" @@ -16,10 +15,8 @@ import ( "github.com/bnema/zerowrap" "go.opentelemetry.io/otel" "go.opentelemetry.io/otel/attribute" - "go.opentelemetry.io/otel/metric" "go.opentelemetry.io/otel/trace" - "github.com/bnema/gordon/internal/adapters/out/telemetry" "github.com/bnema/gordon/internal/boundaries/out" "github.com/bnema/gordon/internal/domain" "github.com/bnema/gordon/internal/usecase/registrystate" @@ -30,17 +27,16 @@ var registryTracer = otel.Tracer("gordon.registry") // Service implements the RegistryService interface. type Service struct { - blobStorage out.BlobStorage - manifestStorage out.ManifestStorage - eventBus out.EventPublisher - metrics *telemetry.Metrics - suppressedImages sync.Map // imageName -> *time.Timer - mutationMu *sync.RWMutex - registryState *registrystate.State + blobStorage out.BlobStorage + manifestStorage out.ManifestStorage + eventBus out.EventPublisher + metrics out.Metrics + mutationMu *sync.RWMutex + registryState *registrystate.State } // SetMetrics sets the telemetry metrics for the registry service. -func (s *Service) SetMetrics(m *telemetry.Metrics) { +func (s *Service) SetMetrics(m out.Metrics) { s.metrics = m } @@ -64,85 +60,6 @@ func NewService( } } -// SuppressDeployEvent marks an image name to skip image.pushed events. -// The suppression auto-expires after 2 minutes to prevent leaks. -func (s *Service) SuppressDeployEvent(imageName string) { - imageName = ExtractImageName(strings.TrimSpace(imageName)) - if imageName == "" { - return - } - - var timer *time.Timer - timer = time.AfterFunc(2*time.Minute, func() { - // Only delete if this timer is still the current one, preventing an - // old timer's callback from removing a newer suppression entry. - if v, ok := s.suppressedImages.Load(imageName); ok && v == timer { - s.suppressedImages.Delete(imageName) - } - }) - if existing, loaded := s.suppressedImages.LoadOrStore(imageName, timer); loaded { - existing.(*time.Timer).Stop() - s.suppressedImages.Store(imageName, timer) - } -} - -// ClearDeployEventSuppression removes event suppression for an image. -func (s *Service) ClearDeployEventSuppression(imageName string) { - imageName = ExtractImageName(strings.TrimSpace(imageName)) - if imageName == "" { - return - } - - if v, loaded := s.suppressedImages.LoadAndDelete(imageName); loaded { - v.(*time.Timer).Stop() - } -} - -// ExtractImageName returns just the repository path of a container image -// reference, stripping any registry host prefix, tag, and digest. -// Examples: -// -// "reg.example.com/team/my-app:latest" -> "team/my-app" -// "reg.example.com/my-app@sha256:abc" -> "my-app" -// "my-app:v1.2" -> "my-app" -func ExtractImageName(imageRef string) string { - name := imageRef - // Strip digest - if idx := strings.Index(name, "@"); idx != -1 { - name = name[:idx] - } - // Strip tag - // Find the last colon, but only strip it if it comes after any slash - // (to avoid treating a port number in the host as a tag). - if idx := strings.LastIndex(name, ":"); idx != -1 { - slashIdx := strings.LastIndex(name, "/") - if idx > slashIdx { - name = name[:idx] - } - } - // Strip registry host: if the first segment contains a dot or colon it is - // a registry hostname; remove it. - parts := strings.SplitN(name, "/", 2) - if len(parts) == 2 { - host := parts[0] - if strings.ContainsAny(host, ".:") || host == "localhost" { - name = parts[1] - } - } - return name -} - -// IsDeployEventSuppressed checks if deploy events are suppressed for an image. -func (s *Service) IsDeployEventSuppressed(imageName string) bool { - imageName = ExtractImageName(strings.TrimSpace(imageName)) - if imageName == "" { - return false - } - - _, exists := s.suppressedImages.Load(imageName) - return exists -} - // GetManifest retrieves a manifest by name and reference. func (s *Service) GetManifest(ctx context.Context, name, reference string) (*domain.Manifest, error) { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ @@ -202,6 +119,10 @@ func (s *Service) PutManifest(ctx context.Context, manifest *domain.Manifest) (s digest = manifest.Reference } + if err := s.validateManifestBlobs(manifest.Name, manifest.Data); err != nil { + return "", log.WrapErr(err, "manifest references unowned content") + } + if err := s.manifestStorage.PutManifest(manifest.Name, manifest.Reference, manifest.ContentType, manifest.Data); err != nil { return "", log.WrapErr(err, "failed to store manifest") } @@ -209,29 +130,20 @@ func (s *Service) PutManifest(ctx context.Context, manifest *domain.Manifest) (s // Record push metrics if s.metrics != nil { - attrs := metric.WithAttributes( - attribute.String("name", manifest.Name), - attribute.String("reference", manifest.Reference), - ) - s.metrics.ImagePushTotal.Add(ctx, 1, attrs) - s.metrics.ImagePushSize.Add(ctx, int64(len(manifest.Data)), attrs) + s.metrics.RecordImagePush(ctx, manifest.Name, manifest.Reference, int64(len(manifest.Data))) } // Publish image pushed event only for tag references (not digests). // A docker push sends manifests by both digest and tag; firing only on // tag prevents duplicate deploy triggers for the same push. if s.eventBus != nil && !validation.IsDigest(manifest.Reference) { - if s.IsDeployEventSuppressed(manifest.Name) { - log.Info().Str("image", manifest.Name).Msg("skipping image.pushed event: CLI deploy intent active") - } else { - if err := s.eventBus.Publish(domain.EventImagePushed, domain.ImagePushedPayload{ - Name: manifest.Name, - Reference: manifest.Reference, - Manifest: manifest.Data, - Annotations: manifest.Annotations, - }); err != nil { - log.Warn().Err(err).Msg("failed to publish image pushed event") - } + if err := s.eventBus.Publish(domain.EventImagePushed, domain.ImagePushedPayload{ + Name: manifest.Name, + Reference: manifest.Reference, + Manifest: manifest.Data, + Annotations: manifest.Annotations, + }); err != nil { + log.Warn().Err(err).Msg("failed to publish image pushed event") } } @@ -277,8 +189,10 @@ func (s *Service) GetBlob(ctx context.Context, digest string) (io.ReadCloser, er return reader, nil } -// GetBlobPath returns the filesystem path to a blob only when a manifest in -// the requested repository references it. +// GetBlobPath returns the filesystem path to a blob only when the +// repository completed an upload of that digest. Ownership is never +// inferred from manifest references: a repository writer controls its own +// manifests, so a manifest naming a foreign digest must not confer access. func (s *Service) GetBlobPath(ctx context.Context, name, digest string) (string, error) { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ zerowrap.FieldLayer: "usecase", @@ -288,11 +202,11 @@ func (s *Service) GetBlobPath(ctx context.Context, name, digest string) (string, }) log := zerowrap.FromCtx(ctx) - referenced, err := s.repositoryReferencesDigest(name, digest) + owned, err := s.blobStorage.BlobOwnedByRepository(name, digest) if err != nil { return "", log.WrapErr(err, "failed to verify blob ownership") } - if !referenced { + if !owned { return "", domain.ErrBlobNotFound } @@ -318,70 +232,54 @@ type manifestReferences struct { const maxManifestTraversal = 10000 -func (s *Service) repositoryReferencesDigest(name, target string) (bool, error) { - tags, err := s.manifestStorage.ListTags(name) - if err != nil { - if errors.Is(err, domain.ErrManifestNotFound) { - return false, nil - } - return false, fmt.Errorf("list tags for repository %s: %w", name, err) +// validateManifestBlobs requires every config/layer blob a manifest +// references to be owned by the repository, and every child manifest to +// already exist in it. A manifest can therefore never manufacture access to +// content the repository did not receive through an authenticated push. +func (s *Service) validateManifestBlobs(name string, data []byte) error { + var refs manifestReferences + if err := json.Unmarshal(data, &refs); err != nil { + return fmt.Errorf("%w: manifest is not a JSON descriptor document", domain.ErrManifestBlobUnknown) } - if len(tags) > maxManifestTraversal { - return false, fmt.Errorf("%w: repository %s exceeds manifest traversal limit", domain.ErrBlobNotFound, name) + blobs := make([]manifestDescriptor, 0, 1+len(refs.Layers)+len(refs.Blobs)) + if refs.Config.Digest != "" { + blobs = append(blobs, refs.Config) } - queue := append([]string(nil), tags...) - seen := make(map[string]struct{}, len(queue)) - for len(queue) > 0 { - reference := queue[0] - queue = queue[1:] - if _, ok := seen[reference]; ok { + blobs = append(blobs, refs.Layers...) + blobs = append(blobs, refs.Blobs...) + for _, descriptor := range blobs { + if descriptor.Digest == "" { continue } - seen[reference] = struct{}{} - if len(seen) > maxManifestTraversal { - return false, fmt.Errorf("%w: repository %s exceeds manifest traversal limit", domain.ErrBlobNotFound, name) - } - - data, _, err := s.manifestStorage.GetManifest(name, reference) + owned, err := s.blobStorage.BlobOwnedByRepository(name, descriptor.Digest) if err != nil { - return false, fmt.Errorf("get manifest %s for repository %s: %w", reference, name, err) + return fmt.Errorf("verify blob ownership: %w", err) } - var refs manifestReferences - if err := json.Unmarshal(data, &refs); err != nil { - return false, fmt.Errorf("decode manifest %s: %w", reference, err) + if !owned { + return fmt.Errorf("%w: %s", domain.ErrManifestBlobUnknown, descriptor.Digest) } + } - if manifestReferencesTarget(refs, target) { - return true, nil - } - nestedManifests := refs.Manifests - if refs.Subject != nil && refs.Subject.Digest != "" { - nestedManifests = append(append([]manifestDescriptor(nil), refs.Manifests...), *refs.Subject) + children := refs.Manifests + if refs.Subject != nil && refs.Subject.Digest != "" { + children = append(append([]manifestDescriptor(nil), children...), *refs.Subject) + } + if len(children) > maxManifestTraversal { + return fmt.Errorf("%w: manifest references too many children", domain.ErrManifestBlobUnknown) + } + for _, descriptor := range children { + if descriptor.Digest == "" { + continue } - if len(seen)+len(queue)+len(nestedManifests) > maxManifestTraversal { - return false, fmt.Errorf("%w: repository %s exceeds manifest traversal limit", domain.ErrBlobNotFound, name) + if err := validation.ValidateDigest(descriptor.Digest); err != nil { + return fmt.Errorf("%w: %s", domain.ErrManifestBlobUnknown, err) } - queue = appendManifestDigests(queue, nestedManifests) - } - return false, nil -} - -func manifestReferencesTarget(refs manifestReferences, target string) bool { - return refs.Config.Digest == target || - descriptorListContains(refs.Layers, target) || - descriptorListContains(refs.Manifests, target) || - descriptorListContains(refs.Blobs, target) || - (refs.Subject != nil && refs.Subject.Digest == target) -} - -func appendManifestDigests(queue []string, descriptors []manifestDescriptor) []string { - for _, descriptor := range descriptors { - if descriptor.Digest != "" { - queue = append(queue, descriptor.Digest) + if _, _, err := s.manifestStorage.GetManifest(name, descriptor.Digest); err != nil { + return fmt.Errorf("%w: %s", domain.ErrManifestBlobUnknown, descriptor.Digest) } } - return queue + return nil } func manifestReferencedDigests(data []byte) []string { @@ -406,15 +304,6 @@ func manifestReferencedDigests(data []byte) []string { return digests } -func descriptorListContains(descriptors []manifestDescriptor, digest string) bool { - for _, descriptor := range descriptors { - if descriptor.Digest == digest { - return true - } - } - return false -} - func manifestDigestMatches(reference string, data []byte) (bool, error) { algorithm, _, ok := strings.Cut(reference, ":") if !ok { @@ -493,8 +382,9 @@ func (s *Service) AppendBlobChunk(ctx context.Context, name, uuid string, data i return length, nil } -// FinishUpload completes a blob upload. -func (s *Service) FinishUpload(ctx context.Context, uuid, digest string) error { +// FinishUpload completes a blob upload for the named repository and records +// the repository/blob association. +func (s *Service) FinishUpload(ctx context.Context, name, uuid, digest string) error { // Keep the transition from upload to blob storage atomic with respect to // registry garbage collection. PruneRegistry holds the exclusive lock, so // it cannot observe a finalized blob before it is marked pending. @@ -504,12 +394,13 @@ func (s *Service) FinishUpload(ctx context.Context, uuid, digest string) error { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ zerowrap.FieldLayer: "usecase", zerowrap.FieldUseCase: "FinishUpload", + "name": name, "uuid": uuid, "digest": digest, }) log := zerowrap.FromCtx(ctx) - if err := s.blobStorage.FinishBlobUpload(uuid, digest); err != nil { + if err := s.blobStorage.FinishBlobUpload(name, uuid, digest); err != nil { return log.WrapErr(err, "failed to finish blob upload") } s.registryState.AddPending(digest, time.Now().UTC()) @@ -519,15 +410,16 @@ func (s *Service) FinishUpload(ctx context.Context, uuid, digest string) error { } // CancelUpload cancels an in-progress upload. -func (s *Service) CancelUpload(ctx context.Context, uuid string) error { +func (s *Service) CancelUpload(ctx context.Context, name, uuid string) error { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ zerowrap.FieldLayer: "usecase", zerowrap.FieldUseCase: "CancelUpload", + "name": name, "uuid": uuid, }) log := zerowrap.FromCtx(ctx) - if err := s.blobStorage.CancelBlobUpload(uuid); err != nil { + if err := s.blobStorage.CancelBlobUpload(name, uuid); err != nil { return log.WrapErr(err, "failed to cancel blob upload") } diff --git a/internal/usecase/registry/service_test.go b/internal/usecase/registry/service_test.go index 063242cc5..8ea423ab4 100644 --- a/internal/usecase/registry/service_test.go +++ b/internal/usecase/registry/service_test.go @@ -88,6 +88,85 @@ func TestService_PutManifest_Success(t *testing.T) { assert.True(t, strings.HasPrefix(digest, "sha256:")) } +func TestService_PutManifest_RecordsPushMetrics(t *testing.T) { + manifestStorage := mocks.NewMockManifestStorage(t) + metrics := mocks.NewMockMetrics(t) + svc := NewService(mocks.NewMockBlobStorage(t), manifestStorage, nil) + svc.SetMetrics(metrics) + + data := []byte(`{"schemaVersion": 2}`) + manifestStorage.EXPECT().PutManifest("myapp", "latest", "application/vnd.oci.image.manifest.v1+json", data).Return(nil) + metrics.EXPECT().RecordImagePush(mock.Anything, "myapp", "latest", int64(len(data))).Return() + + _, err := svc.PutManifest(testContext(), &domain.Manifest{ + Name: "myapp", + Reference: "latest", + ContentType: "application/vnd.oci.image.manifest.v1+json", + Data: data, + }) + + require.NoError(t, err) +} + +func TestService_PutManifest_RejectsUnownedBlob(t *testing.T) { + const digest = "sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4" + blobStorage := mocks.NewMockBlobStorage(t) + manifestStorage := mocks.NewMockManifestStorage(t) + svc := NewService(blobStorage, manifestStorage, nil) + + blobStorage.EXPECT().BlobOwnedByRepository("attacker", digest).Return(false, nil) + + manifest := &domain.Manifest{ + Name: "attacker", + Reference: "latest", + ContentType: "application/vnd.oci.image.manifest.v1+json", + Data: []byte(`{"schemaVersion":2,"layers":[{"digest":"` + digest + `"}]}`), + } + + _, err := svc.PutManifest(testContext(), manifest) + + require.ErrorIs(t, err, domain.ErrManifestBlobUnknown) + manifestStorage.AssertNotCalled(t, "PutManifest", mock.Anything, mock.Anything, mock.Anything, mock.Anything) +} + +func TestService_PutManifest_RejectsUnknownChildManifest(t *testing.T) { + const child = "sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4" + blobStorage := mocks.NewMockBlobStorage(t) + manifestStorage := mocks.NewMockManifestStorage(t) + svc := NewService(blobStorage, manifestStorage, nil) + + manifestStorage.EXPECT().GetManifest("index", child).Return(nil, "", domain.ErrManifestNotFound) + + manifest := &domain.Manifest{ + Name: "index", + Reference: "latest", + ContentType: "application/vnd.oci.image.index.v1+json", + Data: []byte(`{"schemaVersion":2,"manifests":[{"digest":"` + child + `"}]}`), + } + + _, err := svc.PutManifest(testContext(), manifest) + + require.ErrorIs(t, err, domain.ErrManifestBlobUnknown) +} + +func TestService_PutManifest_RejectsMalformedChildDigest(t *testing.T) { + blobStorage := mocks.NewMockBlobStorage(t) + manifestStorage := mocks.NewMockManifestStorage(t) + svc := NewService(blobStorage, manifestStorage, nil) + + manifest := &domain.Manifest{ + Name: "index", + Reference: "latest", + ContentType: "application/vnd.oci.image.index.v1+json", + Data: []byte(`{"schemaVersion":2,"manifests":[{"digest":"../../tags.json"}]}`), + } + + _, err := svc.PutManifest(testContext(), manifest) + + require.ErrorIs(t, err, domain.ErrManifestBlobUnknown) + manifestStorage.AssertNotCalled(t, "GetManifest", mock.Anything, mock.Anything) +} + func TestService_PutManifest_SHA512DigestDoesNotPublishEvent(t *testing.T) { blobStorage := mocks.NewMockBlobStorage(t) manifestStorage := mocks.NewMockManifestStorage(t) @@ -100,6 +179,7 @@ func TestService_PutManifest_SHA512DigestDoesNotPublishEvent(t *testing.T) { manifest := &domain.Manifest{Name: "myapp", Reference: reference, ContentType: "application/vnd.oci.image.manifest.v1+json", Data: data} manifestStorage.EXPECT().PutManifest("myapp", reference, manifest.ContentType, data).Return(nil) + blobStorage.EXPECT().BlobOwnedByRepository("myapp", "sha256:config").Return(true, nil) digest, err := svc.PutManifest(testContext(), manifest) @@ -127,64 +207,56 @@ func TestService_PutManifest_RejectsDigestMismatch(t *testing.T) { assert.ErrorIs(t, err, domain.ErrDigestMismatch) } -func TestService_GetBlobPath_RequiresRepositoryReference(t *testing.T) { +// TestService_GetBlobPath_RequiresRepositoryOwnership proves a blob is +// served only to a repository that completed an upload of it. +func TestService_GetBlobPath_RequiresRepositoryOwnership(t *testing.T) { const digest = "sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4" blobStorage := mocks.NewMockBlobStorage(t) manifestStorage := mocks.NewMockManifestStorage(t) svc := NewService(blobStorage, manifestStorage, nil) - manifestStorage.EXPECT().ListTags("allowed").Return([]string{"latest"}, nil) - manifestStorage.EXPECT().GetManifest("allowed", "latest").Return( - []byte(`{"schemaVersion":2,"layers":[{"digest":"`+digest+`"}]}`), - "application/vnd.oci.image.manifest.v1+json", nil, - ) + blobStorage.EXPECT().BlobOwnedByRepository("allowed", digest).Return(true, nil) blobStorage.EXPECT().GetBlobPath(digest).Return("/registry/blob", nil) path, err := svc.GetBlobPath(testContext(), "allowed", digest) require.NoError(t, err) assert.Equal(t, "/registry/blob", path) - manifestStorage.EXPECT().ListTags("denied").Return([]string{"latest"}, nil) - manifestStorage.EXPECT().GetManifest("denied", "latest").Return( - []byte(`{"schemaVersion":2,"layers":[]}`), - "application/vnd.oci.image.manifest.v1+json", nil, - ) + blobStorage.EXPECT().BlobOwnedByRepository("denied", digest).Return(false, nil) _, err = svc.GetBlobPath(testContext(), "denied", digest) assert.ErrorIs(t, err, domain.ErrBlobNotFound) } -func TestService_GetBlobPath_FollowsSubjectManifest(t *testing.T) { - const target = "sha256:target" +// TestService_GetBlobPath_IgnoresManifestReferences proves a manifest that +// merely names a foreign digest does not confer access to it: ownership is +// never inferred from repository-controlled manifest content. +func TestService_GetBlobPath_IgnoresManifestReferences(t *testing.T) { + const target = "sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4" blobStorage := mocks.NewMockBlobStorage(t) manifestStorage := mocks.NewMockManifestStorage(t) svc := NewService(blobStorage, manifestStorage, nil) - manifestStorage.EXPECT().ListTags("artifacts").Return([]string{"latest"}, nil) - manifestStorage.EXPECT().GetManifest("artifacts", "latest").Return( - []byte(`{"subject":{"digest":"sha256:subject"}}`), "application/vnd.oci.artifact.manifest.v1+json", nil, - ) - manifestStorage.EXPECT().GetManifest("artifacts", "sha256:subject").Return( - []byte(`{"config":{"digest":"`+target+`"}}`), "application/vnd.oci.image.manifest.v1+json", nil, - ) - blobStorage.EXPECT().GetBlobPath(target).Return("/registry/target", nil) + blobStorage.EXPECT().BlobOwnedByRepository("attacker", target).Return(false, nil) - path, err := svc.GetBlobPath(testContext(), "artifacts", target) + _, err := svc.GetBlobPath(testContext(), "attacker", target) - require.NoError(t, err) - assert.Equal(t, "/registry/target", path) + assert.ErrorIs(t, err, domain.ErrBlobNotFound) + blobStorage.AssertNotCalled(t, "GetBlobPath", target) + manifestStorage.AssertNotCalled(t, "GetManifest", mock.Anything, mock.Anything) } -func TestService_GetBlobPath_BoundsManifestTraversal(t *testing.T) { +func TestService_GetBlobPath_OwnershipLookupError(t *testing.T) { + const target = "sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4" blobStorage := mocks.NewMockBlobStorage(t) manifestStorage := mocks.NewMockManifestStorage(t) svc := NewService(blobStorage, manifestStorage, nil) - tags := make([]string, maxManifestTraversal+1) - manifestStorage.EXPECT().ListTags("busy").Return(tags, nil) - _, err := svc.GetBlobPath(testContext(), "busy", "sha256:target") + blobStorage.EXPECT().BlobOwnedByRepository("busy", target).Return(false, errors.New("ownership lookup failed")) - assert.ErrorIs(t, err, domain.ErrBlobNotFound) + _, err := svc.GetBlobPath(testContext(), "busy", target) + + assert.Error(t, err) } func TestService_PutManifest_StorageError(t *testing.T) { @@ -449,9 +521,9 @@ func TestService_FinishUpload_Success(t *testing.T) { svc := NewService(blobStorage, manifestStorage, eventBus) ctx := testContext() - blobStorage.EXPECT().FinishBlobUpload("1234567890-myapp", "sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4").Return(nil) + blobStorage.EXPECT().FinishBlobUpload("myapp", "1234567890-myapp", "sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4").Return(nil) - err := svc.FinishUpload(ctx, "1234567890-myapp", "sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4") + err := svc.FinishUpload(ctx, "myapp", "1234567890-myapp", "sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4") assert.NoError(t, err) } @@ -464,9 +536,9 @@ func TestService_FinishUploadMarksBlobPending(t *testing.T) { svc := NewService(blobStorage, manifestStorage, eventBus, state) const digest = "sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4" - blobStorage.EXPECT().FinishBlobUpload("1234567890-myapp", digest).Return(nil) + blobStorage.EXPECT().FinishBlobUpload("myapp", "1234567890-myapp", digest).Return(nil) - require.NoError(t, svc.FinishUpload(testContext(), "1234567890-myapp", digest)) + require.NoError(t, svc.FinishUpload(testContext(), "myapp", "1234567890-myapp", digest)) assert.Contains(t, state.PendingDigests(time.Now().UTC()), digest) } @@ -479,7 +551,7 @@ func TestService_FinishUploadBlocksRegistryPruning(t *testing.T) { started := make(chan struct{}) release := make(chan struct{}) - blobStorage.EXPECT().FinishBlobUpload("1234567890-myapp", mock.Anything).RunAndReturn(func(string, string) error { + blobStorage.EXPECT().FinishBlobUpload("myapp", "1234567890-myapp", mock.Anything).RunAndReturn(func(string, string, string) error { close(started) <-release return nil @@ -487,7 +559,7 @@ func TestService_FinishUploadBlocksRegistryPruning(t *testing.T) { finished := make(chan error, 1) go func() { - finished <- svc.FinishUpload(testContext(), "1234567890-myapp", "sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4") + finished <- svc.FinishUpload(testContext(), "myapp", "1234567890-myapp", "sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4") }() <-started @@ -526,9 +598,9 @@ func TestService_FinishUpload_Error(t *testing.T) { svc := NewService(blobStorage, manifestStorage, eventBus) ctx := testContext() - blobStorage.EXPECT().FinishBlobUpload("1234567890-myapp", "sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4").Return(errors.New("digest mismatch")) + blobStorage.EXPECT().FinishBlobUpload("myapp", "1234567890-myapp", "sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4").Return(errors.New("digest mismatch")) - err := svc.FinishUpload(ctx, "1234567890-myapp", "sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4") + err := svc.FinishUpload(ctx, "myapp", "1234567890-myapp", "sha256:a3ed95caeb02ffe68cdd9fd84406680ae93d633cb16422d00e8a7c22955b46d4") assert.Error(t, err) assert.Contains(t, err.Error(), "failed to finish blob upload") @@ -542,9 +614,9 @@ func TestService_CancelUpload_Success(t *testing.T) { svc := NewService(blobStorage, manifestStorage, eventBus) ctx := testContext() - blobStorage.EXPECT().CancelBlobUpload("1234567890-myapp").Return(nil) + blobStorage.EXPECT().CancelBlobUpload("myapp", "1234567890-myapp").Return(nil) - err := svc.CancelUpload(ctx, "1234567890-myapp") + err := svc.CancelUpload(ctx, "myapp", "1234567890-myapp") assert.NoError(t, err) } @@ -557,9 +629,9 @@ func TestService_CancelUpload_Error(t *testing.T) { svc := NewService(blobStorage, manifestStorage, eventBus) ctx := testContext() - blobStorage.EXPECT().CancelBlobUpload("1234567890-myapp").Return(errors.New("upload not found")) + blobStorage.EXPECT().CancelBlobUpload("myapp", "1234567890-myapp").Return(errors.New("upload not found")) - err := svc.CancelUpload(ctx, "1234567890-myapp") + err := svc.CancelUpload(ctx, "myapp", "1234567890-myapp") assert.Error(t, err) assert.Contains(t, err.Error(), "failed to cancel blob upload") @@ -636,16 +708,15 @@ func TestService_ListRepositories_Error(t *testing.T) { assert.Contains(t, err.Error(), "failed to list repositories") } -func TestPutManifest_SkipsEventWhenDeployIntentSuppressed(t *testing.T) { +func TestPutManifest_PublishesEventForTagReference(t *testing.T) { blobStorage := mocks.NewMockBlobStorage(t) manifestStorage := mocks.NewMockManifestStorage(t) eventBus := mocks.NewMockEventPublisher(t) svc := NewService(blobStorage, manifestStorage, eventBus) - svc.SuppressDeployEvent("my-app") - manifestStorage.EXPECT().PutManifest("my-app", "latest", "application/vnd.oci.image.manifest.v1+json", mock.Anything).Return(nil) + eventBus.EXPECT().Publish(domain.EventImagePushed, mock.Anything).Return(nil).Once() manifest := &domain.Manifest{ Name: "my-app", @@ -656,20 +727,4 @@ func TestPutManifest_SkipsEventWhenDeployIntentSuppressed(t *testing.T) { _, err := svc.PutManifest(context.Background(), manifest) require.NoError(t, err) - - eventBus.AssertNotCalled(t, "Publish", mock.Anything, mock.Anything) -} - -func TestSuppressDeployEvent_ClearsCorrectly(t *testing.T) { - blobStorage := mocks.NewMockBlobStorage(t) - manifestStorage := mocks.NewMockManifestStorage(t) - eventBus := mocks.NewMockEventPublisher(t) - - svc := NewService(blobStorage, manifestStorage, eventBus) - - svc.SuppressDeployEvent("my-app") - assert.True(t, svc.IsDeployEventSuppressed("my-app")) - - svc.ClearDeployEventSuppression("my-app") - assert.False(t, svc.IsDeployEventSuppressed("my-app")) } diff --git a/internal/usecase/secrets/service.go b/internal/usecase/secrets/service.go deleted file mode 100644 index 1ba448e37..000000000 --- a/internal/usecase/secrets/service.go +++ /dev/null @@ -1,348 +0,0 @@ -// Package secrets implements the secret management use case. -package secrets - -import ( - "context" - "errors" - "sort" - "strings" - - "github.com/bnema/zerowrap" - - "github.com/bnema/gordon/internal/boundaries/out" - "github.com/bnema/gordon/internal/domain" -) - -// Domain validation errors. -var ( - ErrDomainEmpty = errors.New("domain cannot be empty") - ErrDomainTooLong = errors.New("domain exceeds maximum length of 253 characters") - ErrDomainPathTraversal = errors.New("domain contains path traversal sequence") - ErrDomainInvalidChars = errors.New("domain contains invalid characters") - ErrServiceEmpty = errors.New("service name cannot be empty") - ErrInvalidServiceName = errors.New("invalid service name: must start with a letter, contain only lowercase letters, numbers, and hyphens, be at most 63 characters, and not end with a hyphen") -) - -// Service implements the SecretService interface. -type Service struct { - store out.DomainSecretStore - log zerowrap.Logger - eventBus out.EventPublisher -} - -// NewService creates a new secrets service. -func NewService(store out.DomainSecretStore, log zerowrap.Logger, eventBus out.EventPublisher) *Service { - return &Service{ - store: store, - log: log, - eventBus: eventBus, - } -} - -// publishSecretsChanged publishes a secrets.changed event. -// Failures are logged but do not fail the mutation - secret storage -// is the source of truth; the event is a best-effort notification. -func (s *Service) publishSecretsChanged(ctx context.Context, domainName, operation string, keys []string) { - if s.eventBus == nil { - return - } - log := zerowrap.FromCtx(ctx) - payload := domain.SecretsChangedPayload{ - Domain: domainName, - Operation: operation, - Keys: keys, - } - if err := s.eventBus.Publish(domain.EventSecretsChanged, payload); err != nil { - log.Warn().Err(err).Str("domain", domainName).Msg("failed to publish secrets.changed event") - } -} - -// sortedKeys returns the keys of a map in sorted order. -func sortedKeys(m map[string]string) []string { - keys := make([]string, 0, len(m)) - for k := range m { - keys = append(keys, k) - } - sort.Strings(keys) - return keys -} - -// ListKeys returns the list of secret keys for a domain (not values). -func (s *Service) ListKeys(ctx context.Context, domain string) ([]string, error) { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "ListKeys", - "domain": domain, - }) - log := zerowrap.FromCtx(ctx) - - if err := ValidateDomain(domain); err != nil { - log.Warn().Err(err).Msg("domain validation failed") - return nil, err - } - - keys, err := s.store.ListKeys(domain) - if err != nil { - log.Error().Err(err).Msg("failed to list secret keys") - return nil, err - } - - log.Debug().Int("count", len(keys)).Msg("listed secret keys") - return keys, nil -} - -// ListKeysWithAttachments returns the list of secret keys for a domain -// along with any attachment secrets for containers associated with the domain. -func (s *Service) ListKeysWithAttachments(ctx context.Context, domain string) ([]string, []out.AttachmentSecrets, error) { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "ListKeysWithAttachments", - "domain": domain, - }) - log := zerowrap.FromCtx(ctx) - - if err := ValidateDomain(domain); err != nil { - log.Warn().Err(err).Msg("domain validation failed") - return nil, nil, err - } - - // Get domain secrets - keys, err := s.store.ListKeys(domain) - if err != nil { - log.Error().Err(err).Msg("failed to list secret keys") - return nil, nil, err - } - - // Get attachment secrets - attachments, err := s.store.ListAttachmentKeys(domain) - if err != nil { - log.Warn().Err(err).Msg("failed to list attachment secrets, continuing without them") - attachments = nil // Don't fail, just return empty attachments - } - - log.Debug(). - Int("domain_keys", len(keys)). - Int("attachments", len(attachments)). - Msg("listed secret keys with attachments") - - return keys, attachments, nil -} - -// GetAll returns all secrets for a domain as a key-value map. -func (s *Service) GetAll(ctx context.Context, domain string) (map[string]string, error) { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "GetAll", - "domain": domain, - }) - log := zerowrap.FromCtx(ctx) - - if err := ValidateDomain(domain); err != nil { - log.Warn().Err(err).Msg("domain validation failed") - return nil, err - } - - secrets, err := s.store.GetAll(domain) - if err != nil { - log.Error().Err(err).Msg("failed to get secrets") - return nil, err - } - - log.Debug().Int("count", len(secrets)).Msg("retrieved secrets") - return secrets, nil -} - -// Set sets or updates multiple secrets for a domain, merging with existing. -func (s *Service) Set(ctx context.Context, domain string, secrets map[string]string) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "Set", - "domain": domain, - }) - log := zerowrap.FromCtx(ctx) - - if err := ValidateDomain(domain); err != nil { - log.Warn().Err(err).Msg("domain validation failed") - return err - } - - if err := s.store.Set(domain, secrets); err != nil { - log.Error().Err(err).Msg("failed to set secrets") - return err - } - - s.publishSecretsChanged(ctx, domain, "set", sortedKeys(secrets)) - - log.Info().Int("count", len(secrets)).Msg("secrets set") - return nil -} - -// Delete removes a specific secret key from a domain. -func (s *Service) Delete(ctx context.Context, domain, key string) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "Delete", - "domain": domain, - "key": key, - }) - log := zerowrap.FromCtx(ctx) - - if err := ValidateDomain(domain); err != nil { - log.Warn().Err(err).Msg("domain validation failed") - return err - } - - if err := s.store.Delete(domain, key); err != nil { - log.Error().Err(err).Msg("failed to delete secret") - return err - } - - s.publishSecretsChanged(ctx, domain, "delete", []string{key}) - - log.Info().Msg("secret deleted") - return nil -} - -// SetAttachment sets or updates multiple secrets for an attachment container. -func (s *Service) SetAttachment(ctx context.Context, domainName, service string, secrets map[string]string) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "SetAttachment", - "domain": domainName, - "service": service, - }) - log := zerowrap.FromCtx(ctx) - - if err := ValidateDomain(domainName); err != nil { - log.Warn().Err(err).Msg("domain validation failed") - return err - } - - if err := validateServiceName(service); err != nil { - log.Warn().Err(err).Msg("service name validation failed") - return err - } - - containerName := resolveContainerName(domainName, service) - - if err := s.store.SetAttachment(containerName, secrets); err != nil { - log.Error().Err(err).Msg("failed to set attachment secrets") - return err - } - - s.publishSecretsChanged(ctx, domainName, "set", sortedKeys(secrets)) - - log.Info().Int("count", len(secrets)).Str("container", containerName).Msg("attachment secrets set") - return nil -} - -// DeleteAttachment removes a specific secret key from an attachment container. -func (s *Service) DeleteAttachment(ctx context.Context, domainName, service, key string) error { - ctx = zerowrap.CtxWithFields(ctx, map[string]any{ - zerowrap.FieldLayer: "usecase", - zerowrap.FieldUseCase: "DeleteAttachment", - "domain": domainName, - "service": service, - "key": key, - }) - log := zerowrap.FromCtx(ctx) - - if err := ValidateDomain(domainName); err != nil { - log.Warn().Err(err).Msg("domain validation failed") - return err - } - - if err := validateServiceName(service); err != nil { - log.Warn().Err(err).Msg("service name validation failed") - return err - } - - containerName := resolveContainerName(domainName, service) - - if err := s.store.DeleteAttachment(containerName, key); err != nil { - log.Error().Err(err).Msg("failed to delete attachment secret") - return err - } - - s.publishSecretsChanged(ctx, domainName, "delete", []string{key}) - - log.Info().Str("container", containerName).Msg("attachment secret deleted") - return nil -} - -// validateServiceName checks that a service name is safe and follows Docker-style conventions. -func validateServiceName(service string) error { - if service == "" { - return ErrServiceEmpty - } - // DNS labels are limited to 63 characters (RFC 1035) - if len(service) > 63 { - return ErrInvalidServiceName - } - // Docker-style: lowercase letter start, then lowercase alphanumeric + hyphens - for i, c := range service { - if i == 0 { - if c < 'a' || c > 'z' { - return ErrInvalidServiceName - } - continue - } - if (c < 'a' || c > 'z') && (c < '0' || c > '9') && c != '-' { - return ErrInvalidServiceName - } - } - // Trailing hyphen is not allowed in DNS labels - if service[len(service)-1] == '-' { - return ErrInvalidServiceName - } - return nil -} - -// resolveContainerName builds the container name from a domain and service. -// Ensures the total name does not exceed Docker's 255-character limit. -func resolveContainerName(domainName, service string) string { - sanitized := domain.SanitizeDomainForContainer(domainName) - name := "gordon-" + sanitized + "-" + service - - // Docker container names are limited to 255 characters - if len(name) > 255 { - // Truncate the domain part to fit, keeping room for "gordon-", "-", and service name - maxDomainLen := 255 - 7 - 1 - len(service) // 7 for "gordon-", 1 for "-" - if maxDomainLen > 0 { - name = "gordon-" + sanitized[:maxDomainLen] + "-" + service - } - } - return name -} - -// ValidateDomain validates that a domain is safe to use for secret storage. -// This prevents path traversal attacks and ensures the domain is valid. -func ValidateDomain(domainName string) error { - // Check for empty domain - if domainName == "" { - return ErrDomainEmpty - } - - // Check domain length (max DNS name length is 253) - if len(domainName) > 253 { - return ErrDomainTooLong - } - - // Check for path traversal attempts - if strings.Contains(domainName, "..") { - return ErrDomainPathTraversal - } - - // Check for null bytes (can cause issues in file paths) - if strings.ContainsRune(domainName, '\x00') { - return ErrDomainInvalidChars - } - - // Reuse the env-file sanitizer so validation matches the on-disk naming - // scheme and rejects ambiguous names that would collide after sanitization. - if _, err := domain.SanitizeDomainForEnvFile(domainName); err != nil { - return ErrDomainInvalidChars - } - - return nil -} diff --git a/internal/usecase/secrets/service_test.go b/internal/usecase/secrets/service_test.go deleted file mode 100644 index aceeb5a44..000000000 --- a/internal/usecase/secrets/service_test.go +++ /dev/null @@ -1,662 +0,0 @@ -package secrets - -import ( - "context" - "strings" - "testing" - - "github.com/bnema/zerowrap" - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/mock" - - "github.com/bnema/gordon/internal/boundaries/out" - outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" - "github.com/bnema/gordon/internal/domain" -) - -func testLogger() zerowrap.Logger { - return zerowrap.Default() -} - -func TestValidateDomain(t *testing.T) { - tests := []struct { - name string - domain string - wantErr error - }{ - // Valid domains - { - name: "valid simple domain", - domain: "example.com", - wantErr: nil, - }, - { - name: "valid subdomain", - domain: "app.example.com", - wantErr: nil, - }, - { - name: "valid domain with port", - domain: "app.example.com:8080", - wantErr: nil, - }, - { - name: "valid domain with hyphens", - domain: "my-app.example-site.com", - wantErr: nil, - }, - { - name: "valid single label", - domain: "localhost", - wantErr: nil, - }, - // Empty domain - { - name: "empty domain", - domain: "", - wantErr: ErrDomainEmpty, - }, - // Path traversal attempts - { - name: "path traversal with double dots", - domain: "../etc/passwd", - wantErr: ErrDomainPathTraversal, - }, - { - name: "path traversal in middle", - domain: "app/../etc/passwd", - wantErr: ErrDomainPathTraversal, - }, - { - name: "path traversal at end", - domain: "app.example.com/..", - wantErr: ErrDomainPathTraversal, - }, - { - name: "multiple path traversal", - domain: "..../....//etc/passwd", - wantErr: ErrDomainPathTraversal, - }, - { - name: "encoded path traversal still blocked", - domain: "app..example.com", - wantErr: ErrDomainPathTraversal, - }, - // Domain too long - { - name: "domain at max length (253)", - domain: strings.Repeat("a", 253), - wantErr: nil, - }, - { - name: "domain exceeds max length", - domain: strings.Repeat("a", 254), - wantErr: ErrDomainTooLong, - }, - { - name: "domain way too long", - domain: strings.Repeat("a", 1000), - wantErr: ErrDomainTooLong, - }, - // Invalid characters - { - name: "null byte in domain", - domain: "app\x00.example.com", - wantErr: ErrDomainInvalidChars, - }, - { - name: "null byte at start", - domain: "\x00example.com", - wantErr: ErrDomainInvalidChars, - }, - { - name: "underscore rejected to avoid env filename collision", - domain: "app_example.com", - wantErr: ErrDomainInvalidChars, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - err := ValidateDomain(tt.domain) - if tt.wantErr != nil { - assert.ErrorIs(t, err, tt.wantErr) - } else { - assert.NoError(t, err) - } - }) - } -} - -func TestValidateDomain_DistinguishesSeparatorsForEnvStorage(t *testing.T) { - for _, domain := range []string{"app.example.com", "app:example:com", "app/example/com"} { - assert.NoError(t, ValidateDomain(domain), "expected %q to remain valid", domain) - } -} - -func TestService_ListKeys_ValidationErrors(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - tests := []struct { - name string - domain string - wantErr error - }{ - { - name: "empty domain", - domain: "", - wantErr: ErrDomainEmpty, - }, - { - name: "path traversal", - domain: "../etc/passwd", - wantErr: ErrDomainPathTraversal, - }, - { - name: "domain too long", - domain: strings.Repeat("a", 300), - wantErr: ErrDomainTooLong, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - keys, err := svc.ListKeys(context.Background(), tt.domain) - assert.Nil(t, keys) - assert.ErrorIs(t, err, tt.wantErr) - }) - } -} - -func TestService_ListKeys_Success(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - expectedKeys := []string{"API_KEY", "DB_PASSWORD"} - store.EXPECT().ListKeys("app.example.com").Return(expectedKeys, nil) - - keys, err := svc.ListKeys(context.Background(), "app.example.com") - assert.NoError(t, err) - assert.Equal(t, expectedKeys, keys) -} - -func TestService_ListKeysWithAttachments_ValidationErrors(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - tests := []struct { - name string - domain string - wantErr error - }{ - { - name: "empty domain", - domain: "", - wantErr: ErrDomainEmpty, - }, - { - name: "path traversal", - domain: "../etc/passwd", - wantErr: ErrDomainPathTraversal, - }, - { - name: "domain too long", - domain: strings.Repeat("a", 300), - wantErr: ErrDomainTooLong, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - keys, attachments, err := svc.ListKeysWithAttachments(context.Background(), tt.domain) - assert.Nil(t, keys) - assert.Nil(t, attachments) - assert.ErrorIs(t, err, tt.wantErr) - }) - } -} - -func TestService_ListKeysWithAttachments_Success(t *testing.T) { - t.Run("returns both domain and attachment secrets", func(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - expectedKeys := []string{"API_KEY", "DB_PASSWORD"} - expectedAttachments := []out.AttachmentSecrets{ - {Service: "postgres", Keys: []string{"POSTGRES_PASSWORD"}}, - {Service: "redis", Keys: []string{"REDIS_PASSWORD", "REDIS_USER"}}, - } - - store.EXPECT().ListKeys("app.example.com").Return(expectedKeys, nil) - store.EXPECT().ListAttachmentKeys("app.example.com").Return(expectedAttachments, nil) - - keys, attachments, err := svc.ListKeysWithAttachments(context.Background(), "app.example.com") - assert.NoError(t, err) - assert.Equal(t, expectedKeys, keys) - assert.Equal(t, expectedAttachments, attachments) - }) - - t.Run("returns empty attachments when none exist", func(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - expectedKeys := []string{"API_KEY"} - - store.EXPECT().ListKeys("app.example.com").Return(expectedKeys, nil) - store.EXPECT().ListAttachmentKeys("app.example.com").Return(nil, nil) - - keys, attachments, err := svc.ListKeysWithAttachments(context.Background(), "app.example.com") - assert.NoError(t, err) - assert.Equal(t, expectedKeys, keys) - assert.Nil(t, attachments) - }) - - t.Run("continues without attachments on attachment error", func(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - expectedKeys := []string{"API_KEY"} - attachmentErr := assert.AnError - - store.EXPECT().ListKeys("app.example.com").Return(expectedKeys, nil) - store.EXPECT().ListAttachmentKeys("app.example.com").Return(nil, attachmentErr) - - // Should still succeed, just without attachments - keys, attachments, err := svc.ListKeysWithAttachments(context.Background(), "app.example.com") - assert.NoError(t, err) - assert.Equal(t, expectedKeys, keys) - assert.Nil(t, attachments) - }) -} - -func TestService_ListKeysWithAttachments_StoreError(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - storeErr := assert.AnError - store.EXPECT().ListKeys("app.example.com").Return(nil, storeErr) - - keys, attachments, err := svc.ListKeysWithAttachments(context.Background(), "app.example.com") - assert.Nil(t, keys) - assert.Nil(t, attachments) - assert.ErrorIs(t, err, storeErr) -} - -func TestService_GetAll_ValidationErrors(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - tests := []struct { - name string - domain string - wantErr error - }{ - { - name: "empty domain", - domain: "", - wantErr: ErrDomainEmpty, - }, - { - name: "path traversal", - domain: "app/../secret", - wantErr: ErrDomainPathTraversal, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - secrets, err := svc.GetAll(context.Background(), tt.domain) - assert.Nil(t, secrets) - assert.ErrorIs(t, err, tt.wantErr) - }) - } -} - -func TestService_GetAll_Success(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - expectedSecrets := map[string]string{"API_KEY": "secret123", "DB_PASSWORD": "pass456"} - store.EXPECT().GetAll("app.example.com").Return(expectedSecrets, nil) - - secrets, err := svc.GetAll(context.Background(), "app.example.com") - assert.NoError(t, err) - assert.Equal(t, expectedSecrets, secrets) -} - -func TestService_Set_ValidationErrors(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - secrets := map[string]string{"API_KEY": "secret123"} - - tests := []struct { - name string - domain string - wantErr error - }{ - { - name: "empty domain", - domain: "", - wantErr: ErrDomainEmpty, - }, - { - name: "path traversal", - domain: "../../../etc/passwd", - wantErr: ErrDomainPathTraversal, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - err := svc.Set(context.Background(), tt.domain, secrets) - assert.ErrorIs(t, err, tt.wantErr) - }) - } -} - -func TestService_Set_Success(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - eventBus := outmocks.NewMockEventPublisher(t) - svc := NewService(store, testLogger(), eventBus) - - secrets := map[string]string{"API_KEY": "secret123"} - store.EXPECT().Set("app.example.com", secrets).Return(nil) - eventBus.EXPECT().Publish(domain.EventSecretsChanged, mock.AnythingOfType("domain.SecretsChangedPayload")).Return(nil) - - err := svc.Set(context.Background(), "app.example.com", secrets) - assert.NoError(t, err) -} - -func TestService_Set_PublishesSecretsChangedEvent(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - eventBus := outmocks.NewMockEventPublisher(t) - svc := NewService(store, testLogger(), eventBus) - - secrets := map[string]string{"API_KEY": "secret123", "DB_PASSWORD": "pass456"} - store.EXPECT().Set("app.example.com", secrets).Return(nil) - eventBus.EXPECT().Publish(domain.EventSecretsChanged, mock.MatchedBy(func(payload domain.SecretsChangedPayload) bool { - assert.Equal(t, "app.example.com", payload.Domain) - assert.Equal(t, "set", payload.Operation) - assert.ElementsMatch(t, []string{"API_KEY", "DB_PASSWORD"}, payload.Keys) - return true - })).Return(nil) - - err := svc.Set(context.Background(), "app.example.com", secrets) - assert.NoError(t, err) -} - -func TestService_Delete_ValidationErrors(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - tests := []struct { - name string - domain string - wantErr error - }{ - { - name: "empty domain", - domain: "", - wantErr: ErrDomainEmpty, - }, - { - name: "path traversal", - domain: "..%2F..%2Fetc%2Fpasswd", - wantErr: ErrDomainPathTraversal, - }, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - err := svc.Delete(context.Background(), tt.domain, "API_KEY") - assert.ErrorIs(t, err, tt.wantErr) - }) - } -} - -func TestService_Delete_Success(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - eventBus := outmocks.NewMockEventPublisher(t) - svc := NewService(store, testLogger(), eventBus) - - store.EXPECT().Delete("app.example.com", "API_KEY").Return(nil) - eventBus.EXPECT().Publish(domain.EventSecretsChanged, mock.AnythingOfType("domain.SecretsChangedPayload")).Return(nil) - - err := svc.Delete(context.Background(), "app.example.com", "API_KEY") - assert.NoError(t, err) -} - -func TestService_Delete_PublishesSecretsChangedEvent(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - eventBus := outmocks.NewMockEventPublisher(t) - svc := NewService(store, testLogger(), eventBus) - - store.EXPECT().Delete("app.example.com", "API_KEY").Return(nil) - eventBus.EXPECT().Publish(domain.EventSecretsChanged, mock.MatchedBy(func(payload domain.SecretsChangedPayload) bool { - assert.Equal(t, "app.example.com", payload.Domain) - assert.Equal(t, "delete", payload.Operation) - assert.Equal(t, []string{"API_KEY"}, payload.Keys) - return true - })).Return(nil) - - err := svc.Delete(context.Background(), "app.example.com", "API_KEY") - assert.NoError(t, err) -} - -func TestService_Set_SucceedsEvenIfEventPublishFails(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - eventBus := outmocks.NewMockEventPublisher(t) - svc := NewService(store, testLogger(), eventBus) - - secrets := map[string]string{"API_KEY": "secret123"} - store.EXPECT().Set("app.example.com", secrets).Return(nil) - eventBus.EXPECT().Publish(domain.EventSecretsChanged, mock.AnythingOfType("domain.SecretsChangedPayload")).Return(assert.AnError) - - err := svc.Set(context.Background(), "app.example.com", secrets) - assert.NoError(t, err) -} - -func TestService_NilEventPublisher_DoesNotPanic(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - secrets := map[string]string{"API_KEY": "secret123"} - store.EXPECT().Set("app.example.com", secrets).Return(nil) - - assert.NotPanics(t, func() { - err := svc.Set(context.Background(), "app.example.com", secrets) - assert.NoError(t, err) - }) -} - -func TestService_StoreError_PropagatedCorrectly(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - storeErr := assert.AnError - store.EXPECT().ListKeys("app.example.com").Return(nil, storeErr) - - keys, err := svc.ListKeys(context.Background(), "app.example.com") - assert.Nil(t, keys) - assert.ErrorIs(t, err, storeErr) -} - -// TestPathTraversalVariants tests various path traversal attack patterns. -func TestPathTraversalVariants(t *testing.T) { - tests := []struct { - name string - domain string - }{ - {"simple parent dir", ".."}, - {"parent dir prefix", "../secret"}, - {"parent dir suffix", "secret/.."}, - {"parent dir in middle", "app/../secret"}, - {"multiple parent dirs", "../../etc/passwd"}, - {"windows style", "..\\..\\etc\\passwd"}, - {"mixed slashes with dots", "app/./../../etc"}, - {"double dots only", "...."}, - {"triple dots", "..."}, - {"dot dot in subdomain style", "app..evil.com"}, - } - - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - err := ValidateDomain(tt.domain) - // All these should be rejected (either path traversal or other validation) - // The main goal is that ".." patterns are caught - if strings.Contains(tt.domain, "..") { - assert.ErrorIs(t, err, ErrDomainPathTraversal, "domain %q should be rejected for path traversal", tt.domain) - } - }) - } -} - -// TestService_ValidationHappensBeforeStore ensures validation runs before any store calls. -func TestService_ValidationHappensBeforeStore(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - // With path traversal domain, store methods should never be called - // We don't set up any expectations, so if store is called, the test will fail - - _, err := svc.ListKeys(context.Background(), "../etc/passwd") - assert.ErrorIs(t, err, ErrDomainPathTraversal) - - _, err = svc.GetAll(context.Background(), "../etc/passwd") - assert.ErrorIs(t, err, ErrDomainPathTraversal) - - err = svc.Set(context.Background(), "../etc/passwd", map[string]string{"key": "value"}) - assert.ErrorIs(t, err, ErrDomainPathTraversal) - - err = svc.Delete(context.Background(), "../etc/passwd", "key") - assert.ErrorIs(t, err, ErrDomainPathTraversal) - - // Verify no store methods were called (mockery would fail if unexpected calls happened) - mock.AssertExpectationsForObjects(t, store) -} - -func TestService_SetAttachment(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - eventBus := outmocks.NewMockEventPublisher(t) - svc := NewService(store, testLogger(), eventBus) - - secrets := map[string]string{"POSTGRES_PASSWORD": "secret123"} - store.EXPECT().SetAttachment("gordon-app__example__com-postgres", secrets).Return(nil) - eventBus.EXPECT().Publish(domain.EventSecretsChanged, mock.AnythingOfType("domain.SecretsChangedPayload")).Return(nil) - - err := svc.SetAttachment(context.Background(), "app.example.com", "postgres", secrets) - assert.NoError(t, err) -} - -func TestService_DeleteAttachment(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - eventBus := outmocks.NewMockEventPublisher(t) - svc := NewService(store, testLogger(), eventBus) - - store.EXPECT().DeleteAttachment("gordon-app__example__com-postgres", "OLD_KEY").Return(nil) - eventBus.EXPECT().Publish(domain.EventSecretsChanged, mock.AnythingOfType("domain.SecretsChangedPayload")).Return(nil) - - err := svc.DeleteAttachment(context.Background(), "app.example.com", "postgres", "OLD_KEY") - assert.NoError(t, err) -} - -func TestService_SetAttachment_InvalidDomain(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - err := svc.SetAttachment(context.Background(), "", "postgres", map[string]string{"KEY": "val"}) - assert.ErrorIs(t, err, ErrDomainEmpty) -} - -func TestService_SetAttachment_EmptyService(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - err := svc.SetAttachment(context.Background(), "app.example.com", "", map[string]string{"KEY": "val"}) - assert.ErrorIs(t, err, ErrServiceEmpty) -} - -func TestService_DeleteAttachment_InvalidDomain(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - err := svc.DeleteAttachment(context.Background(), "", "postgres", "SOME_KEY") - assert.ErrorIs(t, err, ErrDomainEmpty) -} - -func TestService_DeleteAttachment_EmptyService(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - err := svc.DeleteAttachment(context.Background(), "app.example.com", "", "SOME_KEY") - assert.ErrorIs(t, err, ErrServiceEmpty) -} - -func TestService_SetAttachment_InvalidServiceName(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - err := svc.SetAttachment(context.Background(), "app.example.com", "../evil", map[string]string{"K": "V"}) - assert.ErrorIs(t, err, ErrInvalidServiceName) -} - -func TestService_DeleteAttachment_InvalidServiceName(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - err := svc.DeleteAttachment(context.Background(), "app.example.com", "foo/bar", "SOME_KEY") - assert.ErrorIs(t, err, ErrInvalidServiceName) -} - -func TestService_SetAttachment_ServiceNameTooLong(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - // Service name exceeds DNS label limit of 63 characters - longService := strings.Repeat("a", 64) - err := svc.SetAttachment(context.Background(), "app.example.com", longService, map[string]string{"K": "V"}) - assert.ErrorIs(t, err, ErrInvalidServiceName) -} - -func TestService_SetAttachment_ServiceNameTrailingHyphen(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - err := svc.SetAttachment(context.Background(), "app.example.com", "postgres-", map[string]string{"K": "V"}) - assert.ErrorIs(t, err, ErrInvalidServiceName) -} - -func TestService_DeleteAttachment_ServiceNameTooLong(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - // Service name exceeds DNS label limit of 63 characters - longService := strings.Repeat("a", 64) - err := svc.DeleteAttachment(context.Background(), "app.example.com", longService, "SOME_KEY") - assert.ErrorIs(t, err, ErrInvalidServiceName) -} - -func TestService_DeleteAttachment_ServiceNameTrailingHyphen(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - err := svc.DeleteAttachment(context.Background(), "app.example.com", "redis-", "SOME_KEY") - assert.ErrorIs(t, err, ErrInvalidServiceName) -} - -func TestService_AttachmentValidationHappensBeforeStore(t *testing.T) { - store := outmocks.NewMockDomainSecretStore(t) - svc := NewService(store, testLogger(), nil) - - // With invalid service name, store methods should never be called - err := svc.SetAttachment(context.Background(), "app.example.com", "../evil", map[string]string{"key": "value"}) - assert.ErrorIs(t, err, ErrInvalidServiceName) - - err = svc.DeleteAttachment(context.Background(), "app.example.com", "../evil", "key") - assert.ErrorIs(t, err, ErrInvalidServiceName) - - // Verify no store methods were called - mock.AssertExpectationsForObjects(t, store) -} diff --git a/internal/usecase/services/service.go b/internal/usecase/services/service.go deleted file mode 100644 index 1b9f22f97..000000000 --- a/internal/usecase/services/service.go +++ /dev/null @@ -1,640 +0,0 @@ -package services - -import ( - "bufio" - "bytes" - "context" - "crypto/sha256" - "encoding/hex" - "encoding/json" - "errors" - "fmt" - "io" - "maps" - "net" - "os" - "sort" - "strconv" - "strings" - "time" - - "github.com/bnema/gordon/internal/boundaries/out" - "github.com/bnema/gordon/internal/domain" -) - -const defaultVolumePrefix = "gordon-service" - -type Service struct { - runtime out.ContainerRuntime - secretProvider out.SecretProvider - volumePrefix string -} - -func NewService(runtime out.ContainerRuntime) *Service { - return &Service{runtime: runtime, volumePrefix: defaultVolumePrefix} -} - -func NewServiceWithSecretProvider(runtime out.ContainerRuntime, secretProvider out.SecretProvider) *Service { - return &Service{runtime: runtime, secretProvider: secretProvider, volumePrefix: defaultVolumePrefix} -} - -func NewServiceWithVolumePrefix(runtime out.ContainerRuntime, volumePrefix string) *Service { - if volumePrefix == "" { - volumePrefix = defaultVolumePrefix - } - return &Service{runtime: runtime, volumePrefix: volumePrefix} -} - -func (s *Service) Reconcile(ctx context.Context, configured []domain.StandaloneService) error { - containers, err := s.runtime.ListContainers(ctx, true) - if err != nil { - return fmt.Errorf("list standalone service containers: %w", err) - } - existing := managedServiceContainers(containers) - configuredNames := make(map[string]struct{}, len(configured)) - for _, svc := range configured { - configuredNames[svc.Name] = struct{}{} - if err := s.reconcileOne(ctx, svc, existing[svc.Name]); err != nil { - return err - } - } - for name, containers := range existing { - if _, ok := configuredNames[name]; ok { - continue - } - if err := s.stopRemoved(ctx, name, containers); err != nil { - return err - } - } - return nil -} - -func (s *Service) Status(ctx context.Context) ([]domain.StandaloneServiceStatus, error) { - containers, err := s.runtime.ListContainers(ctx, true) - if err != nil { - return nil, fmt.Errorf("list standalone service containers: %w", err) - } - statuses := make([]domain.StandaloneServiceStatus, 0) - for _, container := range containers { - if container.Labels[domain.LabelService] != "true" { - continue - } - statuses = append(statuses, domain.StandaloneServiceStatus{ - Name: container.Labels[domain.LabelServiceName], - ContainerID: container.ID, - ContainerName: container.Name, - Status: containerStatus(container), - ConfigHash: container.Labels[domain.LabelServiceConfigHash], - }) - } - sort.Slice(statuses, func(i, j int) bool { return statuses[i].Name < statuses[j].Name }) - return statuses, nil -} - -func (s *Service) reconcileOne(ctx context.Context, svc domain.StandaloneService, existing []*domain.Container) error { - cleanup := normalizeCleanup(svc.Cleanup) - if !svc.Enabled { - return s.stopDisabled(ctx, svc.Name, cleanup, existing) - } - env, err := s.serviceEnv(ctx, svc) - if err != nil { - return fmt.Errorf("resolve standalone service %q environment: %w", svc.Name, err) - } - hash, err := serviceConfigHashWithEnv(svc, env) - if err != nil { - return fmt.Errorf("hash standalone service %q config: %w", svc.Name, err) - } - if len(existing) == 0 { - return s.createAndStart(ctx, svc, hash, env) - } - sort.SliceStable(existing, func(i, j int) bool { - leftRunning := containerStatus(existing[i]) == domain.ContainerStatusRunning - rightRunning := containerStatus(existing[j]) == domain.ContainerStatusRunning - if leftRunning != rightRunning { - return leftRunning - } - return existing[i].ID < existing[j].ID - }) - current := existing[0] - if current.Labels[domain.LabelServiceConfigHash] != hash { - if err := s.recreate(ctx, svc, cleanup, existing, hash, env); err != nil { - return err - } - return nil - } - for _, duplicate := range existing[1:] { - if err := s.cleanupContainer(ctx, svc.Name, "duplicate", cleanup, duplicate); err != nil { - return err - } - } - if containerStatus(current) != domain.ContainerStatusRunning { - if err := s.runtime.StartContainer(ctx, current.ID); err != nil { - return fmt.Errorf("start standalone service %q container: %w", svc.Name, err) - } - } - if err := s.waitReadiness(ctx, current.ID, svc); err != nil { - return err - } - return nil -} - -func (s *Service) recreate(ctx context.Context, svc domain.StandaloneService, cleanup domain.StandaloneServiceCleanup, existing []*domain.Container, hash string, env []string) error { - for _, container := range existing { - if containerStatus(container) == domain.ContainerStatusRunning { - if err := s.runtime.StopContainer(ctx, container.ID); err != nil { - return fmt.Errorf("stop stale standalone service %q container: %w", svc.Name, err) - } - } - if cleanup.RemoveContainer { - if err := s.runtime.RemoveContainer(ctx, container.ID, true); err != nil { - return fmt.Errorf("remove stale standalone service %q container: %w", svc.Name, err) - } - } - } - return s.createAndStart(ctx, svc, hash, env) -} - -func (s *Service) stopDisabled(ctx context.Context, name string, cleanup domain.StandaloneServiceCleanup, existing []*domain.Container) error { - for _, container := range existing { - if err := s.cleanupContainer(ctx, name, "disabled", cleanup, container); err != nil { - return err - } - } - return nil -} - -func (s *Service) stopRemoved(ctx context.Context, name string, existing []*domain.Container) error { - for _, container := range existing { - cleanup := cleanupFromLabels(container.Labels) - if err := s.cleanupContainer(ctx, name, "removed", cleanup, container); err != nil { - return err - } - } - return nil -} - -func (s *Service) cleanupContainer(ctx context.Context, name, reason string, cleanup domain.StandaloneServiceCleanup, container *domain.Container) error { - if containerStatus(container) == domain.ContainerStatusRunning { - if err := s.runtime.StopContainer(ctx, container.ID); err != nil { - return fmt.Errorf("stop %s standalone service %q container: %w", reason, name, err) - } - } - if cleanup.RemoveContainer { - if err := s.runtime.RemoveContainer(ctx, container.ID, true); err != nil { - return fmt.Errorf("remove %s standalone service %q container: %w", reason, name, err) - } - } - if err := s.removeManagedVolumes(ctx, name, cleanup, container); err != nil { - return err - } - return nil -} - -func (s *Service) removeManagedVolumes(ctx context.Context, name string, cleanup domain.StandaloneServiceCleanup, container *domain.Container) error { - if cleanup.PreserveVolumes { - return nil - } - managed := managedVolumeSet(container.Labels) - for _, mount := range container.VolumeMounts { - if mount.Type != "volume" || mount.Name == "" { - continue - } - if _, ok := managed[mount.Name]; !ok { - continue - } - if err := s.runtime.RemoveVolume(ctx, mount.Name, true); err != nil { - return fmt.Errorf("remove standalone service %q volume %q: %w", name, mount.Name, err) - } - } - return nil -} - -func managedVolumeSet(labels map[string]string) map[string]struct{} { - values := strings.Split(labels[domain.LabelServiceManagedVolumes], ",") - managed := make(map[string]struct{}, len(values)) - for _, value := range values { - name := strings.TrimSpace(value) - if name == "" { - continue - } - managed[name] = struct{}{} - } - return managed -} - -func cleanupFromLabels(labels map[string]string) domain.StandaloneServiceCleanup { - cleanup := domain.StandaloneServiceCleanup{PreserveVolumes: true, RemoveContainer: true} - if value, ok := labels[domain.LabelServiceCleanupPreserveVolumes]; ok { - if parsed, err := strconv.ParseBool(value); err == nil { - cleanup.PreserveVolumes = parsed - } - } - if value, ok := labels[domain.LabelServiceCleanupRemoveContainer]; ok { - if parsed, err := strconv.ParseBool(value); err == nil { - cleanup.RemoveContainer = parsed - } - } - return cleanup -} - -func (s *Service) createAndStart(ctx context.Context, svc domain.StandaloneService, hash string, env []string) error { - config, err := s.containerConfig(ctx, svc, hash, env) - if err != nil { - return err - } - container, err := s.runtime.CreateContainer(ctx, config) - if err != nil { - return fmt.Errorf("create standalone service %q container: %w", svc.Name, err) - } - if err := s.runtime.StartContainer(ctx, container.ID); err != nil { - return fmt.Errorf("start standalone service %q container: %w", svc.Name, err) - } - if err := s.waitReadiness(ctx, container.ID, svc); err != nil { - return err - } - return nil -} - -func (s *Service) containerConfig(ctx context.Context, svc domain.StandaloneService, hash string, env []string) (*domain.ContainerConfig, error) { - var imageVolumes []string - if len(svc.Volumes) == 0 { - var err error - imageVolumes, err = s.runtime.InspectImageVolumes(ctx, svc.Image) - if err != nil { - return nil, fmt.Errorf("inspect standalone service %q image volumes: %w", svc.Name, err) - } - } - mounts := ResolveVolumeMounts(s.volumePrefix, svc.Name, svc.Volumes, imageVolumes) - volumes := make(map[string]string) - readOnlyVolumes := make(map[string]string) - managedVolumes := make([]string, 0) - for _, mount := range mounts { - if mount.Managed { - managedVolumes = append(managedVolumes, mount.Source) - } - if mount.ReadOnly { - readOnlyVolumes[mount.Target] = mount.Source - continue - } - volumes[mount.Target] = mount.Source - } - publishes, err := portPublishes(svc) - if err != nil { - return nil, err - } - return &domain.ContainerConfig{ - Image: svc.Image, - Name: serviceContainerName(svc.Name), - Env: env, - PortPublishes: publishes, - Labels: serviceLabels(svc.Name, hash, normalizeCleanup(svc.Cleanup), managedVolumes), - AutoRemove: false, - RestartPolicy: domain.RestartPolicyAlways, - Volumes: emptyToNil(volumes), - ReadOnlyVolumes: emptyToNil(readOnlyVolumes), - }, nil -} - -func managedServiceContainers(containers []*domain.Container) map[string][]*domain.Container { - managed := make(map[string][]*domain.Container) - for _, container := range containers { - if container == nil || container.Labels[domain.LabelService] != "true" { - continue - } - name := container.Labels[domain.LabelServiceName] - if name == "" { - continue - } - managed[name] = append(managed[name], container) - } - return managed -} - -func serviceLabels(name, hash string, cleanup domain.StandaloneServiceCleanup, managedVolumes []string) map[string]string { - sort.Strings(managedVolumes) - return map[string]string{ - domain.LabelManaged: "true", - domain.LabelService: "true", - domain.LabelServiceName: name, - domain.LabelServiceConfigHash: hash, - domain.LabelServiceManagedVolumes: strings.Join(managedVolumes, ","), - domain.LabelServiceCleanupPreserveVolumes: strconv.FormatBool(cleanup.PreserveVolumes), - domain.LabelServiceCleanupRemoveContainer: strconv.FormatBool(cleanup.RemoveContainer), - } -} - -func portPublishes(svc domain.StandaloneService) ([]domain.ContainerPortPublish, error) { - publishes := make([]domain.ContainerPortPublish, 0, len(svc.Ports)) - for _, port := range svc.Ports { - if port.Publish == "" { - continue - } - host, hostPort, err := net.SplitHostPort(port.Publish) - if err != nil { - return nil, fmt.Errorf("parse standalone service %q port %q publish: %w", svc.Name, port.Name, err) - } - hostPortNumber, err := strconv.Atoi(hostPort) - if err != nil { - return nil, fmt.Errorf("parse standalone service %q port %q publish port: %w", svc.Name, port.Name, err) - } - publishes = append(publishes, domain.ContainerPortPublish{ - HostIP: host, HostPort: hostPortNumber, ContainerPort: port.Container, Protocol: port.Protocol, - }) - } - return publishes, nil -} - -func serviceConfigHash(svc domain.StandaloneService) (string, error) { - return NewService(nil).serviceConfigHash(context.Background(), svc) -} - -func (s *Service) serviceConfigHash(ctx context.Context, svc domain.StandaloneService) (string, error) { - resolvedEnv, err := s.serviceEnv(ctx, svc) - if err != nil { - return "", err - } - return serviceConfigHashWithEnv(svc, resolvedEnv) -} - -func serviceConfigHashWithEnv(svc domain.StandaloneService, resolvedEnv []string) (string, error) { - payload := struct { - Image string - ResolvedEnv []string - Readiness domain.StandaloneServiceReadiness - Ports []domain.StandaloneServicePort - Volumes []domain.StandaloneServiceVolume - Cleanup domain.StandaloneServiceCleanup - }{svc.Image, append([]string(nil), resolvedEnv...), svc.Readiness, append([]domain.StandaloneServicePort(nil), svc.Ports...), append([]domain.StandaloneServiceVolume(nil), svc.Volumes...), normalizeCleanup(svc.Cleanup)} - bytes, err := json.Marshal(payload) - if err != nil { - return "", err - } - sum := sha256.Sum256(bytes) - return hex.EncodeToString(sum[:]), nil -} - -func (s *Service) serviceEnv(ctx context.Context, svc domain.StandaloneService) ([]string, error) { - envMap := make(map[string]string) - if svc.EnvFile != "" { - fileEnv, err := loadEnvFile(ctx, svc.EnvFile) - if err != nil { - return nil, fmt.Errorf("load standalone service %q env file: %w", svc.Name, err) - } - maps.Copy(envMap, fileEnv) - } - for _, entry := range svc.Env { - key, value, ok := strings.Cut(entry, "=") - if !ok || key == "" { - return nil, fmt.Errorf("parse standalone service %q env entry", svc.Name) - } - envMap[key] = value - } - if len(svc.Secrets) > 0 && s.secretProvider == nil { - return nil, fmt.Errorf("resolve standalone service %q secrets: secret provider is not configured", svc.Name) - } - for _, secret := range svc.Secrets { - value, err := s.secretProvider.GetSecret(ctx, serviceSecretPath(svc.Name, secret.Name)) - if err != nil { - return nil, fmt.Errorf("resolve standalone service %q secret %q: %w", svc.Name, secret.Name, err) - } - envMap[secret.Key] = value - } - return envMapToList(envMap), nil -} - -func loadEnvFile(ctx context.Context, path string) (map[string]string, error) { - if err := ctx.Err(); err != nil { - return nil, err - } - file, err := os.Open(path) - if err != nil { - return nil, err - } - defer file.Close() - values := make(map[string]string) - scanner := bufio.NewScanner(file) - for scanner.Scan() { - if err := ctx.Err(); err != nil { - return nil, err - } - line := strings.TrimSpace(scanner.Text()) - if line == "" || strings.HasPrefix(line, "#") { - continue - } - key, value, ok := strings.Cut(line, "=") - if !ok || strings.TrimSpace(key) == "" { - return nil, fmt.Errorf("invalid env file line") - } - values[strings.TrimSpace(key)] = strings.Trim(strings.TrimSpace(value), `"'`) - } - if err := scanner.Err(); err != nil { - return nil, err - } - return values, nil -} - -func serviceSecretPath(serviceName, secretName string) string { - if strings.Contains(secretName, ":") { - return secretName - } - return "service/" + serviceName + "/" + secretName -} - -func envMapToList(values map[string]string) []string { - if len(values) == 0 { - return nil - } - keys := make([]string, 0, len(values)) - for key := range values { - keys = append(keys, key) - } - sort.Strings(keys) - env := make([]string, 0, len(keys)) - for _, key := range keys { - env = append(env, key+"="+values[key]) - } - return env -} - -func (s *Service) waitReadiness(ctx context.Context, containerID string, svc domain.StandaloneService) error { - readiness := svc.Readiness - if readiness.Type == "" || readiness.Type == domain.StandaloneServiceReadinessNone { - return nil - } - readyCtx := ctx - cancel := func() {} - if readiness.Timeout > 0 { - readyCtx, cancel = context.WithTimeout(ctx, readiness.Timeout) - } - defer cancel() - switch readiness.Type { - case domain.StandaloneServiceReadinessTCP: - return waitTCPReadiness(readyCtx, svc) - case domain.StandaloneServiceReadinessLog: - return s.waitLogReadiness(readyCtx, containerID, svc) - default: - return fmt.Errorf("unsupported standalone service %q readiness type %q", svc.Name, readiness.Type) - } -} - -func waitTCPReadiness(ctx context.Context, svc domain.StandaloneService) error { - address, err := tcpReadinessAddress(svc) - if err != nil { - return err - } - ticker := time.NewTicker(10 * time.Millisecond) - defer ticker.Stop() - for { - if err := ctx.Err(); err != nil { - return fmt.Errorf("standalone service %q tcp readiness timed out: %w", svc.Name, err) - } - dialer := net.Dialer{Timeout: 100 * time.Millisecond} - conn, err := dialer.DialContext(ctx, "tcp", address) - if err == nil { - _ = conn.Close() - return nil - } - select { - case <-ctx.Done(): - return fmt.Errorf("standalone service %q tcp readiness timed out: %w", svc.Name, ctx.Err()) - case <-ticker.C: - } - } -} - -func tcpReadinessAddress(svc domain.StandaloneService) (string, error) { - for _, port := range svc.Ports { - if port.Protocol != domain.NetworkProtocolTCP || port.Publish == "" { - continue - } - host, hostPort, err := net.SplitHostPort(port.Publish) - if err != nil { - return "", fmt.Errorf("parse standalone service %q tcp readiness publish: %w", svc.Name, err) - } - switch host { - case "::": - host = "::1" - case "0.0.0.0", "": - host = "127.0.0.1" - } - return net.JoinHostPort(host, hostPort), nil - } - return "", fmt.Errorf("standalone service %q tcp readiness requires a published tcp port", svc.Name) -} - -func (s *Service) waitLogReadiness(ctx context.Context, containerID string, svc domain.StandaloneService) error { - ticker := time.NewTicker(250 * time.Millisecond) - defer ticker.Stop() - var lastErr error - for { - if err := ctx.Err(); err != nil { - return logReadinessTimeoutError(svc.Name, err, lastErr) - } - found, err := s.logContains(ctx, containerID, svc.Readiness.Path, svc.Readiness.Contains) - if err == nil && found { - return nil - } - if errors.Is(err, domain.ErrReadinessLogSizeExceeded) { - return err - } - if err != nil { - lastErr = err - } - // Continue polling until timeout; many services create the log after start. - select { - case <-ctx.Done(): - return logReadinessTimeoutError(svc.Name, ctx.Err(), lastErr) - case <-ticker.C: - } - } -} - -func logReadinessTimeoutError(serviceName string, timeoutErr, lastErr error) error { - if lastErr != nil { - return fmt.Errorf("standalone service %q log readiness timed out: %w; last read error: %w", serviceName, timeoutErr, lastErr) - } - return fmt.Errorf("standalone service %q log readiness timed out: %w", serviceName, timeoutErr) -} - -const maxReadinessLogSize = 1 << 20 // 1 MiB - -func (s *Service) logContains(ctx context.Context, containerID, path, contains string) (bool, error) { - reader, err := s.runtime.CopyFromContainer(ctx, containerID, path) - if err != nil { - return false, err - } - defer reader.Close() - if contains == "" { - return true, nil - } - - needle := []byte(contains) - buffer := make([]byte, 32*1024) - carry := make([]byte, 0, len(needle)-1) - remaining := maxReadinessLogSize - for remaining > 0 { - readSize := min(len(buffer), remaining) - n, readErr := reader.Read(buffer[:readSize]) - if n > 0 { - window := append(carry, buffer[:n]...) - if bytes.Contains(window, needle) { - return true, nil - } - overlap := min(len(needle)-1, len(window)) - carry = append(carry[:0], window[len(window)-overlap:]...) - remaining -= n - } - if errors.Is(readErr, io.EOF) { - return false, nil - } - if readErr != nil { - return false, readErr - } - } - - var extra [1]byte - if _, err := reader.Read(extra[:]); errors.Is(err, io.EOF) { - return false, nil - } else if err != nil { - return false, err - } - return false, fmt.Errorf("%w: limit is %d bytes", domain.ErrReadinessLogSizeExceeded, maxReadinessLogSize) -} - -func normalizeCleanup(cleanup domain.StandaloneServiceCleanup) domain.StandaloneServiceCleanup { - if !cleanup.PreserveVolumes && !cleanup.RemoveContainer { - return domain.StandaloneServiceCleanup{PreserveVolumes: true, RemoveContainer: true} - } - return cleanup -} - -func containerStatus(container *domain.Container) domain.ContainerStatus { - status := strings.ToLower(container.Status) - if strings.Contains(status, string(domain.ContainerStatusRunning)) { - return domain.ContainerStatusRunning - } - if strings.Contains(status, string(domain.ContainerStatusExited)) { - return domain.ContainerStatusExited - } - if strings.Contains(status, string(domain.ContainerStatusCreated)) { - return domain.ContainerStatusCreated - } - if strings.Contains(status, string(domain.ContainerStatusPaused)) { - return domain.ContainerStatusPaused - } - if strings.Contains(status, string(domain.ContainerStatusStopped)) { - return domain.ContainerStatusStopped - } - return domain.ContainerStatusUnknown -} - -func serviceContainerName(name string) string { - return "gordon-service-" + strings.NewReplacer(".", "-", "_", "-", "/", "-").Replace(name) -} - -func emptyToNil(values map[string]string) map[string]string { - if len(values) == 0 { - return nil - } - return values -} diff --git a/internal/usecase/services/service_test.go b/internal/usecase/services/service_test.go deleted file mode 100644 index 0679cfe93..000000000 --- a/internal/usecase/services/service_test.go +++ /dev/null @@ -1,482 +0,0 @@ -package services - -import ( - "context" - "errors" - "io" - "net" - "os" - "strings" - "testing" - "time" - - "github.com/stretchr/testify/assert" - "github.com/stretchr/testify/mock" - "github.com/stretchr/testify/require" - - "github.com/bnema/gordon/internal/adapters/out/secrets" - outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" - "github.com/bnema/gordon/internal/domain" -) - -func TestService_ReconcileCreatesMissingEnabledService(t *testing.T) { - rt := outmocks.NewMockContainerRuntime(t) - svc := sampleService() - var created *domain.ContainerConfig - rt.On("ListContainers", mock.Anything, true).Return([]*domain.Container{}, nil).Once() - rt.On("InspectImageVolumes", mock.Anything, svc.Image).Return([]string{"/data"}, nil).Once() - rt.On("CreateContainer", mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")).Run(func(args mock.Arguments) { - created = args.Get(1).(*domain.ContainerConfig) - }).Return(&domain.Container{ID: "created-1"}, nil).Once() - rt.On("StartContainer", mock.Anything, "created-1").Return(nil).Once() - - err := NewService(rt).Reconcile(context.Background(), []domain.StandaloneService{svc}) - - require.NoError(t, err) - require.NotNil(t, created) - assert.Equal(t, "game:latest", created.Image) - assert.Equal(t, "gordon-service-game", created.Name) - assert.Equal(t, []string{"PUBLIC=value"}, created.Env) - assert.Equal(t, domain.RestartPolicyAlways, created.RestartPolicy) - assert.Equal(t, "true", created.Labels[domain.LabelManaged]) - assert.Equal(t, "true", created.Labels[domain.LabelService]) - assert.Equal(t, "game", created.Labels[domain.LabelServiceName]) - assert.Equal(t, "true", created.Labels[domain.LabelServiceCleanupPreserveVolumes]) - assert.Equal(t, "true", created.Labels[domain.LabelServiceCleanupRemoveContainer]) - assert.NotEmpty(t, created.Labels[domain.LabelServiceConfigHash]) - assert.Equal(t, "gordon-service-game-data", created.Labels[domain.LabelServiceManagedVolumes]) - assert.Equal(t, []domain.ContainerPortPublish{{HostIP: "127.0.0.1", HostPort: 38015, ContainerPort: 28015, Protocol: domain.NetworkProtocolUDP}}, created.PortPublishes) - assert.Equal(t, map[string]string{"/data": "gordon-service-game-data"}, created.Volumes) -} - -func TestService_ReconcileStartsStoppedExistingService(t *testing.T) { - rt := outmocks.NewMockContainerRuntime(t) - svc := sampleService() - hash, err := serviceConfigHash(svc) - require.NoError(t, err) - rt.On("ListContainers", mock.Anything, true).Return([]*domain.Container{managedContainer("existing-1", svc.Name, hash, "exited")}, nil).Once() - rt.On("StartContainer", mock.Anything, "existing-1").Return(nil).Once() - - err = NewService(rt).Reconcile(context.Background(), []domain.StandaloneService{svc}) - - require.NoError(t, err) -} - -func TestService_ReconcileChecksReadinessForExistingMatchingService(t *testing.T) { - rt := outmocks.NewMockContainerRuntime(t) - svc := sampleService() - svc.Readiness = domain.StandaloneServiceReadiness{Type: domain.StandaloneServiceReadinessLog, Path: "/logs/server.log", Contains: "ready", Timeout: time.Second} - hash, err := serviceConfigHash(svc) - require.NoError(t, err) - rt.On("ListContainers", mock.Anything, true).Return([]*domain.Container{managedContainer("existing-1", svc.Name, hash, "running")}, nil).Once() - rt.On("CopyFromContainer", mock.Anything, "existing-1", "/logs/server.log").Return(io.NopCloser(strings.NewReader("ready")), nil).Once() - - err = NewService(rt).Reconcile(context.Background(), []domain.StandaloneService{svc}) - - require.NoError(t, err) -} - -func TestService_ReconcileRecreatesStaleService(t *testing.T) { - rt := outmocks.NewMockContainerRuntime(t) - svc := sampleService() - rt.On("ListContainers", mock.Anything, true).Return([]*domain.Container{managedContainer("old-1", svc.Name, "old-hash", "running")}, nil).Once() - rt.On("StopContainer", mock.Anything, "old-1").Return(nil).Once() - rt.On("RemoveContainer", mock.Anything, "old-1", true).Return(nil).Once() - rt.On("InspectImageVolumes", mock.Anything, svc.Image).Return([]string{}, nil).Once() - rt.On("CreateContainer", mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")).Return(&domain.Container{ID: "new-1"}, nil).Once() - rt.On("StartContainer", mock.Anything, "new-1").Return(nil).Once() - - err := NewService(rt).Reconcile(context.Background(), []domain.StandaloneService{svc}) - - require.NoError(t, err) -} - -func TestService_ReconcileRecreateDoesNotRemoveManagedVolumes(t *testing.T) { - rt := outmocks.NewMockContainerRuntime(t) - svc := sampleService() - svc.Cleanup = domain.StandaloneServiceCleanup{PreserveVolumes: false, RemoveContainer: true} - old := managedContainer("old-1", svc.Name, "old-hash", "running") - old.Labels[domain.LabelServiceManagedVolumes] = "gordon-service-game-data" - old.VolumeMounts = []domain.ContainerVolumeMount{{Name: "gordon-service-game-data", Type: "volume", Destination: "/data"}} - rt.On("ListContainers", mock.Anything, true).Return([]*domain.Container{old}, nil).Once() - rt.On("StopContainer", mock.Anything, "old-1").Return(nil).Once() - rt.On("RemoveContainer", mock.Anything, "old-1", true).Return(nil).Once() - rt.On("InspectImageVolumes", mock.Anything, svc.Image).Return([]string{}, nil).Once() - rt.On("CreateContainer", mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")).Return(&domain.Container{ID: "new-1"}, nil).Once() - rt.On("StartContainer", mock.Anything, "new-1").Return(nil).Once() - - err := NewService(rt).Reconcile(context.Background(), []domain.StandaloneService{svc}) - - require.NoError(t, err) -} - -func TestService_ReconcileRemovesDuplicateMatchingServiceContainers(t *testing.T) { - rt := outmocks.NewMockContainerRuntime(t) - svc := sampleService() - hash, err := serviceConfigHash(svc) - require.NoError(t, err) - older := managedContainer("existing-1", svc.Name, hash, "running") - duplicate := managedContainer("existing-2", svc.Name, hash, "running") - rt.On("ListContainers", mock.Anything, true).Return([]*domain.Container{duplicate, older}, nil).Once() - rt.On("StopContainer", mock.Anything, "existing-2").Return(nil).Once() - rt.On("RemoveContainer", mock.Anything, "existing-2", true).Return(nil).Once() - - err = NewService(rt).Reconcile(context.Background(), []domain.StandaloneService{svc}) - - require.NoError(t, err) -} - -func TestService_ReconcileKeepsRunningDuplicateBeforeLowerStaleID(t *testing.T) { - rt := outmocks.NewMockContainerRuntime(t) - svc := sampleService() - hash, err := serviceConfigHash(svc) - require.NoError(t, err) - staleLowerID := managedContainer("existing-1", svc.Name, hash, "exited") - runningPrimary := managedContainer("existing-2", svc.Name, hash, "running") - rt.On("ListContainers", mock.Anything, true).Return([]*domain.Container{staleLowerID, runningPrimary}, nil).Once() - rt.On("RemoveContainer", mock.Anything, "existing-1", true).Return(nil).Once() - - err = NewService(rt).Reconcile(context.Background(), []domain.StandaloneService{svc}) - - require.NoError(t, err) -} - -func TestService_ReconcileStopsAndRemovesDisabledService(t *testing.T) { - rt := outmocks.NewMockContainerRuntime(t) - svc := sampleService() - svc.Enabled = false - rt.On("ListContainers", mock.Anything, true).Return([]*domain.Container{managedContainer("disabled-1", svc.Name, "old-hash", "running")}, nil).Once() - rt.On("StopContainer", mock.Anything, "disabled-1").Return(nil).Once() - rt.On("RemoveContainer", mock.Anything, "disabled-1", true).Return(nil).Once() - - err := NewService(rt).Reconcile(context.Background(), []domain.StandaloneService{svc}) - - require.NoError(t, err) -} - -func TestService_ReconcileStopsAndRemovesOmittedManagedService(t *testing.T) { - rt := outmocks.NewMockContainerRuntime(t) - container := managedContainer("removed-1", "removed", "old-hash", "running") - rt.On("ListContainers", mock.Anything, true).Return([]*domain.Container{container}, nil).Once() - rt.On("StopContainer", mock.Anything, "removed-1").Return(nil).Once() - rt.On("RemoveContainer", mock.Anything, "removed-1", true).Return(nil).Once() - - err := NewService(rt).Reconcile(context.Background(), nil) - - require.NoError(t, err) -} - -func TestService_ReconcileRemovesNamedVolumeMountsWhenCleanupDoesNotPreserveVolumes(t *testing.T) { - rt := outmocks.NewMockContainerRuntime(t) - svc := sampleService() - svc.Enabled = false - svc.Cleanup = domain.StandaloneServiceCleanup{PreserveVolumes: false, RemoveContainer: true} - container := managedContainer("disabled-1", svc.Name, "old-hash", "running") - container.Labels[domain.LabelServiceManagedVolumes] = "gordon-service-game-data" - container.VolumeMounts = []domain.ContainerVolumeMount{ - {Name: "gordon-service-game-data", Type: "volume", Destination: "/data"}, - {Name: "user-shared-data", Type: "volume", Destination: "/shared"}, - {Name: "", Type: "volume", Destination: "/anonymous"}, - {Name: "host-data", Type: "bind", Destination: "/host"}, - } - rt.On("ListContainers", mock.Anything, true).Return([]*domain.Container{container}, nil).Once() - rt.On("StopContainer", mock.Anything, "disabled-1").Return(nil).Once() - rt.On("RemoveContainer", mock.Anything, "disabled-1", true).Return(nil).Once() - rt.On("RemoveVolume", mock.Anything, "gordon-service-game-data", true).Return(nil).Once() - - err := NewService(rt).Reconcile(context.Background(), []domain.StandaloneService{svc}) - - require.NoError(t, err) -} - -func TestService_ReconcileUsesCleanupLabelsForOmittedService(t *testing.T) { - rt := outmocks.NewMockContainerRuntime(t) - container := managedContainer("removed-1", "removed", "old-hash", "exited") - container.Labels[domain.LabelServiceCleanupPreserveVolumes] = "false" - container.Labels[domain.LabelServiceCleanupRemoveContainer] = "true" - container.Labels[domain.LabelServiceManagedVolumes] = "gordon-service-removed-data" - container.VolumeMounts = []domain.ContainerVolumeMount{{Name: "gordon-service-removed-data", Type: "volume", Destination: "/data"}} - rt.On("ListContainers", mock.Anything, true).Return([]*domain.Container{container}, nil).Once() - rt.On("RemoveContainer", mock.Anything, "removed-1", true).Return(nil).Once() - rt.On("RemoveVolume", mock.Anything, "gordon-service-removed-data", true).Return(nil).Once() - - err := NewService(rt).Reconcile(context.Background(), nil) - - require.NoError(t, err) -} - -func TestService_ReconcileDefaultsOmittedServiceCleanupToPreserveVolumes(t *testing.T) { - rt := outmocks.NewMockContainerRuntime(t) - container := managedContainer("removed-1", "removed", "old-hash", "exited") - container.VolumeMounts = []domain.ContainerVolumeMount{{Name: "gordon-service-removed-data", Type: "volume", Destination: "/data"}} - rt.On("ListContainers", mock.Anything, true).Return([]*domain.Container{container}, nil).Once() - rt.On("RemoveContainer", mock.Anything, "removed-1", true).Return(nil).Once() - - err := NewService(rt).Reconcile(context.Background(), nil) - - require.NoError(t, err) -} - -func TestService_ReconcileUsesExplicitReadOnlyVolumes(t *testing.T) { - rt := outmocks.NewMockContainerRuntime(t) - svc := sampleService() - svc.Volumes = []domain.StandaloneServiceVolume{{Source: "cfg", Target: "/cfg", ReadOnly: true}} - var created *domain.ContainerConfig - rt.On("ListContainers", mock.Anything, true).Return([]*domain.Container{}, nil).Once() - rt.On("CreateContainer", mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")).Run(func(args mock.Arguments) { - created = args.Get(1).(*domain.ContainerConfig) - }).Return(&domain.Container{ID: "created-1"}, nil).Once() - rt.On("StartContainer", mock.Anything, "created-1").Return(nil).Once() - - err := NewService(rt).Reconcile(context.Background(), []domain.StandaloneService{svc}) - - require.NoError(t, err) - assert.Nil(t, created.Volumes) - assert.Equal(t, map[string]string{"/cfg": "cfg"}, created.ReadOnlyVolumes) -} - -func TestService_StatusReturnsManagedServices(t *testing.T) { - rt := outmocks.NewMockContainerRuntime(t) - rt.On("ListContainers", mock.Anything, true).Return([]*domain.Container{ - managedContainer("id-1", "alpha", "hash-1", "running"), - {ID: "other", Labels: map[string]string{}}, - }, nil).Once() - - statuses, err := NewService(rt).Status(context.Background()) - - require.NoError(t, err) - require.Len(t, statuses, 1) - assert.Equal(t, "alpha", statuses[0].Name) - assert.Equal(t, "id-1", statuses[0].ContainerID) - assert.Equal(t, domain.ContainerStatusRunning, statuses[0].Status) - assert.Equal(t, "hash-1", statuses[0].ConfigHash) -} - -func TestServiceEnvLoadsEnvFileAndMergesInlineEnv(t *testing.T) { - envFile := writeTempEnvFile(t, "FROM_FILE=file\nOVERRIDE=file\n# ignored\n") - svc := sampleService() - svc.EnvFile = envFile - svc.Env = []string{"OVERRIDE=inline", "INLINE=value"} - - env, err := NewService(nil).serviceEnv(context.Background(), svc) - - require.NoError(t, err) - assert.Equal(t, []string{"FROM_FILE=file", "INLINE=value", "OVERRIDE=inline"}, env) -} - -func TestServiceConfigHashIncludesResolvedEnvFileAndSecretValues(t *testing.T) { - envFile := writeTempEnvFile(t, "FROM_FILE=one\n") - provider := outmocks.NewMockSecretProvider(t) - svc := sampleService() - svc.EnvFile = envFile - svc.Secrets = []domain.StandaloneServiceSecretRef{{Name: "rcon", Key: "RCON_PASSWORD"}} - provider.On("GetSecret", mock.Anything, "service/game/rcon").Return("secret-one", nil).Once() - first, err := NewServiceWithSecretProvider(nil, provider).serviceConfigHash(context.Background(), svc) - require.NoError(t, err) - - require.NoError(t, os.WriteFile(envFile, []byte("FROM_FILE=two\n"), 0o600)) - provider.On("GetSecret", mock.Anything, "service/game/rcon").Return("secret-two", nil).Once() - second, err := NewServiceWithSecretProvider(nil, provider).serviceConfigHash(context.Background(), svc) - require.NoError(t, err) - - assert.NotEqual(t, first, second) -} - -func TestService_ReconcileResolvesSecretEnvOnceForHashAndContainer(t *testing.T) { - rt := outmocks.NewMockContainerRuntime(t) - provider := outmocks.NewMockSecretProvider(t) - svc := sampleService() - svc.Secrets = []domain.StandaloneServiceSecretRef{{Name: "rcon", Key: "RCON_PASSWORD"}} - var created *domain.ContainerConfig - provider.On("GetSecret", mock.Anything, "service/game/rcon").Return("secret-value", nil).Once() - rt.On("ListContainers", mock.Anything, true).Return([]*domain.Container{}, nil).Once() - rt.On("InspectImageVolumes", mock.Anything, svc.Image).Return([]string{}, nil).Once() - rt.On("CreateContainer", mock.Anything, mock.AnythingOfType("*domain.ContainerConfig")).Run(func(args mock.Arguments) { - created = args.Get(1).(*domain.ContainerConfig) - }).Return(&domain.Container{ID: "created-1"}, nil).Once() - rt.On("StartContainer", mock.Anything, "created-1").Return(nil).Once() - - err := NewServiceWithSecretProvider(rt, provider).Reconcile(context.Background(), []domain.StandaloneService{svc}) - - require.NoError(t, err) - require.NotNil(t, created) - assert.Contains(t, created.Env, "RCON_PASSWORD=secret-value") - assert.NotEmpty(t, created.Labels[domain.LabelServiceConfigHash]) -} - -func TestServiceEnvResolvesServiceScopedSecretsWithoutLeakingValues(t *testing.T) { - rt := outmocks.NewMockContainerRuntime(t) - provider := outmocks.NewMockSecretProvider(t) - svc := sampleService() - svc.Secrets = []domain.StandaloneServiceSecretRef{{Name: "rcon", Key: "RCON_PASSWORD"}} - provider.On("GetSecret", mock.Anything, "service/game/rcon").Return("super-secret-value", errors.New("provider failed")).Once() - - env, err := NewServiceWithSecretProvider(rt, provider).serviceEnv(context.Background(), svc) - - require.Error(t, err) - assert.Nil(t, env) - assert.NotContains(t, err.Error(), "super-secret-value") -} - -func TestServiceSecretPathIsPassProviderSafe(t *testing.T) { - path := serviceSecretPath("game", "rcon") - - require.NoError(t, secrets.ValidatePath(path)) - assert.Equal(t, "service/game/rcon", path) -} - -func TestServiceSecretPathPreservesExplicitProviderPath(t *testing.T) { - path := serviceSecretPath("game", "secrets/service.yaml:game.rcon") - - assert.Equal(t, "secrets/service.yaml:game.rcon", path) -} - -func TestServiceEnvUsesExplicitProviderSecretPaths(t *testing.T) { - provider := outmocks.NewMockSecretProvider(t) - svc := sampleService() - svc.Secrets = []domain.StandaloneServiceSecretRef{{Name: "secrets/service.yaml:game.rcon", Key: "RCON_PASSWORD"}} - provider.On("GetSecret", mock.Anything, "secrets/service.yaml:game.rcon").Return("secret-value", nil).Once() - - env, err := NewServiceWithSecretProvider(nil, provider).serviceEnv(context.Background(), svc) - - require.NoError(t, err) - assert.Contains(t, env, "RCON_PASSWORD=secret-value") -} - -func TestTCPReadinessAddressMapsIPv6WildcardToIPv6Loopback(t *testing.T) { - svc := sampleService() - svc.Ports = []domain.StandaloneServicePort{{Name: "admin", Container: 28016, Protocol: domain.NetworkProtocolTCP, Publish: "[::]:38016"}} - - address, err := tcpReadinessAddress(svc) - - require.NoError(t, err) - assert.Equal(t, "[::1]:38016", address) -} - -func TestWaitTCPReadinessDialsResolvedLoopbackPublish(t *testing.T) { - listener, err := net.Listen("tcp", "127.0.0.1:0") - require.NoError(t, err) - defer listener.Close() - accepted := make(chan struct{}) - go func() { - conn, acceptErr := listener.Accept() - if acceptErr == nil { - _ = conn.Close() - close(accepted) - } - }() - svc := sampleService() - svc.Readiness = domain.StandaloneServiceReadiness{Type: domain.StandaloneServiceReadinessTCP, Timeout: time.Second} - svc.Ports = []domain.StandaloneServicePort{{Name: "admin", Container: 28016, Protocol: domain.NetworkProtocolTCP, Publish: listener.Addr().String()}} - - err = NewService(nil).waitReadiness(context.Background(), "container-1", svc) - - require.NoError(t, err) - <-accepted -} - -func TestWaitLogReadinessReadsContainerFileUntilTextAppears(t *testing.T) { - rt := outmocks.NewMockContainerRuntime(t) - svc := sampleService() - svc.Readiness = domain.StandaloneServiceReadiness{Type: domain.StandaloneServiceReadinessLog, Path: "/logs/server.log", Contains: "ready", Timeout: time.Second} - rt.On("CopyFromContainer", mock.Anything, "container-1", "/logs/server.log").Return(io.NopCloser(strings.NewReader("starting")), nil).Once() - rt.On("CopyFromContainer", mock.Anything, "container-1", "/logs/server.log").Return(io.NopCloser(strings.NewReader("server ready")), nil).Once() - - err := NewService(rt).waitReadiness(context.Background(), "container-1", svc) - - require.NoError(t, err) -} - -func TestLogContainsStopsAtMatchBeforeSizeLimit(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - svc := NewService(runtime) - runtime.EXPECT().CopyFromContainer(mock.Anything, "container-1", "/logs/server.log").Return( - io.NopCloser(strings.NewReader("ready\n"+strings.Repeat("x", maxReadinessLogSize))), nil, - ) - - found, err := svc.logContains(context.Background(), "container-1", "/logs/server.log", "ready") - - require.NoError(t, err) - assert.True(t, found) -} - -func TestLogContainsReturnsSentinelAtSizeLimit(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - svc := NewService(runtime) - runtime.EXPECT().CopyFromContainer(mock.Anything, "container-1", "/logs/server.log").Return( - io.NopCloser(strings.NewReader(strings.Repeat("x", maxReadinessLogSize+1))), nil, - ) - - found, err := svc.logContains(context.Background(), "container-1", "/logs/server.log", "ready") - - assert.False(t, found) - assert.ErrorIs(t, err, domain.ErrReadinessLogSizeExceeded) -} - -func TestWaitReadinessHonorsContextTimeout(t *testing.T) { - svc := sampleService() - svc.Readiness = domain.StandaloneServiceReadiness{Type: domain.StandaloneServiceReadinessTCP} - svc.Ports = []domain.StandaloneServicePort{{Name: "admin", Container: 28016, Protocol: domain.NetworkProtocolTCP, Publish: "127.0.0.1:1"}} - ctx, cancel := context.WithTimeout(context.Background(), time.Nanosecond) - defer cancel() - - err := NewService(nil).waitReadiness(ctx, "container-1", svc) - - require.Error(t, err) - assert.ErrorIs(t, err, context.DeadlineExceeded) -} - -func TestWaitLogReadinessPreservesLastReadErrorOnTimeout(t *testing.T) { - rt := outmocks.NewMockContainerRuntime(t) - svc := sampleService() - svc.Readiness = domain.StandaloneServiceReadiness{Type: domain.StandaloneServiceReadinessLog, Path: "/logs/server.log", Contains: "ready", Timeout: 25 * time.Millisecond} - readErr := errors.New("copy failed") - rt.On("CopyFromContainer", mock.Anything, "container-1", "/logs/server.log").Return(nil, readErr) - - err := NewService(rt).waitReadiness(context.Background(), "container-1", svc) - - require.Error(t, err) - assert.ErrorIs(t, err, readErr) - assert.ErrorIs(t, err, context.DeadlineExceeded) -} - -func TestNormalizeCleanupDefaultsPreserveVolumes(t *testing.T) { - cleanup := normalizeCleanup(domain.StandaloneServiceCleanup{}) - - assert.True(t, cleanup.PreserveVolumes) - assert.True(t, cleanup.RemoveContainer) -} - -func sampleService() domain.StandaloneService { - return domain.StandaloneService{ - Name: "game", - Image: "game:latest", - Enabled: true, - Env: []string{"PUBLIC=value"}, - Cleanup: domain.StandaloneServiceCleanup{PreserveVolumes: true, RemoveContainer: true}, - Ports: []domain.StandaloneServicePort{{Name: "game", Container: 28015, Protocol: domain.NetworkProtocolUDP, Publish: "127.0.0.1:38015"}}, - } -} - -func managedContainer(id, name, hash, status string) *domain.Container { - return &domain.Container{ - ID: id, - Name: "gordon-service-" + name, - Status: status, - Labels: map[string]string{ - domain.LabelService: "true", - domain.LabelServiceName: name, - domain.LabelServiceConfigHash: hash, - }, - } -} - -func writeTempEnvFile(t *testing.T, content string) string { - t.Helper() - file, err := os.CreateTemp(t.TempDir(), "service-*.env") - require.NoError(t, err) - _, err = file.WriteString(content) - require.NoError(t, err) - require.NoError(t, file.Close()) - return file.Name() -} diff --git a/internal/usecase/volumes/service.go b/internal/usecase/volumes/service.go index aa2a58062..4ed0edea2 100644 --- a/internal/usecase/volumes/service.go +++ b/internal/usecase/volumes/service.go @@ -9,11 +9,17 @@ import ( "github.com/bnema/gordon/internal/boundaries/out" "github.com/bnema/gordon/internal/domain" + "github.com/bnema/gordon/internal/usecase/pruneguard" ) // Service implements the VolumeService interface. type Service struct { runtime out.ContainerRuntime + // protection, pruneRuntime, and barrier are the prune ports. All + // three are required for PruneVolumes; a missing port fails closed. + protection out.PruneProtectionStore + pruneRuntime out.PruneRuntime + barrier out.GCBarrier } // NewService creates a new volume service. @@ -21,6 +27,15 @@ func NewService(runtime out.ContainerRuntime) *Service { return &Service{runtime: runtime} } +// WithPrunePorts wires the prune ports: the coherent protection store, +// the runtime inventory/deletion port, and the GC barrier. +func (s *Service) WithPrunePorts(protection out.PruneProtectionStore, pruneRuntime out.PruneRuntime, barrier out.GCBarrier) *Service { + s.protection = protection + s.pruneRuntime = pruneRuntime + s.barrier = barrier + return s +} + // ListVolumes returns all volumes with usage status. func (s *Service) ListVolumes(ctx context.Context) ([]*domain.VolumeInfo, error) { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ @@ -35,8 +50,13 @@ func (s *Service) ListVolumes(ctx context.Context) ([]*domain.VolumeInfo, error) return vols, nil } -// PruneVolumes removes volumes not mounted by any container. -// Returns a report and the list of volumes that were (or would be) removed. +// PruneVolumes plans (and unless dryRun executes) one volume prune run. +// +// A volume is removed only when a durable ownership record explicitly +// records it as released, the runtime labels agree with that record, and +// no container uses it. Every other volume is protected or unknown. +// A valid plan with zero eligible volumes succeeds: with current +// metadata a zero-deletion volume prune is a normal outcome. func (s *Service) PruneVolumes(ctx context.Context, dryRun bool) (*domain.VolumePruneReport, []*domain.VolumeInfo, error) { ctx = zerowrap.CtxWithFields(ctx, map[string]any{ zerowrap.FieldLayer: "usecase", @@ -45,37 +65,87 @@ func (s *Service) PruneVolumes(ctx context.Context, dryRun bool) (*domain.Volume }) log := zerowrap.FromCtx(ctx) - vols, err := s.runtime.ListVolumes(ctx) + lease, err := s.acquireExclusive(ctx) if err != nil { - return nil, nil, fmt.Errorf("failed to list volumes: %w", err) + return nil, nil, fmt.Errorf("volumes: acquire prune lease: %w", err) } + defer lease.Release() - report := &domain.VolumePruneReport{} - var removed []*domain.VolumeInfo + if s.protection == nil || s.pruneRuntime == nil { + return nil, nil, fmt.Errorf("volumes: prune requires a protection store and a runtime inventory port: %w", domain.ErrPruneDisabled) + } - for _, vol := range vols { - if !isGordonManagedVolume(vol) || vol.InUse { - continue - } + snapshot, err := s.protection.ProtectionSnapshot(ctx) + if err != nil { + return nil, nil, fmt.Errorf("volumes: read protection snapshot: %w", err) + } + + inventory, err := s.pruneRuntime.InventoryRuntime(ctx) + if err != nil { + return nil, nil, fmt.Errorf("volumes: read runtime inventory: %w", err) + } - if !dryRun { - if err := s.runtime.RemoveVolume(ctx, vol.Name, false); err != nil { - log.Warn().Err(err).Str("volume", vol.Name).Msg("failed to remove volume, skipping") - continue - } + plan := &domain.PrunePlan{ + Gaps: append(append([]domain.InventoryGap(nil), snapshot.Gaps...), inventory.Gaps...), + Volumes: pruneguard.PlanVolumes(pruneguard.VolumePlanInput{ + Volumes: inventory.Volumes, + Snapshot: snapshot, + Complete: inventory.Complete(), + }), + } + if err := plan.Validate(); err != nil { + return nil, nil, fmt.Errorf("volumes: refusing to execute an invalid prune plan: %w", err) + } + + report := &domain.VolumePruneReport{Plan: *domain.NewPruneReport(plan, !dryRun)} + if dryRun { + return report, nil, nil + } + + byName := make(map[string]*domain.VolumeInfo, len(inventory.Volumes)) + for _, volume := range inventory.Volumes { + if volume != nil { + byName[volume.Name] = volume } + } - removed = append(removed, vol) + var removed []*domain.VolumeInfo + for _, ref := range plan.EligibleVolumes() { + if err := s.pruneRuntime.RemoveVolumeExact(ctx, ref); err != nil { + log.Warn().Err(err).Str("volume", ref.Name).Msg("failed to remove volume, skipping") + report.Plan.Failures = append(report.Plan.Failures, domain.PruneFailure{ + Kind: domain.PruneResourceVolume, Ref: ref.Name, Err: err.Error(), + }) + continue + } + report.Plan.Deleted = append(report.Plan.Deleted, domain.PruneCandidateReport{ + Kind: domain.PruneResourceVolume, Ref: ref.Name, Verdict: domain.PruneVerdictEligible, + }) report.VolumesRemoved++ - report.SpaceReclaimed += vol.Size + if volume, ok := byName[ref.Name]; ok { + report.SpaceReclaimed += volume.Size + report.Plan.ReclaimedBytes += volume.Size + // The driver reported this size for the volume it removed. + report.Plan.ReclaimedKnown = true + removed = append(removed, volume) + continue + } + removed = append(removed, &domain.VolumeInfo{Name: ref.Name}) } return report, removed, nil } -func isGordonManagedVolume(vol *domain.VolumeInfo) bool { - if vol == nil || vol.Labels == nil { - return false +// acquireExclusive takes the GC exclusive lease for the whole prune run, +// or a no-op lease when no barrier is wired. +func (s *Service) acquireExclusive(ctx context.Context) (out.GCLease, error) { + if s.barrier == nil { + return noopGCLease{}, nil } - return vol.Labels[domain.LabelManaged] == "true" + return s.barrier.AcquireExclusive(ctx) } + +// noopGCLease is the lease used when no barrier is wired. +type noopGCLease struct{} + +func (noopGCLease) Release() {} diff --git a/internal/usecase/volumes/service_test.go b/internal/usecase/volumes/service_test.go index c9039bdf3..0f2550500 100644 --- a/internal/usecase/volumes/service_test.go +++ b/internal/usecase/volumes/service_test.go @@ -6,12 +6,54 @@ import ( "testing" "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" "github.com/stretchr/testify/require" outmocks "github.com/bnema/gordon/internal/boundaries/out/mocks" "github.com/bnema/gordon/internal/domain" ) +// appVolumeLabels are the runtime labels a stamped app volume carries. +func appVolumeLabels(app, service string) map[string]string { + return map[string]string{ + domain.LabelManaged: "true", + domain.LabelApp: app, + domain.LabelAppID: app + "-uuid", + domain.LabelAppService: service, + } +} + +// newPruneService wires a volume service against a scripted snapshot and +// runtime inventory. +func newPruneService( + t *testing.T, + volumes []*domain.VolumeInfo, + snapshot *domain.PruneProtectionSnapshot, + inventoryGaps ...domain.InventoryGap, +) (*Service, *outmocks.MockContainerRuntime, *outmocks.MockPruneRuntime, *outmocks.MockPruneProtectionStore, *outmocks.MockGCBarrier) { + t.Helper() + runtime := outmocks.NewMockContainerRuntime(t) + protection := outmocks.NewMockPruneProtectionStore(t) + pruneRuntime := outmocks.NewMockPruneRuntime(t) + barrier := outmocks.NewMockGCBarrier(t) + + if snapshot == nil { + snapshot = &domain.PruneProtectionSnapshot{} + } + protection.EXPECT().ProtectionSnapshot(context.Background()).Return(snapshot, nil) + pruneRuntime.EXPECT().InventoryRuntime(context.Background()).Return(&domain.RuntimeInventory{ + Volumes: volumes, + Gaps: inventoryGaps, + }, nil) + + lease := outmocks.NewMockGCLease(t) + barrier.EXPECT().AcquireExclusive(context.Background()).Return(lease, nil) + lease.EXPECT().Release() + + svc := NewService(runtime).WithPrunePorts(protection, pruneRuntime, barrier) + return svc, runtime, pruneRuntime, protection, barrier +} + func TestService_ListVolumes(t *testing.T) { runtime := outmocks.NewMockContainerRuntime(t) @@ -27,123 +69,174 @@ func TestService_ListVolumes(t *testing.T) { assert.Len(t, result, 2) } -func TestService_PruneVolumes_RemovesOrphaned(t *testing.T) { +func TestService_ListVolumes_PropagatesError(t *testing.T) { runtime := outmocks.NewMockContainerRuntime(t) + runtime.EXPECT().ListVolumes(context.Background()).Return(nil, fmt.Errorf("connection refused")) - vols := []*domain.VolumeInfo{ - {Name: "vol-in-use", InUse: true, Containers: []string{"web"}, Labels: map[string]string{domain.LabelManaged: "true"}}, - {Name: "vol-orphaned", InUse: false, Size: 1024, Labels: map[string]string{domain.LabelManaged: "true"}}, + svc := NewService(runtime) + _, err := svc.ListVolumes(context.Background()) + require.Error(t, err) + assert.Contains(t, err.Error(), "connection refused") +} + +// TestService_PruneVolumes_RemovesOnlyExplicitlyReleasedVolumes is the +// central safety case: managed, app-labelled, and legacy volumes all +// survive; only a released, label-agreeing, unused volume is removed. +func TestService_PruneVolumes_RemovesOnlyExplicitlyReleasedVolumes(t *testing.T) { + const released = "gordon-shop--web--vol--data" + volumes := []*domain.VolumeInfo{ + {Name: released, Size: 1024, Labels: appVolumeLabels("shop", "web")}, + {Name: "gordon-shop--api--vol--data", Size: 200, Labels: appVolumeLabels("shop", "api")}, + {Name: "gordon-old--web--vol--data", Size: 300, Labels: appVolumeLabels("old", "web")}, + {Name: "gordon-legacy", Size: 400, Labels: map[string]string{domain.LabelManaged: "true"}}, + {Name: "pgdata", Size: 500}, + {Name: "gordon-shop--cache--vol--data", Size: 600, Labels: appVolumeLabels("shop", "web"), InUse: true}, } - runtime.EXPECT().ListVolumes(context.Background()).Return(vols, nil) - runtime.EXPECT().RemoveVolume(context.Background(), "vol-orphaned", false).Return(nil) + snapshot := &domain.PruneProtectionSnapshot{VolumeClaims: []domain.VolumeClaim{ + {Name: released, App: "shop", AppID: "shop-uuid", Service: "web", State: domain.VolumeClaimReleased}, + {Name: "gordon-shop--api--vol--data", App: "shop", AppID: "shop-uuid", Service: "api", State: domain.VolumeClaimAttached}, + {Name: "gordon-old--web--vol--data", App: "old", AppID: "old-uuid", Service: "web", State: domain.VolumeClaimRetained}, + {Name: "gordon-shop--cache--vol--data", App: "shop", AppID: "shop-uuid", Service: "web", State: domain.VolumeClaimAttached}, + }} + + svc, _, pruneRuntime, _, _ := newPruneService(t, volumes, snapshot) + pruneRuntime.EXPECT().RemoveVolumeExact(context.Background(), domain.RuntimeVolumeRef{Name: released}).Return(nil) - svc := NewService(runtime) report, removed, err := svc.PruneVolumes(context.Background(), false) require.NoError(t, err) assert.Equal(t, 1, report.VolumesRemoved) assert.Equal(t, int64(1024), report.SpaceReclaimed) - assert.Len(t, removed, 1) - assert.Equal(t, "vol-orphaned", removed[0].Name) -} + require.Len(t, removed, 1) + assert.Equal(t, released, removed[0].Name) + assert.True(t, report.Plan.Applied) -func TestService_PruneVolumes_DryRunDoesNotRemove(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) + protected := report.Plan.CountByVerdict(domain.PruneVerdictProtected) + assert.GreaterOrEqual(t, protected, 4, "attached, retained, legacy, unknown and in-use volumes must be protected") +} - vols := []*domain.VolumeInfo{ - {Name: "vol-orphaned", InUse: false, Size: 2048, Labels: map[string]string{domain.LabelManaged: "true"}}, +func TestService_PruneVolumes_UnknownAndContradictoryVolumesSurvive(t *testing.T) { + volumes := []*domain.VolumeInfo{ + // App labels but no durable record: ownership history missing. + {Name: "gordon-ghost--web--vol--data", Labels: appVolumeLabels("ghost", "web")}, + // Durable released record but contradicting labels. + {Name: "gordon-shop--web--vol--other", Labels: appVolumeLabels("other", "web")}, } - runtime.EXPECT().ListVolumes(context.Background()).Return(vols, nil) + snapshot := &domain.PruneProtectionSnapshot{VolumeClaims: []domain.VolumeClaim{ + {Name: "gordon-shop--web--vol--other", App: "shop", AppID: "shop-uuid", Service: "web", State: domain.VolumeClaimReleased}, + }} - svc := NewService(runtime) - report, removed, err := svc.PruneVolumes(context.Background(), true) + svc, _, pruneRuntime, _, _ := newPruneService(t, volumes, snapshot) + pruneRuntime.EXPECT().RemoveVolumeExact(mock.Anything, mock.Anything).Maybe().Return(nil) + + report, removed, err := svc.PruneVolumes(context.Background(), false) require.NoError(t, err) - assert.Equal(t, 1, report.VolumesRemoved) - assert.Equal(t, int64(2048), report.SpaceReclaimed) - assert.Len(t, removed, 1) + assert.Zero(t, report.VolumesRemoved) + assert.Empty(t, removed) + assert.Equal(t, 2, report.Plan.CountByVerdict(domain.PruneVerdictUnknown)) } -func TestService_PruneVolumes_OnlyRemovesUnusedGordonManagedVolumes(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) +func TestService_PruneVolumes_DryRunPlansWithoutDeleting(t *testing.T) { + const released = "gordon-shop--web--vol--data" + volumes := []*domain.VolumeInfo{{Name: released, Size: 2048, Labels: appVolumeLabels("shop", "web")}} + snapshot := &domain.PruneProtectionSnapshot{VolumeClaims: []domain.VolumeClaim{ + {Name: released, App: "shop", AppID: "shop-uuid", Service: "web", State: domain.VolumeClaimReleased}, + }} - vols := []*domain.VolumeInfo{ - {Name: "gordon-used", InUse: true, Size: 100, Labels: map[string]string{domain.LabelManaged: "true"}}, - {Name: "gordon-unused", InUse: false, Size: 200, Labels: map[string]string{domain.LabelManaged: "true"}}, - {Name: "other-unused", InUse: false, Size: 300}, - {Name: "other-labelled-false", InUse: false, Size: 400, Labels: map[string]string{domain.LabelManaged: "false"}}, - {Name: "other-used", InUse: true, Size: 500}, - } - runtime.EXPECT().ListVolumes(context.Background()).Return(vols, nil) - runtime.EXPECT().RemoveVolume(context.Background(), "gordon-unused", false).Return(nil) + svc, _, pruneRuntime, _, _ := newPruneService(t, volumes, snapshot) + pruneRuntime.EXPECT().RemoveVolumeExact(mock.Anything, mock.Anything).Maybe().Return(nil) - svc := NewService(runtime) - report, removed, err := svc.PruneVolumes(context.Background(), false) + report, removed, err := svc.PruneVolumes(context.Background(), true) require.NoError(t, err) - assert.Equal(t, 1, report.VolumesRemoved) - assert.Equal(t, int64(200), report.SpaceReclaimed) - require.Len(t, removed, 1) - assert.Equal(t, "gordon-unused", removed[0].Name) + assert.Zero(t, report.VolumesRemoved) + assert.Empty(t, removed, "a dry run must report only what would be removed") + assert.False(t, report.Plan.Applied) + assert.Equal(t, 1, report.Plan.CountByVerdict(domain.PruneVerdictEligible)) + + // The dry-run plan and the executed plan must agree. + executed, _, err := svc.PruneVolumes(context.Background(), false) + require.NoError(t, err) + assert.Equal(t, report.Plan.Candidates, executed.Plan.Candidates) } -func TestService_PruneVolumes_DryRunReportsOnlyUnusedGordonManagedVolumes(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) +func TestService_PruneVolumes_ZeroDeletionSucceeds(t *testing.T) { + volumes := []*domain.VolumeInfo{ + {Name: "pgdata"}, + {Name: "gordon-shop--web--vol--data", Labels: appVolumeLabels("shop", "web"), InUse: true}, + } + svc, _, _, _, _ := newPruneService(t, volumes, nil) - vols := []*domain.VolumeInfo{ - {Name: "gordon-used", InUse: true, Size: 100, Labels: map[string]string{domain.LabelManaged: "true"}}, - {Name: "gordon-unused", InUse: false, Size: 200, Labels: map[string]string{domain.LabelManaged: "true"}}, - {Name: "other-unused", InUse: false, Size: 300}, + report, removed, err := svc.PruneVolumes(context.Background(), false) + require.NoError(t, err, "with current metadata a zero-deletion prune is valid") + assert.Zero(t, report.VolumesRemoved) + assert.Empty(t, removed) + assert.False(t, report.Plan.ReclaimedKnown) +} + +func TestService_PruneVolumes_FailedRemovalIsIsolated(t *testing.T) { + first := "gordon-a--web--vol--data" + second := "gordon-b--web--vol--data" + volumes := []*domain.VolumeInfo{ + {Name: first, Size: 1024, Labels: appVolumeLabels("a", "web")}, + {Name: second, Size: 2048, Labels: appVolumeLabels("b", "web")}, } - runtime.EXPECT().ListVolumes(context.Background()).Return(vols, nil) + snapshot := &domain.PruneProtectionSnapshot{VolumeClaims: []domain.VolumeClaim{ + {Name: first, App: "a", AppID: "a-uuid", Service: "web", State: domain.VolumeClaimReleased}, + {Name: second, App: "b", AppID: "b-uuid", Service: "web", State: domain.VolumeClaimReleased}, + }} - svc := NewService(runtime) - report, removed, err := svc.PruneVolumes(context.Background(), true) + svc, _, pruneRuntime, _, _ := newPruneService(t, volumes, snapshot) + pruneRuntime.EXPECT().RemoveVolumeExact(context.Background(), domain.RuntimeVolumeRef{Name: first}).Return(fmt.Errorf("volume in use")) + pruneRuntime.EXPECT().RemoveVolumeExact(context.Background(), domain.RuntimeVolumeRef{Name: second}).Return(nil) + + report, removed, err := svc.PruneVolumes(context.Background(), false) require.NoError(t, err) assert.Equal(t, 1, report.VolumesRemoved) - assert.Equal(t, int64(200), report.SpaceReclaimed) + assert.Equal(t, int64(2048), report.SpaceReclaimed) require.Len(t, removed, 1) - assert.Equal(t, "gordon-unused", removed[0].Name) + assert.Equal(t, second, removed[0].Name) + require.Len(t, report.Plan.Failures, 1) + assert.Equal(t, first, report.Plan.Failures[0].Ref) } -func TestService_PruneVolumes_AllInUse(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - - vols := []*domain.VolumeInfo{ - {Name: "vol1", InUse: true, Containers: []string{"web"}}, +func TestService_PruneVolumes_IncompleteInventoryProtectsEverything(t *testing.T) { + const released = "gordon-shop--web--vol--data" + volumes := []*domain.VolumeInfo{{Name: released, Labels: appVolumeLabels("shop", "web")}} + snapshot := &domain.PruneProtectionSnapshot{VolumeClaims: []domain.VolumeClaim{ + {Name: released, App: "shop", AppID: "shop-uuid", Service: "web", State: domain.VolumeClaimReleased}, + }} + gap := domain.InventoryGap{ + Source: domain.InventorySourceRuntimeContainers, + Reason: domain.PruneReasonUnknownContainerUse, + Detail: "list containers failed", } - runtime.EXPECT().ListVolumes(context.Background()).Return(vols, nil) - svc := NewService(runtime) + svc, _, _, _, _ := newPruneService(t, volumes, snapshot, gap) report, removed, err := svc.PruneVolumes(context.Background(), false) require.NoError(t, err) - assert.Equal(t, 0, report.VolumesRemoved) assert.Empty(t, removed) + assert.Equal(t, 1, report.Plan.CountByVerdict(domain.PruneVerdictUnknown)) + require.Len(t, report.Plan.Gaps, 1) } -func TestService_ListVolumes_PropagatesError(t *testing.T) { - runtime := outmocks.NewMockContainerRuntime(t) - runtime.EXPECT().ListVolumes(context.Background()).Return(nil, fmt.Errorf("connection refused")) - - svc := NewService(runtime) - _, err := svc.ListVolumes(context.Background()) +func TestService_PruneVolumes_RequiresPrunePorts(t *testing.T) { + svc := NewService(outmocks.NewMockContainerRuntime(t)) + _, _, err := svc.PruneVolumes(context.Background(), false) require.Error(t, err) - assert.Contains(t, err.Error(), "connection refused") + assert.ErrorIs(t, err, domain.ErrPruneDisabled) } -func TestService_PruneVolumes_SkipsFailedRemoval(t *testing.T) { +func TestService_PruneVolumes_ProtectionSnapshotFailureFailsClosed(t *testing.T) { runtime := outmocks.NewMockContainerRuntime(t) - - vols := []*domain.VolumeInfo{ - {Name: "vol-fail", InUse: false, Size: 1024, Labels: map[string]string{domain.LabelManaged: "true"}}, - {Name: "vol-ok", InUse: false, Size: 2048, Labels: map[string]string{domain.LabelManaged: "true"}}, - } - runtime.EXPECT().ListVolumes(context.Background()).Return(vols, nil) - runtime.EXPECT().RemoveVolume(context.Background(), "vol-fail", false).Return(fmt.Errorf("volume in use")) - runtime.EXPECT().RemoveVolume(context.Background(), "vol-ok", false).Return(nil) - - svc := NewService(runtime) - report, removed, err := svc.PruneVolumes(context.Background(), false) - require.NoError(t, err) - assert.Equal(t, 1, report.VolumesRemoved) - assert.Equal(t, int64(2048), report.SpaceReclaimed) - assert.Len(t, removed, 1) - assert.Equal(t, "vol-ok", removed[0].Name) + protection := outmocks.NewMockPruneProtectionStore(t) + pruneRuntime := outmocks.NewMockPruneRuntime(t) + barrier := outmocks.NewMockGCBarrier(t) + lease := outmocks.NewMockGCLease(t) + barrier.EXPECT().AcquireExclusive(context.Background()).Return(lease, nil) + lease.EXPECT().Release() + protection.EXPECT().ProtectionSnapshot(context.Background()).Return(nil, fmt.Errorf("unreadable state")) + + svc := NewService(runtime).WithPrunePorts(protection, pruneRuntime, barrier) + _, _, err := svc.PruneVolumes(context.Background(), false) + require.Error(t, err) + assert.Contains(t, err.Error(), "protection snapshot") } diff --git a/main.go b/main.go index d28d12add..ad6011cae 100644 --- a/main.go +++ b/main.go @@ -12,12 +12,13 @@ var ( version = "dev" commit = "unknown" date = "unknown" + dirty = "unknown" ) func main() { // Set version information globally and for the CLI - versionpkg.Set(version, commit, date) - cli.SetVersionInfo(version, commit, date) + versionpkg.Set(version, commit, date, dirty) + cli.SetVersionInfo(version, commit, date, dirty) if err := cli.NewRootCmd().Execute(); err != nil { os.Exit(1) diff --git a/pkg/registrypush/push.go b/pkg/registrypush/push.go index 0e39e5664..5c3513fe2 100644 --- a/pkg/registrypush/push.go +++ b/pkg/registrypush/push.go @@ -39,6 +39,7 @@ type Pusher struct { timeout time.Duration transport http.RoundTripper insecureTLS bool + plainHTTP bool imageSource func(ctx context.Context, ref string) (v1.Image, func(), error) cleanupFn func() progress io.Writer @@ -69,6 +70,13 @@ func WithInsecureTLS(insecure bool) Option { return func(p *Pusher) { p.insecureTLS = insecure } } +// WithPlainHTTP forces the registry base URL to use http:// instead of +// https://. This is independent from WithInsecureTLS, which only disables +// TLS certificate verification while still using https://. +func WithPlainHTTP(plain bool) Option { + return func(p *Pusher) { p.plainHTTP = plain } +} + // WithImageSource overrides the default daemon-based image reader. func WithImageSource(src ImageSource) Option { return func(p *Pusher) { @@ -162,7 +170,7 @@ func (p *Pusher) Push(ctx context.Context, ref string) error { defer userCleanup() } - baseURL := registryBaseURL(parsedRef.Context().RegistryStr()) + baseURL := p.registryBaseURL(parsedRef.Context().RegistryStr()) repo := parsedRef.Context().RepositoryStr() if err := p.uploadImageLayers(ctx, img, baseURL, repo, authHeader); err != nil { @@ -383,7 +391,10 @@ func resolveAuthHeader(ref name.Reference) (string, error) { return "", nil } -func registryBaseURL(host string) string { +func (p *Pusher) registryBaseURL(host string) string { + if p.plainHTTP { + return "http://" + host + } if strings.HasPrefix(host, "localhost:") || strings.HasPrefix(host, "127.0.0.1:") || host == "localhost" || host == "127.0.0.1" { return "http://" + host } diff --git a/pkg/registrypush/push_test.go b/pkg/registrypush/push_test.go index 7adb95d4e..05dc16fa6 100644 --- a/pkg/registrypush/push_test.go +++ b/pkg/registrypush/push_test.go @@ -124,6 +124,24 @@ func TestPusher_UploadBlob(t *testing.T) { } } +func TestPusher_PlainHTTP(t *testing.T) { + img, err := random.Image(1024, 1) + require.NoError(t, err) + + server := httptest.NewServer(newFakePushRegistry(t, img, map[string]bool{}).handler()) + defer server.Close() + serverURL, err := url.Parse(server.URL) + require.NoError(t, err) + + p := registrypush.New( + registrypush.WithPlainHTTP(true), + registrypush.WithImageSource(func(context.Context, string) (v1.Image, error) { + return img, nil + }), + ) + require.NoError(t, p.Push(context.Background(), serverURL.Host+"/demo/app:v1")) +} + func TestPusher_Push(t *testing.T) { testCases := []struct { name string diff --git a/pkg/runtime/interface.go b/pkg/runtime/interface.go index 09c1d5c87..73a2d40d3 100644 --- a/pkg/runtime/interface.go +++ b/pkg/runtime/interface.go @@ -41,14 +41,6 @@ type ContainerConfig struct { Aliases []string // Additional network aliases } -// PruneReport represents the result of an image prune operation. -type PruneReport struct { - // DeletedIDs contains runtime-provided image identifiers removed by prune. - DeletedIDs []string - // SpaceReclaimed is the number of bytes reclaimed by prune. - SpaceReclaimed int64 -} - // ImageDetail represents detailed metadata for an image. type ImageDetail struct { // ID is the runtime image identifier (for example, image ID/digest). @@ -66,9 +58,7 @@ type Runtime interface { // Container lifecycle CreateContainer(ctx context.Context, config *ContainerConfig) (*Container, error) StartContainer(ctx context.Context, containerID string) error - StopContainer(ctx context.Context, containerID string) error WaitForContainer(ctx context.Context, containerID string) error - RestartContainer(ctx context.Context, containerID string) error RemoveContainer(ctx context.Context, containerID string, force bool) error // Container inspection @@ -83,9 +73,6 @@ type Runtime interface { ListImages(ctx context.Context) ([]string, error) // ListImagesDetailed returns metadata for all images visible to the runtime. ListImagesDetailed(ctx context.Context) ([]ImageDetail, error) - // PruneImages removes images eligible for prune; when danglingOnly is true, - // only dangling images are pruned. - PruneImages(ctx context.Context, danglingOnly bool) (PruneReport, error) // Runtime information Ping(ctx context.Context) error @@ -103,7 +90,10 @@ type Runtime interface { // Volume management InspectImageVolumes(ctx context.Context, imageRef string) ([]string, error) VolumeExists(ctx context.Context, volumeName string) (bool, error) - CreateVolume(ctx context.Context, volumeName string) error + // CreateVolume creates one named volume with the given labels. + // Labels carry ownership provenance; a caller that omits them + // creates an unmanaged volume that prune never adopts. + CreateVolume(ctx context.Context, volumeName string, labels map[string]string) error RemoveVolume(ctx context.Context, volumeName string, force bool) error // Environment inspection diff --git a/pkg/validation/image.go b/pkg/validation/image.go index 318273938..fae69d5b0 100644 --- a/pkg/validation/image.go +++ b/pkg/validation/image.go @@ -1,6 +1,26 @@ package validation -import "strings" +import ( + "fmt" + "strings" +) + +const sha256DigestLength = len("sha256:") + 64 + +// ValidateImageDigest validates the only digest format accepted for image +// references and resolved deployment pins: sha256 followed by exactly 64 +// lowercase hexadecimal characters. +func ValidateImageDigest(digest string) error { + if len(digest) != sha256DigestLength || !strings.HasPrefix(digest, "sha256:") { + return fmt.Errorf("image digest must be sha256:<64 hex chars>") + } + for _, c := range digest[len("sha256:"):] { + if (c < '0' || c > '9') && (c < 'a' || c > 'f') { + return fmt.Errorf("image digest must be sha256:<64 hex chars>") + } + } + return nil +} // ParseImageReference parses an image reference into name and tag/digest. // Supports formats: diff --git a/pkg/validation/image_test.go b/pkg/validation/image_test.go index 180e664bf..256bee146 100644 --- a/pkg/validation/image_test.go +++ b/pkg/validation/image_test.go @@ -1,11 +1,29 @@ package validation import ( + "strings" "testing" "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" ) +func TestValidateImageDigest(t *testing.T) { + validLower := "sha256:" + strings.Repeat("a", 64) + require.NoError(t, ValidateImageDigest(validLower)) + + for _, digest := range []string{ + "sha256:" + strings.Repeat("A", 64), + "", + "sha256:" + strings.Repeat("a", 63), + "sha256:" + strings.Repeat("a", 65), + "sha256:" + strings.Repeat("g", 64), + "sha512:" + strings.Repeat("a", 64), + } { + assert.Error(t, ValidateImageDigest(digest), digest) + } +} + func TestParseImageReference(t *testing.T) { tests := []struct { name string diff --git a/pkg/version/version.go b/pkg/version/version.go index 289934d9b..d0c4e993d 100644 --- a/pkg/version/version.go +++ b/pkg/version/version.go @@ -1,5 +1,6 @@ // Package version holds build-time version info for Gordon. -// Set via main using Set(), read from anywhere via Version(), Commit(), or BuildDate(). +// Set via main using Set(), read from anywhere via Version(), Commit(), +// BuildDate(), or Dirty(). package version // Build information, populated by Set() at startup. @@ -7,13 +8,15 @@ var ( version = "dev" commit = "unknown" buildDate = "unknown" + dirty = "unknown" ) // Set stores build-time version info. Call once from main. -func Set(v, c, d string) { +func Set(v, c, d, isDirty string) { version = v commit = c buildDate = d + dirty = isDirty } // Version returns the build version string. @@ -24,3 +27,6 @@ func Commit() string { return commit } // BuildDate returns the build date string. func BuildDate() string { return buildDate } + +// Dirty reports whether the source checkout had uncommitted changes. +func Dirty() string { return dirty } diff --git a/pkg/version/version_test.go b/pkg/version/version_test.go new file mode 100644 index 000000000..8a87a5c09 --- /dev/null +++ b/pkg/version/version_test.go @@ -0,0 +1,23 @@ +package version + +import "testing" + +func TestSetStoresAllBuildInfo(t *testing.T) { + previousVersion, previousCommit, previousDate, previousDirty := Version(), Commit(), BuildDate(), Dirty() + t.Cleanup(func() { Set(previousVersion, previousCommit, previousDate, previousDirty) }) + + Set("v1.2.3", "abc1234", "2026-09-13", "true") + + if got := Version(); got != "v1.2.3" { + t.Errorf("Version() = %q, want %q", got, "v1.2.3") + } + if got := Commit(); got != "abc1234" { + t.Errorf("Commit() = %q, want %q", got, "abc1234") + } + if got := BuildDate(); got != "2026-09-13" { + t.Errorf("BuildDate() = %q, want %q", got, "2026-09-13") + } + if got := Dirty(); got != "true" { + t.Errorf("Dirty() = %q, want %q", got, "true") + } +} diff --git a/tests/install_next_test.sh b/tests/install_next_test.sh new file mode 100644 index 000000000..3ec492897 --- /dev/null +++ b/tests/install_next_test.sh @@ -0,0 +1,254 @@ +#!/bin/sh +set -eu + +ROOT=$(CDPATH='' cd -- "$(dirname -- "$0")/.." && pwd) +PASS=0 +FAIL=0 +SHA=0123456789abcdef0123456789abcdef01234567 + +fail() { + echo "not ok - $1" + FAIL=$((FAIL + 1)) +} + +pass() { + echo "ok - $1" + PASS=$((PASS + 1)) +} + +run_case() { + shift + case_dir=$(mktemp -d) + mkdir -p "$case_dir/bin" "$case_dir/home" "$case_dir/install" + + cat >"$case_dir/bin/uname" <<'EOF' +#!/bin/sh +[ "$1" = "-s" ] && echo Linux || echo x86_64 +EOF + cat >"$case_dir/bin/id" <<'EOF' +#!/bin/sh +[ "$1" = "-u" ] && { echo 1000; exit 0; } +exec /usr/bin/id "$@" +EOF + cat >"$case_dir/bin/curl" <>'$case_dir/curl.log' +case "\$*" in + *api.github.com/repos/bnema/gordon/commits/next*) printf '%s\n' '{"sha":"$SHA"}' ;; + *codeload.github.com/bnema/gordon/tar.gz/$SHA*) : >'$case_dir/source.tar.gz' ;; + *) exit 90 ;; +esac +EOF + cat >"$case_dir/bin/tar" <>'$case_dir/tar.log' +case "\$1" in + -tzf) printf '%s\n' 'gordon-$SHA/' 'gordon-$SHA/go.mod' 'gordon-$SHA/main.go' ;; + -tvzf) printf '%s\n' 'drwxr-xr-x root/root 0 date gordon-$SHA/' '-rw-r--r-- root/root 0 date gordon-$SHA/go.mod' '-rw-r--r-- root/root 0 date gordon-$SHA/main.go' ;; + -xzf) + while [ "\$#" -gt 0 ]; do + if [ "\$1" = -C ]; then + shift + mkdir -p "\$1/gordon-$SHA" + printf 'module example\n\ngo 1.27\n' >"\$1/gordon-$SHA/go.mod" + : >"\$1/gordon-$SHA/main.go" + exit 0 + fi + shift + done + ;; +esac +EOF + cat >"$case_dir/bin/go" <>'$case_dir/go.log' +if [ "\$1 \$2" = 'env GOVERSION' ]; then echo go1.27.1; exit 0; fi +if [ "\$1" = build ]; then + while [ "\$#" -gt 0 ]; do [ "\$1" = -o ] && { shift; : >"\$1"; chmod 755 "\$1"; exit 0; }; shift; done +fi +exit 91 +EOF + cat >"$case_dir/bin/install" <>'$case_dir/install.log' +cp "\$3" "\$4" +EOF + cat >"$case_dir/bin/mv" <>'$case_dir/mv.log' +[ "\$1" = -f ] && shift +/bin/mv "\$1" "\$2" +EOF + chmod +x "$case_dir/bin/"* + mkdir -p "$case_dir/tmp" + + if (PATH="$case_dir/bin:/usr/bin:/bin" HOME="$case_dir/home" SHELL=/bin/bash TMPDIR="$case_dir/tmp" GORDON_UPDATE_PATH=0 "$@") >"$case_dir/out" 2>&1; then + status=0 + else + status=$? + fi +} + +test_next_build() { + run_case next-build env GORDON_CHANNEL=next sh "$ROOT/install.sh" + [ "$status" -eq 0 ] || { cat "$case_dir/out"; fail "next channel builds pinned source"; return; } + if grep -q "commits/next" "$case_dir/curl.log" && + grep -q "codeload.github.com/bnema/gordon/tar.gz/$SHA" "$case_dir/curl.log" && + grep -q -- "-X main.version=next-$SHA" "$case_dir/go.log" && + grep -q "GOOS=linux GOARCH=amd64 GOTOOLCHAIN=local" "$case_dir/go.log" && + grep -q "UNVERIFIED DEVELOPMENT BUILD" "$case_dir/out" && + [ -x "$case_dir/home/.local/bin/gordon" ]; then + pass "next channel builds pinned source" + else + fail "next channel builds pinned source" + fi +} + +test_conflicts() { + run_case version-conflict env GORDON_CHANNEL=next GORDON_VERSION=v1.2.3 sh "$ROOT/install.sh" + if [ "$status" -ne 0 ] && grep -q "GORDON_VERSION cannot be used" "$case_dir/out"; then + pass "next rejects GORDON_VERSION" + else + fail "next rejects GORDON_VERSION" + fi + + run_case prerelease-conflict env GORDON_CHANNEL=next GORDON_PRERELEASE=1 sh "$ROOT/install.sh" + if [ "$status" -ne 0 ] && grep -q "GORDON_PRERELEASE cannot be used" "$case_dir/out"; then + pass "next rejects GORDON_PRERELEASE" + else + fail "next rejects GORDON_PRERELEASE" + fi +} + +test_go_version() { + run_case old-go env GORDON_CHANNEL=next MOCK_GO_VERSION=unused sh "$ROOT/install.sh" + # Replace the mock after setup, then run once more in the same fixture. + cat >"$case_dir/bin/go" <<'EOF' +#!/bin/sh +[ "$1 $2" = 'env GOVERSION' ] && { echo go1.26.9; exit 0; } +exit 91 +EOF + chmod +x "$case_dir/bin/go" + if (PATH="$case_dir/bin:/usr/bin:/bin" HOME="$case_dir/home" SHELL=/bin/bash TMPDIR="$case_dir/tmp" GORDON_UPDATE_PATH=0 GORDON_CHANNEL=next sh "$ROOT/install.sh") >"$case_dir/out" 2>&1; then status=0; else status=$?; fi + if [ "$status" -ne 0 ] && grep -q "requires Go 1.27 or newer" "$case_dir/out"; then + pass "next enforces go.mod Go version" + else + fail "next enforces go.mod Go version" + fi +} + +test_default_and_path_modes() { + run_case default-dir env GORDON_CHANNEL=next sh "$ROOT/install.sh" + if [ "$status" -eq 0 ] && [ -x "$case_dir/home/.local/bin/gordon" ] && + grep -q "export PATH='${case_dir}/home/.local/bin':\$PATH" "$case_dir/out"; then + pass "default install is user-local and prints current-shell instruction" + else + fail "default install is user-local and prints current-shell instruction" + fi + + run_case path-present env GORDON_CHANNEL=next sh "$ROOT/install.sh" + if (PATH="$case_dir/bin:$case_dir/home/.local/bin:/usr/bin:/bin" HOME="$case_dir/home" SHELL=/bin/bash TMPDIR="$case_dir/tmp" GORDON_UPDATE_PATH=0 GORDON_CHANNEL=next sh "$ROOT/install.sh") >"$case_dir/out" 2>&1; then status=0; else status=$?; fi + if [ "$status" -eq 0 ] && ! grep -q "Add Gordon to the current shell" "$case_dir/out"; then + pass "existing PATH needs no update" + else + fail "existing PATH needs no update" + fi + + run_case forced-no env GORDON_CHANNEL=next GORDON_UPDATE_PATH=0 sh "$ROOT/install.sh" + if [ "$status" -eq 0 ] && [ ! -e "$case_dir/home/.bashrc" ]; then + pass "forced no does not modify shell config" + else + fail "forced no does not modify shell config" + fi + + run_case invalid-update env GORDON_CHANNEL=next GORDON_UPDATE_PATH=maybe sh "$ROOT/install.sh" + if [ "$status" -ne 0 ] && grep -q "must be 0 or 1" "$case_dir/out"; then + pass "invalid PATH mode is rejected" + else + fail "invalid PATH mode is rejected" + fi + + run_case unsafe-dir env GORDON_CHANNEL=next GORDON_INSTALL_DIR="/tmp/unsafe +path" sh "$ROOT/install.sh" + if [ "$status" -ne 0 ] && grep -q "unsafe control character" "$case_dir/out"; then + pass "control characters in install directory are rejected" + else + fail "control characters in install directory are rejected" + fi + + run_case trailing-newline env GORDON_CHANNEL=next GORDON_INSTALL_DIR="/tmp/unsafe +" sh "$ROOT/install.sh" + if [ "$status" -ne 0 ] && grep -q "unsafe control character" "$case_dir/out"; then + pass "trailing newline in install directory is rejected" + else + fail "trailing newline in install directory is rejected" + fi +} + +test_shell_updates() { + for shell in bash zsh fish; do + run_case "shell-$shell" env GORDON_CHANNEL=next GORDON_UPDATE_PATH=1 SHELL="/usr/bin/$shell" sh "$ROOT/install.sh" + case "$shell" in + bash) config=$case_dir/home/.bashrc; expected="export PATH='${case_dir}/home/.local/bin':\$PATH" ;; + zsh) config=$case_dir/home/.zshrc; expected="export PATH='${case_dir}/home/.local/bin':\$PATH" ;; + fish) config=$case_dir/home/.config/fish/config.fish; expected="fish_add_path '${case_dir}/home/.local/bin'" ;; + esac + if [ "$status" -eq 0 ] && grep -Fq "$expected" "$config"; then + pass "$shell PATH update" + else + fail "$shell PATH update" + fi + done + + run_case idempotent env GORDON_CHANNEL=next GORDON_UPDATE_PATH=1 SHELL=/bin/bash sh "$ROOT/install.sh" + first_case=$case_dir + if (PATH="$case_dir/bin:/usr/bin:/bin" HOME="$case_dir/home" SHELL=/bin/bash TMPDIR="$case_dir/tmp" GORDON_UPDATE_PATH=1 GORDON_CHANNEL=next sh "$ROOT/install.sh") >>"$case_dir/out" 2>&1 && + [ "$(grep -c '^# >>> Gordon installer PATH >>>$' "$case_dir/home/.bashrc")" -eq 1 ]; then + pass "PATH marker block is idempotent" + else + cat "$first_case/out" + fail "PATH marker block is idempotent" + fi + + run_case unsupported env GORDON_CHANNEL=next GORDON_UPDATE_PATH=1 SHELL=/bin/ksh sh "$ROOT/install.sh" + if [ "$status" -eq 0 ] && grep -q "unsupported or missing SHELL" "$case_dir/out"; then + pass "unsupported shell gets manual instructions" + else + fail "unsupported shell gets manual instructions" + fi +} + +test_explicit_global_override() { + run_case home-bin env GORDON_CHANNEL=next sh "$ROOT/install.sh" + if (PATH="$case_dir/bin:/usr/bin:/bin" HOME="$case_dir/home" SHELL=/bin/bash TMPDIR="$case_dir/tmp" GORDON_UPDATE_PATH=1 GORDON_INSTALL_DIR="$case_dir/home/bin" GORDON_CHANNEL=next sh "$ROOT/install.sh") >"$case_dir/out" 2>&1; then status=0; else status=$?; fi + if [ "$status" -eq 0 ] && [ -x "$case_dir/home/bin/gordon" ] && + grep -Fq "export PATH='${case_dir}/home/bin':\$PATH" "$case_dir/home/.bashrc"; then + pass "HOME bin override is installed and added to PATH" + else + fail "HOME bin override is installed and added to PATH" + fi + + run_case global env GORDON_CHANNEL=next sh "$ROOT/install.sh" + if (PATH="$case_dir/bin:/usr/bin:/bin" HOME="$case_dir/home" SHELL=/bin/bash TMPDIR="$case_dir/tmp" GORDON_UPDATE_PATH=0 GORDON_INSTALL_DIR="$case_dir/install" GORDON_CHANNEL=next sh "$ROOT/install.sh") >"$case_dir/out" 2>&1; then status=0; else status=$?; fi + if [ "$status" -eq 0 ] && [ -x "$case_dir/install/gordon" ]; then + pass "explicit global install override is preserved" + else + fail "explicit global install override is preserved" + fi + + run_case sudo-default env SUDO_USER=test-user GORDON_CHANNEL=next sh "$ROOT/install.sh" + if [ "$status" -ne 0 ] && grep -q "Refusing a default user-local install" "$case_dir/out" && [ ! -e "$case_dir/home/.local/bin/gordon" ]; then + pass "sudo default install is refused" + else + fail "sudo default install is refused" + fi +} + +test_next_build +test_conflicts +test_go_version +test_default_and_path_modes +test_shell_updates +test_explicit_global_override +printf '%s passed, %s failed\n' "$PASS" "$FAIL" +[ "$FAIL" -eq 0 ] diff --git a/wiki/agents/deploy.md b/wiki/agents/deploy.md index c57da5b53..6d2a7fcdf 100644 --- a/wiki/agents/deploy.md +++ b/wiki/agents/deploy.md @@ -1,101 +1,51 @@ # AI Agent Deployment Guide -This guide provides a structured workflow for AI coding assistants to help users deploy applications with Gordon. +Use Gordon's declarative app workflow. Do not edit daemon state or infer workload identity from domains. -## Deployment Workflow +## Required files -### Key Paths (VPS) - -``` -~/.config/gordon/gordon.toml # Main config (routes, registry) -~/.gordon/env/ # Env files per domain -~/.gordon/logs/containers/ # Container logs -``` - -### Step 1: Add route in gordon.toml - -```toml -[routes] -"app.example.com" = "my-app:latest" -``` - -### Step 2: Create env file - -Naming convention: domain with dots replaced by underscores. - -Example: `app.example.com` → `~/.gordon/env/app_example_com.env` - -```bash -cat > ~/.gordon/env/app_example_com.env << 'EOF' -NODE_ENV=production -PUBLIC_API_URL=https://api.example.com -EOF -``` - -### Step 3: Generate registry token (on VPS) - -```bash -gordon auth token generate --subject username --expiry 720h -``` - -### Step 4: Login to registry (local machine) - -```bash -echo "TOKEN" | docker login -u username --password-stdin reg.example.com:5000 +```text +~/.config/gordon/gordon.toml # Daemon configuration +app.toml # Declarative app manifest ``` -### Step 5: Build and push - -Tag images to match git tags + latest for bleeding edge: - -```bash -# Get current git tag (falls back to "latest" if no tag) -TAG=$(git describe --tags --exact-match 2>/dev/null || echo "latest") +## Workflow -# Build with version tag and latest -docker buildx build --platform linux/amd64 \ - -t reg.example.com:5000/my-app:$TAG \ - -t reg.example.com:5000/my-app:latest \ - --push . -``` - -Gordon auto-deploys when it receives the pushed image. +1. Define services, routes, networks, volumes, secrets, readiness, and backups in `app.toml`. +2. Validate and persist the manifest: -### Step 6: Verify + ```bash + gordon apps apply --file app.toml + ``` -```bash -curl -I https://app.example.com -ssh user@vps "cat ~/.gordon/logs/containers/app_example_com.log" -``` +3. Register required secret values without placing them in the manifest: -## Framework Notes + ```bash + gordon apps secrets set APP --service SERVICE KEY + ``` -### SvelteKit / Next.js +4. Deploy the accepted revision: -Static env vars (`$env/static/public`, `NEXT_PUBLIC_*`) must be available at build time: + ```bash + gordon apps deploy APP + ``` -```dockerfile -ARG PUBLIC_API_URL -ENV PUBLIC_API_URL=$PUBLIC_API_URL -``` +5. Verify state and workload logs: -Build with: -```bash -docker buildx build --build-arg PUBLIC_API_URL=https://api.example.com --push . -``` + ```bash + gordon apps show APP + gordon apps logs APP --service SERVICE + ``` -## Secrets with pass +The local CLI uses the authenticated owner-only Unix socket. To target another Gordon instance, use the configured authenticated HTTP/TLS remote transport. -Use pass integration for sensitive values: +## Registry authentication -```bash -DATABASE_PASSWORD=${pass:gordon/app/db_password} -``` +Generate a scoped token on the server, then pass it to the registry client through standard input. Never place tokens in manifests, command history, fixtures, or logs. -## Troubleshooting +## Safety rules -| Issue | Solution | -|-------|----------| -| HTTPS error on push | Add registry to `insecure-registries` in daemon.json | -| Container not starting | Check env file name matches domain pattern | -| Env vars not working | SvelteKit static env = build args, not runtime env | +- Use app and service names as workload identity; domains are routing addresses only. +- Keep secret values outside manifests and app state. +- Reuse the same idempotency key only to inspect or retry the same mutation request. +- Do not delete retained volumes or secrets during app removal. diff --git a/wiki/examples/production.md b/wiki/examples/production.md index bb802d830..91ce0ac4d 100644 --- a/wiki/examples/production.md +++ b/wiki/examples/production.md @@ -37,16 +37,8 @@ max_size = 100 max_backups = 10 max_age = 90 -[logging.container_logs] -enabled = true # enabled by default -dir = "~/.gordon/logs/containers" -max_size = 100 -max_backups = 10 -max_age = 90 - -# Environment directory -[env] -dir = "~/.gordon/env" +# Workload logs are read from the container runtime with +# `gordon apps logs APP --service SERVICE`. # Volume settings (all enabled by default) [volumes] @@ -59,22 +51,8 @@ preserve = true enabled = true network_prefix = "prod" -# Application routes with pinned versions -[routes] -"app.company.com" = "company-app:v2.1.0" -"api.company.com" = "company-api:v1.5.2" -"admin.company.com" = "admin-panel:v1.0.1" -"docs.company.com" = "company-docs:latest" - -# Network groups for shared services -[network_groups] -"backend" = ["app.company.com", "api.company.com"] - -# Service attachments -[attachments] -"backend" = ["company-redis:latest"] -"app.company.com" = ["company-postgres:latest"] -"api.company.com" = ["company-postgres:latest"] +# Applications, routes, services, and shared networks are declared in +# separate app manifest files and applied with `gordon apps apply`. ``` ## Setup Steps @@ -100,26 +78,14 @@ openssl rand -base64 32 | pass insert -m gordon/auth/token_secret gordon auth token generate --subject ci-bot --scopes push,pull --expiry 0 ``` -### 4. Create Environment Files +### 4. Set App Secrets + +Declare public values under `[env]` and secret names under `[services..secrets]` in each app file, apply it, then set the values: ```bash -# App environment -cat > ~/.gordon/env/app_company_com.env < ~/.gordon/env/api_company_com.env < [!CAUTION] -> `--insecure` and `insecure_tls = true` disable TLS certificate verification. -> This removes server identity validation and increases man-in-the-middle risk. -> Use this only for private tailnet/VPN deployments with strict network access controls (restricted CIDRs, firewall allowlists, and no public ingress to admin endpoints). -> Safer alternatives: -> - Use a publicly trusted certificate (for example, Let's Encrypt). -> - Use an internal CA and trust that CA on operator machines. -> - Use Tailscale certificate generation (`tailscale cert`) and serve Gordon with that cert. - -```bash -# Save remote once -gordon remotes add tailnet-reg https://gordon.example.com --token-env GORDON_TOKEN --insecure -gordon remotes use tailnet-reg - -# Use it for auth/admin commands -gordon auth login -gordon routes list -``` - -Equivalent `remotes.toml` entry: - -```toml -active = "tailnet-reg" - -[remotes.tailnet-reg] -url = "https://gordon.example.com" -token_env = "GORDON_TOKEN" -insecure_tls = true -``` - -Use this mode when registry/auth/admin access is intentionally restricted to tailnet CIDRs. - -## Quick Start - -### One-off Remote Command - -```bash -gordon routes list --remote https://gordon.mydomain.com --token $TOKEN -``` - -### Using Environment Variables - -```bash -export GORDON_REMOTE=https://gordon.mydomain.com -export GORDON_TOKEN=$TOKEN -gordon routes list -``` - -### Using Saved Remotes - -```bash -# Add a remote -gordon remotes add prod https://gordon.mydomain.com --token-env PROD_TOKEN - -# Set as active -gordon remotes use prod - -# Now use without flags -gordon routes list -``` - -## Global Flags - -These flags are available on all commands: - -| Flag | Description | -|------|-------------| -| `--remote ` | Remote Gordon URL | -| `--token ` | Authentication token | -| `--insecure` | Skip TLS verification (self-signed/private certs) | - -```bash -# List routes on remote -gordon routes list --remote https://gordon.mydomain.com --token $TOKEN - -# Manage secrets on remote -gordon secrets list myapp.example.com --remote https://gordon.mydomain.com --token $TOKEN -``` - -## Environment Variables - -| Variable | Description | -|----------|-------------| -| `GORDON_REMOTE` | Remote Gordon URL | -| `GORDON_TOKEN` | Authentication token | - -Environment variables are useful for CI/CD pipelines and shell sessions: - -```bash -# Set for current session -export GORDON_REMOTE=https://gordon.mydomain.com -export GORDON_TOKEN=eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9... - -# Commands automatically use these -gordon routes list -gordon secrets list myapp.example.com -``` - -## Saved Remotes - -Save frequently used remotes to avoid repeating URLs and tokens. - -### Configuration File - -Remotes are stored in `~/.config/gordon/remotes.toml`: - -```toml -active = "prod" - -[remotes.prod] -url = "https://gordon.mydomain.com" -token_env = "PROD_TOKEN" - -[remotes.staging] -url = "https://staging.mydomain.com" -token = "eyJ..." -``` - -### Managing Remotes - -#### List Remotes - -```bash -gordon remotes list -``` - -Output: - -``` -Saved Remotes - -Name URL Token Status -────────────────────────────────────────────────────────────────────────────── -prod https://gordon.mydomain.com $PROD_TOKEN active -staging https://staging.mydomain.com set -dev https://dev.mydomain.com none -``` - -#### Add a Remote - -```bash -# Basic (no token) -gordon remotes add prod https://gordon.mydomain.com - -# With direct token -gordon remotes add prod https://gordon.mydomain.com --token eyJ... - -# With environment variable reference (recommended) -gordon remotes add prod https://gordon.mydomain.com --token-env PROD_TOKEN -``` - -| Flag | Description | -|------|-------------| -| `--token ` | Store token directly in config | -| `--token-env ` | Store env variable name (token resolved at runtime) | - -#### Authenticate with Password Auth - -For servers using password authentication, you can log in interactively: - -```bash -# Login to active remote -gordon auth login - -# Login to specific remote -gordon auth login --remote prod - -# Pre-fill username -gordon auth login --username admin -``` - -This prompts for your username and password, authenticates with the remote server, and stores the returned token automatically. - -> **Note:** Only works with servers that have password authentication enabled (`username` + `password_hash` configured). For token-only servers, use `gordon remotes set-token` instead. - -#### Set Token Manually - -For servers using token-based auth, or to update a token manually: - -```bash -# Set token directly -gordon remotes set-token prod eyJhbGciOiJIUzI1NiIs... - -# Set token from file -gordon remotes set-token staging $(cat token.txt) -``` - -#### Remove a Remote - -```bash -# With confirmation prompt -gordon remotes remove staging - -# Skip confirmation -gordon remotes remove staging --force -``` - -#### Set Active Remote +Verify connectivity: ```bash -gordon remotes use prod +gordon daemon status --remote https://gordon.example.com --token "$GORDON_TOKEN" ``` -When a remote is active, it's used automatically for all remote-capable commands: +## App management -```bash -gordon remotes use prod -gordon routes list # Uses prod remote -gordon secrets list app.com # Uses prod remote -``` - -## Resolution Precedence - -When multiple sources specify remote or token, the CLI uses this priority order: - -**Remote URL:** -1. `--remote` flag -2. `GORDON_REMOTE` environment variable -3. Active remote from `remotes.toml` - -**Token:** -1. `--token` flag -2. `GORDON_TOKEN` environment variable -3. Token from active remote in `remotes.toml` - -This allows overriding specific values while keeping defaults: +The app name is the canonical workload identity. Domains are routing addresses declared by the manifest. ```bash -# Active remote is prod, but use different token -gordon routes list --token $TEMPORARY_TOKEN -``` - -## Commands Supporting Remote - -These commands work with remote targeting: - -| Command | Description | -|---------|-------------| -| `gordon routes list` | List all routes | -| `gordon routes add ` | Add a route | -| `gordon routes remove ` | Remove a route | -| `gordon routes deploy ` | Deploy/redeploy a route | -| `gordon attachments list [target]` | List all or targeted attachments | -| `gordon attachments add ` | Add an attachment | -| `gordon attachments remove ` | Remove an attachment | -| `gordon secrets list ` | List secrets for a domain | -| `gordon secrets set KEY=value` | Set secrets | -| `gordon secrets remove ` | Remove a secret | -| `gordon status` | Show server status | -| `gordon reload` | Reload config and start new routes | -| `gordon logs` | View Gordon process logs | -| `gordon logs ` | View container logs | - -### Remote Logs Requirement - -The `gordon logs` command requires **file logging to be enabled** on the remote server. Without it, you'll get a "failed to get logs" error. - -On the remote Gordon server, add to `config.toml`: - -```toml -[logging.file] -enabled = true -# path is optional - defaults to {data_dir}/logs/gordon.log -# path = "/var/log/gordon/gordon.log" +gordon apps apply --file app.toml --remote https://gordon.example.com --token "$GORDON_TOKEN" +gordon apps list --remote https://gordon.example.com --token "$GORDON_TOKEN" +gordon apps show APP --remote https://gordon.example.com --token "$GORDON_TOKEN" +gordon apps deploy APP --remote https://gordon.example.com --token "$GORDON_TOKEN" +gordon apps logs APP --service SERVICE --remote https://gordon.example.com --token "$GORDON_TOKEN" ``` -Then restart Gordon. The log directory will be created automatically in your data directory (e.g., `~/.gordon/logs/`). - -If you specify a custom path, ensure the directory exists and is writable: +Lifecycle commands use the same transport: ```bash -sudo mkdir -p /var/log/gordon -sudo chown gordon:gordon /var/log/gordon # adjust user as needed +gordon apps stop APP +gordon apps start APP +gordon apps restart APP --service SERVICE +gordon apps remove APP ``` -> **Note:** When running as a systemd service, Gordon logs to journalctl by default. However, the remote admin API cannot read from journalctl—it reads from the configured log file. Both can be used simultaneously. - -## Token Security - -### Option 1: Environment Variable Reference (Recommended) +## Backups -Store the environment variable name, not the actual token: +Backup targets come from the ACTIVE app manifest and use app, service, and resource identity: ```bash -gordon remotes add prod https://gordon.mydomain.com --token-env PROD_TOKEN +gordon backups run APP --service SERVICE --database DATABASE +gordon backups volume run APP --service SERVICE --volume VOLUME +gordon backups status +gordon backups volume status ``` -The `remotes.toml` stores only: +## Logs -```toml -[remotes.prod] -url = "https://gordon.mydomain.com" -token_env = "PROD_TOKEN" -``` - -At runtime, Gordon reads `$PROD_TOKEN` from the environment. - -**Benefits:** -- Token not stored in plaintext config -- Works with secret managers that inject env vars -- Easy rotation without editing config - -### Option 2: Direct Token Storage +`gordon daemon logs` streams daemon process logs. Workload logs use the app command: ```bash -gordon remotes add prod https://gordon.mydomain.com --token eyJ... -``` - -The token is stored directly in `remotes.toml`: - -```toml -[remotes.prod] -url = "https://gordon.mydomain.com" -token = "eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9..." -``` - -**Security note:** Ensure `remotes.toml` has restricted permissions: - -```bash -chmod 600 ~/.config/gordon/remotes.toml -``` - -### Option 3: Environment Variables Only - -For maximum security, don't save tokens at all: - -```bash -# Add remote without token -gordon remotes add prod https://gordon.mydomain.com - -# Set token via environment -export GORDON_TOKEN=$TOKEN - -# Use normally -gordon remotes use prod -gordon routes list -``` - -## Workflow Examples - -### Development Workflow - -```bash -# Add dev and staging remotes -gordon remotes add dev https://gordon.dev.local --token-env DEV_TOKEN -gordon remotes add staging https://gordon.staging.example.com --token-env STAGING_TOKEN - -# Work with dev -gordon remotes use dev -gordon routes list -gordon routes add myapp.dev.local myapp:latest - -# Switch to staging -gordon remotes use staging -gordon routes list -``` - -### CI/CD Pipeline - -```yaml -# GitHub Actions example -env: - GORDON_REMOTE: ${{ secrets.GORDON_URL }} - GORDON_TOKEN: ${{ secrets.GORDON_TOKEN }} - -steps: - - name: Deploy to Gordon - run: | - gordon routes deploy myapp.example.com +gordon daemon logs --remote https://gordon.example.com --token "$GORDON_TOKEN" +gordon apps logs APP --service SERVICE --remote https://gordon.example.com --token "$GORDON_TOKEN" ``` -```bash -# GitLab CI example -deploy: - script: - - export GORDON_REMOTE=$GORDON_URL - - export GORDON_TOKEN=$GORDON_TOKEN - - gordon routes deploy myapp.example.com -``` - -### Multi-Environment Management - -```bash -# Add all environments -gordon remotes add prod https://gordon.example.com --token-env PROD_TOKEN -gordon remotes add staging https://gordon.staging.example.com --token-env STAGING_TOKEN -gordon remotes add dev https://gordon.dev.example.com --token-env DEV_TOKEN - -# Compare routes across environments -gordon routes list --remote https://gordon.example.com --token $PROD_TOKEN -gordon routes list --remote https://gordon.staging.example.com --token $STAGING_TOKEN - -# Or switch active -gordon remotes use prod && gordon routes list -gordon remotes use staging && gordon routes list -``` - -## Troubleshooting - -### "unauthorized" Error - -Token is missing, expired, or invalid. - -```bash -# Check if token is set -echo $GORDON_TOKEN - -# Verify token works -gordon routes list --remote https://gordon.mydomain.com --token $TOKEN -``` - -### "connection refused" Error - -Remote Gordon isn't running or URL is wrong. - -```bash -# Verify URL is correct -curl https://gordon.mydomain.com/health - -# Check if admin API is enabled on remote -# The remote gordon.toml needs: -# [admin] -# enabled = true -``` - -### Remote Not Found - -The specified remote doesn't exist in `remotes.toml`. - -```bash -# List available remotes -gordon remotes list - -# Add the remote -gordon remotes add myremote https://gordon.mydomain.com -``` - -### Active Remote Cleared - -After removing the active remote, you need to set a new active: - -```bash -gordon remotes use prod -``` - -Or use explicit flags: - -```bash -gordon routes list --remote https://gordon.mydomain.com --token $TOKEN -``` - -### "failed to get logs" Error - -The remote Gordon server doesn't have file logging enabled. - -```bash -# Error: failed to get process logs: 500 Internal Server Error: failed to get logs -``` - -On the remote server, enable file logging in `config.toml`: - -```toml -[logging.file] -enabled = true -path = "/var/log/gordon/gordon.log" -``` - -See [Remote Logs Requirement](#remote-logs-requirement) for details. - -## Related +## Security -- [CLI Commands](/docs/cli/index.md) -- [Authentication](/docs/config/auth.md) -- [Attachments](/docs/config/attachments.md) +- Use HTTPS for remote access. +- Scope and rotate tokens. +- Keep the local admin socket owner-only. +- Never expose the local socket through a public proxy. +- Prefer environment variables or a protected credential store over literal tokens. diff --git a/wiki/guides/secrets-pass.md b/wiki/guides/secrets-pass.md index 814dc8798..3c5e15337 100644 --- a/wiki/guides/secrets-pass.md +++ b/wiki/guides/secrets-pass.md @@ -106,17 +106,17 @@ token_secret = "gordon/auth/token_secret" # password_hash = "gordon/auth/password_hash" ``` -### Using Route Secrets +### Using App Secrets -With the pass backend, per-domain secrets are stored in pass, not `.env` files. +Declare secret names in the app file under `[services..secrets]`, apply it, then set the values. Values are stored in pass under `gordon/apps///`. ```bash -# Store secrets for a domain -gordon secrets set app.mydomain.com DATABASE_URL "postgresql://user:pass@postgres:5432/app" -gordon secrets set app.mydomain.com API_KEY "your-api-key" +gordon apps apply --file ./blog.toml +printf 'DATABASE_URL=%s\n' "$(pass show myapp/database-url)" \ + | gordon apps secrets set blog --service web --stdin ``` -Gordon migrates existing `.env` files on startup and renames them to `.env.migrated`. +See [App Manifest](/docs/config/apps.md) and [Apps CLI](/docs/cli/apps.md). ## Organizing Secrets diff --git a/wiki/guides/secrets-sops.md b/wiki/guides/secrets-sops.md index 2314f9f97..e7ce4b518 100644 --- a/wiki/guides/secrets-sops.md +++ b/wiki/guides/secrets-sops.md @@ -208,17 +208,7 @@ The path format is `file:key.path` where: - `file` is the SOPS-encrypted file path - `key.path` is dot-notation to the value -Route secrets remain in `.env` files with `${sops:...}` references. The SOPS backend is used for auth secrets (like `token_secret`) and provider lookups. - -### Using Secrets in Environment Files - -Reference SOPS secrets in your app's environment files: - -```bash -# ~/.gordon/env/app_mydomain_com.env -DATABASE_URL=postgresql://user:${sops:secrets.yaml:database.password}@postgres:5432/app -API_KEY=${sops:secrets.yaml:api.key} -``` +The SOPS backend is used for auth secrets such as `token_secret`. App secret values are managed with `gordon apps secrets` (see [Apps CLI](/docs/cli/apps.md)). ## File Organization diff --git a/wiki/guides/secure-vps-setup.md b/wiki/guides/secure-vps-setup.md index 421707042..ebaaa9035 100644 --- a/wiki/guides/secure-vps-setup.md +++ b/wiki/guides/secure-vps-setup.md @@ -175,7 +175,7 @@ gordon remotes use tailnet-reg # Then use commands normally (auth + admin API) gordon auth login gordon routes list -gordon backup status +gordon backups status ``` `~/.config/gordon/remotes.toml` should contain an entry like: diff --git a/wiki/tutorials/first-deploy.md b/wiki/tutorials/first-deploy.md index db06d9c9b..16f819c6f 100644 --- a/wiki/tutorials/first-deploy.md +++ b/wiki/tutorials/first-deploy.md @@ -87,7 +87,7 @@ On your Gordon server, edit `~/.config/gordon/gordon.toml`: Reload Gordon: ```bash -gordon reload +gordon daemon reload ``` ### 5. Push to Deploy @@ -130,7 +130,7 @@ docker login -u deploy -p registry.mydomain.com 2. Check Gordon logs: ```bash - gordon logs -f + gordon daemon logs -f ``` 3. Ensure DNS points to your server @@ -155,10 +155,10 @@ docker tag my-first-app registry.mydomain.com/my-first-app:latest docker push registry.mydomain.com/my-first-app:latest ``` -Gordon automatically deploys the update with zero downtime. +Gordon replaces the running container: it withdraws traffic, stops and removes the old container, starts the new one, waits for it to pass its readiness probe, and routes traffic to it. Expect a short interruption at that point. ## Next Steps -- [Add environment variables](/docs/config/env.md) +- [Add environment variables](/docs/config/apps.md) - [Add a database](./postgres-service.md) - [Set up CI/CD](/docs/deployment/github-actions.md) diff --git a/wiki/tutorials/postgres-service.md b/wiki/tutorials/postgres-service.md index fb3675a4c..6d95015ae 100644 --- a/wiki/tutorials/postgres-service.md +++ b/wiki/tutorials/postgres-service.md @@ -56,7 +56,7 @@ enabled = true Reload Gordon: ```bash -gordon reload +gordon daemon reload ``` ### 4. Configure Database Password