diff --git a/.dockerignore b/.dockerignore index d37127f..15adda8 100644 --- a/.dockerignore +++ b/.dockerignore @@ -11,5 +11,7 @@ **/*.db **/*.db-* **/.env* +**/*.env +**/*.env.* **/*.user **/Properties/launchSettings.json diff --git a/OpenNest.Server/Dockerfile b/OpenNest.Server/Dockerfile index 488e72f..a74d914 100644 --- a/OpenNest.Server/Dockerfile +++ b/OpenNest.Server/Dockerfile @@ -1,5 +1,10 @@ # Build from the repository root: docker build -f OpenNest.Server/Dockerfile -t opennest-server . -FROM mcr.microsoft.com/dotnet/sdk:8.0 AS build +# Release builds pass --build-arg VERSION=X.Y.Z and SOURCE_REVISION=; +# the defaults mark an unversioned development build. Base images may be pinned by digest. +ARG SDK_IMAGE=mcr.microsoft.com/dotnet/sdk:8.0 +ARG RUNTIME_IMAGE=mcr.microsoft.com/dotnet/aspnet:8.0 + +FROM ${SDK_IMAGE} AS build WORKDIR /src COPY OpenNest.Server/OpenNest.Server.csproj OpenNest.Server/ @@ -10,16 +15,44 @@ RUN dotnet restore OpenNest.Server/OpenNest.Server.csproj COPY OpenNest.Server/ OpenNest.Server/ COPY OpenNest.Data/ OpenNest.Data/ COPY OpenNest.Core/ OpenNest.Core/ -RUN dotnet publish OpenNest.Server/OpenNest.Server.csproj -c Release -o /app --no-restore +ARG VERSION=0.0.0-dev +ARG SOURCE_REVISION=unknown +RUN dotnet publish OpenNest.Server/OpenNest.Server.csproj -c Release -o /app --no-restore \ + -p:Version="$VERSION" -p:SourceRevisionId="$SOURCE_REVISION" -FROM mcr.microsoft.com/dotnet/aspnet:8.0 AS runtime +FROM ${RUNTIME_IMAGE} AS runtime +# curl is installed only for the HEALTHCHECK below. +RUN apt-get update \ + && apt-get install -y --no-install-recommends curl \ + && rm -rf /var/lib/apt/lists/* WORKDIR /app +# Runtime files stay root-owned and read-only to the app user. COPY --from=build /app . -# SQLite database + .nest blobs live here; mount a volume at this path. -VOLUME /app/data +# SQLite database + .nest blobs live here; mount a volume at this path. The directory +# belongs to the base image's non-root app user (APP_UID 1654), so a fresh named volume +# inherits that ownership. Existing root-owned data needs a one-time scoped chown. +RUN mkdir -p /app/data && chown "$APP_UID:$APP_UID" /app/data && chmod 0750 /app/data ENV OPENNEST_DB=/app/data/nests.db +# Listen on the explicit 8090 URL; clear the base image's 8080 port default. ENV ASPNETCORE_URLS=http://+:8090 +ENV ASPNETCORE_HTTP_PORTS= EXPOSE 8090 +USER $APP_UID +VOLUME /app/data + +HEALTHCHECK --interval=30s --timeout=5s --start-period=10s --start-interval=2s --retries=3 \ + CMD ["curl", "--fail", "--silent", "--show-error", "http://127.0.0.1:8090/healthz"] + +ARG VERSION=0.0.0-dev +ARG SOURCE_REVISION=unknown +ARG RUNTIME_IMAGE +LABEL org.opencontainers.image.title="OpenNest.Server" \ + org.opencontainers.image.description="OpenNest nest-storage HTTP service (SQLite)" \ + org.opencontainers.image.source="https://github.com/ajisaacs/OpenNest" \ + org.opencontainers.image.licenses="MIT" \ + org.opencontainers.image.version="$VERSION" \ + org.opencontainers.image.revision="$SOURCE_REVISION" \ + org.opencontainers.image.base.name="$RUNTIME_IMAGE" ENTRYPOINT ["dotnet", "OpenNest.Server.dll"] diff --git a/OpenNest.Server/server.env.example b/OpenNest.Server/server.env.example new file mode 100644 index 0000000..aad5a32 --- /dev/null +++ b/OpenNest.Server/server.env.example @@ -0,0 +1,13 @@ +# Copy to a local, untracked env file next to compose.server.yaml and edit it. +# Never put credentials in this file or the Compose file. + +# A tested image version or digest, for example ghcr.io/ajisaacs/opennest-server@sha256:. +OPENNEST_SERVER_IMAGE=ghcr.io/ajisaacs/opennest-server:REPLACE-WITH-TESTED-VERSION + +# Loopback for a local test. For shop PCs, use this host's trusted LAN address and +# confirm the firewall admits only trusted clients; the service has no authentication. +OPENNEST_BIND_ADDRESS=127.0.0.1 +OPENNEST_HOST_PORT=8090 + +# Docker volume holding nests.db; change only to switch to a verified restore. +OPENNEST_DATA_VOLUME=opennest-data diff --git a/compose.server.yaml b/compose.server.yaml new file mode 100644 index 0000000..d9a0863 --- /dev/null +++ b/compose.server.yaml @@ -0,0 +1,23 @@ +# OpenNest.Server deployment example: one instance, one local data volume, LAN-only. +# Usage: docker compose --env-file -f compose.server.yaml up -d +# Start from OpenNest.Server/server.env.example. See docs/nest-storage.md before deploying. +name: opennest-server +services: + opennest-server: + image: ${OPENNEST_SERVER_IMAGE:?Set a tested version or digest} + restart: unless-stopped + ports: + - "${OPENNEST_BIND_ADDRESS:?Set 127.0.0.1 for testing or a trusted LAN address}:${OPENNEST_HOST_PORT:-8090}:8090" + environment: + ASPNETCORE_URLS: http://+:8090 + OPENNEST_DB: /app/data/nests.db + volumes: + - opennest-data:/app/data + security_opt: + - no-new-privileges:true + cap_drop: + - ALL +volumes: + opennest-data: + # A restore switches to a separately verified volume instead of overwriting this one. + name: ${OPENNEST_DATA_VOLUME:-opennest-data} diff --git a/docs/nest-storage.md b/docs/nest-storage.md index 0e6492d..2db4601 100644 --- a/docs/nest-storage.md +++ b/docs/nest-storage.md @@ -106,6 +106,8 @@ dotnet run --project OpenNest.Server/OpenNest.Server.csproj Listens on `ASPNETCORE_URLS` (default Kestrel ports) unless overridden. +### Docker image + Docker must be built from the repository root: the server references `OpenNest.Data`, which references `OpenNest.Core`. The Dockerfile copies all three project files before restore and their source/resources before publish, including @@ -116,29 +118,155 @@ runtime to run; IO/Engine are used only by the host-side smoke tool, not the ima ```sh docker build --pull -f OpenNest.Server/Dockerfile -t opennest-server:local . +# A release build stamps its version and exact source commit: +docker build --pull -f OpenNest.Server/Dockerfile --build-arg VERSION=X.Y.Z \ + --build-arg SOURCE_REVISION="$(git rev-parse HEAD)" -t opennest-server:X.Y.Z . ``` -The image listens on `:8090` and keeps `OPENNEST_DB=/app/data/nests.db`. Mount -`/app/data` to preserve the SQLite database and archive blobs across recreation. -For an intentional local run (not needed for the isolated tests below): +Without build arguments the image is labeled version `0.0.0-dev`, revision +`unknown`: a development build, never a release. The OCI labels are +`org.opencontainers.image.source`, `version`, `revision` and `base.name`. +`SDK_IMAGE`/`RUNTIME_IMAGE` build arguments accept digest-pinned base images; +for a release, record the base digests the build log resolved. + +Runtime contract: + +- Container port 8090 (`ASPNETCORE_URLS=http://+:8090`). Change the host port + mapping, not the container port: the healthcheck probes 8090. +- Runs as the .NET base image's non-root `app` user (UID/GID 1654). Application + files are root-owned and read-only to it; within `/app` only `/app/data` is + writable, and a fresh named volume inherits that directory's ownership. The + database stays `OPENNEST_DB=/app/data/nests.db`. The server also needs the + container's writable `/tmp`: multipart uploads over 64 KiB are buffered there. +- `HEALTHCHECK` runs `curl --fail http://127.0.0.1:8090/healthz` (interval 30 s, + timeout 5 s, start period 10 s, retries 3, and a 2 s start interval during the + start period on Docker 25+). curl is in the image only for it. +- A read-only or unwritable data location fails startup: the process exits + rather than serving requests it cannot store. +- Uploads: Kestrel's default 30,000,000-byte request limit covers the whole + multipart request, so an archive must be slightly smaller (the smoke stores a + 29,000,000-byte one). A larger request gets 413 and nothing is stored. The + desktop client streams without `Expect: 100-continue`, so it typically reports + a closed connection instead of the 413 status. The save still fails. + +Data written by a root-run image or a root-owned bind mount is not writable by +UID 1654, so startup fails. With the operator's approval, change ownership of +that data location only, once, for example +`docker run --rm --user 0 --entrypoint chown -v :/app/data -R 1654:1654 /app/data` +(or `chown -R 1654:1654` on the bind-mounted directory itself, never its +parents). The image has no root entrypoint that changes ownership. + +### Deploying with Compose + +`compose.server.yaml` runs a published image (no local build) with +`no-new-privileges`, all capabilities dropped, and a named data volume. Copy it and +`OpenNest.Server/server.env.example` to a deployment directory, save the env file +under a local name such as `server.env`, and set: + +- `OPENNEST_SERVER_IMAGE`: a tested version or, preferably, a digest. +- `OPENNEST_BIND_ADDRESS`: `127.0.0.1` for testing on the host; this host's + trusted LAN address for shop PCs. Do not use `0.0.0.0`. The bind address is only + one layer: verify that the network/firewall admits only trusted clients. +- `OPENNEST_HOST_PORT` (default 8090) and `OPENNEST_DATA_VOLUME` (default `opennest-data`). ```sh -docker run -d -p 127.0.0.1:8090:8090 -v opennest-data:/app/data opennest-server:local +compose="docker compose --env-file server.env -f compose.server.yaml" +$compose up -d +$compose ps # STATUS shows (healthy) +curl --fail http://:/healthz # {"status":"ok"} ``` -This example binds only to loopback. Remote shop clients require an explicitly -chosen trusted-interface binding and network access controls; do not casually -replace it with an all-interface publish. The service has no authentication. -These build/smoke instructions do not establish release readiness, non-root -hardening, or a production deployment. +In the desktop app choose **File > Storage Mode...**, Database, and enter the base +URL `http://:`, without `/healthz` or `/api/nests`. -Point the desktop app's **File > Storage Mode...** server URL at -`http://:8090` (or `http://127.0.0.1:8090` for local use). +### Backup, restore, and upgrade + +SQLite runs in WAL mode, so copying `nests.db` from a running service is not a +backup. Back up during a quiet period with no saves. Each step below is a Bash +function that checks every command and stops at the first failure, however it is +called; define them in the shell where `compose` is set: + +```bash +url=http://: + +# " " for every record, sorted by id. Fails on any failed request. +nest_manifest() ( + set -o pipefail + list=$(curl -fsS "$1/api/nests") || exit 1 + for id in $(printf '%s' "$list" | grep -o '"id":"[0-9a-f-]*"' | cut -d'"' -f4 | sort); do + hash=$(curl -fsS "$1/api/nests/$id/file" | sha256sum) || exit 1 + printf '%s %s\n' "$id" "${hash%% *}" + done +) + +# Records the expected manifest and metadata, then archives the stopped data directory. +# The service is restarted even if the archive fails. +backup_nests() ( # usage: backup_nests opennest-data-YYYY-MM-DD + set -o pipefail + fail() { echo "STOP: $*" >&2; exit 1; } + [[ ! -e "$1.tar" ]] || fail "$1.tar already exists" + nest_manifest "$url" > "$1.manifest" || fail "could not record the manifest" + curl -fsS "$url/api/nests" > "$1.metadata.json" || fail "could not record the metadata" + $compose stop || fail "could not stop the service" + $compose run --rm --no-deps -T --entrypoint tar opennest-server -C /app/data -cf - . > "$1.tar" + status=$? + $compose start || fail "could not restart the service" + curl -fs --retry 30 --retry-all-errors --retry-delay 1 "$url/healthz" > /dev/null \ + || fail "the restarted service is not healthy" + [[ $status == 0 ]] || fail "the archive failed; $1.tar is incomplete" + echo "Backed up $(wc -l < "$1.manifest") records to $1.tar" +) + +backup_nests opennest-data-YYYY-MM-DD +``` + +Restore only into a volume created for that restore, never into an existing one +(least of all the active volume). `restore_check` refuses an existing volume, +extracts the backup into a new one, starts a temporary loopback-only container on +it, and requires the exact metadata list and every archive hash to match. It +removes only the container it started; the new volume is kept either way: + +```bash +restore_check() ( # usage: restore_check opennest-data-YYYY-MM-DD [new-volume] + set -o pipefail + fail() { echo "STOP: $*" >&2; exit 1; } + restore=${3:-opennest-data-restore-$(date +%Y%m%d-%H%M%S)} + ! docker volume inspect "$restore" > /dev/null 2>&1 || fail "volume $restore already exists" + docker volume create "$restore" > /dev/null || fail "could not create volume $restore" + echo "Restoring into new volume $restore" + docker run --rm -i --network none --entrypoint tar -v "$restore:/app/data" "$2" \ + -C /app/data -xf - < "$1.tar" || fail "extraction into $restore failed" + check=$(docker create -p 127.0.0.1::8090 -v "$restore:/app/data" "$2") \ + || fail "could not create the check container" + trap 'docker rm -f "$check" > /dev/null' EXIT + docker start "$check" > /dev/null || fail "the check container did not start" + port=$(docker port "$check" 8090/tcp | cut -d: -f2) && [[ -n $port ]] || fail "no check port" + curl -fs --retry 30 --retry-all-errors --retry-delay 1 "http://127.0.0.1:$port/healthz" > /dev/null \ + || fail "the restored service is not healthy" + curl -fsS "http://127.0.0.1:$port/api/nests" | cmp - "$1.metadata.json" || fail "metadata differs" + nest_manifest "http://127.0.0.1:$port" | diff - "$1.manifest" || fail "archives differ" + echo "Verified $restore. Set OPENNEST_DATA_VOLUME=$restore in server.env, then run: \$compose up -d" +) + +restore_check opennest-data-YYYY-MM-DD +``` + +Switch only after `restore_check` prints `Verified`. If it stops, inspect or remove +the new volume it named; the active volume is untouched. + +To upgrade, take a stopped-service backup and keep a copy of the current +`server.env`. Prefer a digest in `OPENNEST_SERVER_IMAGE` so that file records +exactly what ran; for a tag, record +`docker image inspect --format '{{join .RepoDigests " "}}' ` first. Change +only `OPENNEST_SERVER_IMAGE` and run `$compose up -d`. To roll back, restore the +previous `server.env` and run `$compose up -d`; if the new version wrote data the +old one cannot read, also restore the backup as above. Never run +`docker compose down -v`: it deletes the data volume. ## Isolated container smoke Prerequisites: a local Linux Docker daemon on the default Unix-socket context, -Bash, curl, GNU `timeout`, `mktemp`, and a .NET SDK able to build `net8.0` projects +Bash, curl, GNU `timeout`, `mktemp`, `tar`, and a .NET SDK able to build `net8.0` projects (.NET 8 or newer). Restore needs NuGet access or cached packages. Run from the repository root after building the image: @@ -150,9 +278,17 @@ scripts/Test-ServerContainer.sh --image opennest-server:local --results /path/to The wrapper accepts an image, not a URL or an existing volume/container. It uses only the local default Docker context, creates uniquely labeled disposable -resources, and publishes to `127.0.0.1` on a Docker-allocated port. It builds the -smoke tool once into a private temporary directory, honoring `TMPDIR`; readiness, -HTTP requests, and child commands have watchdogs. +resources, and publishes to `127.0.0.1` on a Docker-allocated port. Every container +runs with the Compose example's `--cap-drop ALL` and `no-new-privileges`. It builds +the smoke tool once into a private temporary directory, honoring `TMPDIR`; +readiness, HTTP requests, and child commands have watchdogs. + +Before starting a container it checks the image's numeric non-root `USER`, the +documented `HEALTHCHECK` command and timings (including the start interval), and +the OCI source/version/revision labels. A service is ready only when Docker reports the image's own healthcheck +`healthy` and the host receives `{"status":"ok"}`. Each service then checks that +the server process (PID 1) runs as the image user with no effective capabilities, +owns `nests.db`, and can write `/app/data` but not `/app` or the server DLL. The test-only `scripts/Server.Tests/Server.Tests.csproj` console exercises the existing .NET storage client; it is not included in the server image. @@ -166,21 +302,33 @@ and counts. `reject` snapshots every record's metadata and archive hash, then checks 400 responses for missing/invalid/null metadata and missing/empty files on both upload and file-update routes; unknown metadata/file IDs must return 404. Every rejected operation must leave the snapshot unchanged. +`limits` stores, downloads byte-for-byte, and deletes a 29,000,000-byte synthetic +archive (buffered to disk by the non-root runtime), then requires 413 for a +request over 30,000,000 bytes sent with `Expect: 100-continue`, and a reported +failure from the real client's oversized upload and file update. The snapshot +must be unchanged after each. The persistence stage stops and removes the first container while retaining its -named volume, starts a replacement on another allocated loopback port, and runs +fresh named volume, starts a replacement on another allocated loopback port, and runs `verify` against saved IDs, **all** metadata (including server-assigned `savedAt`), -archive hashes, and exact list membership. A missing/invalid state file or failed -assertion exits nonzero. Direct tool use is test-only: it accepts plain loopback -URLs, and `seed` refuses a nonempty server unless `--test-allow-nonempty` is -explicitly supplied for an owned disposable target. Prefer the wrapper; loopback -alone is not proof that an existing server is disposable. +archive hashes, and exact list membership. It then follows the documented +stopped-service backup: `tar` of the whole data directory through the image, +restored into a second owned volume whose service must pass the same `verify`. +The original volume mounted read-only, and a root-owned `tmpfs`, must each make +the server exit nonzero at startup with a SQLite error and never report healthy. + +A missing/invalid state file or failed assertion exits nonzero. Direct tool use is +test-only: it accepts plain loopback URLs, and `seed` refuses a nonempty server +unless `--test-allow-nonempty` is explicitly supplied for an owned disposable +target. Prefer the wrapper; loopback alone is not proof that an existing server is +disposable. Each invocation writes a fresh `run.*` results directory containing build, smoke, -container, ownership, and cleanup logs. On failure those logs remain for diagnosis. -The exit trap removes only invocation-owned containers and the named volume; -transient state, synthetic databases, and build artifacts are removed on success -or failure. It never prunes Docker or alters preexisting services/volumes. +container, image, runtime, ownership, and cleanup logs. On failure those logs remain +for diagnosis. The exit trap removes only invocation-owned containers and named +volumes; transient state, the backup archive, synthetic databases, and build +artifacts are removed on success or failure. It never prunes Docker or alters +preexisting services/volumes. Cleanup requires successful empty Docker inventory readback, including for an already-removed container. Unresolved ownership, lookup, removal, or readback errors make the wrapper fail without deleting unproven resources; inspect diff --git a/scripts/Server.Tests/Program.cs b/scripts/Server.Tests/Program.cs index 35017f4..57d9283 100644 --- a/scripts/Server.Tests/Program.cs +++ b/scripts/Server.Tests/Program.cs @@ -29,7 +29,8 @@ internal static class Program using var http = new HttpClient(handler) { Timeout = TimeSpan.FromSeconds(10), - MaxResponseContentBufferSize = 16 * 1024 * 1024, + // Only the limits mode downloads its near-cap synthetic archive. + MaxResponseContentBufferSize = (options.Mode == "limits" ? 32 : 16) * 1024 * 1024, }; using var repository = new RemoteNestRepository(http, options.Url); switch (options.Mode) @@ -43,6 +44,9 @@ internal static class Program case "reject": await Reject(repository, http, options.Url, watchdog.Token); break; + case "limits": + await Limits(repository, http, options.Url, watchdog.Token); + break; } Console.WriteLine($"PASS {options.Mode}; raw camelCase/string-enum responses checked: {handler.CheckedResponses}"); return 0; @@ -163,6 +167,77 @@ internal static class Program Console.WriteLine("Unknown metadata/file ids: 404; all metadata and hashes unchanged."); } + // Kestrel's default MaxRequestBodySize (30,000,000 bytes) bounds a whole multipart request. + private const int RequestBodyLimit = 30_000_000; + + private static async Task Limits(RemoteNestRepository repository, HttpClient http, string url, CancellationToken ct) + { + var before = await Snapshot(repository, ct); + Check(before.Count > 0, "Limit checks require seeded records on an owned server."); + + // Opaque synthetic bytes (the server does not parse archives) above the 64 KiB form + // memory threshold, so the non-root runtime must buffer the part to its temp directory. + var accepted = new byte[29_000_000]; + RandomNumberGenerator.Fill(accepted); + var created = await repository.UploadAsync(accepted, new NestRecord + { + Name = "Synthetic upload-limit probe", + Comments = "Random bytes; deleted by the same smoke stage", + }, ct); + Check(created.Id != Guid.Empty && created.FileSize == accepted.LongLength, + "A near-limit upload must be stored with its exact size."); + var downloaded = await repository.GetFileAsync(created.Id, ct); + Check(downloaded is not null && downloaded.AsSpan().SequenceEqual(accepted), + "A near-limit archive must download byte-for-byte."); + await CheckMembership(repository, before.Select(s => s.Metadata.Id).Append(created.Id), ct); + await repository.DeleteAsync(created.Id, ct); + await CheckSnapshot(repository, before, ct); + Console.WriteLine($"Accepted, downloaded, and deleted a {accepted.Length}-byte archive; all other records unchanged."); + + // Kestrel rejects an oversized declared length before the app reads the form. With + // Expect: 100-continue the client waits for, and must receive, that 413 response. + var oversized = new byte[RequestBodyLimit]; + var oversizedRecord = new NestRecord { Name = "Synthetic oversized upload" }; + var metadata = JsonSerializer.Serialize(oversizedRecord, JsonOptions); + foreach (var method in new[] { HttpMethod.Post, HttpMethod.Put }) + { + var path = method == HttpMethod.Post ? "api/nests" : $"api/nests/{before[0].Metadata.Id}/file"; + using var content = new MultipartFormDataContent(); + content.Add(new StringContent(metadata), "metadata"); + content.Add(new ByteArrayContent(oversized), "file", "synthetic.nest"); + using var request = new HttpRequestMessage(method, new Uri(new Uri(url + "/"), path)) { Content = content }; + request.Headers.ExpectContinue = true; + using var response = await http.SendAsync(request, ct); + Check(response.StatusCode == HttpStatusCode.RequestEntityTooLarge, + $"{method} of a request over {RequestBodyLimit} bytes must return 413."); + await CheckSnapshot(repository, before, ct); + Console.WriteLine($"Rejected {method} over-limit request: 413; all metadata and hashes unchanged."); + } + + // RemoteNestRepository streams without Expect: 100-continue, so the server may close the + // connection before the client reads the 413. The real client must still report failure. + var saves = new (string Name, Func Save)[] + { + ("upload", () => repository.UploadAsync(oversized, oversizedRecord, ct)), + ("file update", () => repository.UpdateFileAsync(before[0].Metadata.Id, oversized, oversizedRecord, ct)), + }; + foreach (var (name, save) in saves) + { + var outcome = "succeeded"; + try + { + await save(); + } + catch (Exception ex) when (ex is IOException or HttpRequestException) + { + outcome = ex.GetType().Name; + } + Check(outcome != "succeeded", $"Real-client oversized {name} must fail."); + await CheckSnapshot(repository, before, ct); + Console.WriteLine($"Real-client oversized {name} failed ({outcome}); all metadata and hashes unchanged."); + } + } + private static async Task ReadSaved(RemoteNestRepository repository, Nest nest, byte[] archive, NestRecord returned, CancellationToken ct) { @@ -310,7 +385,8 @@ internal static class Program { public static Options Parse(string[] args) { - Check(args.Length > 0 && args[0] is "seed" or "verify" or "reject", "Mode must be seed, verify, or reject."); + Check(args.Length > 0 && args[0] is "seed" or "verify" or "reject" or "limits", + "Mode must be seed, verify, reject, or limits."); string? url = null; string? state = null; var allowNonempty = false; @@ -329,8 +405,8 @@ internal static class Program && uri.Host == "127.0.0.1" && uri.Port > 0 && uri.AbsolutePath == "/" && uri.UserInfo == "" && uri.Query == "" && uri.Fragment == "", "Smoke requires a plain http://127.0.0.1: URL without credentials, path, query, or fragment."); - Check(args[0] == "reject" ? state is null && !allowNonempty : !string.IsNullOrWhiteSpace(state), - "Seed/verify require --state; reject does not accept state or an override."); + Check(args[0] is "reject" or "limits" ? state is null && !allowNonempty : !string.IsNullOrWhiteSpace(state), + "Seed/verify require --state; reject/limits do not accept state or an override."); Check(args[0] == "seed" || !allowNonempty, "Nonempty override is TEST-only and seed-only."); return new Options(args[0], uri!.GetLeftPart(UriPartial.Authority), state, allowNonempty); } @@ -372,8 +448,9 @@ internal static class Program var response = await base.SendAsync(request, ct); try { - if (response.IsSuccessStatusCode && !(request.Method == HttpMethod.Get - && request.RequestUri!.AbsolutePath.EndsWith("/file", StringComparison.Ordinal))) + if (response.IsSuccessStatusCode && response.StatusCode != HttpStatusCode.NoContent + && !(request.Method == HttpMethod.Get + && request.RequestUri!.AbsolutePath.EndsWith("/file", StringComparison.Ordinal))) { using var json = JsonDocument.Parse(await response.Content.ReadAsStringAsync(ct)); if (json.RootElement.ValueKind == JsonValueKind.Array) diff --git a/scripts/Test-ServerContainer.sh b/scripts/Test-ServerContainer.sh index d6330b6..357c6d1 100755 --- a/scripts/Test-ServerContainer.sh +++ b/scripts/Test-ServerContainer.sh @@ -6,7 +6,7 @@ umask 077 # A whole-run bound also leaves time for the exit trap before an outer CI watchdog. if [[ ${OPENNEST_SMOKE_WATCHDOG:-} != 1 ]]; then command -v timeout >/dev/null || { printf 'Missing prerequisite: timeout\n' >&2; exit 2; } - exec timeout --signal=TERM --kill-after=30s 450s env OPENNEST_SMOKE_WATCHDOG=1 \ + exec timeout --signal=TERM --kill-after=30s 600s env OPENNEST_SMOKE_WATCHDOG=1 \ bash "${BASH_SOURCE[0]}" "$@" fi @@ -23,7 +23,7 @@ while (($#)); do esac done [[ -n "$image" && "$image" != -* ]] || { usage; exit 2; } -for command in docker dotnet curl timeout mktemp; do +for command in docker dotnet curl timeout mktemp tar; do command -v "$command" >/dev/null || { printf 'Missing prerequisite: %s\n' "$command" >&2; exit 2; } done root=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd) @@ -38,23 +38,38 @@ run_results=$(mktemp -d "$results/run.XXXXXXXXXX") scratch=$(mktemp -d "$tmp_root/opennest-server-smoke.XXXXXXXXXX") token=${scratch##*/} owner_label='com.opennest.server-smoke.owner' -volume="$token-data" -container_names=("$token-first" "$token-replacement") -cidfiles=("$scratch/first.cid" "$scratch/replacement.cid") -volume_attempted=false -container_attempted=(false false) +# Two service generations on one data volume, a stopped-service backup and restore into +# a second volume, its service, then two storage locations that must fail startup. +container_roles=(first replacement backup restore restored readonly unwritable) +container_names=() +cidfiles=() +container_attempted=() +for role in "${container_roles[@]}"; do + container_names+=("$token-$role") + cidfiles+=("$scratch/$role.cid") + container_attempted+=(false) +done +volumes=("$token-data" "$token-restore-data") +volume_attempted=(false false) # Explicit local default context, ignoring environment overrides for remote daemons. docker_local() { timeout 20s env -u DOCKER_HOST -u DOCKER_CONTEXT docker --context default "$@"; } owned_container() { [[ $(docker_local inspect --format "{{index .Config.Labels \"$owner_label\"}}" "$1") == "$token" ]] } +generated_name() { + local name + for name in "${container_names[@]}"; do + [[ "$1" == "$name" ]] && return 0 + done + return 1 +} cleanup_absent() { local kind=$1 target=$2 remaining filter local -a inventory if [[ "$kind" == container ]]; then filter="id=$target" - if [[ "$target" == "${container_names[0]}" || "$target" == "${container_names[1]}" ]]; then + if generated_name "$target"; then # Generated names contain only alphanumerics, hyphens, and dots. filter="name=^/${target//./\\.}$" fi @@ -95,8 +110,8 @@ cleanup_resource() { fi if [[ "$kind" == container ]]; then # Name recovery resolves a full owned ID before removal, even without a cidfile. - if [[ ! "$resource" =~ ^[0-9a-f]{64}$ || ( "$target" != "${container_names[0]}" \ - && "$target" != "${container_names[1]}" && "$resource" != "$target" ) ]]; then + if [[ ! "$resource" =~ ^[0-9a-f]{64}$ ]] \ + || { ! generated_name "$target" && [[ "$resource" != "$target" ]]; }; then printf 'Cleanup unknown: container %s identity mismatch; refusing removal.\n' \ "$target" >> "$run_results/cleanup.log" return 1 @@ -125,7 +140,7 @@ cleanup() { local status=$? cleanup_failed=0 cid index trap - EXIT INT TERM set +e - for index in 0 1; do + for index in "${!container_names[@]}"; do [[ ${container_attempted[$index]} == true ]] || continue cid='' if [[ -s ${cidfiles[$index]} ]]; then @@ -134,9 +149,10 @@ cleanup() { cleanup_resource container "${cid:-${container_names[$index]}}" "$index" || cleanup_failed=1 done # Docker volume create may return a preexisting name: delete only a matching owner label. - if [[ "$volume_attempted" == true ]]; then - cleanup_resource volume "$volume" || cleanup_failed=1 - fi + for index in "${!volumes[@]}"; do + [[ ${volume_attempted[$index]} == true ]] || continue + cleanup_resource volume "${volumes[$index]}" || cleanup_failed=1 + done rm -rf -- "$scratch" if ((cleanup_failed != 0 && status == 0)); then status=1; fi printf 'exit=%s cleanup_failed=%s\n' "$status" "$cleanup_failed" > "$run_results/outcome.log" @@ -147,64 +163,176 @@ trap cleanup EXIT trap 'exit 130' INT trap 'exit 143' TERM +fail() { printf '%s\n' "$1" >&2; exit 1; } + endpoint=$(docker_local context inspect default --format '{{.Endpoints.docker.Host}}') [[ "$endpoint" == 'unix:///var/run/docker.sock' ]] || { printf 'Refusing a non-local default Docker endpoint.\n' >&2; exit 2; } -docker_local info --format 'daemon={{.Name}} os={{.OSType}}' > "$run_results/daemon.log" +docker_local info --format 'daemon={{.Name}} os={{.OSType}} server={{.ServerVersion}}' > "$run_results/daemon.log" + +# Image contract: numeric non-root user, the HTTP healthcheck, and the source label. docker_local image inspect --format 'image={{.Id}}' "$image" > "$run_results/image.log" +image_user=$(docker_local image inspect --format '{{.Config.User}}' "$image") +healthcheck=$(docker_local image inspect --format '{{.Config.Healthcheck.Test}} {{.Config.Healthcheck.Interval}} {{.Config.Healthcheck.Timeout}} {{.Config.Healthcheck.StartPeriod}} {{.Config.Healthcheck.StartInterval}} {{.Config.Healthcheck.Retries}}' "$image") +labels=$(docker_local image inspect --format '{{index .Config.Labels "org.opencontainers.image.source"}} {{index .Config.Labels "org.opencontainers.image.version"}} {{index .Config.Labels "org.opencontainers.image.revision"}}' "$image") +printf 'user=%s\nhealthcheck=%s\nsource/version/revision=%s\n' "$image_user" "$healthcheck" "$labels" >> "$run_results/image.log" +[[ "$image_user" =~ ^[1-9][0-9]*$ ]] || fail 'Image must declare a numeric non-root USER.' +[[ "$healthcheck" == '[CMD curl --fail --silent --show-error http://127.0.0.1:8090/healthz] 30s 5s 10s 2s 3' ]] \ + || fail 'Image HEALTHCHECK does not match the documented probe.' +[[ "$labels" == 'https://github.com/ajisaacs/OpenNest '?*' '?* ]] || fail 'Image OCI source/version/revision labels are missing.' + printf 'host_sdk=%s\n' "$(dotnet --version)" > "$run_results/toolchain.log" -printf 'owner=%s volume=%s\n' "$token" "$volume" > "$run_results/ownership.log" +printf 'owner=%s volumes=%s\n' "$token" "${volumes[*]}" > "$run_results/ownership.log" timeout 180s dotnet build "$root/scripts/Server.Tests/Server.Tests.csproj" -c Release \ --artifacts-path "$scratch/build" -m:1 /nodeReuse:false > "$run_results/build.log" 2>&1 dll="$scratch/build/bin/Server.Tests/release/Server.Tests.dll" -[[ -f "$dll" ]] || { printf 'Smoke build did not produce the expected DLL.\n' >&2; exit 1; } -# Refuse collisions before creation; the post-create label is the authoritative ownership proof. -if docker_local volume inspect "$volume" >/dev/null 2>&1; then - printf 'Refusing a preexisting volume name.\n' >&2; exit 1 -fi -volume_attempted=true -docker_local volume create --label "$owner_label=$token" "$volume" > "$run_results/volume.log" -[[ $(docker_local volume inspect --format "{{index .Labels \"$owner_label\"}}" "$volume") == "$token" ]] \ - || { printf 'Volume ownership could not be established.\n' >&2; exit 1; } +[[ -f "$dll" ]] || fail 'Smoke build did not produce the expected DLL.' -start_container() { - local index=$1 cid binding - if docker_local inspect "${container_names[$index]}" >/dev/null 2>&1; then +create_volume() { + local index=$1 name=${volumes[$1]} + # Refuse collisions before creation; the post-create label is the authoritative ownership proof. + if docker_local volume inspect "$name" >/dev/null 2>&1; then + printf 'Refusing a preexisting volume name.\n' >&2; return 1 + fi + volume_attempted[$index]=true + docker_local volume create --label "$owner_label=$token" "$name" >> "$run_results/volume.log" + [[ $(docker_local volume inspect --format "{{index .Labels \"$owner_label\"}}" "$name") == "$token" ]] \ + || { printf 'Volume ownership could not be established.\n' >&2; return 1; } +} + +# Sets the global cid. Every container gets the Compose example's privilege hardening. +create_container() { + local index=$1 + shift + if docker_local container inspect "${container_names[$index]}" >/dev/null 2>&1; then printf 'Refusing a preexisting container name.\n' >&2; return 1 fi container_attempted[$index]=true docker_local create --name "${container_names[$index]}" --cidfile "${cidfiles[$index]}" \ - --label "$owner_label=$token" --publish 127.0.0.1::8090 \ - --mount "type=volume,src=$volume,dst=/app/data" "$image" > "$run_results/create-$index.log" + --label "$owner_label=$token" --cap-drop ALL --security-opt no-new-privileges:true \ + "$@" > "$run_results/create-$index.log" + cid='' read -r cid < "${cidfiles[$index]}" || [[ -n "$cid" ]] owned_container "$cid" || { printf 'Container ownership could not be established.\n' >&2; return 1; } + printf 'index=%s name=%s id=%s\n' "$index" "${container_names[$index]}" "$cid" >> "$run_results/ownership.log" +} + +# PID 1 itself, not merely an exec session, must be the image user without capabilities. +check_runtime() { + local index=$1 key a b c d uids='' cap_eff='' no_new_privs='' db_owner + docker_local exec "$cid" cat /proc/1/status > "$scratch/status-$index" + while IFS=$'\t' read -r key a b c d; do + case "$key" in + Uid:) uids="$a $b $c $d" ;; + CapEff:) cap_eff=$a ;; + NoNewPrivs:) no_new_privs=$a ;; + esac + done < "$scratch/status-$index" + db_owner=$(docker_local exec "$cid" stat -c '%u' /app/data/nests.db) + printf 'index=%s uids=%s cap_eff=%s no_new_privs=%s db_owner=%s\n' \ + "$index" "$uids" "$cap_eff" "$no_new_privs" "$db_owner" >> "$run_results/runtime.log" + [[ "$uids" == "$image_user $image_user $image_user $image_user" ]] || fail 'Server process is not the image user.' + [[ "$cap_eff" =~ ^0+$ && "$no_new_privs" == 1 ]] || fail 'Server process retained capabilities or privilege escalation.' + [[ "$db_owner" == "$image_user" ]] || fail 'Database is not owned by the image user.' + docker_local exec "$cid" sh -c 'test -w /app/data && test ! -w /app && test ! -w /app/OpenNest.Server.dll' \ + || fail 'Within /app, only the data directory may be writable by the server user.' +} + +start_service() { + local index=$1 volume_name=$2 binding state deadline + create_container "$index" --publish 127.0.0.1::8090 \ + --mount "type=volume,src=$volume_name,dst=/app/data" "$image" docker_local start "$cid" > "$run_results/start-$index.log" binding=$(docker_local port "$cid" 8090/tcp) [[ "$binding" =~ ^127\.0\.0\.1:([0-9]+)$ ]] || { printf 'Unexpected port binding.\n' >&2; return 1; } url="http://127.0.0.1:${BASH_REMATCH[1]}" - printf 'id=%s binding=%s\n' "$cid" "$binding" >> "$run_results/ownership.log" - local deadline=$((SECONDS + 45)) + printf 'index=%s binding=%s\n' "$index" "$binding" >> "$run_results/ownership.log" + # Ready means Docker's own in-image probe reports healthy and the host sees the same body. + deadline=$((SECONDS + 60)) while ((SECONDS < deadline)); do - [[ $(docker_local inspect --format '{{.State.Running}}' "$cid") == true ]] \ - || { printf 'Container exited before readiness.\n' >&2; return 1; } - if curl --noproxy '*' --fail --silent --show-error --connect-timeout 1 --max-time 2 \ - "$url/healthz" > "$scratch/health.json" 2>> "$run_results/readiness-$index.log"; then - [[ $(< "$scratch/health.json") == '{"status":"ok"}' ]] && return 0 + state=$(docker_local inspect --format '{{.State.Running}} {{.State.Health.Status}}' "$cid") + [[ "$state" == 'true '* ]] || { printf 'Container exited before readiness.\n' >&2; return 1; } + [[ "$state" != 'true unhealthy' ]] || { printf 'Container reported unhealthy.\n' >&2; return 1; } + if [[ "$state" == 'true healthy' ]] && curl --noproxy '*' --fail --silent --show-error \ + --connect-timeout 1 --max-time 2 "$url/healthz" > "$scratch/health.json" \ + 2>> "$run_results/readiness-$index.log"; then + if [[ $(< "$scratch/health.json") == '{"status":"ok"}' ]]; then + check_runtime "$index" + return 0 + fi fi sleep 1 done printf 'Container readiness timed out.\n' >&2; return 1 } -start_container 0 -timeout 100s dotnet "$dll" seed --url "$url" --state "$scratch/state.json" > "$run_results/seed.log" 2>&1 -timeout 100s dotnet "$dll" reject --url "$url" > "$run_results/reject.log" 2>&1 -read -r first_cid < "${cidfiles[0]}" || [[ -n "$first_cid" ]] -docker_local logs "$first_cid" > "$run_results/container-0.log" 2>&1 -docker_local inspect --format 'id={{.Id}} running={{.State.Running}} exit={{.State.ExitCode}}' \ - "$first_cid" > "$run_results/container-0-state.log" -docker_local stop --time 10 "$first_cid" > "$run_results/stop-0.log" -docker_local rm "$first_cid" > "$run_results/remove-0.log" +retire_service() { + local index=$1 retired + read -r retired < "${cidfiles[$index]}" || [[ -n "$retired" ]] + docker_local logs "$retired" > "$run_results/container-$index.log" 2>&1 + docker_local inspect --format 'id={{.Id}} running={{.State.Running}} exit={{.State.ExitCode}}' \ + "$retired" > "$run_results/container-$index-state.log" + docker_local stop --time 10 "$retired" > "$run_results/stop-$index.log" + docker_local rm "$retired" > "$run_results/remove-$index.log" +} + +# An unusable data location must stop the service, never leave a healthy but unusable one. +expect_startup_failure() { + local index=$1 state='' running exit_code health deadline + shift + create_container "$index" --network none "$@" "$image" + docker_local start "$cid" > "$run_results/start-$index.log" + deadline=$((SECONDS + 45)) + while ((SECONDS < deadline)); do + state=$(docker_local inspect --format '{{.State.Running}} {{.State.ExitCode}} {{.State.Health.Status}}' "$cid") + [[ "$state" != *' healthy' ]] || fail 'A service with unusable storage reported healthy.' + [[ "$state" == 'false '* ]] && break + sleep 1 + done + read -r running exit_code health <<< "$state" + docker_local logs "$cid" > "$run_results/startup-failure-$index.log" 2>&1 + printf 'index=%s running=%s exit=%s health=%s\n' "$index" "$running" "$exit_code" "$health" \ + >> "$run_results/startup-failures.log" + [[ "$running" == false && "$exit_code" != 0 && "$health" != healthy ]] \ + || fail 'A service with unusable storage did not fail startup.' + [[ $(< "$run_results/startup-failure-$index.log") == *SqliteException* ]] \ + || fail 'Startup failed for a reason other than storage.' +} + +run_tool() { + local mode=$1 + shift + timeout 100s dotnet "$dll" "$mode" --url "$url" "$@" > "$run_results/$mode-$service.log" 2>&1 +} + +# A fresh named volume must be writable by the non-root image user. +create_volume 0 +service=first +start_service 0 "${volumes[0]}" +run_tool seed --state "$scratch/state.json" +run_tool reject +run_tool limits +retire_service 0 # The replacement gets another Docker-allocated loopback port; only the volume is reused. -start_container 1 -timeout 100s dotnet "$dll" verify --url "$url" --state "$scratch/state.json" > "$run_results/verify.log" 2>&1 -printf 'PASS: real-client create/update/copy, rejections, and persistent recreation.\n' +service=replacement +start_service 1 "${volumes[0]}" +run_tool verify --state "$scratch/state.json" +retire_service 1 + +# Back up the whole data directory from the stopped service (SQLite WAL makes a live +# nests.db copy insufficient), then restore into a separate volume and verify it. +create_container 2 --rm --network none --entrypoint tar \ + --mount "type=volume,src=${volumes[0]},dst=/app/data,readonly" "$image" -C /app/data -cf - . +docker_local start --attach "$cid" > "$scratch/data-backup.tar" 2> "$run_results/backup.log" +tar -tvf "$scratch/data-backup.tar" > "$run_results/backup-contents.log" +backup_entries=$(tar -tf "$scratch/data-backup.tar") +[[ $'\n'"$backup_entries"$'\n' == *$'\n./nests.db\n'* ]] || fail 'Backup does not contain nests.db.' +expect_startup_failure 5 --mount "type=volume,src=${volumes[0]},dst=/app/data,readonly" +expect_startup_failure 6 --mount type=tmpfs,dst=/app/data,tmpfs-mode=0755 +create_volume 1 +create_container 3 --rm --interactive --network none --entrypoint tar \ + --mount "type=volume,src=${volumes[1]},dst=/app/data" "$image" -C /app/data -xf - +docker_local start --attach --interactive "$cid" < "$scratch/data-backup.tar" > "$run_results/restore.log" 2>&1 +service=restored +start_service 4 "${volumes[1]}" +run_tool verify --state "$scratch/state.json" +printf 'PASS: non-root healthy runtime, real-client round trips, rejections, upload limit, persistent recreation, stopped-service backup/restore, and storage startup failures.\n' diff --git a/scripts/test_server_container_cleanup.py b/scripts/test_server_container_cleanup.py index 9d1437b..0375ed7 100644 --- a/scripts/test_server_container_cleanup.py +++ b/scripts/test_server_container_cleanup.py @@ -20,6 +20,9 @@ CID = "b" * 64 FIRST_NAME = "cleanup-unit.first" NAME = "cleanup-unit.replacement" VOLUME = "cleanup-unit-data" +EXTRA_ID = "c" * 64 +EXTRA_NAME = "cleanup-unit.restored" +EXTRA_VOLUME = "cleanup-unit-restore" def mock_docker(state_path, args): @@ -95,7 +98,7 @@ def mock_docker(state_path, args): class CleanupTests(unittest.TestCase): def run_cleanup(self, *, missing_cidfile=False, absent_first=False, absent_all=False, unowned=False, attempted=True, - original_status=0, **faults): + original_status=0, extra=False, **faults): source = Path(__file__).with_name("Test-ServerContainer.sh").read_text() # Extract the actual candidate functions, never a copied cleanup implementation. functions = source[source.index("owned_container() {"):source.index("trap cleanup EXIT")] @@ -111,26 +114,48 @@ class CleanupTests(unittest.TestCase): (scratch / "replacement.cid").write_text(CID + "\n") if absent_first: (scratch / "first.cid").write_text(FIRST_ID + "\n") + if extra: + (scratch / "restored.cid").write_text(EXTRA_ID + "\n") owner = "someone-else" if unowned else TOKEN state_path = root / "virtual-docker.json" + containers = {} if absent_all else {CID: {"name": NAME, "owner": owner}} + volumes = {} if absent_all else {VOLUME: {"owner": owner}} + names = [FIRST_NAME, NAME] + cidfiles = [scratch / "first.cid", scratch / "replacement.cid"] + container_attempted = [absent_first and attempted, attempted] + volume_names = [VOLUME] + if extra: + containers[EXTRA_ID] = {"name": EXTRA_NAME, "owner": owner} + volumes[EXTRA_VOLUME] = {"owner": owner} + names.append(EXTRA_NAME) + cidfiles.append(scratch / "restored.cid") + container_attempted.append(attempted) + volume_names.append(EXTRA_VOLUME) state = { - "containers": {} if absent_all else {CID: {"name": NAME, "owner": owner}}, - "volumes": {} if absent_all else {VOLUME: {"owner": owner}}, + "containers": containers, + "volumes": volumes, "calls": [], "removals": [], **faults, } state_path.write_text(json.dumps(state)) q = shlex.quote + + def bash_array(values): + return "(" + " ".join(q(str(value)) for value in values) + ")" + + def bash_flags(values): + return "(" + " ".join(str(value).lower() for value in values) + ")" + script = "\n".join([ "set -Eeuo pipefail", f"scratch={q(str(scratch))}", f"run_results={q(str(logs))}", "owner_label='com.opennest.server-smoke.owner'", f"token={q(TOKEN)}", - f"volume={q(VOLUME)}", - f"volume_attempted={str(attempted).lower()}", - f"container_attempted=({str(absent_first and attempted).lower()} {str(attempted).lower()})", - f"container_names=({q(FIRST_NAME)} {q(NAME)})", - f"cidfiles=({q(str(scratch / 'first.cid'))} {q(str(scratch / 'replacement.cid'))})", + f"volumes={bash_array(volume_names)}", + f"volume_attempted={bash_flags([attempted] * len(volume_names))}", + f"container_attempted={bash_flags(container_attempted)}", + f"container_names={bash_array(names)}", + f"cidfiles={bash_array(cidfiles)}", f'docker_local() {{ {q(sys.executable)} {q(str(Path(__file__).resolve()))} ' f'--mock-docker {q(str(state_path))} "$@"; }}', functions, @@ -235,6 +260,22 @@ class CleanupTests(unittest.TestCase): self.assert_failed(result, status=23) self.assertIn("Injected Docker container lookup failure status=124", result["log"]) + def test_every_attempted_container_and_volume_is_removed(self): + result = self.run_cleanup(extra=True) + self.assertEqual(0, result["status"], result) + self.assertEqual("exit=0 cleanup_failed=0\n", result["outcome"]) + self.assertEqual([["container", CID], ["container", EXTRA_ID], ["volume", VOLUME], + ["volume", EXTRA_VOLUME]], result["state"]["removals"]) + self.assertFalse(result["state"]["containers"]) + self.assertFalse(result["state"]["volumes"]) + + def test_one_failed_volume_does_not_skip_or_hide_the_others(self): + result = self.run_cleanup(extra=True, fault_kind="volume", remove_mode="error") + self.assert_failed(result) + self.assertEqual([["container", CID], ["container", EXTRA_ID], ["volume", VOLUME], + ["volume", EXTRA_VOLUME]], result["state"]["removals"]) + self.assertFalse(result["state"]["containers"]) + if __name__ == "__main__": if len(sys.argv) > 1 and sys.argv[1] == "--mock-docker":