feat(server): harden Docker runtime and document persistent deployment

- Run as the .NET image's non-root app user (UID 1654) with root-owned
  application files and an app-owned /app/data that fresh named volumes inherit.
- Add a curl HEALTHCHECK on /healthz (30s/5s/10s/3, 2s start interval), keep the
  explicit 8090 URL and clear the base image's 8080 port default.
- Parameterize VERSION/SOURCE_REVISION and base images; add OCI labels with
  development defaults.
- Add an image-only Compose example (required image and bind address, named
  volume, cap_drop ALL, no-new-privileges) and an env template.
- Document deployment, the upload limit, scoped ownership preparation, and
  checked backup/restore/upgrade functions that refuse existing destinations and
  verify the metadata list and every archive hash before switching volumes.
- Extend the container smoke: image user/healthcheck/label contract, PID 1 UID,
  capabilities and writable paths, Docker-reported health, near-limit and
  oversized uploads, stopped-service backup restored into a second volume, and
  startup failure on read-only and root-owned data mounts.
This commit is contained in:
aj committed 2026-10-02 17:23:36 -04:00
1 parent 142c77a27f
commit d428357c8b
8 files changed
+558 -93

No files matched your search

+2
View File
@@ -11,5 +11,7 @@
**/*.db
**/*.db-*
**/.env*
**/*.env
**/*.env.*
**/*.user
**/Properties/launchSettings.json
+38 -5
View File
@@ -1,5 +1,10 @@
# Build from the repository root: docker build -f OpenNest.Server/Dockerfile -t opennest-server .
FROM mcr.microsoft.com/dotnet/sdk:8.0 AS build
# Release builds pass --build-arg VERSION=X.Y.Z and SOURCE_REVISION=<full commit SHA>;
# the defaults mark an unversioned development build. Base images may be pinned by digest.
ARG SDK_IMAGE=mcr.microsoft.com/dotnet/sdk:8.0
ARG RUNTIME_IMAGE=mcr.microsoft.com/dotnet/aspnet:8.0
FROM ${SDK_IMAGE} AS build
WORKDIR /src
COPY OpenNest.Server/OpenNest.Server.csproj OpenNest.Server/
@@ -10,16 +15,44 @@ RUN dotnet restore OpenNest.Server/OpenNest.Server.csproj
COPY OpenNest.Server/ OpenNest.Server/
COPY OpenNest.Data/ OpenNest.Data/
COPY OpenNest.Core/ OpenNest.Core/
RUN dotnet publish OpenNest.Server/OpenNest.Server.csproj -c Release -o /app --no-restore
ARG VERSION=0.0.0-dev
ARG SOURCE_REVISION=unknown
RUN dotnet publish OpenNest.Server/OpenNest.Server.csproj -c Release -o /app --no-restore \
-p:Version="$VERSION" -p:SourceRevisionId="$SOURCE_REVISION"
FROM mcr.microsoft.com/dotnet/aspnet:8.0 AS runtime
FROM ${RUNTIME_IMAGE} AS runtime
# curl is installed only for the HEALTHCHECK below.
RUN apt-get update \
&& apt-get install -y --no-install-recommends curl \
&& rm -rf /var/lib/apt/lists/*
WORKDIR /app
# Runtime files stay root-owned and read-only to the app user.
COPY --from=build /app .
# SQLite database + .nest blobs live here; mount a volume at this path.
VOLUME /app/data
# SQLite database + .nest blobs live here; mount a volume at this path. The directory
# belongs to the base image's non-root app user (APP_UID 1654), so a fresh named volume
# inherits that ownership. Existing root-owned data needs a one-time scoped chown.
RUN mkdir -p /app/data && chown "$APP_UID:$APP_UID" /app/data && chmod 0750 /app/data
ENV OPENNEST_DB=/app/data/nests.db
# Listen on the explicit 8090 URL; clear the base image's 8080 port default.
ENV ASPNETCORE_URLS=http://+:8090
ENV ASPNETCORE_HTTP_PORTS=
EXPOSE 8090
USER $APP_UID
VOLUME /app/data
HEALTHCHECK --interval=30s --timeout=5s --start-period=10s --start-interval=2s --retries=3 \
CMD ["curl", "--fail", "--silent", "--show-error", "http://127.0.0.1:8090/healthz"]
ARG VERSION=0.0.0-dev
ARG SOURCE_REVISION=unknown
ARG RUNTIME_IMAGE
LABEL org.opencontainers.image.title="OpenNest.Server" \
org.opencontainers.image.description="OpenNest nest-storage HTTP service (SQLite)" \
org.opencontainers.image.source="https://github.com/ajisaacs/OpenNest" \
org.opencontainers.image.licenses="MIT" \
org.opencontainers.image.version="$VERSION" \
org.opencontainers.image.revision="$SOURCE_REVISION" \
org.opencontainers.image.base.name="$RUNTIME_IMAGE"
ENTRYPOINT ["dotnet", "OpenNest.Server.dll"]
+13
View File
@@ -0,0 +1,13 @@
# Copy to a local, untracked env file next to compose.server.yaml and edit it.
# Never put credentials in this file or the Compose file.
# A tested image version or digest, for example ghcr.io/ajisaacs/opennest-server@sha256:<digest>.
OPENNEST_SERVER_IMAGE=ghcr.io/ajisaacs/opennest-server:REPLACE-WITH-TESTED-VERSION
# Loopback for a local test. For shop PCs, use this host's trusted LAN address and
# confirm the firewall admits only trusted clients; the service has no authentication.
OPENNEST_BIND_ADDRESS=127.0.0.1
OPENNEST_HOST_PORT=8090
# Docker volume holding nests.db; change only to switch to a verified restore.
OPENNEST_DATA_VOLUME=opennest-data
+23
View File
@@ -0,0 +1,23 @@
# OpenNest.Server deployment example: one instance, one local data volume, LAN-only.
# Usage: docker compose --env-file <local-env> -f compose.server.yaml up -d
# Start from OpenNest.Server/server.env.example. See docs/nest-storage.md before deploying.
name: opennest-server
services:
opennest-server:
image: ${OPENNEST_SERVER_IMAGE:?Set a tested version or digest}
restart: unless-stopped
ports:
- "${OPENNEST_BIND_ADDRESS:?Set 127.0.0.1 for testing or a trusted LAN address}:${OPENNEST_HOST_PORT:-8090}:8090"
environment:
ASPNETCORE_URLS: http://+:8090
OPENNEST_DB: /app/data/nests.db
volumes:
- opennest-data:/app/data
security_opt:
- no-new-privileges:true
cap_drop:
- ALL
volumes:
opennest-data:
# A restore switches to a separately verified volume instead of overwriting this one.
name: ${OPENNEST_DATA_VOLUME:-opennest-data}
+173 -25
View File
@@ -106,6 +106,8 @@ dotnet run --project OpenNest.Server/OpenNest.Server.csproj
Listens on `ASPNETCORE_URLS` (default Kestrel ports) unless overridden.
### Docker image
Docker must be built from the repository root: the server references
`OpenNest.Data`, which references `OpenNest.Core`. The Dockerfile copies all three
project files before restore and their source/resources before publish, including
@@ -116,29 +118,155 @@ runtime to run; IO/Engine are used only by the host-side smoke tool, not the ima
```sh
docker build --pull -f OpenNest.Server/Dockerfile -t opennest-server:local .
# A release build stamps its version and exact source commit:
docker build --pull -f OpenNest.Server/Dockerfile --build-arg VERSION=X.Y.Z \
--build-arg SOURCE_REVISION="$(git rev-parse HEAD)" -t opennest-server:X.Y.Z .
```
The image listens on `:8090` and keeps `OPENNEST_DB=/app/data/nests.db`. Mount
`/app/data` to preserve the SQLite database and archive blobs across recreation.
For an intentional local run (not needed for the isolated tests below):
Without build arguments the image is labeled version `0.0.0-dev`, revision
`unknown`: a development build, never a release. The OCI labels are
`org.opencontainers.image.source`, `version`, `revision` and `base.name`.
`SDK_IMAGE`/`RUNTIME_IMAGE` build arguments accept digest-pinned base images;
for a release, record the base digests the build log resolved.
Runtime contract:
- Container port 8090 (`ASPNETCORE_URLS=http://+:8090`). Change the host port
mapping, not the container port: the healthcheck probes 8090.
- Runs as the .NET base image's non-root `app` user (UID/GID 1654). Application
files are root-owned and read-only to it; within `/app` only `/app/data` is
writable, and a fresh named volume inherits that directory's ownership. The
database stays `OPENNEST_DB=/app/data/nests.db`. The server also needs the
container's writable `/tmp`: multipart uploads over 64 KiB are buffered there.
- `HEALTHCHECK` runs `curl --fail http://127.0.0.1:8090/healthz` (interval 30 s,
timeout 5 s, start period 10 s, retries 3, and a 2 s start interval during the
start period on Docker 25+). curl is in the image only for it.
- A read-only or unwritable data location fails startup: the process exits
rather than serving requests it cannot store.
- Uploads: Kestrel's default 30,000,000-byte request limit covers the whole
multipart request, so an archive must be slightly smaller (the smoke stores a
29,000,000-byte one). A larger request gets 413 and nothing is stored. The
desktop client streams without `Expect: 100-continue`, so it typically reports
a closed connection instead of the 413 status. The save still fails.
Data written by a root-run image or a root-owned bind mount is not writable by
UID 1654, so startup fails. With the operator's approval, change ownership of
that data location only, once, for example
`docker run --rm --user 0 --entrypoint chown -v <volume>:/app/data <image> -R 1654:1654 /app/data`
(or `chown -R 1654:1654` on the bind-mounted directory itself, never its
parents). The image has no root entrypoint that changes ownership.
### Deploying with Compose
`compose.server.yaml` runs a published image (no local build) with
`no-new-privileges`, all capabilities dropped, and a named data volume. Copy it and
`OpenNest.Server/server.env.example` to a deployment directory, save the env file
under a local name such as `server.env`, and set:
- `OPENNEST_SERVER_IMAGE`: a tested version or, preferably, a digest.
- `OPENNEST_BIND_ADDRESS`: `127.0.0.1` for testing on the host; this host's
trusted LAN address for shop PCs. Do not use `0.0.0.0`. The bind address is only
one layer: verify that the network/firewall admits only trusted clients.
- `OPENNEST_HOST_PORT` (default 8090) and `OPENNEST_DATA_VOLUME` (default `opennest-data`).
```sh
docker run -d -p 127.0.0.1:8090:8090 -v opennest-data:/app/data opennest-server:local
compose="docker compose --env-file server.env -f compose.server.yaml"
$compose up -d
$compose ps # STATUS shows (healthy)
curl --fail http://<bind-address>:<port>/healthz # {"status":"ok"}
```
This example binds only to loopback. Remote shop clients require an explicitly
chosen trusted-interface binding and network access controls; do not casually
replace it with an all-interface publish. The service has no authentication.
These build/smoke instructions do not establish release readiness, non-root
hardening, or a production deployment.
In the desktop app choose **File > Storage Mode...**, Database, and enter the base
URL `http://<bind-address>:<port>`, without `/healthz` or `/api/nests`.
Point the desktop app's **File > Storage Mode...** server URL at
`http://<trusted-host>:8090` (or `http://127.0.0.1:8090` for local use).
### Backup, restore, and upgrade
SQLite runs in WAL mode, so copying `nests.db` from a running service is not a
backup. Back up during a quiet period with no saves. Each step below is a Bash
function that checks every command and stops at the first failure, however it is
called; define them in the shell where `compose` is set:
```bash
url=http://<bind-address>:<port>
# "<id> <archive SHA-256>" for every record, sorted by id. Fails on any failed request.
nest_manifest() (
set -o pipefail
list=$(curl -fsS "$1/api/nests") || exit 1
for id in $(printf '%s' "$list" | grep -o '"id":"[0-9a-f-]*"' | cut -d'"' -f4 | sort); do
hash=$(curl -fsS "$1/api/nests/$id/file" | sha256sum) || exit 1
printf '%s %s\n' "$id" "${hash%% *}"
done
)
# Records the expected manifest and metadata, then archives the stopped data directory.
# The service is restarted even if the archive fails.
backup_nests() ( # usage: backup_nests opennest-data-YYYY-MM-DD
set -o pipefail
fail() { echo "STOP: $*" >&2; exit 1; }
[[ ! -e "$1.tar" ]] || fail "$1.tar already exists"
nest_manifest "$url" > "$1.manifest" || fail "could not record the manifest"
curl -fsS "$url/api/nests" > "$1.metadata.json" || fail "could not record the metadata"
$compose stop || fail "could not stop the service"
$compose run --rm --no-deps -T --entrypoint tar opennest-server -C /app/data -cf - . > "$1.tar"
status=$?
$compose start || fail "could not restart the service"
curl -fs --retry 30 --retry-all-errors --retry-delay 1 "$url/healthz" > /dev/null \
|| fail "the restarted service is not healthy"
[[ $status == 0 ]] || fail "the archive failed; $1.tar is incomplete"
echo "Backed up $(wc -l < "$1.manifest") records to $1.tar"
)
backup_nests opennest-data-YYYY-MM-DD
```
Restore only into a volume created for that restore, never into an existing one
(least of all the active volume). `restore_check` refuses an existing volume,
extracts the backup into a new one, starts a temporary loopback-only container on
it, and requires the exact metadata list and every archive hash to match. It
removes only the container it started; the new volume is kept either way:
```bash
restore_check() ( # usage: restore_check opennest-data-YYYY-MM-DD <image> [new-volume]
set -o pipefail
fail() { echo "STOP: $*" >&2; exit 1; }
restore=${3:-opennest-data-restore-$(date +%Y%m%d-%H%M%S)}
! docker volume inspect "$restore" > /dev/null 2>&1 || fail "volume $restore already exists"
docker volume create "$restore" > /dev/null || fail "could not create volume $restore"
echo "Restoring into new volume $restore"
docker run --rm -i --network none --entrypoint tar -v "$restore:/app/data" "$2" \
-C /app/data -xf - < "$1.tar" || fail "extraction into $restore failed"
check=$(docker create -p 127.0.0.1::8090 -v "$restore:/app/data" "$2") \
|| fail "could not create the check container"
trap 'docker rm -f "$check" > /dev/null' EXIT
docker start "$check" > /dev/null || fail "the check container did not start"
port=$(docker port "$check" 8090/tcp | cut -d: -f2) && [[ -n $port ]] || fail "no check port"
curl -fs --retry 30 --retry-all-errors --retry-delay 1 "http://127.0.0.1:$port/healthz" > /dev/null \
|| fail "the restored service is not healthy"
curl -fsS "http://127.0.0.1:$port/api/nests" | cmp - "$1.metadata.json" || fail "metadata differs"
nest_manifest "http://127.0.0.1:$port" | diff - "$1.manifest" || fail "archives differ"
echo "Verified $restore. Set OPENNEST_DATA_VOLUME=$restore in server.env, then run: \$compose up -d"
)
restore_check opennest-data-YYYY-MM-DD <image>
```
Switch only after `restore_check` prints `Verified`. If it stops, inspect or remove
the new volume it named; the active volume is untouched.
To upgrade, take a stopped-service backup and keep a copy of the current
`server.env`. Prefer a digest in `OPENNEST_SERVER_IMAGE` so that file records
exactly what ran; for a tag, record
`docker image inspect --format '{{join .RepoDigests " "}}' <image>` first. Change
only `OPENNEST_SERVER_IMAGE` and run `$compose up -d`. To roll back, restore the
previous `server.env` and run `$compose up -d`; if the new version wrote data the
old one cannot read, also restore the backup as above. Never run
`docker compose down -v`: it deletes the data volume.
## Isolated container smoke
Prerequisites: a local Linux Docker daemon on the default Unix-socket context,
Bash, curl, GNU `timeout`, `mktemp`, and a .NET SDK able to build `net8.0` projects
Bash, curl, GNU `timeout`, `mktemp`, `tar`, and a .NET SDK able to build `net8.0` projects
(.NET 8 or newer). Restore needs NuGet access or cached packages. Run from the
repository root after building the image:
@@ -150,9 +278,17 @@ scripts/Test-ServerContainer.sh --image opennest-server:local --results /path/to
The wrapper accepts an image, not a URL or an existing volume/container. It uses
only the local default Docker context, creates uniquely labeled disposable
resources, and publishes to `127.0.0.1` on a Docker-allocated port. It builds the
smoke tool once into a private temporary directory, honoring `TMPDIR`; readiness,
HTTP requests, and child commands have watchdogs.
resources, and publishes to `127.0.0.1` on a Docker-allocated port. Every container
runs with the Compose example's `--cap-drop ALL` and `no-new-privileges`. It builds
the smoke tool once into a private temporary directory, honoring `TMPDIR`;
readiness, HTTP requests, and child commands have watchdogs.
Before starting a container it checks the image's numeric non-root `USER`, the
documented `HEALTHCHECK` command and timings (including the start interval), and
the OCI source/version/revision labels. A service is ready only when Docker reports the image's own healthcheck
`healthy` and the host receives `{"status":"ok"}`. Each service then checks that
the server process (PID 1) runs as the image user with no effective capabilities,
owns `nests.db`, and can write `/app/data` but not `/app` or the server DLL.
The test-only `scripts/Server.Tests/Server.Tests.csproj` console exercises the
existing .NET storage client; it is not included in the server image.
@@ -166,21 +302,33 @@ and counts. `reject` snapshots every record's metadata and archive hash, then
checks 400 responses for missing/invalid/null metadata and missing/empty files on
both upload and file-update routes; unknown metadata/file IDs must return 404.
Every rejected operation must leave the snapshot unchanged.
`limits` stores, downloads byte-for-byte, and deletes a 29,000,000-byte synthetic
archive (buffered to disk by the non-root runtime), then requires 413 for a
request over 30,000,000 bytes sent with `Expect: 100-continue`, and a reported
failure from the real client's oversized upload and file update. The snapshot
must be unchanged after each.
The persistence stage stops and removes the first container while retaining its
named volume, starts a replacement on another allocated loopback port, and runs
fresh named volume, starts a replacement on another allocated loopback port, and runs
`verify` against saved IDs, **all** metadata (including server-assigned `savedAt`),
archive hashes, and exact list membership. A missing/invalid state file or failed
assertion exits nonzero. Direct tool use is test-only: it accepts plain loopback
URLs, and `seed` refuses a nonempty server unless `--test-allow-nonempty` is
explicitly supplied for an owned disposable target. Prefer the wrapper; loopback
alone is not proof that an existing server is disposable.
archive hashes, and exact list membership. It then follows the documented
stopped-service backup: `tar` of the whole data directory through the image,
restored into a second owned volume whose service must pass the same `verify`.
The original volume mounted read-only, and a root-owned `tmpfs`, must each make
the server exit nonzero at startup with a SQLite error and never report healthy.
A missing/invalid state file or failed assertion exits nonzero. Direct tool use is
test-only: it accepts plain loopback URLs, and `seed` refuses a nonempty server
unless `--test-allow-nonempty` is explicitly supplied for an owned disposable
target. Prefer the wrapper; loopback alone is not proof that an existing server is
disposable.
Each invocation writes a fresh `run.*` results directory containing build, smoke,
container, ownership, and cleanup logs. On failure those logs remain for diagnosis.
The exit trap removes only invocation-owned containers and the named volume;
transient state, synthetic databases, and build artifacts are removed on success
or failure. It never prunes Docker or alters preexisting services/volumes.
container, image, runtime, ownership, and cleanup logs. On failure those logs remain
for diagnosis. The exit trap removes only invocation-owned containers and named
volumes; transient state, the backup archive, synthetic databases, and build
artifacts are removed on success or failure. It never prunes Docker or alters
preexisting services/volumes.
Cleanup requires successful empty Docker inventory readback, including for an
already-removed container. Unresolved ownership, lookup, removal, or readback
errors make the wrapper fail without deleting unproven resources; inspect
+83 -6
View File
@@ -29,7 +29,8 @@ internal static class Program
using var http = new HttpClient(handler)
{
Timeout = TimeSpan.FromSeconds(10),
MaxResponseContentBufferSize = 16 * 1024 * 1024,
// Only the limits mode downloads its near-cap synthetic archive.
MaxResponseContentBufferSize = (options.Mode == "limits" ? 32 : 16) * 1024 * 1024,
};
using var repository = new RemoteNestRepository(http, options.Url);
switch (options.Mode)
@@ -43,6 +44,9 @@ internal static class Program
case "reject":
await Reject(repository, http, options.Url, watchdog.Token);
break;
case "limits":
await Limits(repository, http, options.Url, watchdog.Token);
break;
}
Console.WriteLine($"PASS {options.Mode}; raw camelCase/string-enum responses checked: {handler.CheckedResponses}");
return 0;
@@ -163,6 +167,77 @@ internal static class Program
Console.WriteLine("Unknown metadata/file ids: 404; all metadata and hashes unchanged.");
}
// Kestrel's default MaxRequestBodySize (30,000,000 bytes) bounds a whole multipart request.
private const int RequestBodyLimit = 30_000_000;
private static async Task Limits(RemoteNestRepository repository, HttpClient http, string url, CancellationToken ct)
{
var before = await Snapshot(repository, ct);
Check(before.Count > 0, "Limit checks require seeded records on an owned server.");
// Opaque synthetic bytes (the server does not parse archives) above the 64 KiB form
// memory threshold, so the non-root runtime must buffer the part to its temp directory.
var accepted = new byte[29_000_000];
RandomNumberGenerator.Fill(accepted);
var created = await repository.UploadAsync(accepted, new NestRecord
{
Name = "Synthetic upload-limit probe",
Comments = "Random bytes; deleted by the same smoke stage",
}, ct);
Check(created.Id != Guid.Empty && created.FileSize == accepted.LongLength,
"A near-limit upload must be stored with its exact size.");
var downloaded = await repository.GetFileAsync(created.Id, ct);
Check(downloaded is not null && downloaded.AsSpan().SequenceEqual(accepted),
"A near-limit archive must download byte-for-byte.");
await CheckMembership(repository, before.Select(s => s.Metadata.Id).Append(created.Id), ct);
await repository.DeleteAsync(created.Id, ct);
await CheckSnapshot(repository, before, ct);
Console.WriteLine($"Accepted, downloaded, and deleted a {accepted.Length}-byte archive; all other records unchanged.");
// Kestrel rejects an oversized declared length before the app reads the form. With
// Expect: 100-continue the client waits for, and must receive, that 413 response.
var oversized = new byte[RequestBodyLimit];
var oversizedRecord = new NestRecord { Name = "Synthetic oversized upload" };
var metadata = JsonSerializer.Serialize(oversizedRecord, JsonOptions);
foreach (var method in new[] { HttpMethod.Post, HttpMethod.Put })
{
var path = method == HttpMethod.Post ? "api/nests" : $"api/nests/{before[0].Metadata.Id}/file";
using var content = new MultipartFormDataContent();
content.Add(new StringContent(metadata), "metadata");
content.Add(new ByteArrayContent(oversized), "file", "synthetic.nest");
using var request = new HttpRequestMessage(method, new Uri(new Uri(url + "/"), path)) { Content = content };
request.Headers.ExpectContinue = true;
using var response = await http.SendAsync(request, ct);
Check(response.StatusCode == HttpStatusCode.RequestEntityTooLarge,
$"{method} of a request over {RequestBodyLimit} bytes must return 413.");
await CheckSnapshot(repository, before, ct);
Console.WriteLine($"Rejected {method} over-limit request: 413; all metadata and hashes unchanged.");
}
// RemoteNestRepository streams without Expect: 100-continue, so the server may close the
// connection before the client reads the 413. The real client must still report failure.
var saves = new (string Name, Func<Task> Save)[]
{
("upload", () => repository.UploadAsync(oversized, oversizedRecord, ct)),
("file update", () => repository.UpdateFileAsync(before[0].Metadata.Id, oversized, oversizedRecord, ct)),
};
foreach (var (name, save) in saves)
{
var outcome = "succeeded";
try
{
await save();
}
catch (Exception ex) when (ex is IOException or HttpRequestException)
{
outcome = ex.GetType().Name;
}
Check(outcome != "succeeded", $"Real-client oversized {name} must fail.");
await CheckSnapshot(repository, before, ct);
Console.WriteLine($"Real-client oversized {name} failed ({outcome}); all metadata and hashes unchanged.");
}
}
private static async Task<SavedRecord> ReadSaved(RemoteNestRepository repository, Nest nest,
byte[] archive, NestRecord returned, CancellationToken ct)
{
@@ -310,7 +385,8 @@ internal static class Program
{
public static Options Parse(string[] args)
{
Check(args.Length > 0 && args[0] is "seed" or "verify" or "reject", "Mode must be seed, verify, or reject.");
Check(args.Length > 0 && args[0] is "seed" or "verify" or "reject" or "limits",
"Mode must be seed, verify, reject, or limits.");
string? url = null;
string? state = null;
var allowNonempty = false;
@@ -329,8 +405,8 @@ internal static class Program
&& uri.Host == "127.0.0.1" && uri.Port > 0 && uri.AbsolutePath == "/"
&& uri.UserInfo == "" && uri.Query == "" && uri.Fragment == "",
"Smoke requires a plain http://127.0.0.1:<port> URL without credentials, path, query, or fragment.");
Check(args[0] == "reject" ? state is null && !allowNonempty : !string.IsNullOrWhiteSpace(state),
"Seed/verify require --state; reject does not accept state or an override.");
Check(args[0] is "reject" or "limits" ? state is null && !allowNonempty : !string.IsNullOrWhiteSpace(state),
"Seed/verify require --state; reject/limits do not accept state or an override.");
Check(args[0] == "seed" || !allowNonempty, "Nonempty override is TEST-only and seed-only.");
return new Options(args[0], uri!.GetLeftPart(UriPartial.Authority), state, allowNonempty);
}
@@ -372,8 +448,9 @@ internal static class Program
var response = await base.SendAsync(request, ct);
try
{
if (response.IsSuccessStatusCode && !(request.Method == HttpMethod.Get
&& request.RequestUri!.AbsolutePath.EndsWith("/file", StringComparison.Ordinal)))
if (response.IsSuccessStatusCode && response.StatusCode != HttpStatusCode.NoContent
&& !(request.Method == HttpMethod.Get
&& request.RequestUri!.AbsolutePath.EndsWith("/file", StringComparison.Ordinal)))
{
using var json = JsonDocument.Parse(await response.Content.ReadAsStringAsync(ct));
if (json.RootElement.ValueKind == JsonValueKind.Array)
+177 -49
View File
@@ -6,7 +6,7 @@ umask 077
# A whole-run bound also leaves time for the exit trap before an outer CI watchdog.
if [[ ${OPENNEST_SMOKE_WATCHDOG:-} != 1 ]]; then
command -v timeout >/dev/null || { printf 'Missing prerequisite: timeout\n' >&2; exit 2; }
exec timeout --signal=TERM --kill-after=30s 450s env OPENNEST_SMOKE_WATCHDOG=1 \
exec timeout --signal=TERM --kill-after=30s 600s env OPENNEST_SMOKE_WATCHDOG=1 \
bash "${BASH_SOURCE[0]}" "$@"
fi
@@ -23,7 +23,7 @@ while (($#)); do
esac
done
[[ -n "$image" && "$image" != -* ]] || { usage; exit 2; }
for command in docker dotnet curl timeout mktemp; do
for command in docker dotnet curl timeout mktemp tar; do
command -v "$command" >/dev/null || { printf 'Missing prerequisite: %s\n' "$command" >&2; exit 2; }
done
root=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd)
@@ -38,23 +38,38 @@ run_results=$(mktemp -d "$results/run.XXXXXXXXXX")
scratch=$(mktemp -d "$tmp_root/opennest-server-smoke.XXXXXXXXXX")
token=${scratch##*/}
owner_label='com.opennest.server-smoke.owner'
volume="$token-data"
container_names=("$token-first" "$token-replacement")
cidfiles=("$scratch/first.cid" "$scratch/replacement.cid")
volume_attempted=false
container_attempted=(false false)
# Two service generations on one data volume, a stopped-service backup and restore into
# a second volume, its service, then two storage locations that must fail startup.
container_roles=(first replacement backup restore restored readonly unwritable)
container_names=()
cidfiles=()
container_attempted=()
for role in "${container_roles[@]}"; do
container_names+=("$token-$role")
cidfiles+=("$scratch/$role.cid")
container_attempted+=(false)
done
volumes=("$token-data" "$token-restore-data")
volume_attempted=(false false)
# Explicit local default context, ignoring environment overrides for remote daemons.
docker_local() { timeout 20s env -u DOCKER_HOST -u DOCKER_CONTEXT docker --context default "$@"; }
owned_container() {
[[ $(docker_local inspect --format "{{index .Config.Labels \"$owner_label\"}}" "$1") == "$token" ]]
}
generated_name() {
local name
for name in "${container_names[@]}"; do
[[ "$1" == "$name" ]] && return 0
done
return 1
}
cleanup_absent() {
local kind=$1 target=$2 remaining filter
local -a inventory
if [[ "$kind" == container ]]; then
filter="id=$target"
if [[ "$target" == "${container_names[0]}" || "$target" == "${container_names[1]}" ]]; then
if generated_name "$target"; then
# Generated names contain only alphanumerics, hyphens, and dots.
filter="name=^/${target//./\\.}$"
fi
@@ -95,8 +110,8 @@ cleanup_resource() {
fi
if [[ "$kind" == container ]]; then
# Name recovery resolves a full owned ID before removal, even without a cidfile.
if [[ ! "$resource" =~ ^[0-9a-f]{64}$ || ( "$target" != "${container_names[0]}" \
&& "$target" != "${container_names[1]}" && "$resource" != "$target" ) ]]; then
if [[ ! "$resource" =~ ^[0-9a-f]{64}$ ]] \
|| { ! generated_name "$target" && [[ "$resource" != "$target" ]]; }; then
printf 'Cleanup unknown: container %s identity mismatch; refusing removal.\n' \
"$target" >> "$run_results/cleanup.log"
return 1
@@ -125,7 +140,7 @@ cleanup() {
local status=$? cleanup_failed=0 cid index
trap - EXIT INT TERM
set +e
for index in 0 1; do
for index in "${!container_names[@]}"; do
[[ ${container_attempted[$index]} == true ]] || continue
cid=''
if [[ -s ${cidfiles[$index]} ]]; then
@@ -134,9 +149,10 @@ cleanup() {
cleanup_resource container "${cid:-${container_names[$index]}}" "$index" || cleanup_failed=1
done
# Docker volume create may return a preexisting name: delete only a matching owner label.
if [[ "$volume_attempted" == true ]]; then
cleanup_resource volume "$volume" || cleanup_failed=1
fi
for index in "${!volumes[@]}"; do
[[ ${volume_attempted[$index]} == true ]] || continue
cleanup_resource volume "${volumes[$index]}" || cleanup_failed=1
done
rm -rf -- "$scratch"
if ((cleanup_failed != 0 && status == 0)); then status=1; fi
printf 'exit=%s cleanup_failed=%s\n' "$status" "$cleanup_failed" > "$run_results/outcome.log"
@@ -147,64 +163,176 @@ trap cleanup EXIT
trap 'exit 130' INT
trap 'exit 143' TERM
fail() { printf '%s\n' "$1" >&2; exit 1; }
endpoint=$(docker_local context inspect default --format '{{.Endpoints.docker.Host}}')
[[ "$endpoint" == 'unix:///var/run/docker.sock' ]] || { printf 'Refusing a non-local default Docker endpoint.\n' >&2; exit 2; }
docker_local info --format 'daemon={{.Name}} os={{.OSType}}' > "$run_results/daemon.log"
docker_local info --format 'daemon={{.Name}} os={{.OSType}} server={{.ServerVersion}}' > "$run_results/daemon.log"
# Image contract: numeric non-root user, the HTTP healthcheck, and the source label.
docker_local image inspect --format 'image={{.Id}}' "$image" > "$run_results/image.log"
image_user=$(docker_local image inspect --format '{{.Config.User}}' "$image")
healthcheck=$(docker_local image inspect --format '{{.Config.Healthcheck.Test}} {{.Config.Healthcheck.Interval}} {{.Config.Healthcheck.Timeout}} {{.Config.Healthcheck.StartPeriod}} {{.Config.Healthcheck.StartInterval}} {{.Config.Healthcheck.Retries}}' "$image")
labels=$(docker_local image inspect --format '{{index .Config.Labels "org.opencontainers.image.source"}} {{index .Config.Labels "org.opencontainers.image.version"}} {{index .Config.Labels "org.opencontainers.image.revision"}}' "$image")
printf 'user=%s\nhealthcheck=%s\nsource/version/revision=%s\n' "$image_user" "$healthcheck" "$labels" >> "$run_results/image.log"
[[ "$image_user" =~ ^[1-9][0-9]*$ ]] || fail 'Image must declare a numeric non-root USER.'
[[ "$healthcheck" == '[CMD curl --fail --silent --show-error http://127.0.0.1:8090/healthz] 30s 5s 10s 2s 3' ]] \
|| fail 'Image HEALTHCHECK does not match the documented probe.'
[[ "$labels" == 'https://github.com/ajisaacs/OpenNest '?*' '?* ]] || fail 'Image OCI source/version/revision labels are missing.'
printf 'host_sdk=%s\n' "$(dotnet --version)" > "$run_results/toolchain.log"
printf 'owner=%s volume=%s\n' "$token" "$volume" > "$run_results/ownership.log"
printf 'owner=%s volumes=%s\n' "$token" "${volumes[*]}" > "$run_results/ownership.log"
timeout 180s dotnet build "$root/scripts/Server.Tests/Server.Tests.csproj" -c Release \
--artifacts-path "$scratch/build" -m:1 /nodeReuse:false > "$run_results/build.log" 2>&1
dll="$scratch/build/bin/Server.Tests/release/Server.Tests.dll"
[[ -f "$dll" ]] || { printf 'Smoke build did not produce the expected DLL.\n' >&2; exit 1; }
# Refuse collisions before creation; the post-create label is the authoritative ownership proof.
if docker_local volume inspect "$volume" >/dev/null 2>&1; then
printf 'Refusing a preexisting volume name.\n' >&2; exit 1
fi
volume_attempted=true
docker_local volume create --label "$owner_label=$token" "$volume" > "$run_results/volume.log"
[[ $(docker_local volume inspect --format "{{index .Labels \"$owner_label\"}}" "$volume") == "$token" ]] \
|| { printf 'Volume ownership could not be established.\n' >&2; exit 1; }
[[ -f "$dll" ]] || fail 'Smoke build did not produce the expected DLL.'
start_container() {
local index=$1 cid binding
if docker_local inspect "${container_names[$index]}" >/dev/null 2>&1; then
create_volume() {
local index=$1 name=${volumes[$1]}
# Refuse collisions before creation; the post-create label is the authoritative ownership proof.
if docker_local volume inspect "$name" >/dev/null 2>&1; then
printf 'Refusing a preexisting volume name.\n' >&2; return 1
fi
volume_attempted[$index]=true
docker_local volume create --label "$owner_label=$token" "$name" >> "$run_results/volume.log"
[[ $(docker_local volume inspect --format "{{index .Labels \"$owner_label\"}}" "$name") == "$token" ]] \
|| { printf 'Volume ownership could not be established.\n' >&2; return 1; }
}
# Sets the global cid. Every container gets the Compose example's privilege hardening.
create_container() {
local index=$1
shift
if docker_local container inspect "${container_names[$index]}" >/dev/null 2>&1; then
printf 'Refusing a preexisting container name.\n' >&2; return 1
fi
container_attempted[$index]=true
docker_local create --name "${container_names[$index]}" --cidfile "${cidfiles[$index]}" \
--label "$owner_label=$token" --publish 127.0.0.1::8090 \
--mount "type=volume,src=$volume,dst=/app/data" "$image" > "$run_results/create-$index.log"
--label "$owner_label=$token" --cap-drop ALL --security-opt no-new-privileges:true \
"$@" > "$run_results/create-$index.log"
cid=''
read -r cid < "${cidfiles[$index]}" || [[ -n "$cid" ]]
owned_container "$cid" || { printf 'Container ownership could not be established.\n' >&2; return 1; }
printf 'index=%s name=%s id=%s\n' "$index" "${container_names[$index]}" "$cid" >> "$run_results/ownership.log"
}
# PID 1 itself, not merely an exec session, must be the image user without capabilities.
check_runtime() {
local index=$1 key a b c d uids='' cap_eff='' no_new_privs='' db_owner
docker_local exec "$cid" cat /proc/1/status > "$scratch/status-$index"
while IFS=$'\t' read -r key a b c d; do
case "$key" in
Uid:) uids="$a $b $c $d" ;;
CapEff:) cap_eff=$a ;;
NoNewPrivs:) no_new_privs=$a ;;
esac
done < "$scratch/status-$index"
db_owner=$(docker_local exec "$cid" stat -c '%u' /app/data/nests.db)
printf 'index=%s uids=%s cap_eff=%s no_new_privs=%s db_owner=%s\n' \
"$index" "$uids" "$cap_eff" "$no_new_privs" "$db_owner" >> "$run_results/runtime.log"
[[ "$uids" == "$image_user $image_user $image_user $image_user" ]] || fail 'Server process is not the image user.'
[[ "$cap_eff" =~ ^0+$ && "$no_new_privs" == 1 ]] || fail 'Server process retained capabilities or privilege escalation.'
[[ "$db_owner" == "$image_user" ]] || fail 'Database is not owned by the image user.'
docker_local exec "$cid" sh -c 'test -w /app/data && test ! -w /app && test ! -w /app/OpenNest.Server.dll' \
|| fail 'Within /app, only the data directory may be writable by the server user.'
}
start_service() {
local index=$1 volume_name=$2 binding state deadline
create_container "$index" --publish 127.0.0.1::8090 \
--mount "type=volume,src=$volume_name,dst=/app/data" "$image"
docker_local start "$cid" > "$run_results/start-$index.log"
binding=$(docker_local port "$cid" 8090/tcp)
[[ "$binding" =~ ^127\.0\.0\.1:([0-9]+)$ ]] || { printf 'Unexpected port binding.\n' >&2; return 1; }
url="http://127.0.0.1:${BASH_REMATCH[1]}"
printf 'id=%s binding=%s\n' "$cid" "$binding" >> "$run_results/ownership.log"
local deadline=$((SECONDS + 45))
printf 'index=%s binding=%s\n' "$index" "$binding" >> "$run_results/ownership.log"
# Ready means Docker's own in-image probe reports healthy and the host sees the same body.
deadline=$((SECONDS + 60))
while ((SECONDS < deadline)); do
[[ $(docker_local inspect --format '{{.State.Running}}' "$cid") == true ]] \
|| { printf 'Container exited before readiness.\n' >&2; return 1; }
if curl --noproxy '*' --fail --silent --show-error --connect-timeout 1 --max-time 2 \
"$url/healthz" > "$scratch/health.json" 2>> "$run_results/readiness-$index.log"; then
[[ $(< "$scratch/health.json") == '{"status":"ok"}' ]] && return 0
state=$(docker_local inspect --format '{{.State.Running}} {{.State.Health.Status}}' "$cid")
[[ "$state" == 'true '* ]] || { printf 'Container exited before readiness.\n' >&2; return 1; }
[[ "$state" != 'true unhealthy' ]] || { printf 'Container reported unhealthy.\n' >&2; return 1; }
if [[ "$state" == 'true healthy' ]] && curl --noproxy '*' --fail --silent --show-error \
--connect-timeout 1 --max-time 2 "$url/healthz" > "$scratch/health.json" \
2>> "$run_results/readiness-$index.log"; then
if [[ $(< "$scratch/health.json") == '{"status":"ok"}' ]]; then
check_runtime "$index"
return 0
fi
fi
sleep 1
done
printf 'Container readiness timed out.\n' >&2; return 1
}
start_container 0
timeout 100s dotnet "$dll" seed --url "$url" --state "$scratch/state.json" > "$run_results/seed.log" 2>&1
timeout 100s dotnet "$dll" reject --url "$url" > "$run_results/reject.log" 2>&1
read -r first_cid < "${cidfiles[0]}" || [[ -n "$first_cid" ]]
docker_local logs "$first_cid" > "$run_results/container-0.log" 2>&1
docker_local inspect --format 'id={{.Id}} running={{.State.Running}} exit={{.State.ExitCode}}' \
"$first_cid" > "$run_results/container-0-state.log"
docker_local stop --time 10 "$first_cid" > "$run_results/stop-0.log"
docker_local rm "$first_cid" > "$run_results/remove-0.log"
retire_service() {
local index=$1 retired
read -r retired < "${cidfiles[$index]}" || [[ -n "$retired" ]]
docker_local logs "$retired" > "$run_results/container-$index.log" 2>&1
docker_local inspect --format 'id={{.Id}} running={{.State.Running}} exit={{.State.ExitCode}}' \
"$retired" > "$run_results/container-$index-state.log"
docker_local stop --time 10 "$retired" > "$run_results/stop-$index.log"
docker_local rm "$retired" > "$run_results/remove-$index.log"
}
# An unusable data location must stop the service, never leave a healthy but unusable one.
expect_startup_failure() {
local index=$1 state='' running exit_code health deadline
shift
create_container "$index" --network none "$@" "$image"
docker_local start "$cid" > "$run_results/start-$index.log"
deadline=$((SECONDS + 45))
while ((SECONDS < deadline)); do
state=$(docker_local inspect --format '{{.State.Running}} {{.State.ExitCode}} {{.State.Health.Status}}' "$cid")
[[ "$state" != *' healthy' ]] || fail 'A service with unusable storage reported healthy.'
[[ "$state" == 'false '* ]] && break
sleep 1
done
read -r running exit_code health <<< "$state"
docker_local logs "$cid" > "$run_results/startup-failure-$index.log" 2>&1
printf 'index=%s running=%s exit=%s health=%s\n' "$index" "$running" "$exit_code" "$health" \
>> "$run_results/startup-failures.log"
[[ "$running" == false && "$exit_code" != 0 && "$health" != healthy ]] \
|| fail 'A service with unusable storage did not fail startup.'
[[ $(< "$run_results/startup-failure-$index.log") == *SqliteException* ]] \
|| fail 'Startup failed for a reason other than storage.'
}
run_tool() {
local mode=$1
shift
timeout 100s dotnet "$dll" "$mode" --url "$url" "$@" > "$run_results/$mode-$service.log" 2>&1
}
# A fresh named volume must be writable by the non-root image user.
create_volume 0
service=first
start_service 0 "${volumes[0]}"
run_tool seed --state "$scratch/state.json"
run_tool reject
run_tool limits
retire_service 0
# The replacement gets another Docker-allocated loopback port; only the volume is reused.
start_container 1
timeout 100s dotnet "$dll" verify --url "$url" --state "$scratch/state.json" > "$run_results/verify.log" 2>&1
printf 'PASS: real-client create/update/copy, rejections, and persistent recreation.\n'
service=replacement
start_service 1 "${volumes[0]}"
run_tool verify --state "$scratch/state.json"
retire_service 1
# Back up the whole data directory from the stopped service (SQLite WAL makes a live
# nests.db copy insufficient), then restore into a separate volume and verify it.
create_container 2 --rm --network none --entrypoint tar \
--mount "type=volume,src=${volumes[0]},dst=/app/data,readonly" "$image" -C /app/data -cf - .
docker_local start --attach "$cid" > "$scratch/data-backup.tar" 2> "$run_results/backup.log"
tar -tvf "$scratch/data-backup.tar" > "$run_results/backup-contents.log"
backup_entries=$(tar -tf "$scratch/data-backup.tar")
[[ $'\n'"$backup_entries"$'\n' == *$'\n./nests.db\n'* ]] || fail 'Backup does not contain nests.db.'
expect_startup_failure 5 --mount "type=volume,src=${volumes[0]},dst=/app/data,readonly"
expect_startup_failure 6 --mount type=tmpfs,dst=/app/data,tmpfs-mode=0755
create_volume 1
create_container 3 --rm --interactive --network none --entrypoint tar \
--mount "type=volume,src=${volumes[1]},dst=/app/data" "$image" -C /app/data -xf -
docker_local start --attach --interactive "$cid" < "$scratch/data-backup.tar" > "$run_results/restore.log" 2>&1
service=restored
start_service 4 "${volumes[1]}"
run_tool verify --state "$scratch/state.json"
printf 'PASS: non-root healthy runtime, real-client round trips, rejections, upload limit, persistent recreation, stopped-service backup/restore, and storage startup failures.\n'
+49 -8
View File
@@ -20,6 +20,9 @@ CID = "b" * 64
FIRST_NAME = "cleanup-unit.first"
NAME = "cleanup-unit.replacement"
VOLUME = "cleanup-unit-data"
EXTRA_ID = "c" * 64
EXTRA_NAME = "cleanup-unit.restored"
EXTRA_VOLUME = "cleanup-unit-restore"
def mock_docker(state_path, args):
@@ -95,7 +98,7 @@ def mock_docker(state_path, args):
class CleanupTests(unittest.TestCase):
def run_cleanup(self, *, missing_cidfile=False, absent_first=False,
absent_all=False, unowned=False, attempted=True,
original_status=0, **faults):
original_status=0, extra=False, **faults):
source = Path(__file__).with_name("Test-ServerContainer.sh").read_text()
# Extract the actual candidate functions, never a copied cleanup implementation.
functions = source[source.index("owned_container() {"):source.index("trap cleanup EXIT")]
@@ -111,26 +114,48 @@ class CleanupTests(unittest.TestCase):
(scratch / "replacement.cid").write_text(CID + "\n")
if absent_first:
(scratch / "first.cid").write_text(FIRST_ID + "\n")
if extra:
(scratch / "restored.cid").write_text(EXTRA_ID + "\n")
owner = "someone-else" if unowned else TOKEN
state_path = root / "virtual-docker.json"
containers = {} if absent_all else {CID: {"name": NAME, "owner": owner}}
volumes = {} if absent_all else {VOLUME: {"owner": owner}}
names = [FIRST_NAME, NAME]
cidfiles = [scratch / "first.cid", scratch / "replacement.cid"]
container_attempted = [absent_first and attempted, attempted]
volume_names = [VOLUME]
if extra:
containers[EXTRA_ID] = {"name": EXTRA_NAME, "owner": owner}
volumes[EXTRA_VOLUME] = {"owner": owner}
names.append(EXTRA_NAME)
cidfiles.append(scratch / "restored.cid")
container_attempted.append(attempted)
volume_names.append(EXTRA_VOLUME)
state = {
"containers": {} if absent_all else {CID: {"name": NAME, "owner": owner}},
"volumes": {} if absent_all else {VOLUME: {"owner": owner}},
"containers": containers,
"volumes": volumes,
"calls": [], "removals": [], **faults,
}
state_path.write_text(json.dumps(state))
q = shlex.quote
def bash_array(values):
return "(" + " ".join(q(str(value)) for value in values) + ")"
def bash_flags(values):
return "(" + " ".join(str(value).lower() for value in values) + ")"
script = "\n".join([
"set -Eeuo pipefail",
f"scratch={q(str(scratch))}",
f"run_results={q(str(logs))}",
"owner_label='com.opennest.server-smoke.owner'",
f"token={q(TOKEN)}",
f"volume={q(VOLUME)}",
f"volume_attempted={str(attempted).lower()}",
f"container_attempted=({str(absent_first and attempted).lower()} {str(attempted).lower()})",
f"container_names=({q(FIRST_NAME)} {q(NAME)})",
f"cidfiles=({q(str(scratch / 'first.cid'))} {q(str(scratch / 'replacement.cid'))})",
f"volumes={bash_array(volume_names)}",
f"volume_attempted={bash_flags([attempted] * len(volume_names))}",
f"container_attempted={bash_flags(container_attempted)}",
f"container_names={bash_array(names)}",
f"cidfiles={bash_array(cidfiles)}",
f'docker_local() {{ {q(sys.executable)} {q(str(Path(__file__).resolve()))} '
f'--mock-docker {q(str(state_path))} "$@"; }}',
functions,
@@ -235,6 +260,22 @@ class CleanupTests(unittest.TestCase):
self.assert_failed(result, status=23)
self.assertIn("Injected Docker container lookup failure status=124", result["log"])
def test_every_attempted_container_and_volume_is_removed(self):
result = self.run_cleanup(extra=True)
self.assertEqual(0, result["status"], result)
self.assertEqual("exit=0 cleanup_failed=0\n", result["outcome"])
self.assertEqual([["container", CID], ["container", EXTRA_ID], ["volume", VOLUME],
["volume", EXTRA_VOLUME]], result["state"]["removals"])
self.assertFalse(result["state"]["containers"])
self.assertFalse(result["state"]["volumes"])
def test_one_failed_volume_does_not_skip_or_hide_the_others(self):
result = self.run_cleanup(extra=True, fault_kind="volume", remove_mode="error")
self.assert_failed(result)
self.assertEqual([["container", CID], ["container", EXTRA_ID], ["volume", VOLUME],
["volume", EXTRA_VOLUME]], result["state"]["removals"])
self.assertFalse(result["state"]["containers"])
if __name__ == "__main__":
if len(sys.argv) > 1 and sys.argv[1] == "--mock-docker":