diff --git a/docker/Dockerfile b/docker/Dockerfile index 648ca16..28ec0cd 100644 --- a/docker/Dockerfile +++ b/docker/Dockerfile @@ -12,5 +12,12 @@ RUN chmod 755 /opt/scripts/*.sh # Expose port 380 EXPOSE 380 -ENTRYPOINT /bin/sh -c 'trap exit TERM; while :; do /opt/scripts/1-renew-cert.sh ; sleep 12h & wait $(jobs -p); done;' +# The base image runs certbot itself (ENTRYPOINT ["certbot"]). Reset it, or the +# CMD below is appended as arguments to certbot instead of replacing it. Both +# directives are in exec form: a shell-form ENTRYPOINT makes docker drop CMD +# altogether, and a shell-form CMD would expand into a nested /bin/sh -c. +# Keeping the loop in CMD is what lets a one-off run replace it outright: +# docker run --rm -v letsencrypt:/etc/letsencrypt certbot certificates +ENTRYPOINT [] +CMD ["/opt/scripts/entrypoint.sh"] diff --git a/docker/scripts/entrypoint.sh b/docker/scripts/entrypoint.sh new file mode 100644 index 0000000..27bc113 --- /dev/null +++ b/docker/scripts/entrypoint.sh @@ -0,0 +1,32 @@ +#!/bin/sh + +# The renewal loop: run a pass, sleep 12 hours, repeat. PID 1 of the container. +# +# Deliberately not `set -e`: a pass that fails — Let's Encrypt unreachable, +# HAProxy refusing the certificate — must not end the loop, it must be retried +# on the next one. +# +# The trap is not decoration. PID 1 has no default signal dispositions, so a +# signal with no handler installed is ignored outright and `docker stop` would +# always have to fall through to SIGKILL. +# +# `sleep &` followed by `wait`, rather than a plain `sleep`: a foreground +# command keeps the shell from running a trap until it finishes, so a plain +# sleep would leave the container deaf to SIGTERM for up to 12 hours. `wait` is +# interruptible. Bare `wait` rather than `wait $(jobs -p)` — the command +# substitution runs in a subshell that reports the parent's jobs in bash but +# not in dash, and bare `wait` waits for every background job in either. +# +# One thing this cannot fix: a signal arriving while 1-renew-cert.sh is in the +# foreground is deferred until that script returns, and an issuance talking to +# Let's Encrypt can outlast docker's 10-second stop grace. Give the service a +# longer `stop_grace_period` if that matters. + +trap 'exit 0' TERM INT + +while :; do + /opt/scripts/1-renew-cert.sh \ + || echo "entrypoint: renewal pass failed, retrying in 12h" >&2 + sleep 12h & + wait +done