Build docker image and push to registry.bitdeals.org / main-build-job (push) Successful in 2m21s
The chown at startup was the only thing that ever needed root here, and it served a case this project does not have: keys.dat arriving from a bind mount owned by somebody else. Every deployment uses a named volume, which takes its ownership from the image. So the chown goes, and everything that existed to survive it goes with it. USER bitmessage in the Dockerfile, from PID 1 onwards. drop_privs.py is deleted, the supervisor no longer re-executes itself with a trimmed bounding set, run.sh has no privileged prologue and no wrapper around the eight commands that used to run through one. What is left of run.sh differs from the version before any of this by eleven lines: two chowns gone, seven `gosu bitmessage` prefixes gone, one comment reworded. keys.dat gets its mode 600 at build time instead of on every start, because on every start there is now no root to set it. Its mode and ownership reach a fresh volume from the image, and every volume in service already carries them -- checked on all four live nodes: nothing under /home/bitmessage is owned by anyone but 2000. The setuid strip stays. It is two lines and it closes the one way a taken-over daemon could still have climbed. What the caller sets changes too, and in the right direction: `cap_drop: ALL` with nothing added back, where the previous commit needed seven capabilities handed in. Confinement that used to be split between the image and the caller now sits in one place. The image gives up defending itself when run with no options at all, which is the trade named in the README along with the bind mount it costs.
343 lines
12 KiB
Bash
343 lines
12 KiB
Bash
#!/bin/sh
|
|
|
|
set -eu
|
|
|
|
export BITMESSAGE_API_USER="${BITMESSAGE_API_USER:-bitmessage_api_user}"
|
|
export BITMESSAGE_API_PASSWORD="${BITMESSAGE_API_PASSWORD:-bitmessage_api_password}"
|
|
export BITMESSAGE_SEED_ADDRESSES="${BITMESSAGE_SEED_ADDRESSES:-0}"
|
|
export BITMESSAGE_API_PORT="${BITMESSAGE_API_PORT:-8442}"
|
|
export BITMESSAGE_TTL="${BITMESSAGE_TTL:-172800}"
|
|
export BITMESSAGE_STOPRESENDINGAFTERXDAYS="${BITMESSAGE_STOPRESENDINGAFTERXDAYS:-30}"
|
|
export BITMESSAGE_APIVARIANT="${BITMESSAGE_APIVARIANT:-legacy}"
|
|
export BITMESSAGE_MAXTOTALCONNECTIONS="${BITMESSAGE_MAXTOTALCONNECTIONS:-200}"
|
|
export BITMESSAGE_TRUSTED_PEER="${BITMESSAGE_TRUSTED_PEER:-}"
|
|
export BITMESSAGE_SEND_OUTGOING="${BITMESSAGE_SEND_OUTGOING:-True}"
|
|
export BITMESSAGE_KNOWN_NODES="${BITMESSAGE_KNOWN_NODES:-}"
|
|
# The watchdog at the end of this file. A daemon that has lost every peer does
|
|
# not find its way back on its own, while one that has just started dials hard
|
|
# and does -- so a restart is the cure, and noticing is the whole difference.
|
|
# On by default; False leaves the supervisor holding the daemon and stops it
|
|
# acting. PERIOD is seconds between checks, AFTER how many peerless checks in a
|
|
# row it takes to act, COOLDOWN the floor between two restarts.
|
|
export BITMESSAGE_WATCHDOG="${BITMESSAGE_WATCHDOG:-True}"
|
|
export BITMESSAGE_WATCHDOG_PERIOD="${BITMESSAGE_WATCHDOG_PERIOD:-60}"
|
|
export BITMESSAGE_WATCHDOG_AFTER="${BITMESSAGE_WATCHDOG_AFTER:-5}"
|
|
export BITMESSAGE_WATCHDOG_COOLDOWN="${BITMESSAGE_WATCHDOG_COOLDOWN:-900}"
|
|
|
|
# Reject anything but a plain number: this value is written into keys.dat, and
|
|
# unlike the credentials below it has no business containing characters that
|
|
# esc() would have to neutralise. A typo here would otherwise land in the config
|
|
# as a key the daemon silently ignores.
|
|
case "$BITMESSAGE_MAXTOTALCONNECTIONS" in
|
|
'' | *[!0-9]*)
|
|
echo "BITMESSAGE_MAXTOTALCONNECTIONS must be a positive integer" >&2
|
|
exit 1
|
|
;;
|
|
esac
|
|
|
|
# sendoutgoingconnections is read with safeGetBoolean, which would take "yes" or
|
|
# "1" too; keys.dat is written by hand often enough that it is worth keeping one
|
|
# spelling in it. Anything else is a typo, and a typo here reads as False --
|
|
# a node that quietly never dials out.
|
|
case "$BITMESSAGE_SEND_OUTGOING" in
|
|
[Tt]rue) BITMESSAGE_SEND_OUTGOING=True ;;
|
|
[Ff]alse) BITMESSAGE_SEND_OUTGOING=False ;;
|
|
*)
|
|
echo "BITMESSAGE_SEND_OUTGOING must be True or False" >&2
|
|
exit 1
|
|
;;
|
|
esac
|
|
|
|
# The same two rules again, for the watchdog. Checked here rather than in the
|
|
# loop because a typo would otherwise surface hours later as a supervisor that
|
|
# spins, or one that never acts -- and both look like a working container.
|
|
case "$BITMESSAGE_WATCHDOG" in
|
|
[Tt]rue) BITMESSAGE_WATCHDOG=True ;;
|
|
[Ff]alse) BITMESSAGE_WATCHDOG=False ;;
|
|
*)
|
|
echo "BITMESSAGE_WATCHDOG must be True or False" >&2
|
|
exit 1
|
|
;;
|
|
esac
|
|
|
|
case "$BITMESSAGE_WATCHDOG_PERIOD" in
|
|
'' | *[!0-9]* | 0)
|
|
echo "BITMESSAGE_WATCHDOG_PERIOD must be a positive integer" >&2
|
|
exit 1
|
|
;;
|
|
esac
|
|
|
|
case "$BITMESSAGE_WATCHDOG_AFTER" in
|
|
'' | *[!0-9]* | 0)
|
|
echo "BITMESSAGE_WATCHDOG_AFTER must be a positive integer" >&2
|
|
exit 1
|
|
;;
|
|
esac
|
|
|
|
# Zero is allowed here and means "no floor": restart on every verdict.
|
|
case "$BITMESSAGE_WATCHDOG_COOLDOWN" in
|
|
'' | *[!0-9]*)
|
|
echo "BITMESSAGE_WATCHDOG_COOLDOWN must be a non-negative integer" >&2
|
|
exit 1
|
|
;;
|
|
esac
|
|
|
|
# host:port with a numeric port -- the form both consumers need. PyBitmessage
|
|
# does check trustedpeer itself, but by sys.exit() from a constructor deep in
|
|
# the network thread: the container dies with the reason buried in the daemon
|
|
# log. Fail here, where the message is the first thing in `docker logs`.
|
|
check_peer() {
|
|
case "$1" in
|
|
*:*) ;;
|
|
*) return 1 ;;
|
|
esac
|
|
[ -n "${1%:*}" ] || return 1
|
|
case "${1##*:}" in
|
|
'' | *[!0-9]*) return 1 ;;
|
|
esac
|
|
}
|
|
|
|
if [ -n "$BITMESSAGE_TRUSTED_PEER" ] && ! check_peer "$BITMESSAGE_TRUSTED_PEER"
|
|
then
|
|
echo "BITMESSAGE_TRUSTED_PEER must be host:port" >&2
|
|
exit 1
|
|
fi
|
|
|
|
if [ -z "${BITMESSAGE_SEED_PHRASE:-}" ]
|
|
then
|
|
BITMESSAGE_SEED_PHRASE="$(cat /dev/random | tr -dc "a-z" | head -c32)"
|
|
export BITMESSAGE_SEED_PHRASE
|
|
fi
|
|
|
|
# Escape a value for use on the right-hand side of the sed expressions below.
|
|
# There, a backslash starts an escape, "&" stands for the whole match, and "|"
|
|
# ends the replacement because it is the delimiter. Unescaped, a password
|
|
# containing "&" was silently rewritten into something else and one containing
|
|
# "|" made sed fail outright.
|
|
esc() {
|
|
printf '%s' "$1" | sed -e 's/[\\&|]/\\&/g'
|
|
}
|
|
|
|
# maxtotalconnections is the only brake on a node whose P2P port (8444) is
|
|
# published: it caps inbound sockets at the total minus maxoutboundconnections.
|
|
# The substitution below is a no-op when the key is missing, which would ship a
|
|
# node that looks capped and is not -- and the PyBitmessage clone in the
|
|
# Dockerfile is unpinned, so the stock config is whatever upstream generates
|
|
# today. Add the key rather than trust the substitution alone; line 1 is the
|
|
# [bitmessagesettings] header the daemon reads it from.
|
|
if ! grep -q "^maxtotalconnections = " keys.dat
|
|
then
|
|
sed -i "1a maxtotalconnections = $BITMESSAGE_MAXTOTALCONNECTIONS" keys.dat
|
|
fi
|
|
|
|
# trustedpeer is absent from the stock keys.dat entirely, so the substitution
|
|
# below is a no-op until the key exists -- same trap as maxtotalconnections.
|
|
# The key is added even when the value is empty, which is how it can be taken
|
|
# back off a node that was pinned before: safeGet returns "" and connectionpool
|
|
# falls back to chooseConnection. That empty case is also why the anchors here
|
|
# stop at "=" instead of "= ": with nothing to the right there is no trailing
|
|
# space to match, and the substitution would never fire again.
|
|
if ! grep -q "^trustedpeer =" keys.dat
|
|
then
|
|
sed -i "1a trustedpeer = $(esc "$BITMESSAGE_TRUSTED_PEER")" keys.dat
|
|
fi
|
|
|
|
# Set config values. Every expression is anchored to the start of the line and
|
|
# names its key in the replacement, so no backreference is involved and nothing
|
|
# in another section can match. With set -e a failure here now stops the
|
|
# container instead of leaving the daemon on its previous settings unnoticed --
|
|
# including the case of a bind mount with no keys.dat at all.
|
|
sed -i \
|
|
-e "s|^apiinterface = .*|apiinterface = 0.0.0.0|" \
|
|
-e "s|^apivariant = .*|apivariant = $(esc "$BITMESSAGE_APIVARIANT")|" \
|
|
-e "s|^apiusername = .*|apiusername = $(esc "$BITMESSAGE_API_USER")|" \
|
|
-e "s|^apipassword = .*|apipassword = $(esc "$BITMESSAGE_API_PASSWORD")|" \
|
|
-e "s|^apiport = .*|apiport = $(esc "$BITMESSAGE_API_PORT")|" \
|
|
-e "s|^apienabled = .*|apienabled = True|" \
|
|
-e "s|^ttl = .*|ttl = $(esc "$BITMESSAGE_TTL")|" \
|
|
-e "s|^stopresendingafterxdays = .*|stopresendingafterxdays = $(esc "$BITMESSAGE_STOPRESENDINGAFTERXDAYS")|" \
|
|
-e "s|^maxtotalconnections = .*|maxtotalconnections = $BITMESSAGE_MAXTOTALCONNECTIONS|" \
|
|
-e "s|^trustedpeer =.*|trustedpeer = $(esc "$BITMESSAGE_TRUSTED_PEER")|" \
|
|
-e "s|^sendoutgoingconnections = .*|sendoutgoingconnections = $BITMESSAGE_SEND_OUTGOING|" \
|
|
-e "s|^udp = .*|udp = False|" keys.dat
|
|
|
|
# BITMESSAGE_KNOWN_NODES pins the peers the daemon starts from, and is rewritten
|
|
# on every start: in a private contour the seed *is* the topology, and a file
|
|
# left over from an earlier run names nodes that may no longer exist. Seeding it
|
|
# also switches off the DNS bootstrap -- json_deserialize_knownnodes raises
|
|
# knownNodesActual for any peer that is neither DEFAULT_NODES nor "self", and
|
|
# connectionpool calls startBootstrappers only while that flag is down, so the
|
|
# node never reaches bootstrap8080.bitmessage.org.
|
|
#
|
|
# Writing it "only when the file is missing" would have been a permanent no-op:
|
|
# the image ships a knownnodes.dat, produced by the `pybitmessage -t` run in the
|
|
# Dockerfile, and a named volume inherits it on first use.
|
|
if [ -n "$BITMESSAGE_KNOWN_NODES" ]
|
|
then
|
|
now="$(date +%s)"
|
|
nodes=""
|
|
oldifs="$IFS"
|
|
IFS=","
|
|
for peer in $BITMESSAGE_KNOWN_NODES
|
|
do
|
|
IFS="$oldifs"
|
|
if ! check_peer "$peer"
|
|
then
|
|
echo "BITMESSAGE_KNOWN_NODES entry '$peer' must be host:port" >&2
|
|
exit 1
|
|
fi
|
|
[ -z "$nodes" ] || nodes="$nodes,"
|
|
nodes="$nodes
|
|
{\"stream\": 1, \"peer\": {\"host\": \"${peer%:*}\", \"port\": ${peer##*:}},
|
|
\"info\": {\"lastseen\": $now, \"rating\": 0, \"self\": false}}"
|
|
IFS=","
|
|
done
|
|
IFS="$oldifs"
|
|
printf '[%s\n]\n' "$nodes" > knownnodes.dat
|
|
chmod 600 knownnodes.dat
|
|
fi
|
|
|
|
# generate address from seed
|
|
if [ "$BITMESSAGE_SEED_ADDRESSES" -gt 0 ]
|
|
then
|
|
# Four attempts, not a bash {1..4}: this runs under dash, where brace
|
|
# expansion is literal and the loop would have run once. The call is
|
|
# idempotent (createDeterministicAddresses returns nothing for an address
|
|
# that already exists), so these are retries while the API comes up.
|
|
for i in 1 2 3 4
|
|
do
|
|
sleep 15
|
|
/usr/bin/python /usr/local/bin/seed_addr_gen.py
|
|
done &
|
|
fi
|
|
|
|
# --- the daemon, and the supervisor that owns it --------------------------
|
|
#
|
|
# This file used to end at `exec pybitmessage -d`, which made the daemon PID 1,
|
|
# and it is a poor PID 1. daemonize() double-forks and parks the grandfather in
|
|
# `while True: time.sleep(1)`; the final child then SIGTERMs it to say "ready",
|
|
# and PID 1 drops that signal for want of a handler. Three
|
|
# things followed. The grandfather slept for ever. `docker stop` reached the
|
|
# real daemon only as the SIGKILL ten seconds later -- which is how a startup
|
|
# VACUUM gets cut in half and the node is then trapped retrying it. And a daemon
|
|
# that died on its own left the container Up around a corpse, because what PID 1
|
|
# was doing had nothing to do with whether the daemon was alive.
|
|
#
|
|
# Away from PID 1 that grandfather does die on the ready signal. Measured in
|
|
# this image: the call returns immediately with status 143 and leaves exactly
|
|
# one pybitmessage process behind. So starting the daemon is an ordinary
|
|
# blocking call, and everything below is ordinary shell.
|
|
#
|
|
# What the supervisor does NOT do is act on a daemon whose API is not answering
|
|
# at all. That is the trapped-VACUUM node, a restart does not cure it, and
|
|
# restarting anyway drops the next VACUUM half-done too. watchdog.py reports
|
|
# that case as its own exit code so this loop can leave it alone.
|
|
|
|
#: How long to wait for a daemon to go away before insisting, in seconds. Twice
|
|
#: this is the worst case for a stop, which is what `stop_grace_period` has to
|
|
#: cover -- see the note in the README: a clean PyBitmessage shutdown does not
|
|
#: fit in Docker's default ten seconds, so a compose file that does not raise
|
|
#: the grace period gets the SIGKILL this supervisor exists to avoid.
|
|
STOP_TIMEOUT=30
|
|
|
|
daemon_running() {
|
|
pgrep -f pybitmessage >/dev/null 2>&1
|
|
}
|
|
|
|
start_daemon() {
|
|
set +e
|
|
pybitmessage -d
|
|
rc=$?
|
|
set -e
|
|
# 143 is the ready signal reaching the grandfather, which is this call's
|
|
# ordinary end. Anything but that or a plain 0 never daemonized.
|
|
if [ "$rc" -ne 143 ] && [ "$rc" -ne 0 ]
|
|
then
|
|
echo "watchdog: the daemon did not start (status $rc)" >&2
|
|
return 1
|
|
fi
|
|
echo "watchdog: daemon started"
|
|
}
|
|
|
|
stop_daemon() {
|
|
daemon_running || return 0
|
|
# Through the daemon's own API, which runs doCleanShutdown: the database is
|
|
# closed instead of being cut off mid-write.
|
|
python /usr/local/bin/watchdog.py shutdown || true
|
|
waited=0
|
|
while daemon_running && [ "$waited" -lt "$STOP_TIMEOUT" ]
|
|
do
|
|
sleep 1
|
|
waited=$((waited + 1))
|
|
done
|
|
daemon_running || return 0
|
|
# The API would not answer. TERM, and never KILL: the daemon installs a
|
|
# handler for TERM (setSignalHandler) and shuts down properly on it.
|
|
echo "watchdog: the API did not stop the daemon, sending TERM" >&2
|
|
pkill -TERM -f pybitmessage || true
|
|
waited=0
|
|
while daemon_running && [ "$waited" -lt "$STOP_TIMEOUT" ]
|
|
do
|
|
sleep 1
|
|
waited=$((waited + 1))
|
|
done
|
|
}
|
|
|
|
on_signal() {
|
|
echo "watchdog: stopping"
|
|
stop_daemon
|
|
exit 0
|
|
}
|
|
|
|
trap on_signal TERM INT
|
|
|
|
start_daemon || exit 1
|
|
|
|
streak=0
|
|
last_restart=0
|
|
|
|
while :
|
|
do
|
|
sleep "$BITMESSAGE_WATCHDOG_PERIOD"
|
|
|
|
if ! daemon_running
|
|
then
|
|
echo "watchdog: the daemon is gone, starting it again" >&2
|
|
start_daemon || exit 1
|
|
streak=0
|
|
last_restart="$(date +%s)"
|
|
continue
|
|
fi
|
|
|
|
[ "$BITMESSAGE_WATCHDOG" = True ] || continue
|
|
|
|
set +e
|
|
python /usr/local/bin/watchdog.py peers
|
|
verdict=$?
|
|
set -e
|
|
|
|
# 1 is "answered, and has no peers" -- the only verdict worth acting on.
|
|
# 2 is "did not answer", and the streak starts over rather than carrying an
|
|
# interrupted observation forward.
|
|
if [ "$verdict" -ne 1 ]
|
|
then
|
|
streak=0
|
|
continue
|
|
fi
|
|
|
|
streak=$((streak + 1))
|
|
[ "$streak" -ge "$BITMESSAGE_WATCHDOG_AFTER" ] || continue
|
|
|
|
now="$(date +%s)"
|
|
if [ "$((now - last_restart))" -lt "$BITMESSAGE_WATCHDOG_COOLDOWN" ]
|
|
then
|
|
continue
|
|
fi
|
|
|
|
echo "watchdog: no peers for $streak checks, restarting the daemon" >&2
|
|
stop_daemon
|
|
start_daemon || exit 1
|
|
streak=0
|
|
last_restart="$(date +%s)"
|
|
done
|