diff --git a/.github/scripts/docker-test-replication.sh b/.github/scripts/docker-test-replication.sh new file mode 100755 index 0000000000..f5cdca04ca --- /dev/null +++ b/.github/scripts/docker-test-replication.sh @@ -0,0 +1,718 @@ +#!/usr/bin/env bash +# +# The contents of this file are subject to the terms of the Common Development and +# Distribution License (the License). You may not use this file except in compliance with the +# License. +# +# You can obtain a copy of the License at legal/CDDLv1.0.txt. See the License for the +# specific language governing permission and limitations under the License. +# +# When distributing Covered Software, include this CDDL Header Notice in each file and include +# the License file at legal/CDDLv1.0.txt. If applicable, add the following below the CDDL +# Header, with the fields enclosed by brackets [] replaced by your own identifying +# information: "Portions copyright [year] [name of copyright owner]". +# +# Copyright 2026 3A Systems, LLC. + +# Tests the replication of the Docker image (#1086): the background join of +# OPENDJ_REPLICATION_TYPE=simple (bootstrap/join.sh) and the deprecated one-shot sdsr path of +# bootstrap/replicate.sh. Both image jobs of build.yml run it, each with its own image. +# +# Usage: docker-test-replication.sh + +set -eE -o pipefail + +IMAGE=${1:?usage: $0 } +NODES="dj-0 dj-1 dj-2 dj-x dj-solo dj-bf dj-fi dj-f dj-p0 dj-p1 dj-p2 dj-ipm dj-ipr dj-sdsr dj-shm-probe dj-shm" +VOLUMES="vol-dj-0 vol-dj-1 vol-dj-2 vol-dj-x vol-dj-solo vol-dj-bf vol-dj-fi vol-dj-f vol-dj-ipr vol-dj-p0 vol-dj-p1 vol-dj-p2" +# a bootstrap that imports the base entry and then fails, mounted into dj-bf +FAILING_BOOTSTRAP=$(mktemp) +NETWORK=test_replication +# where dj-sdsr starts: it resolves its own name there, and no other server +NETWORK_ALONE=test_replication_alone + +cleanup() { + rm -f "$FAILING_BOOTSTRAP" + docker rm -f $NODES >/dev/null 2>&1 || true + docker network rm $NETWORK $NETWORK_ALONE >/dev/null 2>&1 || true + docker volume rm -f $VOLUMES >/dev/null 2>&1 || true +} +cleanup +# the container log says that a dsreplication failed and where its detailed log is; the errors +# log of the server and those detailed logs say why. Both are in the instance, which the +# image keeps under /opt/opendj/data, and a stopped container has nothing to exec in +dump_logs() { + local c + for c in $NODES; do + echo "::group::container logs ($c)" + docker logs $c 2>&1 || true + echo "::endgroup::" + [ "$(docker inspect -f '{{.State.Running}}' $c 2>/dev/null)" = true ] || continue + echo "::group::server logs ($c)" + docker exec $c sh -c 'tail -n 1000 /opt/opendj/data/logs/errors; for f in /opt/opendj/data/tmp/opendj-replication-*.log; do [ -f "$f" ] && echo "--- $f" && cat "$f"; done' 2>&1 || true + echo "::endgroup::" + done +} +# set -E hands the trap to every $(...) as well, where a failing command would dump the logs +# into the value being captured and remove the containers under the running test: there the +# failure only ends the subshell, and the main shell decides +trap 'code=$?; [ "$BASH_SUBSHELL" -eq 0 ] || exit $code; dump_logs; cleanup; exit $code' ERR + +fail() { + echo "::error::$1" + false +} + +# grep reads the whole log rather than stopping at the first match (-q), so docker logs never +# writes into a closed pipe, which pipefail would report as a failure of the check +logs_have() { # + docker logs "$1" 2>&1 | grep -F -- "$2" >/dev/null +} + +logs_count() { # + docker logs "$1" 2>&1 | grep -cF -- "$2" || true +} + +# the container logged the text after the last line that holds - after a restart, +# say, which the join of the start before may still have logged into until it was stopped +logs_have_since() { # + docker logs "$1" 2>&1 | awk -v s="$2" 'index($0, s) { buf = ""; next } { buf = buf $0 "\n" } END { printf "%s", buf }' \ + | grep -F -- "$3" >/dev/null +} + +wait_until() { # [...] + local deadline=$((SECONDS + $1)) what=$2 + shift 2 + until "$@"; do + [ "$SECONDS" -lt "$deadline" ] || fail "timed out waiting until $what" + sleep 2 + done +} + +health() { + docker inspect --format='{{.State.Health.Status}}' "$1" +} + +is_healthy() { + [ "$(health "$1")" = healthy ] +} + +wait_healthy() { + wait_until 480 "$1 is healthy" is_healthy "$1" +} + +# A container reports itself healthy only after a probe that finds the marker, and the +# probes run every 30 s: a single look right after a failure reads "starting" whatever the +# join did, so the container is watched over more than two probe cycles +stays_unhealthy() { # + local i + for i in $(seq 1 15); do + is_healthy "$1" && fail "$1 reports itself healthy although $2" + sleep 5 + done +} + +ldaps_has() { # + docker exec "$1" /opt/opendj/bin/ldapsearch --noPropertiesFile --hostname localhost --port 1636 \ + --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll \ + --baseDN "$2" --searchScope base "(objectClass=*)" 1.1 >/dev/null 2>&1 +} + +wait_has() { # + wait_until 60 "$2 is on $1" ldaps_has "$1" "$2" +} + +admin_search() { # [...] + local container=$1 base=$2 scope=$3 filter=$4 + shift 4 + docker exec "$container" /opt/opendj/bin/ldapsearch --noPropertiesFile --hostname localhost --port 4444 \ + --useSsl --trustAll --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" \ + --baseDN "$base" --searchScope "$scope" "$filter" "$@" 2>/dev/null +} + +add_ou() { # + printf 'dn: ou=%s,dc=example,dc=com\nobjectClass: organizationalUnit\nou: %s\n' "$2" "$2" \ + | docker exec -i "$1" /opt/opendj/bin/ldapmodify --noPropertiesFile --hostname localhost --port 1636 \ + --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll --defaultAdd >/dev/null +} + +dsconfig_on() { # ... + local container=$1 + shift + docker exec "$container" /opt/opendj/bin/dsconfig "$@" --hostname localhost --port 4444 \ + --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --trustAll --no-prompt >/dev/null +} + +# the writability-mode the backend of dc=example,dc=com is configured with, empty where its +# entry sets none +backend_mode() { # + admin_search "$1" "ds-cfg-backend-id=userRoot,cn=Backends,cn=config" base "(objectClass=*)" ds-cfg-writability-mode \ + | awk 'tolower($1) == "ds-cfg-writability-mode:" { print $2 }' +} + +# the replication server lists of a container: the one of its replication server and one per +# replication domain (BASE_DN, cn=schema, cn=admin data) +replication_lists() { + admin_search "$1" "cn=config" sub \ + "(|(objectClass=ds-cfg-replication-server)(objectClass=ds-cfg-replication-domain))" ds-cfg-replication-server +} + +lists_lack() { # + ! replication_lists "$1" | grep -F -- "$2" >/dev/null +} + +admin_data_lacks() { # + ! admin_search "$1" "cn=Servers,cn=admin data" one "(objectClass=*)" hostname | grep -F -- "$2" >/dev/null +} + +# the join lock of a server, held by hand as a join of a server named dj-held would hold it +hold_lock() { # + printf 'dn: cn=Docker Join Lock,cn=config\nobjectClass: top\nobjectClass: ds-cfg-branch\nobjectClass: extensibleObject\ncn: Docker Join Lock\ndescription: dj-held %s\n' "$(date +%s)" \ + | docker exec -i "$1" /opt/opendj/bin/ldapmodify --noPropertiesFile --hostname localhost --port 4444 \ + --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll --defaultAdd >/dev/null +} + +drop_lock() { # + printf 'dn: cn=Docker Join Lock,cn=config\nchangetype: delete\n' \ + | docker exec -i "$1" /opt/opendj/bin/ldapmodify --noPropertiesFile --hostname localhost --port 4444 \ + --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll >/dev/null +} + +# hands a lock held by hand over to another holder at once, as a join of that server took it now +relabel_lock() { # + printf 'dn: cn=Docker Join Lock,cn=config\nchangetype: modify\nreplace: description\ndescription: %s %s\n' "$2" "$(date +%s)" \ + | docker exec -i "$1" /opt/opendj/bin/ldapmodify --noPropertiesFile --hostname localhost --port 4444 \ + --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll >/dev/null +} + +# the state the join of the container publishes to its peers +published_state() { # + admin_search "$1" "cn=Docker Join,cn=config" base "(objectClass=*)" description \ + | awk 'tolower($1) == "description:" { print $2 }' +} + +published_ready() { # + [ "$(published_state "$1")" = ready ] +} + +registered_in_dj0() { # + ! admin_data_lacks dj-0 "$1" +} + +# deletes an entry on the container alone: the replication repair control keeps the delete +# from reaching any other server, as an initialize that failed half-way leaves the +# cn=admin data of one server unlike that of the others +delete_here() { # [...] + local container=$1 dn=$2 + shift 2 + docker exec "$container" /opt/opendj/bin/ldapdelete --noPropertiesFile --hostname localhost --port 4444 \ + --useSsl --trustAll --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" \ + --control 1.3.6.1.4.1.26027.1.5.2:true "$@" "$dn" >/dev/null +} + +# the container holds the replication domain of dc=example,dc=com +replicates_example() { # + admin_search "$1" "cn=config" sub "(&(objectClass=ds-cfg-replication-domain)(ds-cfg-base-dn=dc=example,dc=com))" 1.1 \ + | grep "^dn:" >/dev/null +} + +# a client's write to the container is refused with Unwilling to Perform (53), as the backend +# of a server whose replication is down refuses it +refuses_writes() { # + local rc=0 + add_ou "$1" "$2" 2>/dev/null || rc=$? + [ "$rc" -eq 53 ] +} + +# the replication server of the container lists every one of the other names +server_lists() { # ... + local container=$1 list name + shift + list=$(admin_search "$container" "cn=config" sub "(objectClass=ds-cfg-replication-server)" ds-cfg-replication-server) + for name in "$@"; do + grep -F -- "ds-cfg-replication-server: $name:" <<<"$list" >/dev/null || return 1 + done +} + +# every tool reads the root password from a file (#1084, #1092); dsreplication run with -n +# prints no command line, so a password put back on one would pass every check below +rc=0 +docker run --rm --entrypoint grep "$IMAGE" -nE -- '(^|[[:space:]])(-w|--(bindPassword[12]?|adminPassword|rootUserPassword))([[:space:]=]|$)' \ + /opt/opendj/bootstrap/setup.sh /opt/opendj/bootstrap/replicate.sh /opt/opendj/bootstrap/join.sh || rc=$? +[ "$rc" -eq 1 ] || fail "a bootstrap script passes the root password on a command line, or grep could not read them" +# the password files go to /dev/shm, off the writable layer of the container, and the mktemp +# of the image puts them there +docker run --rm --entrypoint grep "$IMAGE" -qF -- 'mktemp -p /dev/shm "opendj-join.$ADMIN_PORT.' /opt/opendj/bootstrap/join.sh \ + || fail "join.sh no longer puts the password file on /dev/shm" +docker run --rm --entrypoint grep "$IMAGE" -qF -- 'mktemp -p /dev/shm "opendj-replicate.$ADMIN_PORT.' /opt/opendj/bootstrap/replicate.sh \ + || fail "replicate.sh no longer puts the password file on /dev/shm" +docker run --rm --entrypoint sh "$IMAGE" -c 'f=$(mktemp -p /dev/shm "opendj-join.$ADMIN_PORT.XXXXXX") && rm -f "$f" && case $f in /dev/shm/opendj-join.4444.*) ;; *) exit 1;; esac' \ + || fail "mktemp in the image does not create the password file on /dev/shm" +# a bounded dsreplication has to take its JVM down with it: the timeout of BusyBox signals +# only the shell script that starts java, that of coreutils the whole process group +docker run --rm --entrypoint sh "$IMAGE" -c 'timeout --version 2>&1 | grep -q "GNU coreutils"' \ + || fail "the image has no timeout of coreutils" + +# a password with a space in it reaches every tool as one value +ROOT_PASSWORD='replication secret' +# a subnet of its own, so that a container can be given a known address +MASTER_ADDRESS=172.30.99.50 +docker network create --subnet 172.30.99.0/24 $NETWORK >/dev/null + +# small retry values keep the seed decision and the failed-join case quick; the volume holds +# the instance so that a container can be replaced with or without its data surviving +start_node() { # [...] + local name=$1 peers=$2 + shift 2 + docker volume create "vol-$name" >/dev/null + docker run -d --memory="512m" --network $NETWORK --name "$name" --hostname "$name" \ + -v "vol-$name:/opt/opendj/data" \ + -e ROOT_PASSWORD="$ROOT_PASSWORD" -e ADD_BASE_ENTRY="--addBaseEntry" \ + -e OPENDJ_REPLICATION_TYPE=simple -e REPLICATION_PEERS="$peers" \ + -e REPLICATION_RETRY_COUNT=10 -e REPLICATION_RETRY_INTERVAL=3 -e REPLICATION_ATTEMPT_TIMEOUT=90 \ + "$@" "$IMAGE" >/dev/null +} + +# the first peer of REPLICATION_PEERS, and only it, seeds a topology its retries could not +# find; it imports entries no other server ever gets but from an initialize +start_node dj-0 dj-0,dj-1 -e SAMPLE_DATA=10 +wait_healthy dj-0 +logs_have dj-0 "seeding it with this server's data" || fail "dj-0 did not seed the topology" + +# a seed that follows a reset lets the writes of clients in again, as a rejoin does: dj-0 is +# left the way a reset cut off after its disable leaves a volume - the backend held, +# $REJOIN_PENDING naming it, no replication domain - and started again while no other peer +# answers, so that it seeds once its retries are exhausted +dsconfig_on dj-0 set-backend-prop --backend-name userRoot --set writability-mode:internal-only +docker exec dj-0 sh -c 'echo userRoot >/opt/opendj/data/.replication-rejoin-pending' +# started once without a join, the volume reports itself healthy - nothing else would ever +# turn it healthy - and its backend stays held, since only a join lets the writes in again: +# the server says which backend that is +docker rm -f dj-0 >/dev/null +docker run -d --memory="512m" --network $NETWORK --name dj-0 --hostname dj-0 -v vol-dj-0:/opt/opendj/data \ + -e ROOT_PASSWORD="$ROOT_PASSWORD" "$IMAGE" >/dev/null +wait_until 300 "dj-0 says that no join lets the writes in" logs_have dj-0 "The backend userRoot may still refuse the writes of clients" +wait_healthy dj-0 +refuses_writes dj-0 held-without-join || fail "dj-0 takes the writes of clients on a backend a rejoin held, without a join to let them in" +docker rm -f dj-0 >/dev/null +start_node dj-0 dj-0,dj-1 +wait_until 300 "the restarted dj-0 leaves its health to the join" logs_have dj-0 "was taken out of its replication topology" +wait_healthy dj-0 +logs_have dj-0 "seeding it with this server's data" || fail "dj-0 did not seed again after the reset" +add_ou dj-0 seeded-after-reset || fail "dj-0 refuses the writes of clients after it seeded" +[ "$(published_state dj-0)" = ready ] || fail "dj-0 does not publish itself ready after it seeded" + +# a joining server tries again while its peer is unreachable +docker network disconnect $NETWORK dj-0 +start_node dj-1 dj-0,dj-1 +wait_until 300 "dj-1 tries again" logs_have dj-1 "trying again in" +docker network connect --alias dj-0 $NETWORK dj-0 +wait_healthy dj-1 +logs_have dj-1 "joined the replication topology through dj-0" || fail "dj-1 did not join through dj-0" +# the bootstrapped volume of dj-1 was initialized from the topology although its BASE_DN held +# the imported base entry - entries in BASE_DN say nothing about who holds the data. The +# sample entries reached dj-0 by import-ldif, never through the changelog, so only an +# initialize carries them +logs_have dj-1 "initializing from dj-0" || fail "dj-1 did not initialize from the topology" +ldaps_has dj-1 "uid=user.0,ou=People,dc=example,dc=com" || fail "dj-1 lacks the entries only an initialize from dj-0 brings" + +# a change made on the seed reaches the replica +add_ou dj-0 replicated +wait_has dj-1 "ou=replicated,dc=example,dc=com" + +# a member is ready again right after a restart, without waiting for its peers - gating it +# on them would deadlock a whole-cluster restart under OrderedReady - and replication still +# flows +docker stop dj-1 >/dev/null +docker restart dj-0 >/dev/null +wait_until 120 "dj-0 is healthy while dj-1 is down" is_healthy dj-0 +docker start dj-1 >/dev/null +wait_healthy dj-1 +add_ou dj-1 replicated2 +wait_has dj-0 "ou=replicated2,dc=example,dc=com" +# and neither of them took its replication down to enable it anew: only a server that no +# peer registers any more does that, or one that its own cn=admin data does not register, +# and a reset followed by a new enable passes every check above +for n in dj-0 dj-1; do + if logs_have $n "taking its replication configuration down"; then fail "$n reset its replication on a plain restart"; fi +done + +# a seed that no peer joined holds the data without a replication domain, and its join binds +# with the ROOT_PASSWORD of the bootstrap: once the root password is changed, a restart is +# still healthy, as it is without replication +start_node dj-solo dj-solo +wait_healthy dj-solo +logs_have dj-solo "seeding it with this server's data" || fail "dj-solo did not seed its topology" +docker exec dj-solo /opt/opendj/bin/ldappasswordmodify --noPropertiesFile --hostname localhost --port 1636 \ + --useSsl --trustAll --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" \ + --authzID "dn:cn=Directory Manager" --currentPassword "$ROOT_PASSWORD" --newPassword "changed $ROOT_PASSWORD" >/dev/null +docker restart dj-solo >/dev/null +wait_until 180 "dj-solo is healthy again with its root password changed" is_healthy dj-solo +# and a peer that refuses the bind with ROOT_PASSWORD is past its bootstrap: it may hold the +# data of the topology, so a first peer that finds no other one gives up beside it rather +# than seed a topology of its own. Its retry interval is written with a leading zero, which +# $(( )) reads as an octal number and rejects for 08: read so, it would abandon the rounds at +# their first pause and give up all the same, without trying again +start_node dj-f dj-f,dj-solo -e REPLICATION_RETRY_COUNT=2 -e REPLICATION_RETRY_INTERVAL=08 +wait_until 600 "dj-f gives up beside a peer that refuses the bind" logs_have dj-f "could not join the replication topology after 2 attempts" +if logs_have dj-f "seeding it with this server's data"; then fail "dj-f seeded next to a peer past its bootstrap"; fi +logs_have dj-f "(1 of 2), trying again in" || fail "dj-f did not try again with REPLICATION_RETRY_INTERVAL=08" +if logs_have dj-f "REPLICATION_RETRY_INTERVAL is not a whole number"; then fail "dj-f did not read REPLICATION_RETRY_INTERVAL=08 as 8"; fi +docker rm -f dj-f >/dev/null +docker volume rm vol-dj-f >/dev/null +docker rm -f dj-solo >/dev/null +docker volume rm vol-dj-solo >/dev/null + +# a Kubernetes pod keeps its /dev/shm across container restarts, shared by all its +# containers: a starting container removes the password files a killed join or replicate.sh +# left there - only those of its own ADMIN_PORT, the files of the other containers of the pod +# are not its to remove +docker run -d --memory="64m" --ipc=shareable --name dj-shm --entrypoint sleep "$IMAGE" 600 >/dev/null +docker exec dj-shm sh -c ': >/dev/shm/opendj-join.4444.killed && : >/dev/shm/opendj-replicate.4444.killed && : >/dev/shm/opendj-join.5444.other' +docker run -d --memory="512m" --network $NETWORK --ipc=container:dj-shm --name dj-shm-probe --hostname dj-shm-probe \ + -e ROOT_PASSWORD="$ROOT_PASSWORD" "$IMAGE" >/dev/null +# the planted files only: the probe's own bootstrap keeps a password file of its ADMIN_PORT +# there while it runs +shm_cleared() { ! docker exec dj-shm sh -c 'test -e /dev/shm/opendj-join.4444.killed || test -e /dev/shm/opendj-replicate.4444.killed'; } +wait_until 60 "run.sh removes the password files of its ADMIN_PORT" shm_cleared +docker exec dj-shm test -e /dev/shm/opendj-join.5444.other || fail "run.sh removed the password file of another container" +docker rm -f dj-shm-probe dj-shm >/dev/null + +# a join that cannot succeed keeps the container from reporting itself healthy - across a +# restart too, where the health marker used to follow the upgrade and a failed join turned +# into a healthy, unreplicated server. dj-x may not seed: it is not the first peer of its list +start_node dj-x dj-absent,dj-x +wait_until 300 "the join of dj-x gives up" logs_have dj-x "could not join the replication topology" +stays_unhealthy dj-x "its join failed" +docker restart dj-x >/dev/null +wait_until 300 "the restarted dj-x waits for its join" logs_have dj-x "never joined its replication topology" +second_failure() { [ "$(logs_count dj-x "could not join the replication topology")" -ge 2 ]; } +wait_until 300 "the second join of dj-x gives up" second_failure +stays_unhealthy dj-x "it never joined" +docker rm -f dj-x >/dev/null +docker volume rm vol-dj-x >/dev/null + +# the seed lost its volume: behind the same name, a fresh dj-0 finds the topology at dj-1 and +# takes its data from it instead of seeding an empty one next to it +docker rm -f dj-0 >/dev/null +docker volume rm vol-dj-0 >/dev/null +start_node dj-0 dj-0,dj-1 +wait_healthy dj-0 +logs_have dj-0 "initializing from dj-1" || fail "the reborn dj-0 did not initialize from dj-1" +if logs_have dj-0 "seeding it with this server's data"; then fail "the reborn dj-0 seeded a topology although dj-1 held it"; fi +ldaps_has dj-0 "ou=replicated,dc=example,dc=com" || fail "the reborn dj-0 lacks the data of the topology" +ldaps_has dj-0 "uid=user.0,ou=People,dc=example,dc=com" || fail "the reborn dj-0 lacks the entries of the seed" + +# a bootstrap that failed after it imported the base entry leaves a volume that never counts +# as holding the data of the topology: once restarted, dj-bf initializes from dj-0 rather +# than joining with what its bootstrap left +printf 'sh /opt/opendj/bootstrap/setup.sh\nexit 1\n' >"$FAILING_BOOTSTRAP" +chmod 644 "$FAILING_BOOTSTRAP" +start_node dj-bf dj-0,dj-1,dj-bf -v "$FAILING_BOOTSTRAP:/opt/opendj/failing-bootstrap.sh:ro" \ + -e BOOTSTRAP=/opt/opendj/failing-bootstrap.sh +wait_until 300 "the bootstrap of dj-bf fails" logs_have dj-bf "failing-bootstrap.sh failed" +docker restart dj-bf >/dev/null +wait_healthy dj-bf +logs_have dj-bf "initializing from dj-0" || fail "dj-bf joined with the data of its failed bootstrap" +ldaps_has dj-bf "uid=user.0,ou=People,dc=example,dc=com" || fail "dj-bf lacks the entries only an initialize from dj-0 brings" +docker rm -f dj-bf >/dev/null +docker volume rm vol-dj-bf >/dev/null + +# an initialize that fails keeps the volume waiting for the data of the topology: a +# one-second bound cuts the dsreplication JVM off before it connects +start_node dj-fi dj-0,dj-fi -e REPLICATION_INITIALIZE_TIMEOUT=1 -e REPLICATION_RETRY_COUNT=2 +wait_until 300 "the initialize of dj-fi is cut off" logs_have dj-fi "initialize from dj-0 exited with 124" +stays_unhealthy dj-fi "its initialize failed" +docker exec dj-fi test -f /opt/opendj/data/.replication-initialize-pending \ + || fail "dj-fi dropped its pending marker after a failed initialize" +docker rm -f dj-fi >/dev/null +docker volume rm vol-dj-fi >/dev/null + +# scale up to three - the peer list is configuration, so the running servers are replaced +# with the longer list before the third one starts, as a rolling update would +docker rm -f dj-0 dj-1 >/dev/null +start_node dj-0 dj-0,dj-1,dj-2 +start_node dj-1 dj-0,dj-1,dj-2 +wait_healthy dj-0 +wait_healthy dj-1 +start_node dj-2 dj-0,dj-1,dj-2 +wait_healthy dj-2 +ldaps_has dj-2 "ou=replicated,dc=example,dc=com" || fail "dj-2 lacks the data of the topology" + +# scale down to two: the survivors, restarted with the shorter list, remove dj-2 from +# cn=admin data and from every replication server list they hold - that of BASE_DN, and +# those of cn=schema and cn=admin data that dsreplication enable configures next to it. +# dsreplication disable cannot, dj-2 being already gone. dj-2 keeps its volume, for the +# scale-up below - and on it, in its cn=config, a join lock held by hand on dj-2 itself +hold_lock dj-2 +docker rm -f dj-2 >/dev/null +docker rm -f dj-0 dj-1 >/dev/null +start_node dj-0 dj-0,dj-1 +start_node dj-1 dj-0,dj-1 +wait_healthy dj-0 +wait_healthy dj-1 +# whichever survivor's join ran first deletes the cn=admin data entry, the delete replicates +# to the other; each survivor prunes its own replication server lists +removed_dj2() { logs_have dj-0 "removing departed server dj-2" || logs_have dj-1 "removing departed server dj-2"; } +wait_until 180 "a survivor removes dj-2" removed_dj2 +for c in dj-0 dj-1; do + wait_until 120 "dj-2 leaves cn=admin data of $c" admin_data_lacks $c dj-2 + wait_until 120 "dj-2 leaves the replication server lists of $c" lists_lack $c dj-2 +done +# replication between the survivors is intact +add_ou dj-0 replicated3 +wait_has dj-1 "ou=replicated3,dc=example,dc=com" + +# scaled up again on the volume it kept: dj-2 still replicates, but the survivors no longer +# register it, and an enable between two servers whose cn=admin data is replicated registers +# nobody - so dj-2 takes its replication configuration down, enables again and registers. +# The join locks are held by hand meanwhile, and dj-2 must wait for each of them rather than +# go on: first its own, which it takes before its configuration goes, then those of both +# survivors, which its enables take. With its replication down it must not report itself +# ready, publish itself ready to its peers, or take the writes of clients, not even after a +# restart. Its enables are bounded generously, so that it does not break the held locks as +# left behind by a killed join - the one on dj-2 is held since the scale-down +hold_lock dj-0 +hold_lock dj-1 +start_node dj-2 dj-0,dj-1,dj-2 -e REPLICATION_RETRY_COUNT=30 -e REPLICATION_ATTEMPT_TIMEOUT=1800 +wait_until 300 "dj-2 finds that the peers no longer register it" logs_have dj-2 "the peers no longer register this server" +wait_until 120 "dj-2 waits for its own lock" logs_have dj-2 "dj-held is enabling replication through this server, waiting for it" +docker exec dj-2 test ! -e /opt/opendj/.bootstrap-complete || fail "dj-2 reports itself ready while it waits to take its replication down" +[ "$(published_state dj-2)" = rejoining ] || fail "dj-2 does not publish that it is rejoining" +# two rounds at most: start_node passes REPLICATION_RETRY_INTERVAL=3, and a round waits up to +# twice as long +sleep 12 +replicates_example dj-2 || fail "dj-2 took its replication down while its own lock was held" +drop_lock dj-2 +wait_until 120 "dj-2 waits for dj-0" logs_have dj-2 "dj-held is enabling replication through dj-0, waiting for it" +wait_until 120 "dj-2 waits for dj-1" logs_have dj-2 "dj-held is enabling replication through dj-1, waiting for it" +docker exec dj-2 test ! -e /opt/opendj/.bootstrap-complete || fail "dj-2 reports itself ready while its replication is down" +refuses_writes dj-2 unreplicated || fail "dj-2 takes the writes of clients while its replication is down" +sleep 12 +admin_data_lacks dj-0 dj-2 || fail "dj-2 enabled while the locks of both peers were held" +docker restart dj-2 >/dev/null +wait_until 300 "the restarted dj-2 leaves its health to the join" logs_have dj-2 "was taken out of its replication topology" +waits_again() { logs_have_since dj-2 "was taken out of its replication topology" "dj-held is enabling replication through dj-0, waiting for it"; } +wait_until 300 "the restarted dj-2 waits for dj-0 again" waits_again +docker exec dj-2 test ! -e /opt/opendj/.bootstrap-complete || fail "the restarted dj-2 reports itself ready while its replication is down" +[ "$(published_state dj-2)" = rejoining ] || fail "the restarted dj-2 publishes itself ready while its replication is down" +refuses_writes dj-2 unreplicated || fail "the restarted dj-2 takes the writes of clients while its replication is down" +# its own lock once more, now with both survivors free: an enable configures both of its +# servers, so it waits for this one as well. The lock is taken as soon as no round of dj-2 +# holds it +wait_until 120 "the lock of dj-2 is held by hand" hold_lock dj-2 +own_waits=$(logs_count dj-2 "dj-held is enabling replication through this server, waiting for it") +drop_lock dj-0 +drop_lock dj-1 +own_waits_again() { [ "$(logs_count dj-2 "dj-held is enabling replication through this server, waiting for it")" -gt "$own_waits" ]; } +wait_until 120 "dj-2 waits for its own lock with both survivors free" own_waits_again +sleep 12 +admin_data_lacks dj-0 dj-2 || fail "dj-2 enabled while its own lock was held" +# a lock under its own name was left by a join of an earlier start of dj-2, a start running +# one join: it is broken at once, not once it is older than an enable may take +self=$(docker exec dj-2 hostname -f | tr '[:upper:]' '[:lower:]') +relabel_lock dj-2 "$self" +wait_until 120 "dj-2 breaks the lock left under its own name" logs_have dj-2 "breaking the lock $self took on localhost" +wait_healthy dj-2 +registered_dj2() { ! admin_data_lacks dj-0 dj-2; } +wait_until 300 "dj-2 registers in the topology again" registered_dj2 +[ "$(published_state dj-2)" = ready ] || fail "the rejoined dj-2 does not publish itself ready" +add_ou dj-2 rejoined +wait_has dj-0 "ou=rejoined,dc=example,dc=com" + +# two servers taken out together and scaled up again together on their volumes: both take +# their replication down, and with it every registration in their cn=admin data. While the +# lock of the survivor is held, neither may enable through the other - an enable between two +# servers without replication registers the two of them only, which passes for membership, +# and they would serve a topology of their own beside dj-0 - so each waits for dj-0 and joins +# through it. The operator made the backend of dj-2 read-only beforehand: its rejoin refuses +# the writes of clients as well, and leaves the mode as it found it. dj-0 keeps the shorter +# list, nothing here starts it again. +# A server that comes back shows its peers the state it had before it left - it is kept in +# its cn=config - until its first round finds that the peers no longer register it. So dj-1 +# starts first and is rejoining before dj-2 starts, and dj-2 keeps a lock held by hand on +# itself from before it left, as in the scale-up above, which no enable through it gets past +# while it still shows that state +dsconfig_on dj-2 set-backend-prop --backend-name userRoot --set writability-mode:internal-only +hold_lock dj-2 +docker rm -f dj-0 dj-1 dj-2 >/dev/null +start_node dj-0 dj-0 +wait_healthy dj-0 +for n in dj-1 dj-2; do + wait_until 180 "$n leaves cn=admin data of dj-0" admin_data_lacks dj-0 $n + wait_until 120 "$n leaves the replication server lists of dj-0" lists_lack dj-0 $n +done +hold_lock dj-0 +start_node dj-1 dj-0,dj-1,dj-2 -e REPLICATION_RETRY_COUNT=30 -e REPLICATION_ATTEMPT_TIMEOUT=1800 +dj1_rejoining() { [ "$(published_state dj-1)" = rejoining ]; } +wait_until 300 "dj-1 publishes that it is rejoining" dj1_rejoining +start_node dj-2 dj-0,dj-1,dj-2 -e REPLICATION_RETRY_COUNT=30 -e REPLICATION_ATTEMPT_TIMEOUT=1800 +wait_until 300 "dj-1 passes over the rejoining dj-2" logs_have dj-1 "dj-2 is rejoining, not a peer to join through" +drop_lock dj-2 +wait_until 180 "dj-2 passes over the rejoining dj-1" logs_have dj-2 "dj-1 is rejoining, not a peer to join through" +sleep 12 +for n in dj-1 dj-2; do + docker exec $n test ! -e /opt/opendj/.bootstrap-complete || fail "$n reports itself ready while its replication is down" +done +admin_data_lacks dj-2 dj-1 || fail "dj-1 and dj-2 registered each other while the lock of dj-0 was held" +admin_data_lacks dj-1 dj-2 || fail "dj-1 and dj-2 registered each other while the lock of dj-0 was held" +refuses_writes dj-1 unreplicated || fail "dj-1 takes the writes of clients while its replication is down" +drop_lock dj-0 +# the health status says nothing yet: dj-1 and dj-2 started on volumes without a reset +# pending, so their run.sh reported them ready, a probe may have found them healthy before +# their joins took the replication down, and the status turns unhealthy only after the +# probe's retries. Their joins publish them ready once the writes are let in again +for n in dj-1 dj-2; do + wait_until 480 "$n rejoins and publishes itself ready" published_ready $n +done +wait_healthy dj-1 +wait_healthy dj-2 +for n in dj-1 dj-2; do + wait_until 300 "$n registers with dj-0 again" registered_in_dj0 $n +done +[ "$(backend_mode dj-2)" = internal-only ] || fail "the rejoin of dj-2 let the writes of clients in on a backend the operator made read-only" +add_ou dj-1 rejoined-together +wait_has dj-0 "ou=rejoined-together,dc=example,dc=com" +wait_has dj-2 "ou=rejoined-together,dc=example,dc=com" + +# an enable that stopped half-way - its initialize of cn=admin data failed - leaves BASE_DN +# replicated and the server registered at its peers, but not in its own cn=admin data, and +# every enable after that exits 5, "already replicated", without registering it there. So a +# server in that state takes its replication down and enables anew, as one that no peer +# registers does. dj-1 lacks its own entry, then dj-2 the whole of cn=Servers, which the +# failed initialize of the CI run behind this case may have left; neither delete reaches a +# peer. One at a time: the disable of a reset changes the peers it replicates with as well +half_enabled_rejoins() { # + local n=$1 members + admin_data_lacks $n $n || fail "$n is still registered in its own cn=admin data" + registered_in_dj0 $n || fail "the delete on $n alone reached dj-0" + members=$(logs_count $n "the health check may probe it") + docker restart $n >/dev/null + wait_until 300 "$n takes its half-enabled replication down" logs_have $n "the peers register this server but its own cn=admin data does not" + member_again() { [ "$(logs_count $n "the health check may probe it")" -gt "$members" ]; } + wait_until 600 "$n rejoins" member_again + published_ready $n || fail "$n does not publish itself ready after it rejoined" + ! admin_data_lacks $n $n || fail "$n is not registered in its own cn=admin data after it rejoined" + registered_in_dj0 $n || fail "$n is not registered with dj-0 after it rejoined" + wait_healthy $n +} +own_entry=$(admin_search dj-1 "cn=Servers,cn=admin data" one "(hostname=dj-1)" 1.1 | awk '/^dn: / { print substr($0, 5) }') +[ -n "$own_entry" ] || fail "dj-1 does not register itself in its cn=admin data" +delete_here dj-1 "$own_entry" || fail "could not delete $own_entry on dj-1 alone" +half_enabled_rejoins dj-1 +delete_here dj-2 "cn=Servers,cn=admin data" --deleteSubtree || fail "could not delete cn=Servers on dj-2 alone" +half_enabled_rejoins dj-2 +add_ou dj-1 rejoined-half-enabled +wait_has dj-0 "ou=rejoined-half-enabled,dc=example,dc=com" +wait_has dj-2 "ou=rejoined-half-enabled,dc=example,dc=com" + +# the deprecated one-shot sdsr path of replicate.sh still bootstraps a replica. On a first +# start a server stops the server its bootstrap started and starts it again, and a replica +# can reach it in between: replicate.sh tries a dsreplication enable that could not connect +# again (exit 8). The replica starts on a network of its own, where dj-0 does not resolve, and +# moves to the network of the topology only once it has said it will try again: taking dj-0 +# off the network for as long instead would leave its replication server holding the dead +# connection of dj-0's own directory server, and route the initialize of the replica into it +docker network create $NETWORK_ALONE >/dev/null +docker run -d --memory="512m" --network $NETWORK_ALONE --name dj-sdsr --hostname dj-sdsr \ + -e ROOT_PASSWORD="$ROOT_PASSWORD" -e MASTER_SERVER=dj-0 -e OPENDJ_REPLICATION_TYPE=sdsr "$IMAGE" >/dev/null +wait_until 420 "dj-sdsr tries again" logs_have dj-sdsr "exited with 8, trying again" +docker network connect $NETWORK dj-sdsr +docker network disconnect $NETWORK_ALONE dj-sdsr +wait_healthy dj-sdsr +wait_has dj-sdsr "ou=replicated3,dc=example,dc=com" +# dj-sdsr publishes no state, as a server of an image before this one does, and it +# replicates BASE_DN, so it may hold the data of the topology: a fresh first peer whose only +# other peer it is gives up rather than seed next to it +start_node dj-f dj-f,dj-sdsr -e REPLICATION_RETRY_COUNT=2 -e REPLICATION_RETRY_INTERVAL=5 +wait_until 300 "dj-f gives up" logs_have dj-f "could not join the replication topology after 2 attempts" +if logs_have dj-f "seeding it with this server's data"; then fail "dj-f seeded next to a member that publishes no state"; fi +stays_unhealthy dj-f "a peer that may hold the data answers" +docker rm -f dj-f >/dev/null +docker volume rm vol-dj-f >/dev/null +# replicate.sh tries dsreplication enable again only when it exits 8, and a failed enable +# ends it: run once more on the replica for a base DN neither server holds, the enable exits +# 5 (REPLICATION_CANNOT_BE_ENABLED_ON_BASEDN) and nothing is tried again or initialized +rc=0 +out=$(docker exec -e BASE_DN=dc=absent,dc=com -e ROOT_USER_DN="cn=Directory Manager" dj-sdsr \ + timeout 90 /opt/opendj/bootstrap/replicate.sh 2>&1) || rc=$? +if [ "$rc" -ne 5 ] || grep -E "trying again|initializing replication" <<<"$out" >/dev/null; then + echo "$out" + fail "replicate.sh for a base DN nobody holds exited with $rc, not with the 5 of its dsreplication enable, or went on after it" +fi + +# the root password shows in no container log, and the files the tools read it from are gone +for c in dj-0 dj-1 dj-2 dj-sdsr; do + if docker logs $c 2>&1 | grep -F -- "$ROOT_PASSWORD"; then fail "the root password is in the log of $c"; fi + left=$(docker exec $c grep -rlsF -- "$ROOT_PASSWORD" /tmp /dev/shm || true) + [ -z "$left" ] || fail "the root password is left in $left of $c" +done +docker rm -f dj-0 dj-1 dj-2 dj-sdsr >/dev/null +docker volume rm vol-dj-0 vol-dj-1 vol-dj-2 >/dev/null + +# MASTER_SERVER may name the master by its address, as the grep of /etc/hosts in the base +# replicate.sh allowed: a master given its own address recognises itself and seeds at once, +# and a replica joins and initializes through that address +for n in dj-ipm dj-ipr; do + if [ $n = dj-ipm ]; then extra="--ip $MASTER_ADDRESS -e SAMPLE_DATA=10"; else extra="-v vol-dj-ipr:/opt/opendj/data"; fi + docker run -d --memory="512m" --network $NETWORK --name $n --hostname $n $extra \ + -e ROOT_PASSWORD="$ROOT_PASSWORD" -e ADD_BASE_ENTRY="--addBaseEntry" \ + -e OPENDJ_REPLICATION_TYPE=simple -e MASTER_SERVER=$MASTER_ADDRESS \ + -e REPLICATION_RETRY_COUNT=10 -e REPLICATION_RETRY_INTERVAL=3 "$IMAGE" >/dev/null + wait_healthy $n +done +logs_have dj-ipm "seeding it with this server's data" || fail "dj-ipm did not recognise itself by its address" +logs_have dj-ipr "initializing from $MASTER_ADDRESS" || fail "dj-ipr did not initialize from the master's address" +ldaps_has dj-ipr "uid=user.0,ou=People,dc=example,dc=com" || fail "dj-ipr lacks the entries of the master" +# moved from MASTER_SERVER to a REPLICATION_PEERS of names, the replica keeps the master that +# is registered by its address - in cn=admin data and in its replication server lists - +# rather than taking it for a server that left the topology +docker rm -f dj-ipr >/dev/null +docker run -d --memory="512m" --network $NETWORK --name dj-ipr --hostname dj-ipr -v vol-dj-ipr:/opt/opendj/data \ + -e ROOT_PASSWORD="$ROOT_PASSWORD" -e OPENDJ_REPLICATION_TYPE=simple -e REPLICATION_PEERS=dj-ipm,dj-ipr \ + -e REPLICATION_RETRY_COUNT=2 -e REPLICATION_RETRY_INTERVAL=3 "$IMAGE" >/dev/null +wait_until 300 "the moved dj-ipr has looked for departed servers" logs_have dj-ipr "the health check may probe it" +if logs_have dj-ipr "removing departed server $MASTER_ADDRESS" || logs_have dj-ipr "removing departed replication server $MASTER_ADDRESS:"; then + fail "dj-ipr removed the master registered by its address" +fi +if admin_data_lacks dj-ipr "$MASTER_ADDRESS"; then fail "the master registered by its address left cn=admin data of dj-ipr"; fi +docker rm -f dj-ipm dj-ipr >/dev/null +docker volume rm vol-dj-ipr >/dev/null + +# three fresh servers started together, as Compose or a StatefulSet with +# podManagementPolicy: Parallel start them: none of them takes the bootstrap data of another +# fresh one for the topology's. The first peer seeds, the others initialize from it - its +# sample entries reach them only that way - and the replication servers know each other +# although the joins ran at the same time +for n in dj-p0 dj-p1 dj-p2; do + if [ $n = dj-p0 ]; then extra="-e SAMPLE_DATA=10"; else extra=; fi + start_node $n dj-p0,dj-p1,dj-p2 -e REPLICATION_RETRY_COUNT=20 $extra +done +for n in dj-p0 dj-p1 dj-p2; do + wait_healthy $n +done +logs_have dj-p0 "seeding it with this server's data" || fail "dj-p0 did not seed the topology" +for n in dj-p1 dj-p2; do + if logs_have $n "seeding it with this server's data"; then fail "$n seeded a topology although it is not the first peer"; fi + logs_have $n "initializing from dj-p0" || fail "$n did not initialize from dj-p0" + if logs_have $n "initializing from dj-p1" || logs_have $n "initializing from dj-p2"; then + fail "$n initialized from a server that holds only bootstrap data" + fi + ldaps_has $n "uid=user.0,ou=People,dc=example,dc=com" || fail "$n lacks the entries of the seed" +done +wait_until 120 "the replication server of dj-p0 knows dj-p1 and dj-p2" server_lists dj-p0 dj-p1 dj-p2 +wait_until 120 "the replication server of dj-p1 knows dj-p0 and dj-p2" server_lists dj-p1 dj-p0 dj-p2 +wait_until 120 "the replication server of dj-p2 knows dj-p0 and dj-p1" server_lists dj-p2 dj-p0 dj-p1 +add_ou dj-p1 parallel +wait_has dj-p0 "ou=parallel,dc=example,dc=com" +wait_has dj-p2 "ou=parallel,dc=example,dc=com" +# dj-p1 and dj-p2 enabled through dj-p0 one at a time, and each removed its lock after +if admin_search dj-p0 "cn=Docker Join Lock,cn=config" base "(objectClass=*)" 1.1 | grep "^dn:" >/dev/null; then + fail "a join left its lock on dj-p0" +fi +for c in dj-p0 dj-p1 dj-p2; do + if docker logs $c 2>&1 | grep -F -- "$ROOT_PASSWORD"; then fail "the root password is in the log of $c"; fi +done + +cleanup +echo "Docker replication test passed" diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index ecab9ef4b5..5932fb93ba 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -478,9 +478,12 @@ jobs: ports: - 5000:5000 steps: + # .github/scripts holds the replication test that both image jobs run - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: - sparse-checkout: .github/benchmark + sparse-checkout: | + .github/benchmark + .github/scripts - name: Download artifacts uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 with: @@ -591,75 +594,10 @@ jobs: docker exec test_uid 'sh' '-c' '/opt/opendj/bin/ldapsearch --hostname localhost --port 1636 --bindDN "cn=Directory Manager" --bindPassword password --useSsl --trustAll --baseDN "dc=example,dc=com" --searchScope base "(objectClass=*)" 1.1' docker kill test_uid - name: Docker test replication + # the background join of OPENDJ_REPLICATION_TYPE=simple and the one-shot sdsr path, the same + # scenarios for both images shell: bash - run: | - IMAGE=localhost:5000/${GITHUB_REPOSITORY,,}:${{ env.release_version }} - REPLICAS="test_replica test_replica_sdsr" - cleanup() { docker rm -f test_master $REPLICAS >/dev/null 2>&1 || true; docker network rm test_replication >/dev/null 2>&1 || true; } - cleanup - trap 'code=$?; for c in test_master $REPLICAS; do echo "::group::container logs ($c)"; docker logs $c 2>&1 || true; echo "::endgroup::"; done; cleanup; exit $code' ERR - # every tool reads the root password from a file (#1084, #1092); dsreplication run with -n prints - # no command line, so a password put back on one would pass every check below - rc=0; docker run --rm --entrypoint grep "$IMAGE" -nE -- '(^|[[:space:]])(-w|--(bindPassword[12]?|adminPassword|rootUserPassword))([[:space:]=]|$)' /opt/opendj/bootstrap/setup.sh /opt/opendj/bootstrap/replicate.sh || rc=$? - if [ $rc -ne 1 ]; then echo "::error::setup.sh or replicate.sh passes the root password on a command line, or grep could not read them"; false; fi - # the password file goes to /dev/shm, off the writable layer of the container, and the mktemp of the image puts it there - docker run --rm --entrypoint grep "$IMAGE" -qF -- 'mktemp -p /dev/shm "opendj-replicate.$ADMIN_PORT.' /opt/opendj/bootstrap/replicate.sh || { echo "::error::replicate.sh no longer puts the password file on /dev/shm"; false; } - docker run --rm --entrypoint sh "$IMAGE" -c 'f=$(mktemp -p /dev/shm "opendj-replicate.$ADMIN_PORT.XXXXXX") && rm -f "$f" && case $f in /dev/shm/opendj-replicate.4444.*) ;; *) exit 1;; esac' || { echo "::error::mktemp in the image does not create the password file on /dev/shm"; false; } - # a password with a space in it reaches every tool as one value - ROOT_PASSWORD='replication secret' - docker network create test_replication - docker run --rm -it -d --memory="512m" --network test_replication --ipc=shareable --name=test_master --hostname=dj-master -e ADD_BASE_ENTRY="--addBaseEntry" -e ROOT_PASSWORD="$ROOT_PASSWORD" "$IMAGE" - timeout 3m bash -c 'until docker inspect --format="{{json .State.Health.Status}}" test_master | grep -q \"healthy\"; do sleep 10; done' - # a replica reports itself healthy only once replicate.sh has succeeded; the sdsr replica joins after - # the simple one, as two dsreplication enable at once would both rewrite the admin data of the master - # a Kubernetes pod keeps its /dev/shm across container restarts: the replica shares the /dev/shm of the master, where - # a password file waits as a killed replicate.sh would have left it, and its run.sh has to remove it (checked below); - # the file of another container of the pod, which listens on another admin port, has to be kept - docker exec test_master sh -c 'printf "%s\n" "$ROOT_PASSWORD" >/dev/shm/opendj-replicate.4444.killed' - docker exec test_master sh -c ': >/dev/shm/opendj-replicate.5444.other' - docker run --rm -it -d --memory="512m" --network test_replication --ipc=container:test_master --name=test_replica --hostname=dj-replica -e ROOT_PASSWORD="$ROOT_PASSWORD" -e MASTER_SERVER=dj-master -e OPENDJ_REPLICATION_TYPE=simple "$IMAGE" - # on a first start the master stops the server its bootstrap started and starts it again, and a replica - # started together with it can reach it in between: replicate.sh tries a dsreplication that could not - # connect again. The master is taken off the network while the replica sleeps before its first try, and - # comes back only once the replica has said it will try again - timeout 5m bash -c 'until docker logs test_replica 2>&1 | grep -q "Will sleep for a bit"; do sleep 0.2; done' - docker network disconnect test_replication test_master - timeout 2m bash -c 'until docker logs test_replica 2>&1 | grep -q "exited with 8, trying again"; do sleep 1; done' - docker network connect --alias dj-master test_replication test_master - timeout 5m bash -c 'until docker inspect --format="{{json .State.Health.Status}}" test_replica | grep -q \"healthy\"; do sleep 10; done' - docker run --rm -it -d --memory="512m" --network test_replication --name=test_replica_sdsr --hostname=dj-replica-sdsr -e ROOT_PASSWORD="$ROOT_PASSWORD" -e MASTER_SERVER=dj-master -e OPENDJ_REPLICATION_TYPE=sdsr "$IMAGE" - # the same for the sdsr replica, whose dsreplication enable is another command of replicate.sh - timeout 5m bash -c 'until docker logs test_replica_sdsr 2>&1 | grep -q "Will sleep for a bit"; do sleep 0.2; done' - docker network disconnect test_replication test_master - timeout 2m bash -c 'until docker logs test_replica_sdsr 2>&1 | grep -q "exited with 8, trying again"; do sleep 1; done' - docker network connect --alias dj-master test_replication test_master - timeout 5m bash -c 'until docker inspect --format="{{json .State.Health.Status}}" test_replica_sdsr | grep -q \"healthy\"; do sleep 10; done' - # the replicas were initialized from the master, and a change made on the master reaches them - for c in $REPLICAS; do - docker exec $c /opt/opendj/bin/ldapsearch --hostname localhost --port 1636 --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll --baseDN "dc=example,dc=com" --searchScope base "(objectClass=*)" 1.1 - done - printf 'dn: ou=replicated,dc=example,dc=com\nobjectClass: organizationalUnit\nou: replicated\n' | docker exec -i test_master /opt/opendj/bin/ldapmodify --hostname localhost --port 1636 --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll --defaultAdd - for c in $REPLICAS; do - timeout 1m bash -c 'until docker exec $1 /opt/opendj/bin/ldapsearch --hostname localhost --port 1636 --bindDN "cn=Directory Manager" --bindPassword "$0" --useSsl --trustAll --baseDN "ou=replicated,dc=example,dc=com" --searchScope base "(objectClass=*)" 1.1; do sleep 5; done' "$ROOT_PASSWORD" $c - done - # replicate.sh tries dsreplication enable again only when it exits 8, and a failed enable ends it: run once more - # on the replica, the enable of a base DN already replicated exits 5 and nothing is tried again or initialized - rc=0; out=$(docker exec -e BASE_DN=dc=example,dc=com -e ROOT_USER_DN="cn=Directory Manager" test_replica timeout 90 /opt/opendj/bootstrap/replicate.sh 2>&1) || rc=$? - if [ $rc -ne 5 ] || grep -qE "trying again|initializing replication" <<<"$out"; then - echo "$out"; echo "::error::a second replicate.sh exited with $rc, not with the 5 of its dsreplication enable, or went on after it"; false - fi - # the root password shows in no container log, and the files setup.sh and replicate.sh passed it in are gone (#1084, #1092) - for c in test_master $REPLICAS; do - if docker logs $c 2>&1 | grep -F "$ROOT_PASSWORD"; then echo "::error::The root password is in the log of $c"; false; fi - done - for c in test_master $REPLICAS; do - # a JVM keeps its command line in /tmp/hsperfdata_* while it runs; the HEALTHCHECK no longer binds as root (#1092), - # so no process left running has the root password on it - left=$(docker exec $c grep -rlsF -- "$ROOT_PASSWORD" /tmp /dev/shm || true) - if [ -n "$left" ]; then echo "::error::The root password is left in $left of $c"; false; fi - done - docker exec test_replica test -e /dev/shm/opendj-replicate.5444.other || { echo "::error::run.sh of test_replica removed the password file of another container"; false; } - cleanup + run: .github/scripts/docker-test-replication.sh "localhost:5000/${GITHUB_REPOSITORY,,}:${{ env.release_version }}" - name: Docker test secret volume # a keystore mounted at SECRET_VOLUME is what LDAPS serves from the first start on, a # renewed one - a new password included - is copied while the server runs and served @@ -933,9 +871,12 @@ jobs: ports: - 5000:5000 steps: + # .github/scripts holds the replication test that both image jobs run - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: - sparse-checkout: .github/benchmark + sparse-checkout: | + .github/benchmark + .github/scripts - name: Download artifacts uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 with: @@ -1047,75 +988,10 @@ jobs: docker exec test_uid 'sh' '-c' '/opt/opendj/bin/ldapsearch --hostname localhost --port 1636 --bindDN "cn=Directory Manager" --bindPassword password --useSsl --trustAll --baseDN "dc=example,dc=com" --searchScope base "(objectClass=*)" 1.1' docker kill test_uid - name: Docker test replication + # the background join of OPENDJ_REPLICATION_TYPE=simple and the one-shot sdsr path, the same + # scenarios for both images shell: bash - run: | - IMAGE=localhost:5000/${GITHUB_REPOSITORY,,}:${{ env.release_version }}-alpine - REPLICAS="test_replica test_replica_sdsr" - cleanup() { docker rm -f test_master $REPLICAS >/dev/null 2>&1 || true; docker network rm test_replication >/dev/null 2>&1 || true; } - cleanup - trap 'code=$?; for c in test_master $REPLICAS; do echo "::group::container logs ($c)"; docker logs $c 2>&1 || true; echo "::endgroup::"; done; cleanup; exit $code' ERR - # every tool reads the root password from a file (#1084, #1092); dsreplication run with -n prints - # no command line, so a password put back on one would pass every check below - rc=0; docker run --rm --entrypoint grep "$IMAGE" -nE -- '(^|[[:space:]])(-w|--(bindPassword[12]?|adminPassword|rootUserPassword))([[:space:]=]|$)' /opt/opendj/bootstrap/setup.sh /opt/opendj/bootstrap/replicate.sh || rc=$? - if [ $rc -ne 1 ]; then echo "::error::setup.sh or replicate.sh passes the root password on a command line, or grep could not read them"; false; fi - # the password file goes to /dev/shm, off the writable layer of the container, and the mktemp of the image puts it there - docker run --rm --entrypoint grep "$IMAGE" -qF -- 'mktemp -p /dev/shm "opendj-replicate.$ADMIN_PORT.' /opt/opendj/bootstrap/replicate.sh || { echo "::error::replicate.sh no longer puts the password file on /dev/shm"; false; } - docker run --rm --entrypoint sh "$IMAGE" -c 'f=$(mktemp -p /dev/shm "opendj-replicate.$ADMIN_PORT.XXXXXX") && rm -f "$f" && case $f in /dev/shm/opendj-replicate.4444.*) ;; *) exit 1;; esac' || { echo "::error::mktemp in the image does not create the password file on /dev/shm"; false; } - # a password with a space in it reaches every tool as one value - ROOT_PASSWORD='replication secret' - docker network create test_replication - docker run --rm -it -d --memory="1g" --network test_replication --ipc=shareable --name=test_master --hostname=dj-master -e ADD_BASE_ENTRY="--addBaseEntry" -e ROOT_PASSWORD="$ROOT_PASSWORD" "$IMAGE" - timeout 3m bash -c 'until docker inspect --format="{{json .State.Health.Status}}" test_master | grep -q \"healthy\"; do sleep 10; done' - # a replica reports itself healthy only once replicate.sh has succeeded; the sdsr replica joins after - # the simple one, as two dsreplication enable at once would both rewrite the admin data of the master - # a Kubernetes pod keeps its /dev/shm across container restarts: the replica shares the /dev/shm of the master, where - # a password file waits as a killed replicate.sh would have left it, and its run.sh has to remove it (checked below); - # the file of another container of the pod, which listens on another admin port, has to be kept - docker exec test_master sh -c 'printf "%s\n" "$ROOT_PASSWORD" >/dev/shm/opendj-replicate.4444.killed' - docker exec test_master sh -c ': >/dev/shm/opendj-replicate.5444.other' - docker run --rm -it -d --memory="1g" --network test_replication --ipc=container:test_master --name=test_replica --hostname=dj-replica -e ROOT_PASSWORD="$ROOT_PASSWORD" -e MASTER_SERVER=dj-master -e OPENDJ_REPLICATION_TYPE=simple "$IMAGE" - # on a first start the master stops the server its bootstrap started and starts it again, and a replica - # started together with it can reach it in between: replicate.sh tries a dsreplication that could not - # connect again. The master is taken off the network while the replica sleeps before its first try, and - # comes back only once the replica has said it will try again - timeout 5m bash -c 'until docker logs test_replica 2>&1 | grep -q "Will sleep for a bit"; do sleep 0.2; done' - docker network disconnect test_replication test_master - timeout 2m bash -c 'until docker logs test_replica 2>&1 | grep -q "exited with 8, trying again"; do sleep 1; done' - docker network connect --alias dj-master test_replication test_master - timeout 5m bash -c 'until docker inspect --format="{{json .State.Health.Status}}" test_replica | grep -q \"healthy\"; do sleep 10; done' - docker run --rm -it -d --memory="1g" --network test_replication --name=test_replica_sdsr --hostname=dj-replica-sdsr -e ROOT_PASSWORD="$ROOT_PASSWORD" -e MASTER_SERVER=dj-master -e OPENDJ_REPLICATION_TYPE=sdsr "$IMAGE" - # the same for the sdsr replica, whose dsreplication enable is another command of replicate.sh - timeout 5m bash -c 'until docker logs test_replica_sdsr 2>&1 | grep -q "Will sleep for a bit"; do sleep 0.2; done' - docker network disconnect test_replication test_master - timeout 2m bash -c 'until docker logs test_replica_sdsr 2>&1 | grep -q "exited with 8, trying again"; do sleep 1; done' - docker network connect --alias dj-master test_replication test_master - timeout 5m bash -c 'until docker inspect --format="{{json .State.Health.Status}}" test_replica_sdsr | grep -q \"healthy\"; do sleep 10; done' - # the replicas were initialized from the master, and a change made on the master reaches them - for c in $REPLICAS; do - docker exec $c /opt/opendj/bin/ldapsearch --hostname localhost --port 1636 --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll --baseDN "dc=example,dc=com" --searchScope base "(objectClass=*)" 1.1 - done - printf 'dn: ou=replicated,dc=example,dc=com\nobjectClass: organizationalUnit\nou: replicated\n' | docker exec -i test_master /opt/opendj/bin/ldapmodify --hostname localhost --port 1636 --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll --defaultAdd - for c in $REPLICAS; do - timeout 1m bash -c 'until docker exec $1 /opt/opendj/bin/ldapsearch --hostname localhost --port 1636 --bindDN "cn=Directory Manager" --bindPassword "$0" --useSsl --trustAll --baseDN "ou=replicated,dc=example,dc=com" --searchScope base "(objectClass=*)" 1.1; do sleep 5; done' "$ROOT_PASSWORD" $c - done - # replicate.sh tries dsreplication enable again only when it exits 8, and a failed enable ends it: run once more - # on the replica, the enable of a base DN already replicated exits 5 and nothing is tried again or initialized - rc=0; out=$(docker exec -e BASE_DN=dc=example,dc=com -e ROOT_USER_DN="cn=Directory Manager" test_replica timeout 90 /opt/opendj/bootstrap/replicate.sh 2>&1) || rc=$? - if [ $rc -ne 5 ] || grep -qE "trying again|initializing replication" <<<"$out"; then - echo "$out"; echo "::error::a second replicate.sh exited with $rc, not with the 5 of its dsreplication enable, or went on after it"; false - fi - # the root password shows in no container log, and the files setup.sh and replicate.sh passed it in are gone (#1084, #1092) - for c in test_master $REPLICAS; do - if docker logs $c 2>&1 | grep -F "$ROOT_PASSWORD"; then echo "::error::The root password is in the log of $c"; false; fi - done - for c in test_master $REPLICAS; do - # a JVM keeps its command line in /tmp/hsperfdata_* while it runs; the HEALTHCHECK no longer binds as root (#1092), - # so no process left running has the root password on it - left=$(docker exec $c grep -rlsF -- "$ROOT_PASSWORD" /tmp /dev/shm || true) - if [ -n "$left" ]; then echo "::error::The root password is left in $left of $c"; false; fi - done - docker exec test_replica test -e /dev/shm/opendj-replicate.5444.other || { echo "::error::run.sh of test_replica removed the password file of another container"; false; } - cleanup + run: .github/scripts/docker-test-replication.sh "localhost:5000/${GITHUB_REPOSITORY,,}:${{ env.release_version }}-alpine" - name: Docker test secret volume # a keystore mounted at SECRET_VOLUME is what LDAPS serves from the first start on, a # renewed one - a new password included - is copied while the server runs and served diff --git a/opendj-packages/opendj-docker/Dockerfile b/opendj-packages/opendj-docker/Dockerfile index 3ffaac8231..0ebaf0a5b9 100644 --- a/opendj-packages/opendj-docker/Dockerfile +++ b/opendj-packages/opendj-docker/Dockerfile @@ -29,6 +29,14 @@ ENV ROOT_USER_DN="cn=Directory Manager" ENV OPENDJ_SSL_OPTIONS="--generateSelfSignedCertificate" #ENV MASTER_SERVER #ENV OPENDJ_REPLICATION_TYPE +# every server of the replication topology, comma separated; the first entry may seed a new +# topology. MASTER_SERVER keeps working as a one-element list. See bootstrap/join.sh +#ENV REPLICATION_PEERS +#ENV REPLICATION_PORT=8989 +#ENV REPLICATION_RETRY_COUNT=30 +#ENV REPLICATION_RETRY_INTERVAL=10 +#ENV REPLICATION_ATTEMPT_TIMEOUT=120 +#ENV REPLICATION_INITIALIZE_TIMEOUT=0 ENV OPENDJ_USER="opendj" #ENV OPENDJ_JAVA_ARGS="" ENV BACKEND_TYPE="je" @@ -68,7 +76,7 @@ COPY --chown=$OPENDJ_USER:0 bootstrap/ /opt/opendj/bootstrap/ COPY --chown=$OPENDJ_USER:0 run.sh /opt/opendj/run.sh COPY --chown=$OPENDJ_USER:0 healthcheck.sh /opt/opendj/healthcheck.sh -RUN chmod +x /opt/opendj/run.sh /opt/opendj/healthcheck.sh /opt/opendj/bootstrap/setup.sh /opt/opendj/bootstrap/replicate.sh +RUN chmod +x /opt/opendj/run.sh /opt/opendj/healthcheck.sh /opt/opendj/bootstrap/setup.sh /opt/opendj/bootstrap/replicate.sh /opt/opendj/bootstrap/join.sh EXPOSE $PORT/tcp $LDAPS_PORT/tcp $ADMIN_PORT/tcp diff --git a/opendj-packages/opendj-docker/Dockerfile-alpine b/opendj-packages/opendj-docker/Dockerfile-alpine index 55658cb1ea..fba79d46fb 100644 --- a/opendj-packages/opendj-docker/Dockerfile-alpine +++ b/opendj-packages/opendj-docker/Dockerfile-alpine @@ -29,6 +29,14 @@ ENV ROOT_USER_DN="cn=Directory Manager" ENV OPENDJ_SSL_OPTIONS="--generateSelfSignedCertificate" #ENV MASTER_SERVER #ENV OPENDJ_REPLICATION_TYPE +# every server of the replication topology, comma separated; the first entry may seed a new +# topology. MASTER_SERVER keeps working as a one-element list. See bootstrap/join.sh +#ENV REPLICATION_PEERS +#ENV REPLICATION_PORT=8989 +#ENV REPLICATION_RETRY_COUNT=30 +#ENV REPLICATION_RETRY_INTERVAL=10 +#ENV REPLICATION_ATTEMPT_TIMEOUT=120 +#ENV REPLICATION_INITIALIZE_TIMEOUT=0 ENV OPENDJ_USER="opendj" #ENV OPENDJ_JAVA_ARGS="" ENV BACKEND_TYPE="je" @@ -50,7 +58,7 @@ WORKDIR /opt RUN apk add --update --no-cache --virtual builddeps curl unzip \ && apk upgrade --update --no-cache \ && if [ "$TARGETARCH" = "386" ]; then JDK=openjdk11-jre; else JDK=openjdk25-jre; fi \ - && apk add bash "$JDK" \ + && apk add bash coreutils "$JDK" \ && if [ -z "$VERSION" ] ; then VERSION="$(curl -i -o - --silent https://api.github.com/repos/OpenIdentityPlatform/OpenDJ/releases/latest | grep -m1 "\"name\"" | cut -d\" -f4)"; fi \ && if [ ! -f "$OPENDJ_DIST_FILENAME" ]; then echo file exists && curl -L https://github.com/OpenIdentityPlatform/OpenDJ/releases/download/$VERSION/opendj-$VERSION.zip --output $OPENDJ_DIST_FILENAME; fi \ && unzip $OPENDJ_DIST_FILENAME \ @@ -72,7 +80,7 @@ COPY --chown=$OPENDJ_USER:0 bootstrap/ /opt/opendj/bootstrap/ COPY --chown=$OPENDJ_USER:0 run.sh /opt/opendj/run.sh COPY --chown=$OPENDJ_USER:0 healthcheck.sh /opt/opendj/healthcheck.sh -RUN chmod +x /opt/opendj/run.sh /opt/opendj/healthcheck.sh /opt/opendj/bootstrap/setup.sh /opt/opendj/bootstrap/replicate.sh +RUN chmod +x /opt/opendj/run.sh /opt/opendj/healthcheck.sh /opt/opendj/bootstrap/setup.sh /opt/opendj/bootstrap/replicate.sh /opt/opendj/bootstrap/join.sh EXPOSE $PORT/tcp $LDAPS_PORT/tcp $ADMIN_PORT/tcp diff --git a/opendj-packages/opendj-docker/README.md b/opendj-packages/opendj-docker/README.md index dcc0f2502c..e4ceac5830 100644 --- a/opendj-packages/opendj-docker/README.md +++ b/opendj-packages/opendj-docker/README.md @@ -16,9 +16,10 @@ docker run -d -p 1389:1389 -p 1636:1636 -p 4444:4444 --name opendj openidentityp The image reports itself `healthy` once the server answers on `LDAPS_PORT` *and* the whole bootstrap has succeeded - the instance, the `userRoot` backend over `BASE_DN`, whatever -`ADD_BASE_ENTRY` and `SAMPLE_DATA` asked to be imported into it, and the replication asked -for by `MASTER_SERVER`. Waiting for that status is therefore enough before the first search -of what the bootstrap was told to create: +`ADD_BASE_ENTRY` and `SAMPLE_DATA` asked to be imported into it, and, where replication is +asked for, the join of the replication topology (see [Replication](#replication)). Waiting +for that status is therefore enough before the first search of what the bootstrap was told +to create: ```bash docker run -d --name opendj -e ADD_BASE_ENTRY=--addBaseEntry openidentityplatform/opendj @@ -52,14 +53,7 @@ set them before the upgrade. The server answering is not enough on a first start: the bootstrap starts the server, and once it is done that server is stopped and started again in the foreground, so a client -that only waits for the port can have its first requests fail in between. A replica set up -with `MASTER_SERVER` tries a master it cannot connect to again, every 10 s for up to 5 -minutes, so a master that is down when the replica reaches it does not fail its replication -setup; one that stops while `dsreplication enable` is writing to it still does. - -With `OPENDJ_REPLICATION_TYPE=srs`, start the directory server replicas one at a time, each -once the previous one is healthy: every replica pushes its data to the replicas connected at -that moment, and one that is restarting at the end of its own first start misses it. +that only waits for the port can have its first requests fail in between. A bootstrap that imports `SAMPLE_DATA` can take minutes on a small container, which is what the start period allows for. A bootstrap that fails - or an upgrade that fails when starting @@ -72,6 +66,122 @@ behind to it - those of a health check that ran past its timeout, say. Run the c with `docker run --init` (`init: true` in Compose) to put a PID 1 in front of the server that reaps them and passes SIGTERM on to it. +## Replication + +With `OPENDJ_REPLICATION_TYPE=simple`, the container joins its replication topology in the +background, next to the running server, on every start - not as a step of the first +bootstrap, so a join that could not complete is tried again on the next start, and +membership that changed while no container ran is repaired. `REPLICATION_PEERS` lists every +server of the topology by DNS name, comma separated; all of them share `BASE_DN`, +`ROOT_USER_DN`, `ROOT_PASSWORD`, `ADMIN_PORT` and `REPLICATION_PORT`. A Kubernetes +StatefulSet derives the list from its ordinals +(`-0.,…,-N-1.`), so it needs no registry; plain +`docker run` passes the container names, each container started with `--hostname` equal to +its `--name` (or with `MYHOSTNAME` set to it). `MASTER_SERVER` keeps working as a +one-element list, and may name the master by an address or an `/etc/hosts` alias as before: +the master recognises itself by those as well. + +A server recognises itself in the list by a name that equals its `hostname -f`, or that is +its `hostname -f` cut at a dot (or the other way round) - a pod whose FQDN is +`-N...svc.cluster.local` is listed as +`-N.` - and by one of its own addresses or a name `/etc/hosts` gives one +of them (an `--add-host` or a `hostAliases` entry for itself). Names are compared whole: +`opendj-1` is not `opendj-10`, and `opendj-0.opendj.east` is not `opendj-0.opendj.west`. +List DNS names rather than addresses in `REPLICATION_PEERS`: the servers register in +`cn=admin data` by name, and a server listed by address alone would be taken for one that +left the topology and removed from it (see below). A server registered by an address - a +master that `MASTER_SERVER=
` named, which its replicas registered under that address +- is never removed that way, as nothing tells it from a listed name: moving such a topology +to a `REPLICATION_PEERS` of names keeps the master, and once it really is gone, it is +removed by hand. + +The join decides from what is there, not from an exit code: this server is a member once +its configuration holds the replication domain for `BASE_DN` and it is registered in +`cn=admin data`. Until then it runs `dsreplication enable` through the listed peers, +`REPLICATION_RETRY_COUNT` times every `REPLICATION_RETRY_INTERVAL` seconds, each enable +bounded by `REPLICATION_ATTEMPT_TIMEOUT` seconds, and the container reports itself healthy +only once the join succeeded and the volume holds the data of the topology - across +restarts too. A server whose volume holds that data is ready as soon as it serves, without +waiting for its peers or for its join, so a whole cluster restart does not deadlock under +`OrderedReady`, and a changed root password does not keep it unready either (the join, +which binds with `ROOT_PASSWORD`, then only logs that it cannot). + +A volume the container bootstraps is marked before the bootstrap starts, and initialized +from the topology once the join succeeds, whether or not `BASE_DN` already holds entries - so +a bootstrap that failed or was killed half-way is initialized as well, never taken for data +of the topology: every fresh volume holds what +`ADD_BASE_ENTRY` or `SAMPLE_DATA` imported, which says nothing about which server holds +the data of the topology. `dsreplication initialize` is a full import of `BASE_DN` and runs +without a bound unless `REPLICATION_INITIALIZE_TIMEOUT` sets one. Each server publishes to +its peers whether its volume still waits for that data (`pending`) or holds it (`ready`, or +`rejoining` while its replication is taken down to be enabled anew, see below), in the local +entry `cn=Docker Join,cn=config`, and a server joins only through a ready one - a pending +server initializes from it as well. So servers may start together - Compose, or a StatefulSet with +`podManagementPolicy: Parallel` - without any of them taking the bootstrap data of another +fresh one for the topology's. Servers whose enables involve the same server take turns: two +`dsreplication enable` runs through one server at once leave the loser a member only in +part, for good. The turn is the entry `cn=Docker Join Lock,cn=config`, which a join adds +before its enable on the peer it enables through and on its own server, and removes after +it; one left behind by a killed join is taken over once it is older than +`REPLICATION_ATTEMPT_TIMEOUT` plus a minute, or at once by a later join of the server that +left it. A join that finds a turn taken tries the next peer, and waits a random part of +`REPLICATION_RETRY_INTERVAL` on top of it between rounds, so two joins that each hold the +other's turn do not keep meeting. Only the first entry of `REPLICATION_PEERS` may decide that +there is no topology yet and seed it with its own data: at once when every other peer +answers and waits for data as well, or once its retries are exhausted without any peer +that could hold the data answering - a ready or rejoining one, a server of an earlier image that +publishes no state but replicates `BASE_DN`, or one that refuses the bind with +`ROOT_PASSWORD` (only a server past its bootstrap has another root password). The residual risk of that rule: with every other server +down *and* the volume of the first peer lost, the first peer seeds an empty topology and the +others initialize from it; keep backups accordingly. A volume that joined but whose +initialize never completed keeps waiting for a ready peer on every start, so under +`OrderedReady` a `-0` in that state stays unready until a peer that holds the data is +started by hand - which is where its data has to come from anyway. + +With `REPLICATION_PEERS` set explicitly, a joined server also removes every server that is +registered in the topology but no longer listed: its entries in `cn=admin data` and its +values in every local replication server list - those of `BASE_DN`, and of `cn=schema` and +`cn=admin data`, which `dsreplication enable` replicates next to it. A scale-down therefore +needs no `preStop` hook - `dsreplication disable` on termination would take the server out +on every rolling restart, and cannot clean up a server that is already gone. It also adds +every listed peer registered in the topology to the local lists that lack it, so joins that +ran at the same time cannot leave two replication servers that know nothing of each other. +With only `MASTER_SERVER` set nothing is removed or added, so servers joined by hand stay. + +A server that was removed this way and is scaled up again on the volume it kept still holds +the data and its replication configuration, but no peer registers it any more, and +`dsreplication enable` would not register it again. Its join takes its replication +configuration down and enables it anew, and from that moment until the enable has registered +it again its clients may not write: writes it took meanwhile would replicate nowhere. The +backend of `BASE_DN` is set to `writability-mode: internal-only` before anything changes - +writes of clients are refused with `Unwilling to Perform`, replication and `dsreplication` +still write - and set back to `enabled` once the server rejoined; a backend that was not +`enabled` before is left as it is. Throughout, the container does not report itself healthy, +across restarts too, and the server publishes `rejoining`, so no peer joins or initializes +through it. It enables only through a peer that publishes `ready`, never through another one +that rejoins or waits for the data: two such servers would register only each other and +serve a topology of their own beside the survivors. The health status itself turns +`unhealthy` only after the probe's retries (a readiness probe after its failure threshold), +which is why the writes are refused rather than left to the health check. A join that gives +up leaves the server unhealthy and read-only until a later start rejoins it. A container +started on such a volume without the replication environment runs no join, reports itself +healthy and logs that the backend may still refuse writes: its `writability-mode` is then +the operator's to set back to `enabled`. It takes the replication down only when at least one peer +that replicates `BASE_DN` answered and none of them registers it - a peer that cannot be +asked decides nothing - or when the peers register it but its own `cn=admin data` does not. +An enable that stopped half-way leaves that state behind (its initialize of `cn=admin data` +failed, say): `BASE_DN` is replicated, and every later enable only reports it as replicated +already, without registering the server where it is missing. + +The one-shot types `srs`, `sdsr` and `rg` keep their previous behaviour - they run once, +during the first bootstrap only - and are deprecated in favour of `simple`. Like `simple`, +they need one `ADMIN_PORT` and one `REPLICATION_PORT` on every server: the tools reach the +master on the ports of the container they run in (images before this one used 4444 and +8989 on both sides, whatever the container was set up with). With +`OPENDJ_REPLICATION_TYPE=srs`, start the directory server replicas one at a time, each +once the previous one is healthy: every replica pushes its data to the replicas connected +at that moment, and one that is restarting at the end of its own first start misses it. + ## Certificates With the default `OPENDJ_SSL_OPTIONS` the instance serves LDAPS and StartTLS with a @@ -150,10 +260,16 @@ with `subPath` when the Secret changes. | ROOT_PASSWORD | password | Initial root user password; the bootstrap fails if it contains a line break (CR or LF) | | SECRET_VOLUME | /var/secrets/opendj | Mounted keystore volume, if present its `key*` and `trust*` files are copied into the instance on every start, see [Certificates](#certificates) | | SECRET_VOLUME_REFRESH | 60 | While the server runs, `SECRET_VOLUME` is checked again every that many seconds and changed files are copied again; `0` copies them on start only | -| MASTER_SERVER | - | Replication master server | +| MASTER_SERVER | - | Replication master server; with `simple` it works as a one-element `REPLICATION_PEERS` | +| REPLICATION_PEERS | value of MASTER_SERVER | every server of the replication topology by DNS name, comma separated; the first entry may seed a new topology, and servers no longer listed are removed from it, see [Replication](#replication) | +| REPLICATION_PORT | 8989 | replication port, the same on every server of the topology | +| REPLICATION_RETRY_COUNT | 30 | rounds of join attempts through every peer before the join gives up: the first peer of `REPLICATION_PEERS` then seeds the topology, any other server stays `unhealthy`; a value that is not a whole number above 0 falls back to 30 | +| REPLICATION_RETRY_INTERVAL | 10 | seconds between rounds of join attempts, plus a random part of as much again; a value that is not a whole number falls back to 10 | +| REPLICATION_ATTEMPT_TIMEOUT | 120 | seconds a single `dsreplication enable` may take before it is killed and tried again; it can hang on a peer that stops mid-operation, so a value that is not a whole number above 0 falls back to 120 | +| REPLICATION_INITIALIZE_TIMEOUT | 0 | seconds a single `dsreplication initialize` - a full import of `BASE_DN` - may take before it is killed and tried again; `0` sets no bound, and so does a value that is not a whole number | | VERSION | - | OpenDJ version | | OPENDJ_USER | opendj | user which runs OpenDJ | -| OPENDJ_REPLICATION_TYPE | - | OpenDJ Replication type, valid values are:
  • simple - standart replication
  • srs - standalone replication servers
  • sdsr - Standalone Directory Server Replicas
  • rg - Replication Groups
Other values will be ignored | +| OPENDJ_REPLICATION_TYPE | - | OpenDJ Replication type, valid values are:
  • simple - standard replication, joined in the background on every start, see [Replication](#replication)
  • srs - standalone replication servers (one-shot, deprecated)
  • sdsr - Standalone Directory Server Replicas (one-shot, deprecated)
  • rg - Replication Groups (one-shot, deprecated)
Other values will be ignored | | OPENDJ_SSL_OPTIONS | --generateSelfSignedCertificate | you can replace ssl options at here, like : "--usePkcs12keyStore /opt/domain.pfx --keyStorePassword domain" | | OPENDJ_JAVA_ARGS | -server | extra instance java args | | BACKEND_TYPE | je | OpenDJ backend type, see [dsconfig create-backend](https://doc.openidentityplatform.org/opendj/reference/dsconfig-subcommands-ref#dsconfig-create-backend) documentation | diff --git a/opendj-packages/opendj-docker/bootstrap/join.sh b/opendj-packages/opendj-docker/bootstrap/join.sh new file mode 100755 index 0000000000..8afa04d8a7 --- /dev/null +++ b/opendj-packages/opendj-docker/bootstrap/join.sh @@ -0,0 +1,976 @@ +#!/usr/bin/env bash +# The contents of this file are subject to the terms of the Common Development and +# Distribution License (the License). You may not use this file except in compliance with the +# License. +# +# You can obtain a copy of the License at legal/CDDLv1.0.txt. See the License for the +# specific language governing permission and limitations under the License. +# +# When distributing Covered Software, include this CDDL Header Notice in each file and include +# the License file at legal/CDDLv1.0.txt. If applicable, add the following below the CDDL +# Header, with the fields enclosed by brackets [] replaced by your own identifying +# information: "Portions copyright [year] [name of copyright owner]". +# +# Copyright 2026 3A Systems, LLC. + +# Joins this server to the replication topology (#1086). run.sh starts it in the background +# next to the server on every start, not as a step of the first bootstrap: a join that could +# not complete is tried again on the next start, and membership that changed while no +# container ran is repaired. It covers OPENDJ_REPLICATION_TYPE=simple; the one-shot srs, sdsr +# and rg paths stay in replicate.sh. +# +# The contract with the operator: REPLICATION_PEERS lists every server of the topology by DNS +# name, and all of them share BASE_DN, ROOT_USER_DN, ROOT_PASSWORD, ADMIN_PORT and +# REPLICATION_PORT. A StatefulSet chart derives the list from its ordinals +# (-0. ... -N-1.), plain docker run passes the container names of +# containers whose --hostname is their --name. MASTER_SERVER keeps working as a one-element +# list. This server recognises itself in the list by a name that equals hostname -f, or is +# hostname -f cut at a dot (a StatefulSet pod whose FQDN is -N...svc. +# is listed as -N.), or the other way round; and, as the grep of /etc/hosts it +# replaces did, by one of its own addresses or a name /etc/hosts gives one of them. Names +# are compared whole - that grep took opendj-1 for opendj-10, a replica for its master when +# an --add-host named the master, and a master that lost its volume for a replica of itself - +# and cut only at a dot, so opendj-0.opendj.east is not opendj-0.opendj.west. +# +# Membership is decided from what is there, never from an exit code: exit 5 of dsreplication +# enable (REPLICATION_CANNOT_BE_ENABLED_ON_BASEDN) means "no suffix was left to enable", +# which covers a base DN that is already replicated and a base DN that one of the two +# servers does not hold - the race with a peer whose bootstrap has not created the backend +# yet. The step is a member once the replication domain for BASE_DN exists in its cn=config +# and the server is registered in cn=admin data; anything else is tried again, whatever the +# exit code, because a repeated enable is safe exactly when it is decided this way. +# +# Whose data the topology carries follows what each volume went through, not whether BASE_DN +# has entries: every fresh volume has entries (setup.sh imports the base entry or +# SAMPLE_DATA), and two freshly bootstrapped volumes even share a generation ID, so +# replication silently carries nothing that reached one of them by import-ldif. run.sh marks +# a volume it bootstrapped with $INITIALIZE_PENDING, and the join publishes that to its peers +# in $STATE_DN, a local entry of cn=config: "pending" while the marker is there, "ready" once +# the volume holds the data of the topology - it was never bootstrapped by run.sh, it was +# initialized from a ready peer, or it seeded the topology - and "rejoining" while a volume +# that holds it has its replication taken down to enable it anew. A server joins only through +# a ready peer, and a pending one initializes only from one, so two fresh servers that start +# together never take each other's bootstrap data for the topology's, and two servers whose +# replication is down never form a topology of their own; where the list is only +# MASTER_SERVER, a peer that publishes no state (an image before this one) counts as ready, as +# the old replicate.sh trusted its master. The seed is only the first entry of +# REPLICATION_PEERS: at once when every other peer answers and is pending (all of them start +# from scratch, together or not), or once the retries are exhausted while no other peer +# answers that may hold the data - a ready one, one of an image before this one that +# publishes no state but replicates BASE_DN, or one that refuses the bind with ROOT_PASSWORD +# (only a server past its bootstrap can have another root password). The residual risk is +# documented in the README: every other server down *and* the first peer's volume lost means +# the first peer seeds empty. +# +# On a volume that waits for the data of the topology, the health marker is written only +# once the join succeeded or the seed rule fired, so a container that never joined stays +# unhealthy - across restarts too, which the bootstrap-only replicate.sh could not say (a +# restart wrote the marker right after upgrade and a failed join turned into a healthy, +# unreplicated server). A volume that holds the data is healthy as soon as it serves (see +# run.sh) - unless the join takes its replication down to enable it anew: from then until the +# enable registered it again it is not, across restarts too ($REJOIN_PENDING), and the +# backend of BASE_DN refuses the writes of clients, which no longer replicate meanwhile. + +cd /opt/opendj || exit 1 +export PATH=/opt/opendj/bin:$PATH + +BASE_DN=${BASE_DN:-"dc=example,dc=com"} +ROOT_USER_DN=${ROOT_USER_DN:-"cn=Directory Manager"} +ROOT_PASSWORD=${ROOT_PASSWORD:-password} +ADMIN_PORT=${ADMIN_PORT:-4444} +REPLICATION_PORT=${REPLICATION_PORT:-8989} +MYHOSTNAME=${MYHOSTNAME:-$(hostname -f)} +BOOTSTRAP_COMPLETE=${BOOTSTRAP_COMPLETE:-/opt/opendj/.bootstrap-complete} +INITIALIZE_PENDING=${INITIALIZE_PENDING:-/opt/opendj/data/.replication-initialize-pending} +REJOIN_PENDING=${REJOIN_PENDING:-/opt/opendj/data/.replication-rejoin-pending} +# not replicated: cn=config is this server's alone, and it lives on the volume next to the +# marker, so the two cannot tell different stories +STATE_DN="cn=Docker Join,cn=config" + +# Sets the variable to its value read as a decimal whole number, or to the default when it +# is not one or is below the least value. $(( )) reads a leading zero as octal and fails on +# 08 - abandoning the whole command it runs in, however far up that is - while test reads 08 +# as 8, so a check by test alone lets the value through to the arithmetic. +whole_number() { # + local value=${!1} + case $value in + '' | *[!0-9]*) value= ;; + *) value=$((10#$value)) ;; + esac + if [ -z "$value" ] || [ "$value" -lt "$3" ]; then + echo "join: $1 is not a whole number of at least $3, using $2" + value=$2 + fi + printf -v "$1" '%s' "$value" +} + +REPLICATION_RETRY_COUNT=${REPLICATION_RETRY_COUNT:-30} +whole_number REPLICATION_RETRY_COUNT 30 1 +REPLICATION_RETRY_INTERVAL=${REPLICATION_RETRY_INTERVAL:-10} +whole_number REPLICATION_RETRY_INTERVAL 10 0 +# dsreplication enable can hang rather than fail (a peer that stops mid-operation), so no +# enable runs without a bound +REPLICATION_ATTEMPT_TIMEOUT=${REPLICATION_ATTEMPT_TIMEOUT:-120} +whole_number REPLICATION_ATTEMPT_TIMEOUT 120 1 +# a total update is a full import of BASE_DN, which takes as long as the data needs; a killed +# dsreplication initialize leaves its task running on the server, and the next attempt asks +# for another full import, so by default it runs without a bound +REPLICATION_INITIALIZE_TIMEOUT=${REPLICATION_INITIALIZE_TIMEOUT:-0} +whole_number REPLICATION_INITIALIZE_TIMEOUT 0 0 + +# Removing servers that left the topology needs the operator's word on who belongs to it: +# with only MASTER_SERVER set, a third server joined by hand would be "not in the list" and +# thrown out. So the cleanup, and the repair of the replication server lists, run only when +# REPLICATION_PEERS itself is set. +PEERS_ARE_EXPLICIT=${REPLICATION_PEERS:+yes} +REPLICATION_PEERS=${REPLICATION_PEERS:-$MASTER_SERVER} + +if [ -z "$REPLICATION_PEERS" ]; then + echo "join: neither REPLICATION_PEERS nor MASTER_SERVER is set, nothing to join" + exit 1 +fi + +IFS=',' read -r -a RAW_PEERS <<<"$REPLICATION_PEERS" +PEERS=() +for peer in "${RAW_PEERS[@]}"; do + peer=$(echo "$peer" | tr -d '[:space:]') + [ -n "$peer" ] && PEERS+=("$peer") +done +if [ "${#PEERS[@]}" -eq 0 ]; then + echo "join: REPLICATION_PEERS holds no peer" + exit 1 +fi + +# The tools read the root password from a file (#1084); the file lives on the tmpfs of +# /dev/shm where there is one, and run.sh removes what a killed join left behind, by the +# name, on every start - in /dev/shm those of its ADMIN_PORT, in /tmp all of them +PASSWORD_FILE=$(mktemp -p /dev/shm "opendj-join.$ADMIN_PORT.XXXXXX" 2>/dev/null \ + || mktemp "/tmp/opendj-join.$ADMIN_PORT.XXXXXX") || exit 1 +trap 'rm -f "$PASSWORD_FILE"' EXIT +printf '%s\n' "$ROOT_PASSWORD" >"$PASSWORD_FILE" || exit 1 + +lower() { + printf '%s' "$1" | tr '[:upper:]' '[:lower:]' +} + +# an IPv4 address is digits and dots, an IPv6 one has a colon +is_address() { + case $1 in + *:*) return 0 ;; + *[!0-9.]* | '') return 1 ;; + *) return 0 ;; + esac +} + +# Two names (lower case) are one server when they are equal, or when one of them is the +# other cut at a dot; addresses only when they are equal +names_match() { + [ "$1" = "$2" ] && return 0 + if is_address "$1" || is_address "$2"; then + return 1 + fi + case $1 in "$2".*) return 0 ;; esac + case $2 in "$1".*) return 0 ;; esac + return 1 +} + +SELF_NAME=$(lower "$MYHOSTNAME") +# the addresses of this container (loopback aside) and the names /etc/hosts gives them - an +# --add-host or a hostAliases entry for itself - as the base replicate.sh recognised them +SELF_ADDRESSES=() +SELF_ALIASES=() +own_names=" $(lower "$(hostname)") $(lower "$(hostname -f 2>/dev/null)") $SELF_NAME " +for address in $(hostname -i 2>/dev/null) \ + $(awk -v names="$own_names" '!/^[[:space:]]*#/ { for (i = 2; i <= NF; i++) if (index(names, " " tolower($i) " ")) { print $1; break } }' /etc/hosts 2>/dev/null); do + case $address in 127.* | ::1 | 0.0.0.0) continue ;; esac + SELF_ADDRESSES+=("$(lower "$address")") +done +if [ "${#SELF_ADDRESSES[@]}" -gt 0 ]; then + for name in $(awk -v addresses=" ${SELF_ADDRESSES[*]} " '!/^[[:space:]]*#/ && index(addresses, " " tolower($1) " ") { for (i = 2; i <= NF; i++) print tolower($i) }' /etc/hosts 2>/dev/null); do + SELF_ALIASES+=("$name") + done +fi + +is_self() { + local peer candidate + peer=$(lower "$1") + names_match "$peer" "$SELF_NAME" && return 0 + for candidate in "${SELF_ADDRESSES[@]}" "${SELF_ALIASES[@]}"; do + [ "$peer" = "$candidate" ] && return 0 + done + return 1 +} + +in_peers() { + local host peer + host=$(lower "$1") + for peer in "${PEERS[@]}"; do + names_match "$(lower "$peer")" "$host" && return 0 + done + return 1 +} + +# every LDAP operation goes through the administration connector, which always serves TLS; +# the root user may change its password after the bootstrap, so unlike the health check +# these tools run only while the join still has the bootstrap's ROOT_PASSWORD to work with +search() { + local host=$1 + shift + ldapsearch --noPropertiesFile --hostname "$host" --port "$ADMIN_PORT" --useSsl --trustAll \ + --bindDN "$ROOT_USER_DN" --bindPasswordFile "$PASSWORD_FILE" "$@" 2>/dev/null +} + +ldapmodify_on() { # [