#!/usr/bin/env bash # # The contents of this file are subject to the terms of the Common Development and # Distribution License (the License). You may not use this file except in compliance with the # License. # # You can obtain a copy of the License at legal/CDDLv1.0.txt. See the License for the # specific language governing permission and limitations under the License. # # When distributing Covered Software, include this CDDL Header Notice in each file and include # the License file at legal/CDDLv1.0.txt. If applicable, add the following below the CDDL # Header, with the fields enclosed by brackets [] replaced by your own identifying # information: "Portions copyright [year] [name of copyright owner]". # # Copyright 2026 3A Systems, LLC. # Tests the replication of the Docker image (#1086): the background join of # OPENDJ_REPLICATION_TYPE=simple (bootstrap/join.sh) and the deprecated one-shot sdsr path of # bootstrap/replicate.sh. Both image jobs of build.yml run it, each with its own image. # # Usage: docker-test-replication.sh set -eE -o pipefail IMAGE=${1:?usage: $0 } NODES="dj-0 dj-1 dj-2 dj-x dj-solo dj-bf dj-fi dj-f dj-p0 dj-p1 dj-p2 dj-ipm dj-ipr dj-sdsr dj-shm-probe dj-shm" VOLUMES="vol-dj-0 vol-dj-1 vol-dj-2 vol-dj-x vol-dj-solo vol-dj-bf vol-dj-fi vol-dj-f vol-dj-ipr vol-dj-p0 vol-dj-p1 vol-dj-p2" # a bootstrap that imports the base entry and then fails, mounted into dj-bf FAILING_BOOTSTRAP=$(mktemp) NETWORK=test_replication # where dj-sdsr starts: it resolves its own name there, and no other server NETWORK_ALONE=test_replication_alone cleanup() { rm -f "$FAILING_BOOTSTRAP" docker rm -f $NODES >/dev/null 2>&1 || true docker network rm $NETWORK $NETWORK_ALONE >/dev/null 2>&1 || true docker volume rm -f $VOLUMES >/dev/null 2>&1 || true } cleanup # the container log says that a dsreplication failed and where its detailed log is; the errors # log of the server and those detailed logs say why. Both are in the instance, which the # image keeps under /opt/opendj/data, and a stopped container has nothing to exec in dump_logs() { local c for c in $NODES; do echo "::group::container logs ($c)" docker logs $c 2>&1 || true echo "::endgroup::" [ "$(docker inspect -f '{{.State.Running}}' $c 2>/dev/null)" = true ] || continue echo "::group::server logs ($c)" docker exec $c sh -c 'tail -n 1000 /opt/opendj/data/logs/errors; for f in /opt/opendj/data/tmp/opendj-replication-*.log; do [ -f "$f" ] && echo "--- $f" && cat "$f"; done' 2>&1 || true echo "::endgroup::" done } # set -E hands the trap to every $(...) as well, where a failing command would dump the logs # into the value being captured and remove the containers under the running test: there the # failure only ends the subshell, and the main shell decides trap 'code=$?; [ "$BASH_SUBSHELL" -eq 0 ] || exit $code; dump_logs; cleanup; exit $code' ERR fail() { echo "::error::$1" false } # grep reads the whole log rather than stopping at the first match (-q), so docker logs never # writes into a closed pipe, which pipefail would report as a failure of the check logs_have() { # docker logs "$1" 2>&1 | grep -F -- "$2" >/dev/null } logs_count() { # docker logs "$1" 2>&1 | grep -cF -- "$2" || true } # the container logged the text after the last line that holds - after a restart, # say, which the join of the start before may still have logged into until it was stopped logs_have_since() { # docker logs "$1" 2>&1 | awk -v s="$2" 'index($0, s) { buf = ""; next } { buf = buf $0 "\n" } END { printf "%s", buf }' \ | grep -F -- "$3" >/dev/null } wait_until() { # [...] local deadline=$((SECONDS + $1)) what=$2 shift 2 until "$@"; do [ "$SECONDS" -lt "$deadline" ] || fail "timed out waiting until $what" sleep 2 done } health() { docker inspect --format='{{.State.Health.Status}}' "$1" } is_healthy() { [ "$(health "$1")" = healthy ] } wait_healthy() { wait_until 480 "$1 is healthy" is_healthy "$1" } # A container reports itself healthy only after a probe that finds the marker, and the # probes run every 30 s: a single look right after a failure reads "starting" whatever the # join did, so the container is watched over more than two probe cycles stays_unhealthy() { # local i for i in $(seq 1 15); do is_healthy "$1" && fail "$1 reports itself healthy although $2" sleep 5 done } ldaps_has() { # docker exec "$1" /opt/opendj/bin/ldapsearch --noPropertiesFile --hostname localhost --port 1636 \ --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll \ --baseDN "$2" --searchScope base "(objectClass=*)" 1.1 >/dev/null 2>&1 } wait_has() { # wait_until 60 "$2 is on $1" ldaps_has "$1" "$2" } admin_search() { # [...] local container=$1 base=$2 scope=$3 filter=$4 shift 4 docker exec "$container" /opt/opendj/bin/ldapsearch --noPropertiesFile --hostname localhost --port 4444 \ --useSsl --trustAll --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" \ --baseDN "$base" --searchScope "$scope" "$filter" "$@" 2>/dev/null } add_ou() { # printf 'dn: ou=%s,dc=example,dc=com\nobjectClass: organizationalUnit\nou: %s\n' "$2" "$2" \ | docker exec -i "$1" /opt/opendj/bin/ldapmodify --noPropertiesFile --hostname localhost --port 1636 \ --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll --defaultAdd >/dev/null } dsconfig_on() { # ... local container=$1 shift docker exec "$container" /opt/opendj/bin/dsconfig "$@" --hostname localhost --port 4444 \ --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --trustAll --no-prompt >/dev/null } # the writability-mode the backend of dc=example,dc=com is configured with, empty where its # entry sets none backend_mode() { # admin_search "$1" "ds-cfg-backend-id=userRoot,cn=Backends,cn=config" base "(objectClass=*)" ds-cfg-writability-mode \ | awk 'tolower($1) == "ds-cfg-writability-mode:" { print $2 }' } # the replication server lists of a container: the one of its replication server and one per # replication domain (BASE_DN, cn=schema, cn=admin data) replication_lists() { admin_search "$1" "cn=config" sub \ "(|(objectClass=ds-cfg-replication-server)(objectClass=ds-cfg-replication-domain))" ds-cfg-replication-server } lists_lack() { # ! replication_lists "$1" | grep -F -- "$2" >/dev/null } admin_data_lacks() { # ! admin_search "$1" "cn=Servers,cn=admin data" one "(objectClass=*)" hostname | grep -F -- "$2" >/dev/null } # the join lock of a server, held by hand as a join of a server named dj-held would hold it hold_lock() { # printf 'dn: cn=Docker Join Lock,cn=config\nobjectClass: top\nobjectClass: ds-cfg-branch\nobjectClass: extensibleObject\ncn: Docker Join Lock\ndescription: dj-held %s\n' "$(date +%s)" \ | docker exec -i "$1" /opt/opendj/bin/ldapmodify --noPropertiesFile --hostname localhost --port 4444 \ --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll --defaultAdd >/dev/null } drop_lock() { # printf 'dn: cn=Docker Join Lock,cn=config\nchangetype: delete\n' \ | docker exec -i "$1" /opt/opendj/bin/ldapmodify --noPropertiesFile --hostname localhost --port 4444 \ --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll >/dev/null } # hands a lock held by hand over to another holder at once, as a join of that server took it now relabel_lock() { # printf 'dn: cn=Docker Join Lock,cn=config\nchangetype: modify\nreplace: description\ndescription: %s %s\n' "$2" "$(date +%s)" \ | docker exec -i "$1" /opt/opendj/bin/ldapmodify --noPropertiesFile --hostname localhost --port 4444 \ --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll >/dev/null } # the state the join of the container publishes to its peers published_state() { # admin_search "$1" "cn=Docker Join,cn=config" base "(objectClass=*)" description \ | awk 'tolower($1) == "description:" { print $2 }' } published_ready() { # [ "$(published_state "$1")" = ready ] } registered_in_dj0() { # ! admin_data_lacks dj-0 "$1" } # deletes an entry on the container alone: the replication repair control keeps the delete # from reaching any other server, as an initialize that failed half-way leaves the # cn=admin data of one server unlike that of the others delete_here() { # [...] local container=$1 dn=$2 shift 2 docker exec "$container" /opt/opendj/bin/ldapdelete --noPropertiesFile --hostname localhost --port 4444 \ --useSsl --trustAll --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" \ --control 1.3.6.1.4.1.26027.1.5.2:true "$@" "$dn" >/dev/null } # the container holds the replication domain of dc=example,dc=com replicates_example() { # admin_search "$1" "cn=config" sub "(&(objectClass=ds-cfg-replication-domain)(ds-cfg-base-dn=dc=example,dc=com))" 1.1 \ | grep "^dn:" >/dev/null } # a client's write to the container is refused with Unwilling to Perform (53), as the backend # of a server whose replication is down refuses it refuses_writes() { # local rc=0 add_ou "$1" "$2" 2>/dev/null || rc=$? [ "$rc" -eq 53 ] } # the replication server of the container lists every one of the other names server_lists() { # ... local container=$1 list name shift list=$(admin_search "$container" "cn=config" sub "(objectClass=ds-cfg-replication-server)" ds-cfg-replication-server) for name in "$@"; do grep -F -- "ds-cfg-replication-server: $name:" <<<"$list" >/dev/null || return 1 done } # every tool reads the root password from a file (#1084, #1092); dsreplication run with -n # prints no command line, so a password put back on one would pass every check below rc=0 docker run --rm --entrypoint grep "$IMAGE" -nE -- '(^|[[:space:]])(-w|--(bindPassword[12]?|adminPassword|rootUserPassword))([[:space:]=]|$)' \ /opt/opendj/bootstrap/setup.sh /opt/opendj/bootstrap/replicate.sh /opt/opendj/bootstrap/join.sh || rc=$? [ "$rc" -eq 1 ] || fail "a bootstrap script passes the root password on a command line, or grep could not read them" # the password files go to /dev/shm, off the writable layer of the container, and the mktemp # of the image puts them there docker run --rm --entrypoint grep "$IMAGE" -qF -- 'mktemp -p /dev/shm "opendj-join.$ADMIN_PORT.' /opt/opendj/bootstrap/join.sh \ || fail "join.sh no longer puts the password file on /dev/shm" docker run --rm --entrypoint grep "$IMAGE" -qF -- 'mktemp -p /dev/shm "opendj-replicate.$ADMIN_PORT.' /opt/opendj/bootstrap/replicate.sh \ || fail "replicate.sh no longer puts the password file on /dev/shm" docker run --rm --entrypoint sh "$IMAGE" -c 'f=$(mktemp -p /dev/shm "opendj-join.$ADMIN_PORT.XXXXXX") && rm -f "$f" && case $f in /dev/shm/opendj-join.4444.*) ;; *) exit 1;; esac' \ || fail "mktemp in the image does not create the password file on /dev/shm" # a bounded dsreplication has to take its JVM down with it: the timeout of BusyBox signals # only the shell script that starts java, that of coreutils the whole process group docker run --rm --entrypoint sh "$IMAGE" -c 'timeout --version 2>&1 | grep -q "GNU coreutils"' \ || fail "the image has no timeout of coreutils" # a password with a space in it reaches every tool as one value ROOT_PASSWORD='replication secret' # a subnet of its own, so that a container can be given a known address MASTER_ADDRESS=172.30.99.50 docker network create --subnet 172.30.99.0/24 $NETWORK >/dev/null # small retry values keep the seed decision and the failed-join case quick; the volume holds # the instance so that a container can be replaced with or without its data surviving start_node() { # [...] local name=$1 peers=$2 shift 2 docker volume create "vol-$name" >/dev/null docker run -d --memory="512m" --network $NETWORK --name "$name" --hostname "$name" \ -v "vol-$name:/opt/opendj/data" \ -e ROOT_PASSWORD="$ROOT_PASSWORD" -e ADD_BASE_ENTRY="--addBaseEntry" \ -e OPENDJ_REPLICATION_TYPE=simple -e REPLICATION_PEERS="$peers" \ -e REPLICATION_RETRY_COUNT=10 -e REPLICATION_RETRY_INTERVAL=3 -e REPLICATION_ATTEMPT_TIMEOUT=90 \ "$@" "$IMAGE" >/dev/null } # the first peer of REPLICATION_PEERS, and only it, seeds a topology its retries could not # find; it imports entries no other server ever gets but from an initialize start_node dj-0 dj-0,dj-1 -e SAMPLE_DATA=10 wait_healthy dj-0 logs_have dj-0 "seeding it with this server's data" || fail "dj-0 did not seed the topology" # a seed that follows a reset lets the writes of clients in again, as a rejoin does: dj-0 is # left the way a reset cut off after its disable leaves a volume - the backend held, # $REJOIN_PENDING naming it, no replication domain - and started again while no other peer # answers, so that it seeds once its retries are exhausted dsconfig_on dj-0 set-backend-prop --backend-name userRoot --set writability-mode:internal-only docker exec dj-0 sh -c 'echo userRoot >/opt/opendj/data/.replication-rejoin-pending' # started once without a join, the volume reports itself healthy - nothing else would ever # turn it healthy - and its backend stays held, since only a join lets the writes in again: # the server says which backend that is docker rm -f dj-0 >/dev/null docker run -d --memory="512m" --network $NETWORK --name dj-0 --hostname dj-0 -v vol-dj-0:/opt/opendj/data \ -e ROOT_PASSWORD="$ROOT_PASSWORD" "$IMAGE" >/dev/null wait_until 300 "dj-0 says that no join lets the writes in" logs_have dj-0 "The backend userRoot may still refuse the writes of clients" wait_healthy dj-0 refuses_writes dj-0 held-without-join || fail "dj-0 takes the writes of clients on a backend a rejoin held, without a join to let them in" docker rm -f dj-0 >/dev/null start_node dj-0 dj-0,dj-1 wait_until 300 "the restarted dj-0 leaves its health to the join" logs_have dj-0 "was taken out of its replication topology" wait_healthy dj-0 logs_have dj-0 "seeding it with this server's data" || fail "dj-0 did not seed again after the reset" add_ou dj-0 seeded-after-reset || fail "dj-0 refuses the writes of clients after it seeded" [ "$(published_state dj-0)" = ready ] || fail "dj-0 does not publish itself ready after it seeded" # a joining server tries again while its peer is unreachable docker network disconnect $NETWORK dj-0 start_node dj-1 dj-0,dj-1 wait_until 300 "dj-1 tries again" logs_have dj-1 "trying again in" docker network connect --alias dj-0 $NETWORK dj-0 wait_healthy dj-1 logs_have dj-1 "joined the replication topology through dj-0" || fail "dj-1 did not join through dj-0" # the bootstrapped volume of dj-1 was initialized from the topology although its BASE_DN held # the imported base entry - entries in BASE_DN say nothing about who holds the data. The # sample entries reached dj-0 by import-ldif, never through the changelog, so only an # initialize carries them logs_have dj-1 "initializing from dj-0" || fail "dj-1 did not initialize from the topology" ldaps_has dj-1 "uid=user.0,ou=People,dc=example,dc=com" || fail "dj-1 lacks the entries only an initialize from dj-0 brings" # a change made on the seed reaches the replica add_ou dj-0 replicated wait_has dj-1 "ou=replicated,dc=example,dc=com" # a member is ready again right after a restart, without waiting for its peers - gating it # on them would deadlock a whole-cluster restart under OrderedReady - and replication still # flows docker stop dj-1 >/dev/null docker restart dj-0 >/dev/null wait_until 120 "dj-0 is healthy while dj-1 is down" is_healthy dj-0 docker start dj-1 >/dev/null wait_healthy dj-1 add_ou dj-1 replicated2 wait_has dj-0 "ou=replicated2,dc=example,dc=com" # and neither of them took its replication down to enable it anew: only a server that no # peer registers any more does that, or one that its own cn=admin data does not register, # and a reset followed by a new enable passes every check above for n in dj-0 dj-1; do if logs_have $n "taking its replication configuration down"; then fail "$n reset its replication on a plain restart"; fi done # a seed that no peer joined holds the data without a replication domain, and its join binds # with the ROOT_PASSWORD of the bootstrap: once the root password is changed, a restart is # still healthy, as it is without replication start_node dj-solo dj-solo wait_healthy dj-solo logs_have dj-solo "seeding it with this server's data" || fail "dj-solo did not seed its topology" docker exec dj-solo /opt/opendj/bin/ldappasswordmodify --noPropertiesFile --hostname localhost --port 1636 \ --useSsl --trustAll --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" \ --authzID "dn:cn=Directory Manager" --currentPassword "$ROOT_PASSWORD" --newPassword "changed $ROOT_PASSWORD" >/dev/null docker restart dj-solo >/dev/null wait_until 180 "dj-solo is healthy again with its root password changed" is_healthy dj-solo # and a peer that refuses the bind with ROOT_PASSWORD is past its bootstrap: it may hold the # data of the topology, so a first peer that finds no other one gives up beside it rather # than seed a topology of its own. Its retry interval is written with a leading zero, which # $(( )) reads as an octal number and rejects for 08: read so, it would abandon the rounds at # their first pause and give up all the same, without trying again start_node dj-f dj-f,dj-solo -e REPLICATION_RETRY_COUNT=2 -e REPLICATION_RETRY_INTERVAL=08 wait_until 600 "dj-f gives up beside a peer that refuses the bind" logs_have dj-f "could not join the replication topology after 2 attempts" if logs_have dj-f "seeding it with this server's data"; then fail "dj-f seeded next to a peer past its bootstrap"; fi logs_have dj-f "(1 of 2), trying again in" || fail "dj-f did not try again with REPLICATION_RETRY_INTERVAL=08" if logs_have dj-f "REPLICATION_RETRY_INTERVAL is not a whole number"; then fail "dj-f did not read REPLICATION_RETRY_INTERVAL=08 as 8"; fi docker rm -f dj-f >/dev/null docker volume rm vol-dj-f >/dev/null docker rm -f dj-solo >/dev/null docker volume rm vol-dj-solo >/dev/null # a Kubernetes pod keeps its /dev/shm across container restarts, shared by all its # containers: a starting container removes the password files a killed join or replicate.sh # left there - only those of its own ADMIN_PORT, the files of the other containers of the pod # are not its to remove docker run -d --memory="64m" --ipc=shareable --name dj-shm --entrypoint sleep "$IMAGE" 600 >/dev/null docker exec dj-shm sh -c ': >/dev/shm/opendj-join.4444.killed && : >/dev/shm/opendj-replicate.4444.killed && : >/dev/shm/opendj-join.5444.other' docker run -d --memory="512m" --network $NETWORK --ipc=container:dj-shm --name dj-shm-probe --hostname dj-shm-probe \ -e ROOT_PASSWORD="$ROOT_PASSWORD" "$IMAGE" >/dev/null # the planted files only: the probe's own bootstrap keeps a password file of its ADMIN_PORT # there while it runs shm_cleared() { ! docker exec dj-shm sh -c 'test -e /dev/shm/opendj-join.4444.killed || test -e /dev/shm/opendj-replicate.4444.killed'; } wait_until 60 "run.sh removes the password files of its ADMIN_PORT" shm_cleared docker exec dj-shm test -e /dev/shm/opendj-join.5444.other || fail "run.sh removed the password file of another container" docker rm -f dj-shm-probe dj-shm >/dev/null # a join that cannot succeed keeps the container from reporting itself healthy - across a # restart too, where the health marker used to follow the upgrade and a failed join turned # into a healthy, unreplicated server. dj-x may not seed: it is not the first peer of its list start_node dj-x dj-absent,dj-x wait_until 300 "the join of dj-x gives up" logs_have dj-x "could not join the replication topology" stays_unhealthy dj-x "its join failed" docker restart dj-x >/dev/null wait_until 300 "the restarted dj-x waits for its join" logs_have dj-x "never joined its replication topology" second_failure() { [ "$(logs_count dj-x "could not join the replication topology")" -ge 2 ]; } wait_until 300 "the second join of dj-x gives up" second_failure stays_unhealthy dj-x "it never joined" docker rm -f dj-x >/dev/null docker volume rm vol-dj-x >/dev/null # the seed lost its volume: behind the same name, a fresh dj-0 finds the topology at dj-1 and # takes its data from it instead of seeding an empty one next to it docker rm -f dj-0 >/dev/null docker volume rm vol-dj-0 >/dev/null start_node dj-0 dj-0,dj-1 wait_healthy dj-0 logs_have dj-0 "initializing from dj-1" || fail "the reborn dj-0 did not initialize from dj-1" if logs_have dj-0 "seeding it with this server's data"; then fail "the reborn dj-0 seeded a topology although dj-1 held it"; fi ldaps_has dj-0 "ou=replicated,dc=example,dc=com" || fail "the reborn dj-0 lacks the data of the topology" ldaps_has dj-0 "uid=user.0,ou=People,dc=example,dc=com" || fail "the reborn dj-0 lacks the entries of the seed" # a bootstrap that failed after it imported the base entry leaves a volume that never counts # as holding the data of the topology: once restarted, dj-bf initializes from dj-0 rather # than joining with what its bootstrap left printf 'sh /opt/opendj/bootstrap/setup.sh\nexit 1\n' >"$FAILING_BOOTSTRAP" chmod 644 "$FAILING_BOOTSTRAP" start_node dj-bf dj-0,dj-1,dj-bf -v "$FAILING_BOOTSTRAP:/opt/opendj/failing-bootstrap.sh:ro" \ -e BOOTSTRAP=/opt/opendj/failing-bootstrap.sh wait_until 300 "the bootstrap of dj-bf fails" logs_have dj-bf "failing-bootstrap.sh failed" docker restart dj-bf >/dev/null wait_healthy dj-bf logs_have dj-bf "initializing from dj-0" || fail "dj-bf joined with the data of its failed bootstrap" ldaps_has dj-bf "uid=user.0,ou=People,dc=example,dc=com" || fail "dj-bf lacks the entries only an initialize from dj-0 brings" docker rm -f dj-bf >/dev/null docker volume rm vol-dj-bf >/dev/null # an initialize that fails keeps the volume waiting for the data of the topology: a # one-second bound cuts the dsreplication JVM off before it connects start_node dj-fi dj-0,dj-fi -e REPLICATION_INITIALIZE_TIMEOUT=1 -e REPLICATION_RETRY_COUNT=2 wait_until 300 "the initialize of dj-fi is cut off" logs_have dj-fi "initialize from dj-0 exited with 124" stays_unhealthy dj-fi "its initialize failed" docker exec dj-fi test -f /opt/opendj/data/.replication-initialize-pending \ || fail "dj-fi dropped its pending marker after a failed initialize" docker rm -f dj-fi >/dev/null docker volume rm vol-dj-fi >/dev/null # scale up to three - the peer list is configuration, so the running servers are replaced # with the longer list before the third one starts, as a rolling update would docker rm -f dj-0 dj-1 >/dev/null start_node dj-0 dj-0,dj-1,dj-2 start_node dj-1 dj-0,dj-1,dj-2 wait_healthy dj-0 wait_healthy dj-1 start_node dj-2 dj-0,dj-1,dj-2 wait_healthy dj-2 ldaps_has dj-2 "ou=replicated,dc=example,dc=com" || fail "dj-2 lacks the data of the topology" # scale down to two: the survivors, restarted with the shorter list, remove dj-2 from # cn=admin data and from every replication server list they hold - that of BASE_DN, and # those of cn=schema and cn=admin data that dsreplication enable configures next to it. # dsreplication disable cannot, dj-2 being already gone. dj-2 keeps its volume, for the # scale-up below - and on it, in its cn=config, a join lock held by hand on dj-2 itself hold_lock dj-2 docker rm -f dj-2 >/dev/null docker rm -f dj-0 dj-1 >/dev/null start_node dj-0 dj-0,dj-1 start_node dj-1 dj-0,dj-1 wait_healthy dj-0 wait_healthy dj-1 # whichever survivor's join ran first deletes the cn=admin data entry, the delete replicates # to the other; each survivor prunes its own replication server lists removed_dj2() { logs_have dj-0 "removing departed server dj-2" || logs_have dj-1 "removing departed server dj-2"; } wait_until 180 "a survivor removes dj-2" removed_dj2 for c in dj-0 dj-1; do wait_until 120 "dj-2 leaves cn=admin data of $c" admin_data_lacks $c dj-2 wait_until 120 "dj-2 leaves the replication server lists of $c" lists_lack $c dj-2 done # replication between the survivors is intact add_ou dj-0 replicated3 wait_has dj-1 "ou=replicated3,dc=example,dc=com" # scaled up again on the volume it kept: dj-2 still replicates, but the survivors no longer # register it, and an enable between two servers whose cn=admin data is replicated registers # nobody - so dj-2 takes its replication configuration down, enables again and registers. # The join locks are held by hand meanwhile, and dj-2 must wait for each of them rather than # go on: first its own, which it takes before its configuration goes, then those of both # survivors, which its enables take. With its replication down it must not report itself # ready, publish itself ready to its peers, or take the writes of clients, not even after a # restart. Its enables are bounded generously, so that it does not break the held locks as # left behind by a killed join - the one on dj-2 is held since the scale-down hold_lock dj-0 hold_lock dj-1 start_node dj-2 dj-0,dj-1,dj-2 -e REPLICATION_RETRY_COUNT=30 -e REPLICATION_ATTEMPT_TIMEOUT=1800 wait_until 300 "dj-2 finds that the peers no longer register it" logs_have dj-2 "the peers no longer register this server" wait_until 120 "dj-2 waits for its own lock" logs_have dj-2 "dj-held is enabling replication through this server, waiting for it" docker exec dj-2 test ! -e /opt/opendj/.bootstrap-complete || fail "dj-2 reports itself ready while it waits to take its replication down" [ "$(published_state dj-2)" = rejoining ] || fail "dj-2 does not publish that it is rejoining" # two rounds at most: start_node passes REPLICATION_RETRY_INTERVAL=3, and a round waits up to # twice as long sleep 12 replicates_example dj-2 || fail "dj-2 took its replication down while its own lock was held" drop_lock dj-2 wait_until 120 "dj-2 waits for dj-0" logs_have dj-2 "dj-held is enabling replication through dj-0, waiting for it" wait_until 120 "dj-2 waits for dj-1" logs_have dj-2 "dj-held is enabling replication through dj-1, waiting for it" docker exec dj-2 test ! -e /opt/opendj/.bootstrap-complete || fail "dj-2 reports itself ready while its replication is down" refuses_writes dj-2 unreplicated || fail "dj-2 takes the writes of clients while its replication is down" sleep 12 admin_data_lacks dj-0 dj-2 || fail "dj-2 enabled while the locks of both peers were held" docker restart dj-2 >/dev/null wait_until 300 "the restarted dj-2 leaves its health to the join" logs_have dj-2 "was taken out of its replication topology" waits_again() { logs_have_since dj-2 "was taken out of its replication topology" "dj-held is enabling replication through dj-0, waiting for it"; } wait_until 300 "the restarted dj-2 waits for dj-0 again" waits_again docker exec dj-2 test ! -e /opt/opendj/.bootstrap-complete || fail "the restarted dj-2 reports itself ready while its replication is down" [ "$(published_state dj-2)" = rejoining ] || fail "the restarted dj-2 publishes itself ready while its replication is down" refuses_writes dj-2 unreplicated || fail "the restarted dj-2 takes the writes of clients while its replication is down" # its own lock once more, now with both survivors free: an enable configures both of its # servers, so it waits for this one as well. The lock is taken as soon as no round of dj-2 # holds it wait_until 120 "the lock of dj-2 is held by hand" hold_lock dj-2 own_waits=$(logs_count dj-2 "dj-held is enabling replication through this server, waiting for it") drop_lock dj-0 drop_lock dj-1 own_waits_again() { [ "$(logs_count dj-2 "dj-held is enabling replication through this server, waiting for it")" -gt "$own_waits" ]; } wait_until 120 "dj-2 waits for its own lock with both survivors free" own_waits_again sleep 12 admin_data_lacks dj-0 dj-2 || fail "dj-2 enabled while its own lock was held" # a lock under its own name was left by a join of an earlier start of dj-2, a start running # one join: it is broken at once, not once it is older than an enable may take self=$(docker exec dj-2 hostname -f | tr '[:upper:]' '[:lower:]') relabel_lock dj-2 "$self" wait_until 120 "dj-2 breaks the lock left under its own name" logs_have dj-2 "breaking the lock $self took on localhost" wait_healthy dj-2 registered_dj2() { ! admin_data_lacks dj-0 dj-2; } wait_until 300 "dj-2 registers in the topology again" registered_dj2 [ "$(published_state dj-2)" = ready ] || fail "the rejoined dj-2 does not publish itself ready" add_ou dj-2 rejoined wait_has dj-0 "ou=rejoined,dc=example,dc=com" # two servers taken out together and scaled up again together on their volumes: both take # their replication down, and with it every registration in their cn=admin data. While the # lock of the survivor is held, neither may enable through the other - an enable between two # servers without replication registers the two of them only, which passes for membership, # and they would serve a topology of their own beside dj-0 - so each waits for dj-0 and joins # through it. The operator made the backend of dj-2 read-only beforehand: its rejoin refuses # the writes of clients as well, and leaves the mode as it found it. dj-0 keeps the shorter # list, nothing here starts it again. # A server that comes back shows its peers the state it had before it left - it is kept in # its cn=config - until its first round finds that the peers no longer register it. So dj-1 # starts first and is rejoining before dj-2 starts, and dj-2 keeps a lock held by hand on # itself from before it left, as in the scale-up above, which no enable through it gets past # while it still shows that state dsconfig_on dj-2 set-backend-prop --backend-name userRoot --set writability-mode:internal-only hold_lock dj-2 docker rm -f dj-0 dj-1 dj-2 >/dev/null start_node dj-0 dj-0 wait_healthy dj-0 for n in dj-1 dj-2; do wait_until 180 "$n leaves cn=admin data of dj-0" admin_data_lacks dj-0 $n wait_until 120 "$n leaves the replication server lists of dj-0" lists_lack dj-0 $n done hold_lock dj-0 start_node dj-1 dj-0,dj-1,dj-2 -e REPLICATION_RETRY_COUNT=30 -e REPLICATION_ATTEMPT_TIMEOUT=1800 dj1_rejoining() { [ "$(published_state dj-1)" = rejoining ]; } wait_until 300 "dj-1 publishes that it is rejoining" dj1_rejoining start_node dj-2 dj-0,dj-1,dj-2 -e REPLICATION_RETRY_COUNT=30 -e REPLICATION_ATTEMPT_TIMEOUT=1800 wait_until 300 "dj-1 passes over the rejoining dj-2" logs_have dj-1 "dj-2 is rejoining, not a peer to join through" drop_lock dj-2 wait_until 180 "dj-2 passes over the rejoining dj-1" logs_have dj-2 "dj-1 is rejoining, not a peer to join through" sleep 12 for n in dj-1 dj-2; do docker exec $n test ! -e /opt/opendj/.bootstrap-complete || fail "$n reports itself ready while its replication is down" done admin_data_lacks dj-2 dj-1 || fail "dj-1 and dj-2 registered each other while the lock of dj-0 was held" admin_data_lacks dj-1 dj-2 || fail "dj-1 and dj-2 registered each other while the lock of dj-0 was held" refuses_writes dj-1 unreplicated || fail "dj-1 takes the writes of clients while its replication is down" drop_lock dj-0 # the health status says nothing yet: dj-1 and dj-2 started on volumes without a reset # pending, so their run.sh reported them ready, a probe may have found them healthy before # their joins took the replication down, and the status turns unhealthy only after the # probe's retries. Their joins publish them ready once the writes are let in again for n in dj-1 dj-2; do wait_until 480 "$n rejoins and publishes itself ready" published_ready $n done wait_healthy dj-1 wait_healthy dj-2 for n in dj-1 dj-2; do wait_until 300 "$n registers with dj-0 again" registered_in_dj0 $n done [ "$(backend_mode dj-2)" = internal-only ] || fail "the rejoin of dj-2 let the writes of clients in on a backend the operator made read-only" add_ou dj-1 rejoined-together wait_has dj-0 "ou=rejoined-together,dc=example,dc=com" wait_has dj-2 "ou=rejoined-together,dc=example,dc=com" # an enable that stopped half-way - its initialize of cn=admin data failed - leaves BASE_DN # replicated and the server registered at its peers, but not in its own cn=admin data, and # every enable after that exits 5, "already replicated", without registering it there. So a # server in that state takes its replication down and enables anew, as one that no peer # registers does. dj-1 lacks its own entry, then dj-2 the whole of cn=Servers, which the # failed initialize of the CI run behind this case may have left; neither delete reaches a # peer. One at a time: the disable of a reset changes the peers it replicates with as well half_enabled_rejoins() { # local n=$1 members admin_data_lacks $n $n || fail "$n is still registered in its own cn=admin data" registered_in_dj0 $n || fail "the delete on $n alone reached dj-0" members=$(logs_count $n "the health check may probe it") docker restart $n >/dev/null wait_until 300 "$n takes its half-enabled replication down" logs_have $n "the peers register this server but its own cn=admin data does not" member_again() { [ "$(logs_count $n "the health check may probe it")" -gt "$members" ]; } wait_until 600 "$n rejoins" member_again published_ready $n || fail "$n does not publish itself ready after it rejoined" ! admin_data_lacks $n $n || fail "$n is not registered in its own cn=admin data after it rejoined" registered_in_dj0 $n || fail "$n is not registered with dj-0 after it rejoined" wait_healthy $n } own_entry=$(admin_search dj-1 "cn=Servers,cn=admin data" one "(hostname=dj-1)" 1.1 | awk '/^dn: / { print substr($0, 5) }') [ -n "$own_entry" ] || fail "dj-1 does not register itself in its cn=admin data" delete_here dj-1 "$own_entry" || fail "could not delete $own_entry on dj-1 alone" half_enabled_rejoins dj-1 delete_here dj-2 "cn=Servers,cn=admin data" --deleteSubtree || fail "could not delete cn=Servers on dj-2 alone" half_enabled_rejoins dj-2 add_ou dj-1 rejoined-half-enabled wait_has dj-0 "ou=rejoined-half-enabled,dc=example,dc=com" wait_has dj-2 "ou=rejoined-half-enabled,dc=example,dc=com" # the deprecated one-shot sdsr path of replicate.sh still bootstraps a replica. On a first # start a server stops the server its bootstrap started and starts it again, and a replica # can reach it in between: replicate.sh tries a dsreplication enable that could not connect # again (exit 8). The replica starts on a network of its own, where dj-0 does not resolve, and # moves to the network of the topology only once it has said it will try again: taking dj-0 # off the network for as long instead would leave its replication server holding the dead # connection of dj-0's own directory server, and route the initialize of the replica into it docker network create $NETWORK_ALONE >/dev/null docker run -d --memory="512m" --network $NETWORK_ALONE --name dj-sdsr --hostname dj-sdsr \ -e ROOT_PASSWORD="$ROOT_PASSWORD" -e MASTER_SERVER=dj-0 -e OPENDJ_REPLICATION_TYPE=sdsr "$IMAGE" >/dev/null wait_until 420 "dj-sdsr tries again" logs_have dj-sdsr "exited with 8, trying again" docker network connect $NETWORK dj-sdsr docker network disconnect $NETWORK_ALONE dj-sdsr wait_healthy dj-sdsr wait_has dj-sdsr "ou=replicated3,dc=example,dc=com" # dj-sdsr publishes no state, as a server of an image before this one does, and it # replicates BASE_DN, so it may hold the data of the topology: a fresh first peer whose only # other peer it is gives up rather than seed next to it start_node dj-f dj-f,dj-sdsr -e REPLICATION_RETRY_COUNT=2 -e REPLICATION_RETRY_INTERVAL=5 wait_until 300 "dj-f gives up" logs_have dj-f "could not join the replication topology after 2 attempts" if logs_have dj-f "seeding it with this server's data"; then fail "dj-f seeded next to a member that publishes no state"; fi stays_unhealthy dj-f "a peer that may hold the data answers" docker rm -f dj-f >/dev/null docker volume rm vol-dj-f >/dev/null # replicate.sh tries dsreplication enable again only when it exits 8, and a failed enable # ends it: run once more on the replica for a base DN neither server holds, the enable exits # 5 (REPLICATION_CANNOT_BE_ENABLED_ON_BASEDN) and nothing is tried again or initialized rc=0 out=$(docker exec -e BASE_DN=dc=absent,dc=com -e ROOT_USER_DN="cn=Directory Manager" dj-sdsr \ timeout 90 /opt/opendj/bootstrap/replicate.sh 2>&1) || rc=$? if [ "$rc" -ne 5 ] || grep -E "trying again|initializing replication" <<<"$out" >/dev/null; then echo "$out" fail "replicate.sh for a base DN nobody holds exited with $rc, not with the 5 of its dsreplication enable, or went on after it" fi # the root password shows in no container log, and the files the tools read it from are gone for c in dj-0 dj-1 dj-2 dj-sdsr; do if docker logs $c 2>&1 | grep -F -- "$ROOT_PASSWORD"; then fail "the root password is in the log of $c"; fi left=$(docker exec $c grep -rlsF -- "$ROOT_PASSWORD" /tmp /dev/shm || true) [ -z "$left" ] || fail "the root password is left in $left of $c" done docker rm -f dj-0 dj-1 dj-2 dj-sdsr >/dev/null docker volume rm vol-dj-0 vol-dj-1 vol-dj-2 >/dev/null # MASTER_SERVER may name the master by its address, as the grep of /etc/hosts in the base # replicate.sh allowed: a master given its own address recognises itself and seeds at once, # and a replica joins and initializes through that address for n in dj-ipm dj-ipr; do if [ $n = dj-ipm ]; then extra="--ip $MASTER_ADDRESS -e SAMPLE_DATA=10"; else extra="-v vol-dj-ipr:/opt/opendj/data"; fi docker run -d --memory="512m" --network $NETWORK --name $n --hostname $n $extra \ -e ROOT_PASSWORD="$ROOT_PASSWORD" -e ADD_BASE_ENTRY="--addBaseEntry" \ -e OPENDJ_REPLICATION_TYPE=simple -e MASTER_SERVER=$MASTER_ADDRESS \ -e REPLICATION_RETRY_COUNT=10 -e REPLICATION_RETRY_INTERVAL=3 "$IMAGE" >/dev/null wait_healthy $n done logs_have dj-ipm "seeding it with this server's data" || fail "dj-ipm did not recognise itself by its address" logs_have dj-ipr "initializing from $MASTER_ADDRESS" || fail "dj-ipr did not initialize from the master's address" ldaps_has dj-ipr "uid=user.0,ou=People,dc=example,dc=com" || fail "dj-ipr lacks the entries of the master" # moved from MASTER_SERVER to a REPLICATION_PEERS of names, the replica keeps the master that # is registered by its address - in cn=admin data and in its replication server lists - # rather than taking it for a server that left the topology docker rm -f dj-ipr >/dev/null docker run -d --memory="512m" --network $NETWORK --name dj-ipr --hostname dj-ipr -v vol-dj-ipr:/opt/opendj/data \ -e ROOT_PASSWORD="$ROOT_PASSWORD" -e OPENDJ_REPLICATION_TYPE=simple -e REPLICATION_PEERS=dj-ipm,dj-ipr \ -e REPLICATION_RETRY_COUNT=2 -e REPLICATION_RETRY_INTERVAL=3 "$IMAGE" >/dev/null wait_until 300 "the moved dj-ipr has looked for departed servers" logs_have dj-ipr "the health check may probe it" if logs_have dj-ipr "removing departed server $MASTER_ADDRESS" || logs_have dj-ipr "removing departed replication server $MASTER_ADDRESS:"; then fail "dj-ipr removed the master registered by its address" fi if admin_data_lacks dj-ipr "$MASTER_ADDRESS"; then fail "the master registered by its address left cn=admin data of dj-ipr"; fi docker rm -f dj-ipm dj-ipr >/dev/null docker volume rm vol-dj-ipr >/dev/null # three fresh servers started together, as Compose or a StatefulSet with # podManagementPolicy: Parallel start them: none of them takes the bootstrap data of another # fresh one for the topology's. The first peer seeds, the others initialize from it - its # sample entries reach them only that way - and the replication servers know each other # although the joins ran at the same time for n in dj-p0 dj-p1 dj-p2; do if [ $n = dj-p0 ]; then extra="-e SAMPLE_DATA=10"; else extra=; fi start_node $n dj-p0,dj-p1,dj-p2 -e REPLICATION_RETRY_COUNT=20 $extra done for n in dj-p0 dj-p1 dj-p2; do wait_healthy $n done logs_have dj-p0 "seeding it with this server's data" || fail "dj-p0 did not seed the topology" for n in dj-p1 dj-p2; do if logs_have $n "seeding it with this server's data"; then fail "$n seeded a topology although it is not the first peer"; fi logs_have $n "initializing from dj-p0" || fail "$n did not initialize from dj-p0" if logs_have $n "initializing from dj-p1" || logs_have $n "initializing from dj-p2"; then fail "$n initialized from a server that holds only bootstrap data" fi ldaps_has $n "uid=user.0,ou=People,dc=example,dc=com" || fail "$n lacks the entries of the seed" done wait_until 120 "the replication server of dj-p0 knows dj-p1 and dj-p2" server_lists dj-p0 dj-p1 dj-p2 wait_until 120 "the replication server of dj-p1 knows dj-p0 and dj-p2" server_lists dj-p1 dj-p0 dj-p2 wait_until 120 "the replication server of dj-p2 knows dj-p0 and dj-p1" server_lists dj-p2 dj-p0 dj-p1 add_ou dj-p1 parallel wait_has dj-p0 "ou=parallel,dc=example,dc=com" wait_has dj-p2 "ou=parallel,dc=example,dc=com" # dj-p1 and dj-p2 enabled through dj-p0 one at a time, and each removed its lock after if admin_search dj-p0 "cn=Docker Join Lock,cn=config" base "(objectClass=*)" 1.1 | grep "^dn:" >/dev/null; then fail "a join left its lock on dj-p0" fi for c in dj-p0 dj-p1 dj-p2; do if docker logs $c 2>&1 | grep -F -- "$ROOT_PASSWORD"; then fail "the root password is in the log of $c"; fi done cleanup echo "Docker replication test passed"