From 947c0c9a191cc5b771411e0d0f03d7a56aeedf1e Mon Sep 17 00:00:00 2001
From: Valery Kharseko <vharseko@3a-systems.ru>
Date: Tue, 06 Oct 2026 06:58:19 +0000
Subject: [PATCH] [#1086] Join replication in the background on every start of the Docker image (#1115)
---
opendj-packages/opendj-docker/run.sh | 76 +
opendj-packages/opendj-docker/Dockerfile-alpine | 12
opendj-packages/opendj-docker/README.md | 142 +++
opendj-packages/opendj-docker/bootstrap/join.sh | 976 +++++++++++++++++++++++++++
.github/workflows/build.yml | 152 ---
opendj-packages/opendj-docker/bootstrap/replicate.sh | 62 -
.github/scripts/docker-test-replication.sh | 718 +++++++++++++++++++
opendj-packages/opendj-docker/Dockerfile | 10
8 files changed, 1,947 insertions(+), 201 deletions(-)
diff --git a/.github/scripts/docker-test-replication.sh b/.github/scripts/docker-test-replication.sh
new file mode 100755
index 0000000..f5cdca0
--- /dev/null
+++ b/.github/scripts/docker-test-replication.sh
@@ -0,0 +1,718 @@
+#!/usr/bin/env bash
+#
+# The contents of this file are subject to the terms of the Common Development and
+# Distribution License (the License). You may not use this file except in compliance with the
+# License.
+#
+# You can obtain a copy of the License at legal/CDDLv1.0.txt. See the License for the
+# specific language governing permission and limitations under the License.
+#
+# When distributing Covered Software, include this CDDL Header Notice in each file and include
+# the License file at legal/CDDLv1.0.txt. If applicable, add the following below the CDDL
+# Header, with the fields enclosed by brackets [] replaced by your own identifying
+# information: "Portions copyright [year] [name of copyright owner]".
+#
+# Copyright 2026 3A Systems, LLC.
+
+# Tests the replication of the Docker image (#1086): the background join of
+# OPENDJ_REPLICATION_TYPE=simple (bootstrap/join.sh) and the deprecated one-shot sdsr path of
+# bootstrap/replicate.sh. Both image jobs of build.yml run it, each with its own image.
+#
+# Usage: docker-test-replication.sh <image>
+
+set -eE -o pipefail
+
+IMAGE=${1:?usage: $0 <image>}
+NODES="dj-0 dj-1 dj-2 dj-x dj-solo dj-bf dj-fi dj-f dj-p0 dj-p1 dj-p2 dj-ipm dj-ipr dj-sdsr dj-shm-probe dj-shm"
+VOLUMES="vol-dj-0 vol-dj-1 vol-dj-2 vol-dj-x vol-dj-solo vol-dj-bf vol-dj-fi vol-dj-f vol-dj-ipr vol-dj-p0 vol-dj-p1 vol-dj-p2"
+# a bootstrap that imports the base entry and then fails, mounted into dj-bf
+FAILING_BOOTSTRAP=$(mktemp)
+NETWORK=test_replication
+# where dj-sdsr starts: it resolves its own name there, and no other server
+NETWORK_ALONE=test_replication_alone
+
+cleanup() {
+ rm -f "$FAILING_BOOTSTRAP"
+ docker rm -f $NODES >/dev/null 2>&1 || true
+ docker network rm $NETWORK $NETWORK_ALONE >/dev/null 2>&1 || true
+ docker volume rm -f $VOLUMES >/dev/null 2>&1 || true
+}
+cleanup
+# the container log says that a dsreplication failed and where its detailed log is; the errors
+# log of the server and those detailed logs say why. Both are in the instance, which the
+# image keeps under /opt/opendj/data, and a stopped container has nothing to exec in
+dump_logs() {
+ local c
+ for c in $NODES; do
+ echo "::group::container logs ($c)"
+ docker logs $c 2>&1 || true
+ echo "::endgroup::"
+ [ "$(docker inspect -f '{{.State.Running}}' $c 2>/dev/null)" = true ] || continue
+ echo "::group::server logs ($c)"
+ docker exec $c sh -c 'tail -n 1000 /opt/opendj/data/logs/errors; for f in /opt/opendj/data/tmp/opendj-replication-*.log; do [ -f "$f" ] && echo "--- $f" && cat "$f"; done' 2>&1 || true
+ echo "::endgroup::"
+ done
+}
+# set -E hands the trap to every $(...) as well, where a failing command would dump the logs
+# into the value being captured and remove the containers under the running test: there the
+# failure only ends the subshell, and the main shell decides
+trap 'code=$?; [ "$BASH_SUBSHELL" -eq 0 ] || exit $code; dump_logs; cleanup; exit $code' ERR
+
+fail() {
+ echo "::error::$1"
+ false
+}
+
+# grep reads the whole log rather than stopping at the first match (-q), so docker logs never
+# writes into a closed pipe, which pipefail would report as a failure of the check
+logs_have() { # <container> <text>
+ docker logs "$1" 2>&1 | grep -F -- "$2" >/dev/null
+}
+
+logs_count() { # <container> <text>
+ docker logs "$1" 2>&1 | grep -cF -- "$2" || true
+}
+
+# the container logged the text after the last line that holds <since> - after a restart,
+# say, which the join of the start before may still have logged into until it was stopped
+logs_have_since() { # <container> <since> <text>
+ docker logs "$1" 2>&1 | awk -v s="$2" 'index($0, s) { buf = ""; next } { buf = buf $0 "\n" } END { printf "%s", buf }' \
+ | grep -F -- "$3" >/dev/null
+}
+
+wait_until() { # <seconds> <what> <command> [<argument>...]
+ local deadline=$((SECONDS + $1)) what=$2
+ shift 2
+ until "$@"; do
+ [ "$SECONDS" -lt "$deadline" ] || fail "timed out waiting until $what"
+ sleep 2
+ done
+}
+
+health() {
+ docker inspect --format='{{.State.Health.Status}}' "$1"
+}
+
+is_healthy() {
+ [ "$(health "$1")" = healthy ]
+}
+
+wait_healthy() {
+ wait_until 480 "$1 is healthy" is_healthy "$1"
+}
+
+# A container reports itself healthy only after a probe that finds the marker, and the
+# probes run every 30 s: a single look right after a failure reads "starting" whatever the
+# join did, so the container is watched over more than two probe cycles
+stays_unhealthy() { # <container> <why>
+ local i
+ for i in $(seq 1 15); do
+ is_healthy "$1" && fail "$1 reports itself healthy although $2"
+ sleep 5
+ done
+}
+
+ldaps_has() { # <container> <DN>
+ docker exec "$1" /opt/opendj/bin/ldapsearch --noPropertiesFile --hostname localhost --port 1636 \
+ --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll \
+ --baseDN "$2" --searchScope base "(objectClass=*)" 1.1 >/dev/null 2>&1
+}
+
+wait_has() { # <container> <DN>
+ wait_until 60 "$2 is on $1" ldaps_has "$1" "$2"
+}
+
+admin_search() { # <container> <base DN> <scope> <filter> [<attribute>...]
+ local container=$1 base=$2 scope=$3 filter=$4
+ shift 4
+ docker exec "$container" /opt/opendj/bin/ldapsearch --noPropertiesFile --hostname localhost --port 4444 \
+ --useSsl --trustAll --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" \
+ --baseDN "$base" --searchScope "$scope" "$filter" "$@" 2>/dev/null
+}
+
+add_ou() { # <container> <ou>
+ printf 'dn: ou=%s,dc=example,dc=com\nobjectClass: organizationalUnit\nou: %s\n' "$2" "$2" \
+ | docker exec -i "$1" /opt/opendj/bin/ldapmodify --noPropertiesFile --hostname localhost --port 1636 \
+ --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll --defaultAdd >/dev/null
+}
+
+dsconfig_on() { # <container> <dsconfig argument>...
+ local container=$1
+ shift
+ docker exec "$container" /opt/opendj/bin/dsconfig "$@" --hostname localhost --port 4444 \
+ --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --trustAll --no-prompt >/dev/null
+}
+
+# the writability-mode the backend of dc=example,dc=com is configured with, empty where its
+# entry sets none
+backend_mode() { # <container>
+ admin_search "$1" "ds-cfg-backend-id=userRoot,cn=Backends,cn=config" base "(objectClass=*)" ds-cfg-writability-mode \
+ | awk 'tolower($1) == "ds-cfg-writability-mode:" { print $2 }'
+}
+
+# the replication server lists of a container: the one of its replication server and one per
+# replication domain (BASE_DN, cn=schema, cn=admin data)
+replication_lists() {
+ admin_search "$1" "cn=config" sub \
+ "(|(objectClass=ds-cfg-replication-server)(objectClass=ds-cfg-replication-domain))" ds-cfg-replication-server
+}
+
+lists_lack() { # <container> <name>
+ ! replication_lists "$1" | grep -F -- "$2" >/dev/null
+}
+
+admin_data_lacks() { # <container> <name>
+ ! admin_search "$1" "cn=Servers,cn=admin data" one "(objectClass=*)" hostname | grep -F -- "$2" >/dev/null
+}
+
+# the join lock of a server, held by hand as a join of a server named dj-held would hold it
+hold_lock() { # <container>
+ printf 'dn: cn=Docker Join Lock,cn=config\nobjectClass: top\nobjectClass: ds-cfg-branch\nobjectClass: extensibleObject\ncn: Docker Join Lock\ndescription: dj-held %s\n' "$(date +%s)" \
+ | docker exec -i "$1" /opt/opendj/bin/ldapmodify --noPropertiesFile --hostname localhost --port 4444 \
+ --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll --defaultAdd >/dev/null
+}
+
+drop_lock() { # <container>
+ printf 'dn: cn=Docker Join Lock,cn=config\nchangetype: delete\n' \
+ | docker exec -i "$1" /opt/opendj/bin/ldapmodify --noPropertiesFile --hostname localhost --port 4444 \
+ --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll >/dev/null
+}
+
+# hands a lock held by hand over to another holder at once, as a join of that server took it now
+relabel_lock() { # <container> <holder>
+ printf 'dn: cn=Docker Join Lock,cn=config\nchangetype: modify\nreplace: description\ndescription: %s %s\n' "$2" "$(date +%s)" \
+ | docker exec -i "$1" /opt/opendj/bin/ldapmodify --noPropertiesFile --hostname localhost --port 4444 \
+ --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll >/dev/null
+}
+
+# the state the join of the container publishes to its peers
+published_state() { # <container>
+ admin_search "$1" "cn=Docker Join,cn=config" base "(objectClass=*)" description \
+ | awk 'tolower($1) == "description:" { print $2 }'
+}
+
+published_ready() { # <container>
+ [ "$(published_state "$1")" = ready ]
+}
+
+registered_in_dj0() { # <name>
+ ! admin_data_lacks dj-0 "$1"
+}
+
+# deletes an entry on the container alone: the replication repair control keeps the delete
+# from reaching any other server, as an initialize that failed half-way leaves the
+# cn=admin data of one server unlike that of the others
+delete_here() { # <container> <DN> [<ldapdelete option>...]
+ local container=$1 dn=$2
+ shift 2
+ docker exec "$container" /opt/opendj/bin/ldapdelete --noPropertiesFile --hostname localhost --port 4444 \
+ --useSsl --trustAll --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" \
+ --control 1.3.6.1.4.1.26027.1.5.2:true "$@" "$dn" >/dev/null
+}
+
+# the container holds the replication domain of dc=example,dc=com
+replicates_example() { # <container>
+ admin_search "$1" "cn=config" sub "(&(objectClass=ds-cfg-replication-domain)(ds-cfg-base-dn=dc=example,dc=com))" 1.1 \
+ | grep "^dn:" >/dev/null
+}
+
+# a client's write to the container is refused with Unwilling to Perform (53), as the backend
+# of a server whose replication is down refuses it
+refuses_writes() { # <container> <ou>
+ local rc=0
+ add_ou "$1" "$2" 2>/dev/null || rc=$?
+ [ "$rc" -eq 53 ]
+}
+
+# the replication server of the container lists every one of the other names
+server_lists() { # <container> <name>...
+ local container=$1 list name
+ shift
+ list=$(admin_search "$container" "cn=config" sub "(objectClass=ds-cfg-replication-server)" ds-cfg-replication-server)
+ for name in "$@"; do
+ grep -F -- "ds-cfg-replication-server: $name:" <<<"$list" >/dev/null || return 1
+ done
+}
+
+# every tool reads the root password from a file (#1084, #1092); dsreplication run with -n
+# prints no command line, so a password put back on one would pass every check below
+rc=0
+docker run --rm --entrypoint grep "$IMAGE" -nE -- '(^|[[:space:]])(-w|--(bindPassword[12]?|adminPassword|rootUserPassword))([[:space:]=]|$)' \
+ /opt/opendj/bootstrap/setup.sh /opt/opendj/bootstrap/replicate.sh /opt/opendj/bootstrap/join.sh || rc=$?
+[ "$rc" -eq 1 ] || fail "a bootstrap script passes the root password on a command line, or grep could not read them"
+# the password files go to /dev/shm, off the writable layer of the container, and the mktemp
+# of the image puts them there
+docker run --rm --entrypoint grep "$IMAGE" -qF -- 'mktemp -p /dev/shm "opendj-join.$ADMIN_PORT.' /opt/opendj/bootstrap/join.sh \
+ || fail "join.sh no longer puts the password file on /dev/shm"
+docker run --rm --entrypoint grep "$IMAGE" -qF -- 'mktemp -p /dev/shm "opendj-replicate.$ADMIN_PORT.' /opt/opendj/bootstrap/replicate.sh \
+ || fail "replicate.sh no longer puts the password file on /dev/shm"
+docker run --rm --entrypoint sh "$IMAGE" -c 'f=$(mktemp -p /dev/shm "opendj-join.$ADMIN_PORT.XXXXXX") && rm -f "$f" && case $f in /dev/shm/opendj-join.4444.*) ;; *) exit 1;; esac' \
+ || fail "mktemp in the image does not create the password file on /dev/shm"
+# a bounded dsreplication has to take its JVM down with it: the timeout of BusyBox signals
+# only the shell script that starts java, that of coreutils the whole process group
+docker run --rm --entrypoint sh "$IMAGE" -c 'timeout --version 2>&1 | grep -q "GNU coreutils"' \
+ || fail "the image has no timeout of coreutils"
+
+# a password with a space in it reaches every tool as one value
+ROOT_PASSWORD='replication secret'
+# a subnet of its own, so that a container can be given a known address
+MASTER_ADDRESS=172.30.99.50
+docker network create --subnet 172.30.99.0/24 $NETWORK >/dev/null
+
+# small retry values keep the seed decision and the failed-join case quick; the volume holds
+# the instance so that a container can be replaced with or without its data surviving
+start_node() { # <name> <peers> [<docker run option>...]
+ local name=$1 peers=$2
+ shift 2
+ docker volume create "vol-$name" >/dev/null
+ docker run -d --memory="512m" --network $NETWORK --name "$name" --hostname "$name" \
+ -v "vol-$name:/opt/opendj/data" \
+ -e ROOT_PASSWORD="$ROOT_PASSWORD" -e ADD_BASE_ENTRY="--addBaseEntry" \
+ -e OPENDJ_REPLICATION_TYPE=simple -e REPLICATION_PEERS="$peers" \
+ -e REPLICATION_RETRY_COUNT=10 -e REPLICATION_RETRY_INTERVAL=3 -e REPLICATION_ATTEMPT_TIMEOUT=90 \
+ "$@" "$IMAGE" >/dev/null
+}
+
+# the first peer of REPLICATION_PEERS, and only it, seeds a topology its retries could not
+# find; it imports entries no other server ever gets but from an initialize
+start_node dj-0 dj-0,dj-1 -e SAMPLE_DATA=10
+wait_healthy dj-0
+logs_have dj-0 "seeding it with this server's data" || fail "dj-0 did not seed the topology"
+
+# a seed that follows a reset lets the writes of clients in again, as a rejoin does: dj-0 is
+# left the way a reset cut off after its disable leaves a volume - the backend held,
+# $REJOIN_PENDING naming it, no replication domain - and started again while no other peer
+# answers, so that it seeds once its retries are exhausted
+dsconfig_on dj-0 set-backend-prop --backend-name userRoot --set writability-mode:internal-only
+docker exec dj-0 sh -c 'echo userRoot >/opt/opendj/data/.replication-rejoin-pending'
+# started once without a join, the volume reports itself healthy - nothing else would ever
+# turn it healthy - and its backend stays held, since only a join lets the writes in again:
+# the server says which backend that is
+docker rm -f dj-0 >/dev/null
+docker run -d --memory="512m" --network $NETWORK --name dj-0 --hostname dj-0 -v vol-dj-0:/opt/opendj/data \
+ -e ROOT_PASSWORD="$ROOT_PASSWORD" "$IMAGE" >/dev/null
+wait_until 300 "dj-0 says that no join lets the writes in" logs_have dj-0 "The backend userRoot may still refuse the writes of clients"
+wait_healthy dj-0
+refuses_writes dj-0 held-without-join || fail "dj-0 takes the writes of clients on a backend a rejoin held, without a join to let them in"
+docker rm -f dj-0 >/dev/null
+start_node dj-0 dj-0,dj-1
+wait_until 300 "the restarted dj-0 leaves its health to the join" logs_have dj-0 "was taken out of its replication topology"
+wait_healthy dj-0
+logs_have dj-0 "seeding it with this server's data" || fail "dj-0 did not seed again after the reset"
+add_ou dj-0 seeded-after-reset || fail "dj-0 refuses the writes of clients after it seeded"
+[ "$(published_state dj-0)" = ready ] || fail "dj-0 does not publish itself ready after it seeded"
+
+# a joining server tries again while its peer is unreachable
+docker network disconnect $NETWORK dj-0
+start_node dj-1 dj-0,dj-1
+wait_until 300 "dj-1 tries again" logs_have dj-1 "trying again in"
+docker network connect --alias dj-0 $NETWORK dj-0
+wait_healthy dj-1
+logs_have dj-1 "joined the replication topology through dj-0" || fail "dj-1 did not join through dj-0"
+# the bootstrapped volume of dj-1 was initialized from the topology although its BASE_DN held
+# the imported base entry - entries in BASE_DN say nothing about who holds the data. The
+# sample entries reached dj-0 by import-ldif, never through the changelog, so only an
+# initialize carries them
+logs_have dj-1 "initializing from dj-0" || fail "dj-1 did not initialize from the topology"
+ldaps_has dj-1 "uid=user.0,ou=People,dc=example,dc=com" || fail "dj-1 lacks the entries only an initialize from dj-0 brings"
+
+# a change made on the seed reaches the replica
+add_ou dj-0 replicated
+wait_has dj-1 "ou=replicated,dc=example,dc=com"
+
+# a member is ready again right after a restart, without waiting for its peers - gating it
+# on them would deadlock a whole-cluster restart under OrderedReady - and replication still
+# flows
+docker stop dj-1 >/dev/null
+docker restart dj-0 >/dev/null
+wait_until 120 "dj-0 is healthy while dj-1 is down" is_healthy dj-0
+docker start dj-1 >/dev/null
+wait_healthy dj-1
+add_ou dj-1 replicated2
+wait_has dj-0 "ou=replicated2,dc=example,dc=com"
+# and neither of them took its replication down to enable it anew: only a server that no
+# peer registers any more does that, or one that its own cn=admin data does not register,
+# and a reset followed by a new enable passes every check above
+for n in dj-0 dj-1; do
+ if logs_have $n "taking its replication configuration down"; then fail "$n reset its replication on a plain restart"; fi
+done
+
+# a seed that no peer joined holds the data without a replication domain, and its join binds
+# with the ROOT_PASSWORD of the bootstrap: once the root password is changed, a restart is
+# still healthy, as it is without replication
+start_node dj-solo dj-solo
+wait_healthy dj-solo
+logs_have dj-solo "seeding it with this server's data" || fail "dj-solo did not seed its topology"
+docker exec dj-solo /opt/opendj/bin/ldappasswordmodify --noPropertiesFile --hostname localhost --port 1636 \
+ --useSsl --trustAll --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" \
+ --authzID "dn:cn=Directory Manager" --currentPassword "$ROOT_PASSWORD" --newPassword "changed $ROOT_PASSWORD" >/dev/null
+docker restart dj-solo >/dev/null
+wait_until 180 "dj-solo is healthy again with its root password changed" is_healthy dj-solo
+# and a peer that refuses the bind with ROOT_PASSWORD is past its bootstrap: it may hold the
+# data of the topology, so a first peer that finds no other one gives up beside it rather
+# than seed a topology of its own. Its retry interval is written with a leading zero, which
+# $(( )) reads as an octal number and rejects for 08: read so, it would abandon the rounds at
+# their first pause and give up all the same, without trying again
+start_node dj-f dj-f,dj-solo -e REPLICATION_RETRY_COUNT=2 -e REPLICATION_RETRY_INTERVAL=08
+wait_until 600 "dj-f gives up beside a peer that refuses the bind" logs_have dj-f "could not join the replication topology after 2 attempts"
+if logs_have dj-f "seeding it with this server's data"; then fail "dj-f seeded next to a peer past its bootstrap"; fi
+logs_have dj-f "(1 of 2), trying again in" || fail "dj-f did not try again with REPLICATION_RETRY_INTERVAL=08"
+if logs_have dj-f "REPLICATION_RETRY_INTERVAL is not a whole number"; then fail "dj-f did not read REPLICATION_RETRY_INTERVAL=08 as 8"; fi
+docker rm -f dj-f >/dev/null
+docker volume rm vol-dj-f >/dev/null
+docker rm -f dj-solo >/dev/null
+docker volume rm vol-dj-solo >/dev/null
+
+# a Kubernetes pod keeps its /dev/shm across container restarts, shared by all its
+# containers: a starting container removes the password files a killed join or replicate.sh
+# left there - only those of its own ADMIN_PORT, the files of the other containers of the pod
+# are not its to remove
+docker run -d --memory="64m" --ipc=shareable --name dj-shm --entrypoint sleep "$IMAGE" 600 >/dev/null
+docker exec dj-shm sh -c ': >/dev/shm/opendj-join.4444.killed && : >/dev/shm/opendj-replicate.4444.killed && : >/dev/shm/opendj-join.5444.other'
+docker run -d --memory="512m" --network $NETWORK --ipc=container:dj-shm --name dj-shm-probe --hostname dj-shm-probe \
+ -e ROOT_PASSWORD="$ROOT_PASSWORD" "$IMAGE" >/dev/null
+# the planted files only: the probe's own bootstrap keeps a password file of its ADMIN_PORT
+# there while it runs
+shm_cleared() { ! docker exec dj-shm sh -c 'test -e /dev/shm/opendj-join.4444.killed || test -e /dev/shm/opendj-replicate.4444.killed'; }
+wait_until 60 "run.sh removes the password files of its ADMIN_PORT" shm_cleared
+docker exec dj-shm test -e /dev/shm/opendj-join.5444.other || fail "run.sh removed the password file of another container"
+docker rm -f dj-shm-probe dj-shm >/dev/null
+
+# a join that cannot succeed keeps the container from reporting itself healthy - across a
+# restart too, where the health marker used to follow the upgrade and a failed join turned
+# into a healthy, unreplicated server. dj-x may not seed: it is not the first peer of its list
+start_node dj-x dj-absent,dj-x
+wait_until 300 "the join of dj-x gives up" logs_have dj-x "could not join the replication topology"
+stays_unhealthy dj-x "its join failed"
+docker restart dj-x >/dev/null
+wait_until 300 "the restarted dj-x waits for its join" logs_have dj-x "never joined its replication topology"
+second_failure() { [ "$(logs_count dj-x "could not join the replication topology")" -ge 2 ]; }
+wait_until 300 "the second join of dj-x gives up" second_failure
+stays_unhealthy dj-x "it never joined"
+docker rm -f dj-x >/dev/null
+docker volume rm vol-dj-x >/dev/null
+
+# the seed lost its volume: behind the same name, a fresh dj-0 finds the topology at dj-1 and
+# takes its data from it instead of seeding an empty one next to it
+docker rm -f dj-0 >/dev/null
+docker volume rm vol-dj-0 >/dev/null
+start_node dj-0 dj-0,dj-1
+wait_healthy dj-0
+logs_have dj-0 "initializing from dj-1" || fail "the reborn dj-0 did not initialize from dj-1"
+if logs_have dj-0 "seeding it with this server's data"; then fail "the reborn dj-0 seeded a topology although dj-1 held it"; fi
+ldaps_has dj-0 "ou=replicated,dc=example,dc=com" || fail "the reborn dj-0 lacks the data of the topology"
+ldaps_has dj-0 "uid=user.0,ou=People,dc=example,dc=com" || fail "the reborn dj-0 lacks the entries of the seed"
+
+# a bootstrap that failed after it imported the base entry leaves a volume that never counts
+# as holding the data of the topology: once restarted, dj-bf initializes from dj-0 rather
+# than joining with what its bootstrap left
+printf 'sh /opt/opendj/bootstrap/setup.sh\nexit 1\n' >"$FAILING_BOOTSTRAP"
+chmod 644 "$FAILING_BOOTSTRAP"
+start_node dj-bf dj-0,dj-1,dj-bf -v "$FAILING_BOOTSTRAP:/opt/opendj/failing-bootstrap.sh:ro" \
+ -e BOOTSTRAP=/opt/opendj/failing-bootstrap.sh
+wait_until 300 "the bootstrap of dj-bf fails" logs_have dj-bf "failing-bootstrap.sh failed"
+docker restart dj-bf >/dev/null
+wait_healthy dj-bf
+logs_have dj-bf "initializing from dj-0" || fail "dj-bf joined with the data of its failed bootstrap"
+ldaps_has dj-bf "uid=user.0,ou=People,dc=example,dc=com" || fail "dj-bf lacks the entries only an initialize from dj-0 brings"
+docker rm -f dj-bf >/dev/null
+docker volume rm vol-dj-bf >/dev/null
+
+# an initialize that fails keeps the volume waiting for the data of the topology: a
+# one-second bound cuts the dsreplication JVM off before it connects
+start_node dj-fi dj-0,dj-fi -e REPLICATION_INITIALIZE_TIMEOUT=1 -e REPLICATION_RETRY_COUNT=2
+wait_until 300 "the initialize of dj-fi is cut off" logs_have dj-fi "initialize from dj-0 exited with 124"
+stays_unhealthy dj-fi "its initialize failed"
+docker exec dj-fi test -f /opt/opendj/data/.replication-initialize-pending \
+ || fail "dj-fi dropped its pending marker after a failed initialize"
+docker rm -f dj-fi >/dev/null
+docker volume rm vol-dj-fi >/dev/null
+
+# scale up to three - the peer list is configuration, so the running servers are replaced
+# with the longer list before the third one starts, as a rolling update would
+docker rm -f dj-0 dj-1 >/dev/null
+start_node dj-0 dj-0,dj-1,dj-2
+start_node dj-1 dj-0,dj-1,dj-2
+wait_healthy dj-0
+wait_healthy dj-1
+start_node dj-2 dj-0,dj-1,dj-2
+wait_healthy dj-2
+ldaps_has dj-2 "ou=replicated,dc=example,dc=com" || fail "dj-2 lacks the data of the topology"
+
+# scale down to two: the survivors, restarted with the shorter list, remove dj-2 from
+# cn=admin data and from every replication server list they hold - that of BASE_DN, and
+# those of cn=schema and cn=admin data that dsreplication enable configures next to it.
+# dsreplication disable cannot, dj-2 being already gone. dj-2 keeps its volume, for the
+# scale-up below - and on it, in its cn=config, a join lock held by hand on dj-2 itself
+hold_lock dj-2
+docker rm -f dj-2 >/dev/null
+docker rm -f dj-0 dj-1 >/dev/null
+start_node dj-0 dj-0,dj-1
+start_node dj-1 dj-0,dj-1
+wait_healthy dj-0
+wait_healthy dj-1
+# whichever survivor's join ran first deletes the cn=admin data entry, the delete replicates
+# to the other; each survivor prunes its own replication server lists
+removed_dj2() { logs_have dj-0 "removing departed server dj-2" || logs_have dj-1 "removing departed server dj-2"; }
+wait_until 180 "a survivor removes dj-2" removed_dj2
+for c in dj-0 dj-1; do
+ wait_until 120 "dj-2 leaves cn=admin data of $c" admin_data_lacks $c dj-2
+ wait_until 120 "dj-2 leaves the replication server lists of $c" lists_lack $c dj-2
+done
+# replication between the survivors is intact
+add_ou dj-0 replicated3
+wait_has dj-1 "ou=replicated3,dc=example,dc=com"
+
+# scaled up again on the volume it kept: dj-2 still replicates, but the survivors no longer
+# register it, and an enable between two servers whose cn=admin data is replicated registers
+# nobody - so dj-2 takes its replication configuration down, enables again and registers.
+# The join locks are held by hand meanwhile, and dj-2 must wait for each of them rather than
+# go on: first its own, which it takes before its configuration goes, then those of both
+# survivors, which its enables take. With its replication down it must not report itself
+# ready, publish itself ready to its peers, or take the writes of clients, not even after a
+# restart. Its enables are bounded generously, so that it does not break the held locks as
+# left behind by a killed join - the one on dj-2 is held since the scale-down
+hold_lock dj-0
+hold_lock dj-1
+start_node dj-2 dj-0,dj-1,dj-2 -e REPLICATION_RETRY_COUNT=30 -e REPLICATION_ATTEMPT_TIMEOUT=1800
+wait_until 300 "dj-2 finds that the peers no longer register it" logs_have dj-2 "the peers no longer register this server"
+wait_until 120 "dj-2 waits for its own lock" logs_have dj-2 "dj-held is enabling replication through this server, waiting for it"
+docker exec dj-2 test ! -e /opt/opendj/.bootstrap-complete || fail "dj-2 reports itself ready while it waits to take its replication down"
+[ "$(published_state dj-2)" = rejoining ] || fail "dj-2 does not publish that it is rejoining"
+# two rounds at most: start_node passes REPLICATION_RETRY_INTERVAL=3, and a round waits up to
+# twice as long
+sleep 12
+replicates_example dj-2 || fail "dj-2 took its replication down while its own lock was held"
+drop_lock dj-2
+wait_until 120 "dj-2 waits for dj-0" logs_have dj-2 "dj-held is enabling replication through dj-0, waiting for it"
+wait_until 120 "dj-2 waits for dj-1" logs_have dj-2 "dj-held is enabling replication through dj-1, waiting for it"
+docker exec dj-2 test ! -e /opt/opendj/.bootstrap-complete || fail "dj-2 reports itself ready while its replication is down"
+refuses_writes dj-2 unreplicated || fail "dj-2 takes the writes of clients while its replication is down"
+sleep 12
+admin_data_lacks dj-0 dj-2 || fail "dj-2 enabled while the locks of both peers were held"
+docker restart dj-2 >/dev/null
+wait_until 300 "the restarted dj-2 leaves its health to the join" logs_have dj-2 "was taken out of its replication topology"
+waits_again() { logs_have_since dj-2 "was taken out of its replication topology" "dj-held is enabling replication through dj-0, waiting for it"; }
+wait_until 300 "the restarted dj-2 waits for dj-0 again" waits_again
+docker exec dj-2 test ! -e /opt/opendj/.bootstrap-complete || fail "the restarted dj-2 reports itself ready while its replication is down"
+[ "$(published_state dj-2)" = rejoining ] || fail "the restarted dj-2 publishes itself ready while its replication is down"
+refuses_writes dj-2 unreplicated || fail "the restarted dj-2 takes the writes of clients while its replication is down"
+# its own lock once more, now with both survivors free: an enable configures both of its
+# servers, so it waits for this one as well. The lock is taken as soon as no round of dj-2
+# holds it
+wait_until 120 "the lock of dj-2 is held by hand" hold_lock dj-2
+own_waits=$(logs_count dj-2 "dj-held is enabling replication through this server, waiting for it")
+drop_lock dj-0
+drop_lock dj-1
+own_waits_again() { [ "$(logs_count dj-2 "dj-held is enabling replication through this server, waiting for it")" -gt "$own_waits" ]; }
+wait_until 120 "dj-2 waits for its own lock with both survivors free" own_waits_again
+sleep 12
+admin_data_lacks dj-0 dj-2 || fail "dj-2 enabled while its own lock was held"
+# a lock under its own name was left by a join of an earlier start of dj-2, a start running
+# one join: it is broken at once, not once it is older than an enable may take
+self=$(docker exec dj-2 hostname -f | tr '[:upper:]' '[:lower:]')
+relabel_lock dj-2 "$self"
+wait_until 120 "dj-2 breaks the lock left under its own name" logs_have dj-2 "breaking the lock $self took on localhost"
+wait_healthy dj-2
+registered_dj2() { ! admin_data_lacks dj-0 dj-2; }
+wait_until 300 "dj-2 registers in the topology again" registered_dj2
+[ "$(published_state dj-2)" = ready ] || fail "the rejoined dj-2 does not publish itself ready"
+add_ou dj-2 rejoined
+wait_has dj-0 "ou=rejoined,dc=example,dc=com"
+
+# two servers taken out together and scaled up again together on their volumes: both take
+# their replication down, and with it every registration in their cn=admin data. While the
+# lock of the survivor is held, neither may enable through the other - an enable between two
+# servers without replication registers the two of them only, which passes for membership,
+# and they would serve a topology of their own beside dj-0 - so each waits for dj-0 and joins
+# through it. The operator made the backend of dj-2 read-only beforehand: its rejoin refuses
+# the writes of clients as well, and leaves the mode as it found it. dj-0 keeps the shorter
+# list, nothing here starts it again.
+# A server that comes back shows its peers the state it had before it left - it is kept in
+# its cn=config - until its first round finds that the peers no longer register it. So dj-1
+# starts first and is rejoining before dj-2 starts, and dj-2 keeps a lock held by hand on
+# itself from before it left, as in the scale-up above, which no enable through it gets past
+# while it still shows that state
+dsconfig_on dj-2 set-backend-prop --backend-name userRoot --set writability-mode:internal-only
+hold_lock dj-2
+docker rm -f dj-0 dj-1 dj-2 >/dev/null
+start_node dj-0 dj-0
+wait_healthy dj-0
+for n in dj-1 dj-2; do
+ wait_until 180 "$n leaves cn=admin data of dj-0" admin_data_lacks dj-0 $n
+ wait_until 120 "$n leaves the replication server lists of dj-0" lists_lack dj-0 $n
+done
+hold_lock dj-0
+start_node dj-1 dj-0,dj-1,dj-2 -e REPLICATION_RETRY_COUNT=30 -e REPLICATION_ATTEMPT_TIMEOUT=1800
+dj1_rejoining() { [ "$(published_state dj-1)" = rejoining ]; }
+wait_until 300 "dj-1 publishes that it is rejoining" dj1_rejoining
+start_node dj-2 dj-0,dj-1,dj-2 -e REPLICATION_RETRY_COUNT=30 -e REPLICATION_ATTEMPT_TIMEOUT=1800
+wait_until 300 "dj-1 passes over the rejoining dj-2" logs_have dj-1 "dj-2 is rejoining, not a peer to join through"
+drop_lock dj-2
+wait_until 180 "dj-2 passes over the rejoining dj-1" logs_have dj-2 "dj-1 is rejoining, not a peer to join through"
+sleep 12
+for n in dj-1 dj-2; do
+ docker exec $n test ! -e /opt/opendj/.bootstrap-complete || fail "$n reports itself ready while its replication is down"
+done
+admin_data_lacks dj-2 dj-1 || fail "dj-1 and dj-2 registered each other while the lock of dj-0 was held"
+admin_data_lacks dj-1 dj-2 || fail "dj-1 and dj-2 registered each other while the lock of dj-0 was held"
+refuses_writes dj-1 unreplicated || fail "dj-1 takes the writes of clients while its replication is down"
+drop_lock dj-0
+# the health status says nothing yet: dj-1 and dj-2 started on volumes without a reset
+# pending, so their run.sh reported them ready, a probe may have found them healthy before
+# their joins took the replication down, and the status turns unhealthy only after the
+# probe's retries. Their joins publish them ready once the writes are let in again
+for n in dj-1 dj-2; do
+ wait_until 480 "$n rejoins and publishes itself ready" published_ready $n
+done
+wait_healthy dj-1
+wait_healthy dj-2
+for n in dj-1 dj-2; do
+ wait_until 300 "$n registers with dj-0 again" registered_in_dj0 $n
+done
+[ "$(backend_mode dj-2)" = internal-only ] || fail "the rejoin of dj-2 let the writes of clients in on a backend the operator made read-only"
+add_ou dj-1 rejoined-together
+wait_has dj-0 "ou=rejoined-together,dc=example,dc=com"
+wait_has dj-2 "ou=rejoined-together,dc=example,dc=com"
+
+# an enable that stopped half-way - its initialize of cn=admin data failed - leaves BASE_DN
+# replicated and the server registered at its peers, but not in its own cn=admin data, and
+# every enable after that exits 5, "already replicated", without registering it there. So a
+# server in that state takes its replication down and enables anew, as one that no peer
+# registers does. dj-1 lacks its own entry, then dj-2 the whole of cn=Servers, which the
+# failed initialize of the CI run behind this case may have left; neither delete reaches a
+# peer. One at a time: the disable of a reset changes the peers it replicates with as well
+half_enabled_rejoins() { # <container>
+ local n=$1 members
+ admin_data_lacks $n $n || fail "$n is still registered in its own cn=admin data"
+ registered_in_dj0 $n || fail "the delete on $n alone reached dj-0"
+ members=$(logs_count $n "the health check may probe it")
+ docker restart $n >/dev/null
+ wait_until 300 "$n takes its half-enabled replication down" logs_have $n "the peers register this server but its own cn=admin data does not"
+ member_again() { [ "$(logs_count $n "the health check may probe it")" -gt "$members" ]; }
+ wait_until 600 "$n rejoins" member_again
+ published_ready $n || fail "$n does not publish itself ready after it rejoined"
+ ! admin_data_lacks $n $n || fail "$n is not registered in its own cn=admin data after it rejoined"
+ registered_in_dj0 $n || fail "$n is not registered with dj-0 after it rejoined"
+ wait_healthy $n
+}
+own_entry=$(admin_search dj-1 "cn=Servers,cn=admin data" one "(hostname=dj-1)" 1.1 | awk '/^dn: / { print substr($0, 5) }')
+[ -n "$own_entry" ] || fail "dj-1 does not register itself in its cn=admin data"
+delete_here dj-1 "$own_entry" || fail "could not delete $own_entry on dj-1 alone"
+half_enabled_rejoins dj-1
+delete_here dj-2 "cn=Servers,cn=admin data" --deleteSubtree || fail "could not delete cn=Servers on dj-2 alone"
+half_enabled_rejoins dj-2
+add_ou dj-1 rejoined-half-enabled
+wait_has dj-0 "ou=rejoined-half-enabled,dc=example,dc=com"
+wait_has dj-2 "ou=rejoined-half-enabled,dc=example,dc=com"
+
+# the deprecated one-shot sdsr path of replicate.sh still bootstraps a replica. On a first
+# start a server stops the server its bootstrap started and starts it again, and a replica
+# can reach it in between: replicate.sh tries a dsreplication enable that could not connect
+# again (exit 8). The replica starts on a network of its own, where dj-0 does not resolve, and
+# moves to the network of the topology only once it has said it will try again: taking dj-0
+# off the network for as long instead would leave its replication server holding the dead
+# connection of dj-0's own directory server, and route the initialize of the replica into it
+docker network create $NETWORK_ALONE >/dev/null
+docker run -d --memory="512m" --network $NETWORK_ALONE --name dj-sdsr --hostname dj-sdsr \
+ -e ROOT_PASSWORD="$ROOT_PASSWORD" -e MASTER_SERVER=dj-0 -e OPENDJ_REPLICATION_TYPE=sdsr "$IMAGE" >/dev/null
+wait_until 420 "dj-sdsr tries again" logs_have dj-sdsr "exited with 8, trying again"
+docker network connect $NETWORK dj-sdsr
+docker network disconnect $NETWORK_ALONE dj-sdsr
+wait_healthy dj-sdsr
+wait_has dj-sdsr "ou=replicated3,dc=example,dc=com"
+# dj-sdsr publishes no state, as a server of an image before this one does, and it
+# replicates BASE_DN, so it may hold the data of the topology: a fresh first peer whose only
+# other peer it is gives up rather than seed next to it
+start_node dj-f dj-f,dj-sdsr -e REPLICATION_RETRY_COUNT=2 -e REPLICATION_RETRY_INTERVAL=5
+wait_until 300 "dj-f gives up" logs_have dj-f "could not join the replication topology after 2 attempts"
+if logs_have dj-f "seeding it with this server's data"; then fail "dj-f seeded next to a member that publishes no state"; fi
+stays_unhealthy dj-f "a peer that may hold the data answers"
+docker rm -f dj-f >/dev/null
+docker volume rm vol-dj-f >/dev/null
+# replicate.sh tries dsreplication enable again only when it exits 8, and a failed enable
+# ends it: run once more on the replica for a base DN neither server holds, the enable exits
+# 5 (REPLICATION_CANNOT_BE_ENABLED_ON_BASEDN) and nothing is tried again or initialized
+rc=0
+out=$(docker exec -e BASE_DN=dc=absent,dc=com -e ROOT_USER_DN="cn=Directory Manager" dj-sdsr \
+ timeout 90 /opt/opendj/bootstrap/replicate.sh 2>&1) || rc=$?
+if [ "$rc" -ne 5 ] || grep -E "trying again|initializing replication" <<<"$out" >/dev/null; then
+ echo "$out"
+ fail "replicate.sh for a base DN nobody holds exited with $rc, not with the 5 of its dsreplication enable, or went on after it"
+fi
+
+# the root password shows in no container log, and the files the tools read it from are gone
+for c in dj-0 dj-1 dj-2 dj-sdsr; do
+ if docker logs $c 2>&1 | grep -F -- "$ROOT_PASSWORD"; then fail "the root password is in the log of $c"; fi
+ left=$(docker exec $c grep -rlsF -- "$ROOT_PASSWORD" /tmp /dev/shm || true)
+ [ -z "$left" ] || fail "the root password is left in $left of $c"
+done
+docker rm -f dj-0 dj-1 dj-2 dj-sdsr >/dev/null
+docker volume rm vol-dj-0 vol-dj-1 vol-dj-2 >/dev/null
+
+# MASTER_SERVER may name the master by its address, as the grep of /etc/hosts in the base
+# replicate.sh allowed: a master given its own address recognises itself and seeds at once,
+# and a replica joins and initializes through that address
+for n in dj-ipm dj-ipr; do
+ if [ $n = dj-ipm ]; then extra="--ip $MASTER_ADDRESS -e SAMPLE_DATA=10"; else extra="-v vol-dj-ipr:/opt/opendj/data"; fi
+ docker run -d --memory="512m" --network $NETWORK --name $n --hostname $n $extra \
+ -e ROOT_PASSWORD="$ROOT_PASSWORD" -e ADD_BASE_ENTRY="--addBaseEntry" \
+ -e OPENDJ_REPLICATION_TYPE=simple -e MASTER_SERVER=$MASTER_ADDRESS \
+ -e REPLICATION_RETRY_COUNT=10 -e REPLICATION_RETRY_INTERVAL=3 "$IMAGE" >/dev/null
+ wait_healthy $n
+done
+logs_have dj-ipm "seeding it with this server's data" || fail "dj-ipm did not recognise itself by its address"
+logs_have dj-ipr "initializing from $MASTER_ADDRESS" || fail "dj-ipr did not initialize from the master's address"
+ldaps_has dj-ipr "uid=user.0,ou=People,dc=example,dc=com" || fail "dj-ipr lacks the entries of the master"
+# moved from MASTER_SERVER to a REPLICATION_PEERS of names, the replica keeps the master that
+# is registered by its address - in cn=admin data and in its replication server lists -
+# rather than taking it for a server that left the topology
+docker rm -f dj-ipr >/dev/null
+docker run -d --memory="512m" --network $NETWORK --name dj-ipr --hostname dj-ipr -v vol-dj-ipr:/opt/opendj/data \
+ -e ROOT_PASSWORD="$ROOT_PASSWORD" -e OPENDJ_REPLICATION_TYPE=simple -e REPLICATION_PEERS=dj-ipm,dj-ipr \
+ -e REPLICATION_RETRY_COUNT=2 -e REPLICATION_RETRY_INTERVAL=3 "$IMAGE" >/dev/null
+wait_until 300 "the moved dj-ipr has looked for departed servers" logs_have dj-ipr "the health check may probe it"
+if logs_have dj-ipr "removing departed server $MASTER_ADDRESS" || logs_have dj-ipr "removing departed replication server $MASTER_ADDRESS:"; then
+ fail "dj-ipr removed the master registered by its address"
+fi
+if admin_data_lacks dj-ipr "$MASTER_ADDRESS"; then fail "the master registered by its address left cn=admin data of dj-ipr"; fi
+docker rm -f dj-ipm dj-ipr >/dev/null
+docker volume rm vol-dj-ipr >/dev/null
+
+# three fresh servers started together, as Compose or a StatefulSet with
+# podManagementPolicy: Parallel start them: none of them takes the bootstrap data of another
+# fresh one for the topology's. The first peer seeds, the others initialize from it - its
+# sample entries reach them only that way - and the replication servers know each other
+# although the joins ran at the same time
+for n in dj-p0 dj-p1 dj-p2; do
+ if [ $n = dj-p0 ]; then extra="-e SAMPLE_DATA=10"; else extra=; fi
+ start_node $n dj-p0,dj-p1,dj-p2 -e REPLICATION_RETRY_COUNT=20 $extra
+done
+for n in dj-p0 dj-p1 dj-p2; do
+ wait_healthy $n
+done
+logs_have dj-p0 "seeding it with this server's data" || fail "dj-p0 did not seed the topology"
+for n in dj-p1 dj-p2; do
+ if logs_have $n "seeding it with this server's data"; then fail "$n seeded a topology although it is not the first peer"; fi
+ logs_have $n "initializing from dj-p0" || fail "$n did not initialize from dj-p0"
+ if logs_have $n "initializing from dj-p1" || logs_have $n "initializing from dj-p2"; then
+ fail "$n initialized from a server that holds only bootstrap data"
+ fi
+ ldaps_has $n "uid=user.0,ou=People,dc=example,dc=com" || fail "$n lacks the entries of the seed"
+done
+wait_until 120 "the replication server of dj-p0 knows dj-p1 and dj-p2" server_lists dj-p0 dj-p1 dj-p2
+wait_until 120 "the replication server of dj-p1 knows dj-p0 and dj-p2" server_lists dj-p1 dj-p0 dj-p2
+wait_until 120 "the replication server of dj-p2 knows dj-p0 and dj-p1" server_lists dj-p2 dj-p0 dj-p1
+add_ou dj-p1 parallel
+wait_has dj-p0 "ou=parallel,dc=example,dc=com"
+wait_has dj-p2 "ou=parallel,dc=example,dc=com"
+# dj-p1 and dj-p2 enabled through dj-p0 one at a time, and each removed its lock after
+if admin_search dj-p0 "cn=Docker Join Lock,cn=config" base "(objectClass=*)" 1.1 | grep "^dn:" >/dev/null; then
+ fail "a join left its lock on dj-p0"
+fi
+for c in dj-p0 dj-p1 dj-p2; do
+ if docker logs $c 2>&1 | grep -F -- "$ROOT_PASSWORD"; then fail "the root password is in the log of $c"; fi
+done
+
+cleanup
+echo "Docker replication test passed"
diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml
index ecab9ef..5932fb9 100644
--- a/.github/workflows/build.yml
+++ b/.github/workflows/build.yml
@@ -478,9 +478,12 @@
ports:
- 5000:5000
steps:
+ # .github/scripts holds the replication test that both image jobs run
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
- sparse-checkout: .github/benchmark
+ sparse-checkout: |
+ .github/benchmark
+ .github/scripts
- name: Download artifacts
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
with:
@@ -591,75 +594,10 @@
docker exec test_uid 'sh' '-c' '/opt/opendj/bin/ldapsearch --hostname localhost --port 1636 --bindDN "cn=Directory Manager" --bindPassword password --useSsl --trustAll --baseDN "dc=example,dc=com" --searchScope base "(objectClass=*)" 1.1'
docker kill test_uid
- name: Docker test replication
+ # the background join of OPENDJ_REPLICATION_TYPE=simple and the one-shot sdsr path, the same
+ # scenarios for both images
shell: bash
- run: |
- IMAGE=localhost:5000/${GITHUB_REPOSITORY,,}:${{ env.release_version }}
- REPLICAS="test_replica test_replica_sdsr"
- cleanup() { docker rm -f test_master $REPLICAS >/dev/null 2>&1 || true; docker network rm test_replication >/dev/null 2>&1 || true; }
- cleanup
- trap 'code=$?; for c in test_master $REPLICAS; do echo "::group::container logs ($c)"; docker logs $c 2>&1 || true; echo "::endgroup::"; done; cleanup; exit $code' ERR
- # every tool reads the root password from a file (#1084, #1092); dsreplication run with -n prints
- # no command line, so a password put back on one would pass every check below
- rc=0; docker run --rm --entrypoint grep "$IMAGE" -nE -- '(^|[[:space:]])(-w|--(bindPassword[12]?|adminPassword|rootUserPassword))([[:space:]=]|$)' /opt/opendj/bootstrap/setup.sh /opt/opendj/bootstrap/replicate.sh || rc=$?
- if [ $rc -ne 1 ]; then echo "::error::setup.sh or replicate.sh passes the root password on a command line, or grep could not read them"; false; fi
- # the password file goes to /dev/shm, off the writable layer of the container, and the mktemp of the image puts it there
- docker run --rm --entrypoint grep "$IMAGE" -qF -- 'mktemp -p /dev/shm "opendj-replicate.$ADMIN_PORT.' /opt/opendj/bootstrap/replicate.sh || { echo "::error::replicate.sh no longer puts the password file on /dev/shm"; false; }
- docker run --rm --entrypoint sh "$IMAGE" -c 'f=$(mktemp -p /dev/shm "opendj-replicate.$ADMIN_PORT.XXXXXX") && rm -f "$f" && case $f in /dev/shm/opendj-replicate.4444.*) ;; *) exit 1;; esac' || { echo "::error::mktemp in the image does not create the password file on /dev/shm"; false; }
- # a password with a space in it reaches every tool as one value
- ROOT_PASSWORD='replication secret'
- docker network create test_replication
- docker run --rm -it -d --memory="512m" --network test_replication --ipc=shareable --name=test_master --hostname=dj-master -e ADD_BASE_ENTRY="--addBaseEntry" -e ROOT_PASSWORD="$ROOT_PASSWORD" "$IMAGE"
- timeout 3m bash -c 'until docker inspect --format="{{json .State.Health.Status}}" test_master | grep -q \"healthy\"; do sleep 10; done'
- # a replica reports itself healthy only once replicate.sh has succeeded; the sdsr replica joins after
- # the simple one, as two dsreplication enable at once would both rewrite the admin data of the master
- # a Kubernetes pod keeps its /dev/shm across container restarts: the replica shares the /dev/shm of the master, where
- # a password file waits as a killed replicate.sh would have left it, and its run.sh has to remove it (checked below);
- # the file of another container of the pod, which listens on another admin port, has to be kept
- docker exec test_master sh -c 'printf "%s\n" "$ROOT_PASSWORD" >/dev/shm/opendj-replicate.4444.killed'
- docker exec test_master sh -c ': >/dev/shm/opendj-replicate.5444.other'
- docker run --rm -it -d --memory="512m" --network test_replication --ipc=container:test_master --name=test_replica --hostname=dj-replica -e ROOT_PASSWORD="$ROOT_PASSWORD" -e MASTER_SERVER=dj-master -e OPENDJ_REPLICATION_TYPE=simple "$IMAGE"
- # on a first start the master stops the server its bootstrap started and starts it again, and a replica
- # started together with it can reach it in between: replicate.sh tries a dsreplication that could not
- # connect again. The master is taken off the network while the replica sleeps before its first try, and
- # comes back only once the replica has said it will try again
- timeout 5m bash -c 'until docker logs test_replica 2>&1 | grep -q "Will sleep for a bit"; do sleep 0.2; done'
- docker network disconnect test_replication test_master
- timeout 2m bash -c 'until docker logs test_replica 2>&1 | grep -q "exited with 8, trying again"; do sleep 1; done'
- docker network connect --alias dj-master test_replication test_master
- timeout 5m bash -c 'until docker inspect --format="{{json .State.Health.Status}}" test_replica | grep -q \"healthy\"; do sleep 10; done'
- docker run --rm -it -d --memory="512m" --network test_replication --name=test_replica_sdsr --hostname=dj-replica-sdsr -e ROOT_PASSWORD="$ROOT_PASSWORD" -e MASTER_SERVER=dj-master -e OPENDJ_REPLICATION_TYPE=sdsr "$IMAGE"
- # the same for the sdsr replica, whose dsreplication enable is another command of replicate.sh
- timeout 5m bash -c 'until docker logs test_replica_sdsr 2>&1 | grep -q "Will sleep for a bit"; do sleep 0.2; done'
- docker network disconnect test_replication test_master
- timeout 2m bash -c 'until docker logs test_replica_sdsr 2>&1 | grep -q "exited with 8, trying again"; do sleep 1; done'
- docker network connect --alias dj-master test_replication test_master
- timeout 5m bash -c 'until docker inspect --format="{{json .State.Health.Status}}" test_replica_sdsr | grep -q \"healthy\"; do sleep 10; done'
- # the replicas were initialized from the master, and a change made on the master reaches them
- for c in $REPLICAS; do
- docker exec $c /opt/opendj/bin/ldapsearch --hostname localhost --port 1636 --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll --baseDN "dc=example,dc=com" --searchScope base "(objectClass=*)" 1.1
- done
- printf 'dn: ou=replicated,dc=example,dc=com\nobjectClass: organizationalUnit\nou: replicated\n' | docker exec -i test_master /opt/opendj/bin/ldapmodify --hostname localhost --port 1636 --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll --defaultAdd
- for c in $REPLICAS; do
- timeout 1m bash -c 'until docker exec $1 /opt/opendj/bin/ldapsearch --hostname localhost --port 1636 --bindDN "cn=Directory Manager" --bindPassword "$0" --useSsl --trustAll --baseDN "ou=replicated,dc=example,dc=com" --searchScope base "(objectClass=*)" 1.1; do sleep 5; done' "$ROOT_PASSWORD" $c
- done
- # replicate.sh tries dsreplication enable again only when it exits 8, and a failed enable ends it: run once more
- # on the replica, the enable of a base DN already replicated exits 5 and nothing is tried again or initialized
- rc=0; out=$(docker exec -e BASE_DN=dc=example,dc=com -e ROOT_USER_DN="cn=Directory Manager" test_replica timeout 90 /opt/opendj/bootstrap/replicate.sh 2>&1) || rc=$?
- if [ $rc -ne 5 ] || grep -qE "trying again|initializing replication" <<<"$out"; then
- echo "$out"; echo "::error::a second replicate.sh exited with $rc, not with the 5 of its dsreplication enable, or went on after it"; false
- fi
- # the root password shows in no container log, and the files setup.sh and replicate.sh passed it in are gone (#1084, #1092)
- for c in test_master $REPLICAS; do
- if docker logs $c 2>&1 | grep -F "$ROOT_PASSWORD"; then echo "::error::The root password is in the log of $c"; false; fi
- done
- for c in test_master $REPLICAS; do
- # a JVM keeps its command line in /tmp/hsperfdata_* while it runs; the HEALTHCHECK no longer binds as root (#1092),
- # so no process left running has the root password on it
- left=$(docker exec $c grep -rlsF -- "$ROOT_PASSWORD" /tmp /dev/shm || true)
- if [ -n "$left" ]; then echo "::error::The root password is left in $left of $c"; false; fi
- done
- docker exec test_replica test -e /dev/shm/opendj-replicate.5444.other || { echo "::error::run.sh of test_replica removed the password file of another container"; false; }
- cleanup
+ run: .github/scripts/docker-test-replication.sh "localhost:5000/${GITHUB_REPOSITORY,,}:${{ env.release_version }}"
- name: Docker test secret volume
# a keystore mounted at SECRET_VOLUME is what LDAPS serves from the first start on, a
# renewed one - a new password included - is copied while the server runs and served
@@ -933,9 +871,12 @@
ports:
- 5000:5000
steps:
+ # .github/scripts holds the replication test that both image jobs run
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
- sparse-checkout: .github/benchmark
+ sparse-checkout: |
+ .github/benchmark
+ .github/scripts
- name: Download artifacts
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
with:
@@ -1047,75 +988,10 @@
docker exec test_uid 'sh' '-c' '/opt/opendj/bin/ldapsearch --hostname localhost --port 1636 --bindDN "cn=Directory Manager" --bindPassword password --useSsl --trustAll --baseDN "dc=example,dc=com" --searchScope base "(objectClass=*)" 1.1'
docker kill test_uid
- name: Docker test replication
+ # the background join of OPENDJ_REPLICATION_TYPE=simple and the one-shot sdsr path, the same
+ # scenarios for both images
shell: bash
- run: |
- IMAGE=localhost:5000/${GITHUB_REPOSITORY,,}:${{ env.release_version }}-alpine
- REPLICAS="test_replica test_replica_sdsr"
- cleanup() { docker rm -f test_master $REPLICAS >/dev/null 2>&1 || true; docker network rm test_replication >/dev/null 2>&1 || true; }
- cleanup
- trap 'code=$?; for c in test_master $REPLICAS; do echo "::group::container logs ($c)"; docker logs $c 2>&1 || true; echo "::endgroup::"; done; cleanup; exit $code' ERR
- # every tool reads the root password from a file (#1084, #1092); dsreplication run with -n prints
- # no command line, so a password put back on one would pass every check below
- rc=0; docker run --rm --entrypoint grep "$IMAGE" -nE -- '(^|[[:space:]])(-w|--(bindPassword[12]?|adminPassword|rootUserPassword))([[:space:]=]|$)' /opt/opendj/bootstrap/setup.sh /opt/opendj/bootstrap/replicate.sh || rc=$?
- if [ $rc -ne 1 ]; then echo "::error::setup.sh or replicate.sh passes the root password on a command line, or grep could not read them"; false; fi
- # the password file goes to /dev/shm, off the writable layer of the container, and the mktemp of the image puts it there
- docker run --rm --entrypoint grep "$IMAGE" -qF -- 'mktemp -p /dev/shm "opendj-replicate.$ADMIN_PORT.' /opt/opendj/bootstrap/replicate.sh || { echo "::error::replicate.sh no longer puts the password file on /dev/shm"; false; }
- docker run --rm --entrypoint sh "$IMAGE" -c 'f=$(mktemp -p /dev/shm "opendj-replicate.$ADMIN_PORT.XXXXXX") && rm -f "$f" && case $f in /dev/shm/opendj-replicate.4444.*) ;; *) exit 1;; esac' || { echo "::error::mktemp in the image does not create the password file on /dev/shm"; false; }
- # a password with a space in it reaches every tool as one value
- ROOT_PASSWORD='replication secret'
- docker network create test_replication
- docker run --rm -it -d --memory="1g" --network test_replication --ipc=shareable --name=test_master --hostname=dj-master -e ADD_BASE_ENTRY="--addBaseEntry" -e ROOT_PASSWORD="$ROOT_PASSWORD" "$IMAGE"
- timeout 3m bash -c 'until docker inspect --format="{{json .State.Health.Status}}" test_master | grep -q \"healthy\"; do sleep 10; done'
- # a replica reports itself healthy only once replicate.sh has succeeded; the sdsr replica joins after
- # the simple one, as two dsreplication enable at once would both rewrite the admin data of the master
- # a Kubernetes pod keeps its /dev/shm across container restarts: the replica shares the /dev/shm of the master, where
- # a password file waits as a killed replicate.sh would have left it, and its run.sh has to remove it (checked below);
- # the file of another container of the pod, which listens on another admin port, has to be kept
- docker exec test_master sh -c 'printf "%s\n" "$ROOT_PASSWORD" >/dev/shm/opendj-replicate.4444.killed'
- docker exec test_master sh -c ': >/dev/shm/opendj-replicate.5444.other'
- docker run --rm -it -d --memory="1g" --network test_replication --ipc=container:test_master --name=test_replica --hostname=dj-replica -e ROOT_PASSWORD="$ROOT_PASSWORD" -e MASTER_SERVER=dj-master -e OPENDJ_REPLICATION_TYPE=simple "$IMAGE"
- # on a first start the master stops the server its bootstrap started and starts it again, and a replica
- # started together with it can reach it in between: replicate.sh tries a dsreplication that could not
- # connect again. The master is taken off the network while the replica sleeps before its first try, and
- # comes back only once the replica has said it will try again
- timeout 5m bash -c 'until docker logs test_replica 2>&1 | grep -q "Will sleep for a bit"; do sleep 0.2; done'
- docker network disconnect test_replication test_master
- timeout 2m bash -c 'until docker logs test_replica 2>&1 | grep -q "exited with 8, trying again"; do sleep 1; done'
- docker network connect --alias dj-master test_replication test_master
- timeout 5m bash -c 'until docker inspect --format="{{json .State.Health.Status}}" test_replica | grep -q \"healthy\"; do sleep 10; done'
- docker run --rm -it -d --memory="1g" --network test_replication --name=test_replica_sdsr --hostname=dj-replica-sdsr -e ROOT_PASSWORD="$ROOT_PASSWORD" -e MASTER_SERVER=dj-master -e OPENDJ_REPLICATION_TYPE=sdsr "$IMAGE"
- # the same for the sdsr replica, whose dsreplication enable is another command of replicate.sh
- timeout 5m bash -c 'until docker logs test_replica_sdsr 2>&1 | grep -q "Will sleep for a bit"; do sleep 0.2; done'
- docker network disconnect test_replication test_master
- timeout 2m bash -c 'until docker logs test_replica_sdsr 2>&1 | grep -q "exited with 8, trying again"; do sleep 1; done'
- docker network connect --alias dj-master test_replication test_master
- timeout 5m bash -c 'until docker inspect --format="{{json .State.Health.Status}}" test_replica_sdsr | grep -q \"healthy\"; do sleep 10; done'
- # the replicas were initialized from the master, and a change made on the master reaches them
- for c in $REPLICAS; do
- docker exec $c /opt/opendj/bin/ldapsearch --hostname localhost --port 1636 --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll --baseDN "dc=example,dc=com" --searchScope base "(objectClass=*)" 1.1
- done
- printf 'dn: ou=replicated,dc=example,dc=com\nobjectClass: organizationalUnit\nou: replicated\n' | docker exec -i test_master /opt/opendj/bin/ldapmodify --hostname localhost --port 1636 --bindDN "cn=Directory Manager" --bindPassword "$ROOT_PASSWORD" --useSsl --trustAll --defaultAdd
- for c in $REPLICAS; do
- timeout 1m bash -c 'until docker exec $1 /opt/opendj/bin/ldapsearch --hostname localhost --port 1636 --bindDN "cn=Directory Manager" --bindPassword "$0" --useSsl --trustAll --baseDN "ou=replicated,dc=example,dc=com" --searchScope base "(objectClass=*)" 1.1; do sleep 5; done' "$ROOT_PASSWORD" $c
- done
- # replicate.sh tries dsreplication enable again only when it exits 8, and a failed enable ends it: run once more
- # on the replica, the enable of a base DN already replicated exits 5 and nothing is tried again or initialized
- rc=0; out=$(docker exec -e BASE_DN=dc=example,dc=com -e ROOT_USER_DN="cn=Directory Manager" test_replica timeout 90 /opt/opendj/bootstrap/replicate.sh 2>&1) || rc=$?
- if [ $rc -ne 5 ] || grep -qE "trying again|initializing replication" <<<"$out"; then
- echo "$out"; echo "::error::a second replicate.sh exited with $rc, not with the 5 of its dsreplication enable, or went on after it"; false
- fi
- # the root password shows in no container log, and the files setup.sh and replicate.sh passed it in are gone (#1084, #1092)
- for c in test_master $REPLICAS; do
- if docker logs $c 2>&1 | grep -F "$ROOT_PASSWORD"; then echo "::error::The root password is in the log of $c"; false; fi
- done
- for c in test_master $REPLICAS; do
- # a JVM keeps its command line in /tmp/hsperfdata_* while it runs; the HEALTHCHECK no longer binds as root (#1092),
- # so no process left running has the root password on it
- left=$(docker exec $c grep -rlsF -- "$ROOT_PASSWORD" /tmp /dev/shm || true)
- if [ -n "$left" ]; then echo "::error::The root password is left in $left of $c"; false; fi
- done
- docker exec test_replica test -e /dev/shm/opendj-replicate.5444.other || { echo "::error::run.sh of test_replica removed the password file of another container"; false; }
- cleanup
+ run: .github/scripts/docker-test-replication.sh "localhost:5000/${GITHUB_REPOSITORY,,}:${{ env.release_version }}-alpine"
- name: Docker test secret volume
# a keystore mounted at SECRET_VOLUME is what LDAPS serves from the first start on, a
# renewed one - a new password included - is copied while the server runs and served
diff --git a/opendj-packages/opendj-docker/Dockerfile b/opendj-packages/opendj-docker/Dockerfile
index 3ffaac8..0ebaf0a 100644
--- a/opendj-packages/opendj-docker/Dockerfile
+++ b/opendj-packages/opendj-docker/Dockerfile
@@ -29,6 +29,14 @@
ENV OPENDJ_SSL_OPTIONS="--generateSelfSignedCertificate"
#ENV MASTER_SERVER
#ENV OPENDJ_REPLICATION_TYPE
+# every server of the replication topology, comma separated; the first entry may seed a new
+# topology. MASTER_SERVER keeps working as a one-element list. See bootstrap/join.sh
+#ENV REPLICATION_PEERS
+#ENV REPLICATION_PORT=8989
+#ENV REPLICATION_RETRY_COUNT=30
+#ENV REPLICATION_RETRY_INTERVAL=10
+#ENV REPLICATION_ATTEMPT_TIMEOUT=120
+#ENV REPLICATION_INITIALIZE_TIMEOUT=0
ENV OPENDJ_USER="opendj"
#ENV OPENDJ_JAVA_ARGS=""
ENV BACKEND_TYPE="je"
@@ -68,7 +76,7 @@
COPY --chown=$OPENDJ_USER:0 run.sh /opt/opendj/run.sh
COPY --chown=$OPENDJ_USER:0 healthcheck.sh /opt/opendj/healthcheck.sh
-RUN chmod +x /opt/opendj/run.sh /opt/opendj/healthcheck.sh /opt/opendj/bootstrap/setup.sh /opt/opendj/bootstrap/replicate.sh
+RUN chmod +x /opt/opendj/run.sh /opt/opendj/healthcheck.sh /opt/opendj/bootstrap/setup.sh /opt/opendj/bootstrap/replicate.sh /opt/opendj/bootstrap/join.sh
EXPOSE $PORT/tcp $LDAPS_PORT/tcp $ADMIN_PORT/tcp
diff --git a/opendj-packages/opendj-docker/Dockerfile-alpine b/opendj-packages/opendj-docker/Dockerfile-alpine
index 55658cb..fba79d4 100644
--- a/opendj-packages/opendj-docker/Dockerfile-alpine
+++ b/opendj-packages/opendj-docker/Dockerfile-alpine
@@ -29,6 +29,14 @@
ENV OPENDJ_SSL_OPTIONS="--generateSelfSignedCertificate"
#ENV MASTER_SERVER
#ENV OPENDJ_REPLICATION_TYPE
+# every server of the replication topology, comma separated; the first entry may seed a new
+# topology. MASTER_SERVER keeps working as a one-element list. See bootstrap/join.sh
+#ENV REPLICATION_PEERS
+#ENV REPLICATION_PORT=8989
+#ENV REPLICATION_RETRY_COUNT=30
+#ENV REPLICATION_RETRY_INTERVAL=10
+#ENV REPLICATION_ATTEMPT_TIMEOUT=120
+#ENV REPLICATION_INITIALIZE_TIMEOUT=0
ENV OPENDJ_USER="opendj"
#ENV OPENDJ_JAVA_ARGS=""
ENV BACKEND_TYPE="je"
@@ -50,7 +58,7 @@
RUN apk add --update --no-cache --virtual builddeps curl unzip \
&& apk upgrade --update --no-cache \
&& if [ "$TARGETARCH" = "386" ]; then JDK=openjdk11-jre; else JDK=openjdk25-jre; fi \
- && apk add bash "$JDK" \
+ && apk add bash coreutils "$JDK" \
&& if [ -z "$VERSION" ] ; then VERSION="$(curl -i -o - --silent https://api.github.com/repos/OpenIdentityPlatform/OpenDJ/releases/latest | grep -m1 "\"name\"" | cut -d\" -f4)"; fi \
&& if [ ! -f "$OPENDJ_DIST_FILENAME" ]; then echo file exists && curl -L https://github.com/OpenIdentityPlatform/OpenDJ/releases/download/$VERSION/opendj-$VERSION.zip --output $OPENDJ_DIST_FILENAME; fi \
&& unzip $OPENDJ_DIST_FILENAME \
@@ -72,7 +80,7 @@
COPY --chown=$OPENDJ_USER:0 run.sh /opt/opendj/run.sh
COPY --chown=$OPENDJ_USER:0 healthcheck.sh /opt/opendj/healthcheck.sh
-RUN chmod +x /opt/opendj/run.sh /opt/opendj/healthcheck.sh /opt/opendj/bootstrap/setup.sh /opt/opendj/bootstrap/replicate.sh
+RUN chmod +x /opt/opendj/run.sh /opt/opendj/healthcheck.sh /opt/opendj/bootstrap/setup.sh /opt/opendj/bootstrap/replicate.sh /opt/opendj/bootstrap/join.sh
EXPOSE $PORT/tcp $LDAPS_PORT/tcp $ADMIN_PORT/tcp
diff --git a/opendj-packages/opendj-docker/README.md b/opendj-packages/opendj-docker/README.md
index dcc0f25..e4ceac5 100644
--- a/opendj-packages/opendj-docker/README.md
+++ b/opendj-packages/opendj-docker/README.md
@@ -16,9 +16,10 @@
The image reports itself `healthy` once the server answers on `LDAPS_PORT` *and* the whole
bootstrap has succeeded - the instance, the `userRoot` backend over `BASE_DN`, whatever
-`ADD_BASE_ENTRY` and `SAMPLE_DATA` asked to be imported into it, and the replication asked
-for by `MASTER_SERVER`. Waiting for that status is therefore enough before the first search
-of what the bootstrap was told to create:
+`ADD_BASE_ENTRY` and `SAMPLE_DATA` asked to be imported into it, and, where replication is
+asked for, the join of the replication topology (see [Replication](#replication)). Waiting
+for that status is therefore enough before the first search of what the bootstrap was told
+to create:
```bash
docker run -d --name opendj -e ADD_BASE_ENTRY=--addBaseEntry openidentityplatform/opendj
@@ -52,14 +53,7 @@
The server answering is not enough on a first start: the bootstrap starts the server, and
once it is done that server is stopped and started again in the foreground, so a client
-that only waits for the port can have its first requests fail in between. A replica set up
-with `MASTER_SERVER` tries a master it cannot connect to again, every 10 s for up to 5
-minutes, so a master that is down when the replica reaches it does not fail its replication
-setup; one that stops while `dsreplication enable` is writing to it still does.
-
-With `OPENDJ_REPLICATION_TYPE=srs`, start the directory server replicas one at a time, each
-once the previous one is healthy: every replica pushes its data to the replicas connected at
-that moment, and one that is restarting at the end of its own first start misses it.
+that only waits for the port can have its first requests fail in between.
A bootstrap that imports `SAMPLE_DATA` can take minutes on a small container, which is what
the start period allows for. A bootstrap that fails - or an upgrade that fails when starting
@@ -72,6 +66,122 @@
with `docker run --init` (`init: true` in Compose) to put a PID 1 in front of the server
that reaps them and passes SIGTERM on to it.
+## Replication
+
+With `OPENDJ_REPLICATION_TYPE=simple`, the container joins its replication topology in the
+background, next to the running server, on every start - not as a step of the first
+bootstrap, so a join that could not complete is tried again on the next start, and
+membership that changed while no container ran is repaired. `REPLICATION_PEERS` lists every
+server of the topology by DNS name, comma separated; all of them share `BASE_DN`,
+`ROOT_USER_DN`, `ROOT_PASSWORD`, `ADMIN_PORT` and `REPLICATION_PORT`. A Kubernetes
+StatefulSet derives the list from its ordinals
+(`<sts>-0.<headless svc>,…,<sts>-N-1.<headless svc>`), so it needs no registry; plain
+`docker run` passes the container names, each container started with `--hostname` equal to
+its `--name` (or with `MYHOSTNAME` set to it). `MASTER_SERVER` keeps working as a
+one-element list, and may name the master by an address or an `/etc/hosts` alias as before:
+the master recognises itself by those as well.
+
+A server recognises itself in the list by a name that equals its `hostname -f`, or that is
+its `hostname -f` cut at a dot (or the other way round) - a pod whose FQDN is
+`<sts>-N.<headless svc>.<namespace>.svc.cluster.local` is listed as
+`<sts>-N.<headless svc>` - and by one of its own addresses or a name `/etc/hosts` gives one
+of them (an `--add-host` or a `hostAliases` entry for itself). Names are compared whole:
+`opendj-1` is not `opendj-10`, and `opendj-0.opendj.east` is not `opendj-0.opendj.west`.
+List DNS names rather than addresses in `REPLICATION_PEERS`: the servers register in
+`cn=admin data` by name, and a server listed by address alone would be taken for one that
+left the topology and removed from it (see below). A server registered by an address - a
+master that `MASTER_SERVER=<address>` named, which its replicas registered under that address
+- is never removed that way, as nothing tells it from a listed name: moving such a topology
+to a `REPLICATION_PEERS` of names keeps the master, and once it really is gone, it is
+removed by hand.
+
+The join decides from what is there, not from an exit code: this server is a member once
+its configuration holds the replication domain for `BASE_DN` and it is registered in
+`cn=admin data`. Until then it runs `dsreplication enable` through the listed peers,
+`REPLICATION_RETRY_COUNT` times every `REPLICATION_RETRY_INTERVAL` seconds, each enable
+bounded by `REPLICATION_ATTEMPT_TIMEOUT` seconds, and the container reports itself healthy
+only once the join succeeded and the volume holds the data of the topology - across
+restarts too. A server whose volume holds that data is ready as soon as it serves, without
+waiting for its peers or for its join, so a whole cluster restart does not deadlock under
+`OrderedReady`, and a changed root password does not keep it unready either (the join,
+which binds with `ROOT_PASSWORD`, then only logs that it cannot).
+
+A volume the container bootstraps is marked before the bootstrap starts, and initialized
+from the topology once the join succeeds, whether or not `BASE_DN` already holds entries - so
+a bootstrap that failed or was killed half-way is initialized as well, never taken for data
+of the topology: every fresh volume holds what
+`ADD_BASE_ENTRY` or `SAMPLE_DATA` imported, which says nothing about which server holds
+the data of the topology. `dsreplication initialize` is a full import of `BASE_DN` and runs
+without a bound unless `REPLICATION_INITIALIZE_TIMEOUT` sets one. Each server publishes to
+its peers whether its volume still waits for that data (`pending`) or holds it (`ready`, or
+`rejoining` while its replication is taken down to be enabled anew, see below), in the local
+entry `cn=Docker Join,cn=config`, and a server joins only through a ready one - a pending
+server initializes from it as well. So servers may start together - Compose, or a StatefulSet with
+`podManagementPolicy: Parallel` - without any of them taking the bootstrap data of another
+fresh one for the topology's. Servers whose enables involve the same server take turns: two
+`dsreplication enable` runs through one server at once leave the loser a member only in
+part, for good. The turn is the entry `cn=Docker Join Lock,cn=config`, which a join adds
+before its enable on the peer it enables through and on its own server, and removes after
+it; one left behind by a killed join is taken over once it is older than
+`REPLICATION_ATTEMPT_TIMEOUT` plus a minute, or at once by a later join of the server that
+left it. A join that finds a turn taken tries the next peer, and waits a random part of
+`REPLICATION_RETRY_INTERVAL` on top of it between rounds, so two joins that each hold the
+other's turn do not keep meeting. Only the first entry of `REPLICATION_PEERS` may decide that
+there is no topology yet and seed it with its own data: at once when every other peer
+answers and waits for data as well, or once its retries are exhausted without any peer
+that could hold the data answering - a ready or rejoining one, a server of an earlier image that
+publishes no state but replicates `BASE_DN`, or one that refuses the bind with
+`ROOT_PASSWORD` (only a server past its bootstrap has another root password). The residual risk of that rule: with every other server
+down *and* the volume of the first peer lost, the first peer seeds an empty topology and the
+others initialize from it; keep backups accordingly. A volume that joined but whose
+initialize never completed keeps waiting for a ready peer on every start, so under
+`OrderedReady` a `-0` in that state stays unready until a peer that holds the data is
+started by hand - which is where its data has to come from anyway.
+
+With `REPLICATION_PEERS` set explicitly, a joined server also removes every server that is
+registered in the topology but no longer listed: its entries in `cn=admin data` and its
+values in every local replication server list - those of `BASE_DN`, and of `cn=schema` and
+`cn=admin data`, which `dsreplication enable` replicates next to it. A scale-down therefore
+needs no `preStop` hook - `dsreplication disable` on termination would take the server out
+on every rolling restart, and cannot clean up a server that is already gone. It also adds
+every listed peer registered in the topology to the local lists that lack it, so joins that
+ran at the same time cannot leave two replication servers that know nothing of each other.
+With only `MASTER_SERVER` set nothing is removed or added, so servers joined by hand stay.
+
+A server that was removed this way and is scaled up again on the volume it kept still holds
+the data and its replication configuration, but no peer registers it any more, and
+`dsreplication enable` would not register it again. Its join takes its replication
+configuration down and enables it anew, and from that moment until the enable has registered
+it again its clients may not write: writes it took meanwhile would replicate nowhere. The
+backend of `BASE_DN` is set to `writability-mode: internal-only` before anything changes -
+writes of clients are refused with `Unwilling to Perform`, replication and `dsreplication`
+still write - and set back to `enabled` once the server rejoined; a backend that was not
+`enabled` before is left as it is. Throughout, the container does not report itself healthy,
+across restarts too, and the server publishes `rejoining`, so no peer joins or initializes
+through it. It enables only through a peer that publishes `ready`, never through another one
+that rejoins or waits for the data: two such servers would register only each other and
+serve a topology of their own beside the survivors. The health status itself turns
+`unhealthy` only after the probe's retries (a readiness probe after its failure threshold),
+which is why the writes are refused rather than left to the health check. A join that gives
+up leaves the server unhealthy and read-only until a later start rejoins it. A container
+started on such a volume without the replication environment runs no join, reports itself
+healthy and logs that the backend may still refuse writes: its `writability-mode` is then
+the operator's to set back to `enabled`. It takes the replication down only when at least one peer
+that replicates `BASE_DN` answered and none of them registers it - a peer that cannot be
+asked decides nothing - or when the peers register it but its own `cn=admin data` does not.
+An enable that stopped half-way leaves that state behind (its initialize of `cn=admin data`
+failed, say): `BASE_DN` is replicated, and every later enable only reports it as replicated
+already, without registering the server where it is missing.
+
+The one-shot types `srs`, `sdsr` and `rg` keep their previous behaviour - they run once,
+during the first bootstrap only - and are deprecated in favour of `simple`. Like `simple`,
+they need one `ADMIN_PORT` and one `REPLICATION_PORT` on every server: the tools reach the
+master on the ports of the container they run in (images before this one used 4444 and
+8989 on both sides, whatever the container was set up with). With
+`OPENDJ_REPLICATION_TYPE=srs`, start the directory server replicas one at a time, each
+once the previous one is healthy: every replica pushes its data to the replicas connected
+at that moment, and one that is restarting at the end of its own first start misses it.
+
## Certificates
With the default `OPENDJ_SSL_OPTIONS` the instance serves LDAPS and StartTLS with a
@@ -150,10 +260,16 @@
| ROOT_PASSWORD | password | Initial root user password; the bootstrap fails if it contains a line break (CR or LF) |
| SECRET_VOLUME | /var/secrets/opendj | Mounted keystore volume, if present its `key*` and `trust*` files are copied into the instance on every start, see [Certificates](#certificates) |
| SECRET_VOLUME_REFRESH | 60 | While the server runs, `SECRET_VOLUME` is checked again every that many seconds and changed files are copied again; `0` copies them on start only |
-| MASTER_SERVER | - | Replication master server |
+| MASTER_SERVER | - | Replication master server; with `simple` it works as a one-element `REPLICATION_PEERS` |
+| REPLICATION_PEERS | value of MASTER_SERVER | every server of the replication topology by DNS name, comma separated; the first entry may seed a new topology, and servers no longer listed are removed from it, see [Replication](#replication) |
+| REPLICATION_PORT | 8989 | replication port, the same on every server of the topology |
+| REPLICATION_RETRY_COUNT | 30 | rounds of join attempts through every peer before the join gives up: the first peer of `REPLICATION_PEERS` then seeds the topology, any other server stays `unhealthy`; a value that is not a whole number above 0 falls back to 30 |
+| REPLICATION_RETRY_INTERVAL | 10 | seconds between rounds of join attempts, plus a random part of as much again; a value that is not a whole number falls back to 10 |
+| REPLICATION_ATTEMPT_TIMEOUT | 120 | seconds a single `dsreplication enable` may take before it is killed and tried again; it can hang on a peer that stops mid-operation, so a value that is not a whole number above 0 falls back to 120 |
+| REPLICATION_INITIALIZE_TIMEOUT | 0 | seconds a single `dsreplication initialize` - a full import of `BASE_DN` - may take before it is killed and tried again; `0` sets no bound, and so does a value that is not a whole number |
| VERSION | - | OpenDJ version |
| OPENDJ_USER | opendj | user which runs OpenDJ |
-| OPENDJ_REPLICATION_TYPE | - | OpenDJ Replication type, valid values are: <ul><li>simple - standart replication</li><li>srs - standalone replication servers</li><li>sdsr - Standalone Directory Server Replicas</li><li>rg - Replication Groups</li></ul>Other values will be ignored |
+| OPENDJ_REPLICATION_TYPE | - | OpenDJ Replication type, valid values are: <ul><li>simple - standard replication, joined in the background on every start, see [Replication](#replication)</li><li>srs - standalone replication servers (one-shot, deprecated)</li><li>sdsr - Standalone Directory Server Replicas (one-shot, deprecated)</li><li>rg - Replication Groups (one-shot, deprecated)</li></ul>Other values will be ignored |
| OPENDJ_SSL_OPTIONS | --generateSelfSignedCertificate | you can replace ssl options at here, like : "--usePkcs12keyStore /opt/domain.pfx --keyStorePassword domain" |
| OPENDJ_JAVA_ARGS | -server | extra instance java args |
| BACKEND_TYPE | je | OpenDJ backend type, see [dsconfig create-backend](https://doc.openidentityplatform.org/opendj/reference/dsconfig-subcommands-ref#dsconfig-create-backend) documentation |
diff --git a/opendj-packages/opendj-docker/bootstrap/join.sh b/opendj-packages/opendj-docker/bootstrap/join.sh
new file mode 100755
index 0000000..8afa04d
--- /dev/null
+++ b/opendj-packages/opendj-docker/bootstrap/join.sh
@@ -0,0 +1,976 @@
+#!/usr/bin/env bash
+# The contents of this file are subject to the terms of the Common Development and
+# Distribution License (the License). You may not use this file except in compliance with the
+# License.
+#
+# You can obtain a copy of the License at legal/CDDLv1.0.txt. See the License for the
+# specific language governing permission and limitations under the License.
+#
+# When distributing Covered Software, include this CDDL Header Notice in each file and include
+# the License file at legal/CDDLv1.0.txt. If applicable, add the following below the CDDL
+# Header, with the fields enclosed by brackets [] replaced by your own identifying
+# information: "Portions copyright [year] [name of copyright owner]".
+#
+# Copyright 2026 3A Systems, LLC.
+
+# Joins this server to the replication topology (#1086). run.sh starts it in the background
+# next to the server on every start, not as a step of the first bootstrap: a join that could
+# not complete is tried again on the next start, and membership that changed while no
+# container ran is repaired. It covers OPENDJ_REPLICATION_TYPE=simple; the one-shot srs, sdsr
+# and rg paths stay in replicate.sh.
+#
+# The contract with the operator: REPLICATION_PEERS lists every server of the topology by DNS
+# name, and all of them share BASE_DN, ROOT_USER_DN, ROOT_PASSWORD, ADMIN_PORT and
+# REPLICATION_PORT. A StatefulSet chart derives the list from its ordinals
+# (<sts>-0.<svc> ... <sts>-N-1.<svc>), plain docker run passes the container names of
+# containers whose --hostname is their --name. MASTER_SERVER keeps working as a one-element
+# list. This server recognises itself in the list by a name that equals hostname -f, or is
+# hostname -f cut at a dot (a StatefulSet pod whose FQDN is <sts>-N.<svc>.<ns>.svc.<domain>
+# is listed as <sts>-N.<svc>), or the other way round; and, as the grep of /etc/hosts it
+# replaces did, by one of its own addresses or a name /etc/hosts gives one of them. Names
+# are compared whole - that grep took opendj-1 for opendj-10, a replica for its master when
+# an --add-host named the master, and a master that lost its volume for a replica of itself -
+# and cut only at a dot, so opendj-0.opendj.east is not opendj-0.opendj.west.
+#
+# Membership is decided from what is there, never from an exit code: exit 5 of dsreplication
+# enable (REPLICATION_CANNOT_BE_ENABLED_ON_BASEDN) means "no suffix was left to enable",
+# which covers a base DN that is already replicated and a base DN that one of the two
+# servers does not hold - the race with a peer whose bootstrap has not created the backend
+# yet. The step is a member once the replication domain for BASE_DN exists in its cn=config
+# and the server is registered in cn=admin data; anything else is tried again, whatever the
+# exit code, because a repeated enable is safe exactly when it is decided this way.
+#
+# Whose data the topology carries follows what each volume went through, not whether BASE_DN
+# has entries: every fresh volume has entries (setup.sh imports the base entry or
+# SAMPLE_DATA), and two freshly bootstrapped volumes even share a generation ID, so
+# replication silently carries nothing that reached one of them by import-ldif. run.sh marks
+# a volume it bootstrapped with $INITIALIZE_PENDING, and the join publishes that to its peers
+# in $STATE_DN, a local entry of cn=config: "pending" while the marker is there, "ready" once
+# the volume holds the data of the topology - it was never bootstrapped by run.sh, it was
+# initialized from a ready peer, or it seeded the topology - and "rejoining" while a volume
+# that holds it has its replication taken down to enable it anew. A server joins only through
+# a ready peer, and a pending one initializes only from one, so two fresh servers that start
+# together never take each other's bootstrap data for the topology's, and two servers whose
+# replication is down never form a topology of their own; where the list is only
+# MASTER_SERVER, a peer that publishes no state (an image before this one) counts as ready, as
+# the old replicate.sh trusted its master. The seed is only the first entry of
+# REPLICATION_PEERS: at once when every other peer answers and is pending (all of them start
+# from scratch, together or not), or once the retries are exhausted while no other peer
+# answers that may hold the data - a ready one, one of an image before this one that
+# publishes no state but replicates BASE_DN, or one that refuses the bind with ROOT_PASSWORD
+# (only a server past its bootstrap can have another root password). The residual risk is
+# documented in the README: every other server down *and* the first peer's volume lost means
+# the first peer seeds empty.
+#
+# On a volume that waits for the data of the topology, the health marker is written only
+# once the join succeeded or the seed rule fired, so a container that never joined stays
+# unhealthy - across restarts too, which the bootstrap-only replicate.sh could not say (a
+# restart wrote the marker right after upgrade and a failed join turned into a healthy,
+# unreplicated server). A volume that holds the data is healthy as soon as it serves (see
+# run.sh) - unless the join takes its replication down to enable it anew: from then until the
+# enable registered it again it is not, across restarts too ($REJOIN_PENDING), and the
+# backend of BASE_DN refuses the writes of clients, which no longer replicate meanwhile.
+
+cd /opt/opendj || exit 1
+export PATH=/opt/opendj/bin:$PATH
+
+BASE_DN=${BASE_DN:-"dc=example,dc=com"}
+ROOT_USER_DN=${ROOT_USER_DN:-"cn=Directory Manager"}
+ROOT_PASSWORD=${ROOT_PASSWORD:-password}
+ADMIN_PORT=${ADMIN_PORT:-4444}
+REPLICATION_PORT=${REPLICATION_PORT:-8989}
+MYHOSTNAME=${MYHOSTNAME:-$(hostname -f)}
+BOOTSTRAP_COMPLETE=${BOOTSTRAP_COMPLETE:-/opt/opendj/.bootstrap-complete}
+INITIALIZE_PENDING=${INITIALIZE_PENDING:-/opt/opendj/data/.replication-initialize-pending}
+REJOIN_PENDING=${REJOIN_PENDING:-/opt/opendj/data/.replication-rejoin-pending}
+# not replicated: cn=config is this server's alone, and it lives on the volume next to the
+# marker, so the two cannot tell different stories
+STATE_DN="cn=Docker Join,cn=config"
+
+# Sets the variable to its value read as a decimal whole number, or to the default when it
+# is not one or is below the least value. $(( )) reads a leading zero as octal and fails on
+# 08 - abandoning the whole command it runs in, however far up that is - while test reads 08
+# as 8, so a check by test alone lets the value through to the arithmetic.
+whole_number() { # <variable> <default> <least>
+ local value=${!1}
+ case $value in
+ '' | *[!0-9]*) value= ;;
+ *) value=$((10#$value)) ;;
+ esac
+ if [ -z "$value" ] || [ "$value" -lt "$3" ]; then
+ echo "join: $1 is not a whole number of at least $3, using $2"
+ value=$2
+ fi
+ printf -v "$1" '%s' "$value"
+}
+
+REPLICATION_RETRY_COUNT=${REPLICATION_RETRY_COUNT:-30}
+whole_number REPLICATION_RETRY_COUNT 30 1
+REPLICATION_RETRY_INTERVAL=${REPLICATION_RETRY_INTERVAL:-10}
+whole_number REPLICATION_RETRY_INTERVAL 10 0
+# dsreplication enable can hang rather than fail (a peer that stops mid-operation), so no
+# enable runs without a bound
+REPLICATION_ATTEMPT_TIMEOUT=${REPLICATION_ATTEMPT_TIMEOUT:-120}
+whole_number REPLICATION_ATTEMPT_TIMEOUT 120 1
+# a total update is a full import of BASE_DN, which takes as long as the data needs; a killed
+# dsreplication initialize leaves its task running on the server, and the next attempt asks
+# for another full import, so by default it runs without a bound
+REPLICATION_INITIALIZE_TIMEOUT=${REPLICATION_INITIALIZE_TIMEOUT:-0}
+whole_number REPLICATION_INITIALIZE_TIMEOUT 0 0
+
+# Removing servers that left the topology needs the operator's word on who belongs to it:
+# with only MASTER_SERVER set, a third server joined by hand would be "not in the list" and
+# thrown out. So the cleanup, and the repair of the replication server lists, run only when
+# REPLICATION_PEERS itself is set.
+PEERS_ARE_EXPLICIT=${REPLICATION_PEERS:+yes}
+REPLICATION_PEERS=${REPLICATION_PEERS:-$MASTER_SERVER}
+
+if [ -z "$REPLICATION_PEERS" ]; then
+ echo "join: neither REPLICATION_PEERS nor MASTER_SERVER is set, nothing to join"
+ exit 1
+fi
+
+IFS=',' read -r -a RAW_PEERS <<<"$REPLICATION_PEERS"
+PEERS=()
+for peer in "${RAW_PEERS[@]}"; do
+ peer=$(echo "$peer" | tr -d '[:space:]')
+ [ -n "$peer" ] && PEERS+=("$peer")
+done
+if [ "${#PEERS[@]}" -eq 0 ]; then
+ echo "join: REPLICATION_PEERS holds no peer"
+ exit 1
+fi
+
+# The tools read the root password from a file (#1084); the file lives on the tmpfs of
+# /dev/shm where there is one, and run.sh removes what a killed join left behind, by the
+# name, on every start - in /dev/shm those of its ADMIN_PORT, in /tmp all of them
+PASSWORD_FILE=$(mktemp -p /dev/shm "opendj-join.$ADMIN_PORT.XXXXXX" 2>/dev/null \
+ || mktemp "/tmp/opendj-join.$ADMIN_PORT.XXXXXX") || exit 1
+trap 'rm -f "$PASSWORD_FILE"' EXIT
+printf '%s\n' "$ROOT_PASSWORD" >"$PASSWORD_FILE" || exit 1
+
+lower() {
+ printf '%s' "$1" | tr '[:upper:]' '[:lower:]'
+}
+
+# an IPv4 address is digits and dots, an IPv6 one has a colon
+is_address() {
+ case $1 in
+ *:*) return 0 ;;
+ *[!0-9.]* | '') return 1 ;;
+ *) return 0 ;;
+ esac
+}
+
+# Two names (lower case) are one server when they are equal, or when one of them is the
+# other cut at a dot; addresses only when they are equal
+names_match() {
+ [ "$1" = "$2" ] && return 0
+ if is_address "$1" || is_address "$2"; then
+ return 1
+ fi
+ case $1 in "$2".*) return 0 ;; esac
+ case $2 in "$1".*) return 0 ;; esac
+ return 1
+}
+
+SELF_NAME=$(lower "$MYHOSTNAME")
+# the addresses of this container (loopback aside) and the names /etc/hosts gives them - an
+# --add-host or a hostAliases entry for itself - as the base replicate.sh recognised them
+SELF_ADDRESSES=()
+SELF_ALIASES=()
+own_names=" $(lower "$(hostname)") $(lower "$(hostname -f 2>/dev/null)") $SELF_NAME "
+for address in $(hostname -i 2>/dev/null) \
+ $(awk -v names="$own_names" '!/^[[:space:]]*#/ { for (i = 2; i <= NF; i++) if (index(names, " " tolower($i) " ")) { print $1; break } }' /etc/hosts 2>/dev/null); do
+ case $address in 127.* | ::1 | 0.0.0.0) continue ;; esac
+ SELF_ADDRESSES+=("$(lower "$address")")
+done
+if [ "${#SELF_ADDRESSES[@]}" -gt 0 ]; then
+ for name in $(awk -v addresses=" ${SELF_ADDRESSES[*]} " '!/^[[:space:]]*#/ && index(addresses, " " tolower($1) " ") { for (i = 2; i <= NF; i++) print tolower($i) }' /etc/hosts 2>/dev/null); do
+ SELF_ALIASES+=("$name")
+ done
+fi
+
+is_self() {
+ local peer candidate
+ peer=$(lower "$1")
+ names_match "$peer" "$SELF_NAME" && return 0
+ for candidate in "${SELF_ADDRESSES[@]}" "${SELF_ALIASES[@]}"; do
+ [ "$peer" = "$candidate" ] && return 0
+ done
+ return 1
+}
+
+in_peers() {
+ local host peer
+ host=$(lower "$1")
+ for peer in "${PEERS[@]}"; do
+ names_match "$(lower "$peer")" "$host" && return 0
+ done
+ return 1
+}
+
+# every LDAP operation goes through the administration connector, which always serves TLS;
+# the root user may change its password after the bootstrap, so unlike the health check
+# these tools run only while the join still has the bootstrap's ROOT_PASSWORD to work with
+search() {
+ local host=$1
+ shift
+ ldapsearch --noPropertiesFile --hostname "$host" --port "$ADMIN_PORT" --useSsl --trustAll \
+ --bindDN "$ROOT_USER_DN" --bindPasswordFile "$PASSWORD_FILE" "$@" 2>/dev/null
+}
+
+ldapmodify_on() { # <host> [<option>...]
+ local host=$1
+ shift
+ ldapmodify --noPropertiesFile --hostname "$host" --port "$ADMIN_PORT" --useSsl --trustAll \
+ --bindDN "$ROOT_USER_DN" --bindPasswordFile "$PASSWORD_FILE" "$@" 2>/dev/null
+}
+
+ldapmodify_local() {
+ ldapmodify_on localhost "$@"
+}
+
+dsconfig_local() {
+ dsconfig "$@" --hostname localhost --port "$ADMIN_PORT" --bindDN "$ROOT_USER_DN" \
+ --bindPasswordFile "$PASSWORD_FILE" --trustAll --no-prompt
+}
+
+# runs a command within that many seconds, or without a bound for 0; timeout executes the
+# command, so it has to be a program, not a function of this script
+bounded() {
+ local limit=$1
+ shift
+ if [ "$limit" -gt 0 ] 2>/dev/null; then
+ timeout "$limit" "$@"
+ else
+ "$@"
+ fi
+}
+
+server_up() {
+ search localhost --baseDN "" --searchScope base "(objectClass=*)" 1.1 >/dev/null
+}
+
+publish_state() {
+ printf 'dn: %s\nchangetype: modify\nreplace: description\ndescription: %s\n' "$STATE_DN" "$1" \
+ | ldapmodify_local >/dev/null && return 0
+ printf 'dn: %s\nobjectClass: top\nobjectClass: ds-cfg-branch\nobjectClass: extensibleObject\ncn: Docker Join\ndescription: %s\n' "$STATE_DN" "$1" \
+ | ldapmodify_local --defaultAdd >/dev/null && return 0
+ echo "join: could not publish the state $1 of this server to its peers"
+ return 1
+}
+
+# ready, pending or rejoining as the peer publishes it, absent when it answers without
+# publishing any, refused when it answers but no longer takes ROOT_PASSWORD, down when it does
+# not answer
+peer_state() {
+ local out state
+ out=$(search "$1" --baseDN "$STATE_DN" --searchScope base "(objectClass=*)" description)
+ case $? in
+ 0)
+ state=$(printf '%s\n' "$out" | awk 'tolower($1) == "description:" { print $2; exit }')
+ case $state in
+ ready | pending | rejoining) echo "$state" ;;
+ *) echo absent ;;
+ esac
+ ;;
+ 32) echo absent ;;
+ # 49: invalidCredentials - the root password of a server is changed in its cn=config,
+ # which is not replicated, and only a server past its bootstrap has another one
+ 49) echo refused ;;
+ *) echo down ;;
+ esac
+}
+
+trusted() {
+ case $1 in
+ ready) return 0 ;;
+ absent) [ "$PEERS_ARE_EXPLICIT" != yes ] ;;
+ *) return 1 ;;
+ esac
+}
+
+# fails when the search does, rather than answering with an empty list
+registered_hosts() { # [<host>, localhost by default]
+ local out
+ out=$(search "${1:-localhost}" --baseDN "cn=Servers,cn=admin data" --searchScope one "(objectClass=*)" hostname) \
+ || return 1
+ printf '%s\n' "$out" | awk 'tolower($1) == "hostname:" { print $2 }'
+}
+
+# a member holds the replication domain for BASE_DN in its configuration and is registered
+# in cn=admin data - under the name the enable that registered it connected with, which for
+# a server that never ran an enable of its own is the name a peer listed it by; both are made
+# by one dsreplication enable, so requiring both keeps a half-done enable from being declared
+# a success, and the next round takes it down (see reset_if_unregistered)
+replicates_base_dn() { # <host>
+ search "$1" --baseDN "cn=config" --searchScope sub \
+ "(&(objectClass=ds-cfg-replication-domain)(ds-cfg-base-dn=$BASE_DN))" 1.1 | grep -q "^dn:"
+}
+
+is_member() {
+ local host
+ replicates_base_dn localhost || return 1
+ for host in $(registered_hosts); do
+ is_self "$host" && return 0
+ done
+ return 1
+}
+
+# cn=admin data is replicated, but a server that was away while the survivors removed it
+# from the topology still holds its own entry until that delete reaches it, and it may not
+# have yet when the join looks: the other peers that answer and replicate BASE_DN have the
+# say. Being unregistered takes this server's replication down (reset_replication), so it is
+# concluded only when at least one of them listed its registrations and none of them lists
+# this server: a search that failed, or a peer whose copy of cn=admin data still lags behind
+# the others, decides nothing while another peer lists it. With no such peer the local view
+# stands - nothing else can be asked.
+registered_with_peers() {
+ local peer host hosts asked=no
+ for peer in "${PEERS[@]}"; do
+ is_self "$peer" && continue
+ replicates_base_dn "$peer" || continue
+ hosts=$(registered_hosts "$peer") || continue
+ asked=yes
+ for host in $hosts; do
+ is_self "$host" && return 0
+ done
+ done
+ [ "$asked" = no ]
+}
+
+member_of_topology() {
+ is_member && registered_with_peers
+}
+
+# whether the cn=admin data of this server registers it: 0 when it does, 1 when a search
+# answers that it does not - cn=Servers missing altogether included, which an initialize of
+# cn=admin data that failed half-way leaves - and 2 when the search failed otherwise, which
+# decides nothing
+registered_here() {
+ local out host
+ out=$(search localhost --baseDN "cn=Servers,cn=admin data" --searchScope one "(objectClass=*)" hostname)
+ case $? in
+ 0) ;;
+ 32) return 1 ;;
+ *) return 2 ;;
+ esac
+ for host in $(printf '%s\n' "$out" | awk 'tolower($1) == "hostname:" { print $2 }'); do
+ is_self "$host" && return 0
+ done
+ return 1
+}
+
+# Two dsreplication enable runs through one peer at the same time break each other: both
+# create the replication server that a seed does not have yet, the loser fails half-way
+# (ManagedObjectAlreadyExistsException, exit 17), and every enable after that exits 5,
+# "already replicated", without ever completing its membership. So an enable holds a lock on
+# the peer it runs through - an entry of the peer's cn=config, which only one add creates -
+# and one on this server: an enable configures both of its servers, and a server that enables
+# through a peer may at the same time be the peer another server enables through. A server
+# that takes its own replication down holds its own lock for as long.
+# A lock whose holder was killed before it removed it is broken once it is older than an
+# enable may take, by a delete that asserts the value it read: two joins that break it at
+# once cannot remove the lock one of them took right after. One that holds this server's own
+# name is broken at once: it was left by a join of an earlier start, and a start runs one.
+JOIN_LOCK_DN="cn=Docker Join Lock,cn=config"
+JOIN_LOCK_TTL=$((REPLICATION_ATTEMPT_TIMEOUT + 60))
+JOIN_LOCK_VALUE=
+
+add_join_lock() { # <host> <value>
+ printf 'dn: %s\nobjectClass: top\nobjectClass: ds-cfg-branch\nobjectClass: extensibleObject\ncn: Docker Join Lock\ndescription: %s\n' "$JOIN_LOCK_DN" "$2" \
+ | ldapmodify_on "$1" --defaultAdd >/dev/null
+}
+
+# Succeeds when this server may go on: it took the lock and left its value in
+# JOIN_LOCK_VALUE, or the server could not be asked for one - a peer that does not answer
+# fails the enable anyway, and one that refuses the entry for another reason than holding it
+# must not keep the join out for good. A lock taken this way is released with
+# release_join_lock <host> "$JOIN_LOCK_VALUE".
+take_join_lock() { # <host>
+ local value rc held holder since age where
+ JOIN_LOCK_VALUE=
+ value="$SELF_NAME $(date +%s)"
+ add_join_lock "$1" "$value"
+ rc=$?
+ if [ "$rc" -eq 0 ]; then
+ JOIN_LOCK_VALUE=$value
+ return 0
+ fi
+ # 68: entryAlreadyExists
+ [ "$rc" -eq 68 ] || return 0
+ held=$(search "$1" --baseDN "$JOIN_LOCK_DN" --searchScope base "(objectClass=*)" description \
+ | awk 'tolower($1) == "description:" { print $2, $3; exit }')
+ holder=${held% *}
+ since=${held##* }
+ age=$(($(date +%s) - ${since:-0}))
+ if [ -n "$held" ] && { [ "$holder" = "$SELF_NAME" ] || [ "$age" -gt "$JOIN_LOCK_TTL" ]; }; then
+ echo "join: breaking the lock $holder took on $1 $age s ago"
+ printf 'dn: %s\nchangetype: delete\n' "$JOIN_LOCK_DN" \
+ | ldapmodify_on "$1" --assertionFilter "(description=$held)" >/dev/null
+ if add_join_lock "$1" "$value"; then
+ JOIN_LOCK_VALUE=$value
+ return 0
+ fi
+ fi
+ if [ "$1" = localhost ]; then where="this server"; else where=$1; fi
+ # a lock released between the add and the search has no holder left to name
+ echo "join: ${holder:-another server} is enabling replication through $where, waiting for it"
+ return 1
+}
+
+release_join_lock() { # <host> <value>
+ [ -n "$2" ] || return 0
+ printf 'dn: %s\nchangetype: delete\n' "$JOIN_LOCK_DN" \
+ | ldapmodify_on "$1" --assertionFilter "(description=$2)" >/dev/null \
+ || echo "join: could not remove the lock on $1, it expires in $JOIN_LOCK_TTL s"
+}
+
+# exits 75 without running the enable while another server enables through the peer or
+# through this one. Neither lock is waited for, so two servers that each hold their own and
+# ask for the other's cannot deadlock; they both give up and try again after pause().
+enable_through() {
+ local rc own peer_lock
+ take_join_lock localhost || return 75
+ own=$JOIN_LOCK_VALUE
+ if ! take_join_lock "$1"; then
+ release_join_lock localhost "$own"
+ return 75
+ fi
+ peer_lock=$JOIN_LOCK_VALUE
+ bounded "$REPLICATION_ATTEMPT_TIMEOUT" dsreplication enable \
+ --host1 "$1" --port1 "$ADMIN_PORT" --bindDN1 "$ROOT_USER_DN" \
+ --bindPasswordFile1 "$PASSWORD_FILE" --replicationPort1 "$REPLICATION_PORT" \
+ --host2 "$MYHOSTNAME" --port2 "$ADMIN_PORT" --bindDN2 "$ROOT_USER_DN" \
+ --bindPasswordFile2 "$PASSWORD_FILE" --replicationPort2 "$REPLICATION_PORT" \
+ --adminUID admin --adminPasswordFile "$PASSWORD_FILE" \
+ --baseDN "$BASE_DN" -X -n
+ rc=$?
+ release_join_lock "$1" "$peer_lock"
+ release_join_lock localhost "$own"
+ return $rc
+}
+
+# the retry interval and up to as much again, at random: two joins that keep missing each
+# other's lock would otherwise try again in step, round after round
+pause() {
+ local delay=$((REPLICATION_RETRY_INTERVAL + RANDOM % (REPLICATION_RETRY_INTERVAL + 1)))
+ echo "$1, trying again in $delay s"
+ sleep "$delay"
+}
+
+initialize_from() {
+ bounded "$REPLICATION_INITIALIZE_TIMEOUT" dsreplication initialize --baseDN "$BASE_DN" \
+ --adminUID admin --adminPasswordFile "$PASSWORD_FILE" \
+ --hostSource "$1" --portSource "$ADMIN_PORT" \
+ --hostDestination "$MYHOSTNAME" --portDestination "$ADMIN_PORT" -X -n
+}
+
+# the generation ID of the replication domain, from the domain's own monitor entry: a
+# replication server also puts domain-name, connected-to and generation-id on the entry of
+# every directory server connected to it, but only the domain publishes replayed-updates
+generation_id() {
+ search "$1" --baseDN "cn=monitor" --searchScope sub \
+ "(&(domain-name=$BASE_DN)(replayed-updates=*))" generation-id \
+ | awk 'tolower($1) == "generation-id:" { print $2; exit }'
+}
+
+cross_check() { # <peer> <what the mismatch means>
+ local local_id peer_id
+ local_id=$(generation_id localhost)
+ peer_id=$(generation_id "$1")
+ if [ -n "$local_id" ] && [ -n "$peer_id" ] && [ "$local_id" != "$peer_id" ]; then
+ echo "join: generation ID $local_id does not match $peer_id of $1$2"
+ fi
+}
+
+# the first other peer, in the order of the list, that holds the data of the topology and
+# replicates BASE_DN - asked of the peer itself, as cn=admin data registers a server under
+# whatever name the enable that registered it used, which need not be the listed one
+trusted_source() {
+ local peer
+ for peer in "${PEERS[@]}"; do
+ is_self "$peer" && continue
+ trusted "$(peer_state "$peer")" && replicates_base_dn "$peer" && { echo "$peer"; return 0; }
+ done
+ return 1
+}
+
+# every other peer answers and waits for the data of the topology itself; true as well when
+# the list names no other server
+all_others_pending() {
+ local peer
+ for peer in "${PEERS[@]}"; do
+ is_self "$peer" && continue
+ [ "$(peer_state "$peer")" = pending ] || return 1
+ done
+ return 0
+}
+
+# some other peer answers and may hold the data of the topology: it is ready, it is rejoining
+# (it holds the data while its replication is down), it refuses ROOT_PASSWORD, or it publishes
+# no state and replicates BASE_DN. Unlike an initialize source,
+# the last counts here even where the peers are explicit - a server of an image before this
+# one, still serving the topology, and seeding next to it would fork it. A server of this
+# image publishes no state only while it bootstraps, and it replicates nothing then, so it
+# does not hold the seed off. A refusing peer cannot be asked what it replicates, but only a
+# server past its bootstrap has a root password other than ROOT_PASSWORD.
+any_other_may_hold_data() {
+ local peer
+ for peer in "${PEERS[@]}"; do
+ is_self "$peer" && continue
+ case $(peer_state "$peer") in
+ ready | rejoining | refused) return 0 ;;
+ absent) replicates_base_dn "$peer" && return 0 ;;
+ esac
+ done
+ return 1
+}
+
+# The values of the replication server lists this server holds: one "<kind>\t<cn>" line for
+# every list, its replication server entry (kind server) and each of its replication domains
+# (kind domain) - BASE_DN, and cn=schema and cn=admin data, which dsreplication enable
+# configures next to it - followed by a "<kind>\t<cn>\t<host:port>" line for every value
+replication_server_lists() {
+ search localhost --baseDN "cn=config" --searchScope sub \
+ "(|(objectClass=ds-cfg-replication-server)(objectClass=ds-cfg-replication-domain))" \
+ objectClass cn ds-cfg-replication-server \
+ | awk -v OFS='\t' '
+ function flush( i) {
+ if (kind != "") { print kind, cn; for (i = 1; i <= n; i++) print kind, cn, values[i] }
+ kind = ""; cn = ""; n = 0
+ }
+ /^dn: / { flush() }
+ tolower($0) == "objectclass: ds-cfg-replication-domain" { kind = "domain" }
+ tolower($0) == "objectclass: ds-cfg-replication-server" { kind = "server" }
+ tolower($1) == "cn:" { cn = substr($0, 5) }
+ tolower($1) == "ds-cfg-replication-server:" { values[++n] = $2 }
+ END { flush() }'
+}
+
+# it runs inside loops that read their lines from stdin, which dsconfig is kept away from
+change_replication_servers() { # <kind> <cn> --add|--remove <host:port>
+ if [ "$1" = server ]; then
+ dsconfig_local set-replication-server-prop --provider-name "Multimaster Synchronization" \
+ "$3" "replication-server:$4" </dev/null
+ else
+ dsconfig_local set-replication-domain-prop --provider-name "Multimaster Synchronization" \
+ --domain-name "$2" "$3" "replication-server:$4" </dev/null
+ fi
+}
+
+# "<dn>\t<hostname>" for every server this server finds in its cn=admin data
+registrations() {
+ search localhost --baseDN "cn=Servers,cn=admin data" --searchScope one "(objectClass=*)" hostname \
+ | awk '/^dn: /{dn=substr($0,5)} tolower($1) == "hostname:" {print dn "\t" $2}'
+}
+
+# removes a server's entry and its group membership from cn=admin data
+unregister() { # <dn>
+ printf 'dn: %s\nchangetype: delete\n' "$1" | ldapmodify_local >/dev/null || return 1
+ printf 'dn: cn=all-servers,cn=Server Groups,cn=admin data\nchangetype: modify\ndelete: uniqueMember\nuniqueMember: %s\n' "${1%%,cn=Servers,cn=admin data}" \
+ | ldapmodify_local >/dev/null || true
+}
+
+# A server that the survivors removed from the topology while it was away, and that comes
+# back on the volume it kept, still replicates BASE_DN and cn=admin data. dsreplication enable
+# registers a server only where the administration data of one of the two is not replicated
+# yet: with both replicated and their registries alike - this server's copy lacks it once the
+# delete of the survivors reached it - the enable updates the replication server lists and
+# leaves cn=admin data as it is, and this server stays unregistered for good. So its own
+# replication configuration is taken down first, and the enable after that registers it
+# anew. A copy of its entry that the delete has not reached yet is dropped as well: the
+# registries would differ, the enable would merge them, and the delete arriving later would
+# take this server out again.
+# From the first change on, writes this server takes get no change number and reach no other
+# server. Removing the health marker is not enough to stop them: new probes fail, but a
+# container turns unhealthy only after its probe's retries (a readiness probe after its
+# failure threshold), by when the enable has usually run, and connections that are open stay
+# open. So before anything changes the backend of BASE_DN refuses the writes of clients (see
+# hold_writes), and joined() or seed() lets them in again. $REJOIN_PENDING keeps run.sh from
+# reporting a restart healthy until then. A volume that holds the data of the topology
+# publishes the state rejoining meanwhile, which keeps the peers from enabling or
+# initializing through it while it has no replication of its own, and keeps a first peer
+# from seeding next to it; one that waits for that data stays pending. It holds its own join
+# lock while its configuration goes, so that no server enables through it then. One attempt:
+# a lock another join holds leaves everything as it was, and the next round tries again.
+reset_replication() { # <rejoining|pending> <why>
+ local dn host own
+ echo "join: $2, taking its replication configuration down to enable it anew"
+ publish_state "$1" || return 1
+ touch "$REJOIN_PENDING"
+ rm -f "$BOOTSTRAP_COMPLETE"
+ if ! take_join_lock localhost; then
+ echo "join: its replication configuration stays as it is until the next round"
+ return 1
+ fi
+ own=$JOIN_LOCK_VALUE
+ if ! hold_writes; then
+ echo "join: could not stop the writes of clients, its replication configuration stays as it is"
+ release_join_lock localhost "$own"
+ return 1
+ fi
+ registrations | while IFS=$'\t' read -r dn host; do
+ [ -n "$host" ] && is_self "$host" || continue
+ echo "join: dropping the stale entry $dn"
+ unregister "$dn" || echo "join: could not remove $dn"
+ done
+ bounded "$REPLICATION_ATTEMPT_TIMEOUT" dsreplication disable --hostname "$MYHOSTNAME" --port "$ADMIN_PORT" \
+ --adminUID admin --adminPasswordFile "$PASSWORD_FILE" --disableAll -X -n \
+ || echo "join: dsreplication disable exited with $?"
+ release_join_lock localhost "$own"
+}
+
+# "<backend-id> <writability-mode>" of the backend that holds BASE_DN; the mode is left out
+# where the entry does not set one, which means enabled
+data_backend() {
+ search localhost --baseDN "cn=Backends,cn=config" --searchScope one \
+ "(&(objectClass=ds-cfg-backend)(ds-cfg-base-dn=$BASE_DN))" ds-cfg-backend-id ds-cfg-writability-mode \
+ | awk 'tolower($1) == "ds-cfg-backend-id:" { id = $2 } tolower($1) == "ds-cfg-writability-mode:" { mode = $2 }
+ END { if (id != "") print id, mode }'
+}
+
+# The backend of BASE_DN takes internal and replication writes only - dsreplication enable
+# and initialize go through those, a client's write is refused (Unwilling to Perform). Only
+# that backend: internal-only on the whole server would refuse the writes dsreplication
+# enable makes to cn=config and cn=admin data. The mode is configuration, so it outlasts a
+# restart; $REJOIN_PENDING names the backend, so that release_writes lets the writes in
+# again on the start that rejoins, and only where this join stopped them: a backend that an
+# operator made read-only already stays as it is.
+hold_writes() {
+ local backend mode
+ read -r backend mode <<<"$(data_backend)"
+ if [ -z "$backend" ]; then
+ echo "join: found no backend that holds $BASE_DN"
+ return 1
+ fi
+ [ "${mode:-enabled}" = enabled ] || return 0
+ echo "join: backend $backend refuses the writes of clients until this server rejoins"
+ # named before the mode changes: a container killed in between lets the writes in again
+ # once it rejoined, rather than leaving them refused for good
+ printf '%s\n' "$backend" >"$REJOIN_PENDING" || return 1
+ dsconfig_local set-backend-prop --backend-name "$backend" --set writability-mode:internal-only </dev/null >/dev/null
+}
+
+release_writes() {
+ local backend
+ backend=$(cat "$REJOIN_PENDING" 2>/dev/null)
+ [ -n "$backend" ] || return 0
+ dsconfig_local set-backend-prop --backend-name "$backend" --set writability-mode:enabled </dev/null >/dev/null || return 1
+ echo "join: backend $backend takes the writes of clients again"
+}
+
+# Removes every server that is registered in the topology but no longer listed in
+# REPLICATION_PEERS: its entry and group membership in cn=admin data (which is replicated,
+# so one survivor's delete reaches the others) and its values in every replication server
+# list of this server (which are configuration of this server alone, so every survivor
+# prunes its own on its next start). dsreplication disable cannot do this for a dead server:
+# it changes only the servers it can reach, and OpenDJ 4 has no cleanup subcommand. Every
+# step tolerates losing the race to another survivor. A server registered by an address - a
+# master that MASTER_SERVER named by one, whose replicas enabled through that address - is
+# left alone: nothing tells it from a server listed by name, and the list moving from
+# MASTER_SERVER to REPLICATION_PEERS would otherwise throw the live master out.
+cleanup_departed() {
+ local dn host kind cn value
+ [ "$PEERS_ARE_EXPLICIT" = yes ] || return 0
+ registrations | while IFS=$'\t' read -r dn host; do
+ [ -z "$host" ] && continue
+ is_address "$host" && continue
+ if ! in_peers "$host" && ! is_self "$host"; then
+ echo "join: removing departed server $host from cn=admin data"
+ unregister "$dn" || echo "join: could not remove $dn (already removed by another survivor?)"
+ fi
+ done
+ replication_server_lists | while IFS=$'\t' read -r kind cn value; do
+ [ -z "$value" ] && continue
+ host=${value%:*}
+ is_address "$host" && continue
+ if ! in_peers "$host" && ! is_self "$host"; then
+ echo "join: removing departed replication server $value from the $kind list of $cn"
+ change_replication_servers "$kind" "$cn" --remove "$value" || true
+ fi
+ done
+}
+
+# Adds every listed peer that is registered in the topology to each replication server list
+# of this server that lacks it, under the name it is registered by - the name dsreplication
+# enable writes as well, so no server is listed twice. Joins that run at the same time each
+# configure the topology as they read it, and two of them can leave a pair of replication
+# servers that know nothing of each other, cutting the updates of one off from the other; a
+# replication server connects to a server added to its list at once. A peer that is not
+# registered yet is waited for, as long as the retries last, after this server already
+# reports itself healthy.
+repair_replication_servers() {
+ local i peer waiting hosts host lists kind cn value found
+ [ "$PEERS_ARE_EXPLICIT" = yes ] || return 0
+ for i in $(seq 1 "$REPLICATION_RETRY_COUNT"); do
+ waiting=
+ # one search a round: every ldapsearch starts a JVM
+ hosts=$(registered_hosts)
+ for peer in "${PEERS[@]}"; do
+ is_self "$peer" && continue
+ found=no
+ for host in $hosts; do
+ names_match "$(lower "$peer")" "$(lower "$host")" && { found=yes; break; }
+ done
+ [ "$found" = yes ] || waiting="$waiting $peer"
+ done
+ [ -z "$waiting" ] && break
+ [ "$i" -eq "$REPLICATION_RETRY_COUNT" ] && break
+ echo "join: waiting for$waiting to register in the topology before the replication server lists are checked"
+ sleep "$REPLICATION_RETRY_INTERVAL"
+ done
+ hosts=$(registered_hosts)
+ lists=$(replication_server_lists)
+ # a here-string of nothing still reads as one empty line
+ [ -n "$lists" ] || return 0
+ while IFS=$'\t' read -r kind cn value; do
+ [ -n "$value" ] && continue
+ for host in $hosts; do
+ is_self "$host" && continue
+ in_peers "$host" || continue
+ found=no
+ while IFS=$'\t' read -r k c v; do
+ [ "$k" = "$kind" ] && [ "$c" = "$cn" ] && [ -n "$v" ] || continue
+ names_match "$(lower "${v%:*}")" "$(lower "$host")" && { found=yes; break; }
+ done <<<"$lists"
+ if [ "$found" = no ]; then
+ echo "join: adding replication server $host:$REPLICATION_PORT to the $kind list of $cn"
+ change_replication_servers "$kind" "$cn" --add "$host:$REPLICATION_PORT" || true
+ fi
+ done
+ done <<<"$lists"
+}
+
+# the backend of BASE_DN refuses the writes of clients because this join stopped them: an
+# operator's read-only backend leaves $REJOIN_PENDING empty, and a hold whose dsconfig failed
+# named the backend without changing its mode
+writes_held() {
+ [ -s "$REJOIN_PENDING" ] && [ "$(data_backend | awk '{ print $2 }')" = internal-only ]
+}
+
+# a server that rejoins keeps $REJOIN_PENDING, and so stays unready on its next start too,
+# until its clients may write again and its peers see it ready. A dsconfig or a publish that
+# fails is tried again here, as long as the retries last: the join that got this far would
+# otherwise end, and only a later start would enable, or initialize, all over again
+leave_rejoin() {
+ local i
+ [ -f "$REJOIN_PENDING" ] || return 0
+ for i in $(seq 1 "$REPLICATION_RETRY_COUNT"); do
+ if release_writes && publish_state ready; then
+ rm -f "$REJOIN_PENDING"
+ return 0
+ fi
+ if [ "$i" -lt "$REPLICATION_RETRY_COUNT" ]; then
+ pause "join: this server is not out of its rejoin yet ($i of $REPLICATION_RETRY_COUNT)"
+ fi
+ done
+ if writes_held; then
+ echo "join: could not let the writes of clients in again after $REPLICATION_RETRY_COUNT attempts, this container will not report itself healthy and serves its data read-only until a later start does"
+ else
+ echo "join: could not publish this server ready after $REPLICATION_RETRY_COUNT attempts, this container will not report itself healthy until a later start does"
+ fi
+ return 1
+}
+
+joined() {
+ cleanup_departed
+ leave_rejoin || exit 1
+ touch "$BOOTSTRAP_COMPLETE"
+ echo "join: this server is a member of the replication topology, the health check may probe it"
+ repair_replication_servers
+}
+
+# the peers learn that this server holds the data before anything else changes: a seed whose
+# state never reached them would be healthy while every pending peer skips it
+seed() { # <why>
+ echo "join: $1, seeding it with this server's data"
+ publish_state ready || return 1
+ leave_rejoin || return 1
+ rm -f "$INITIALIZE_PENDING"
+ touch "$BOOTSTRAP_COMPLETE"
+}
+
+# what giving up means depends on whether the container already reports itself healthy: a
+# volume that holds the data of the topology does so from the start, until a reset of its
+# replication took that back - and the writes of its clients, where the reset stopped them
+give_up() {
+ if [ -f "$BOOTSTRAP_COMPLETE" ]; then
+ echo "join: could not join the replication topology after $REPLICATION_RETRY_COUNT attempts, this server keeps serving its data unreplicated until a later start joins it"
+ elif writes_held; then
+ echo "join: could not join the replication topology after $REPLICATION_RETRY_COUNT attempts, this container will not report itself healthy and serves its data read-only until a later start joins it"
+ else
+ echo "join: could not join the replication topology after $REPLICATION_RETRY_COUNT attempts, this container will not report itself healthy"
+ fi
+ exit 1
+}
+
+echo "join: waiting for the server on the administration connector"
+for i in $(seq 1 150); do
+ server_up && break
+ if [ "$i" -eq 150 ]; then
+ echo "join: the server did not come up, or does not take ROOT_PASSWORD any more, giving up"
+ exit 1
+ fi
+ sleep 2
+done
+
+FIRST=no
+is_self "${PEERS[0]}" && FIRST=yes
+
+# Every round on both roads starts here: a replication configuration that the peers no longer
+# register is taken down (see reset_replication). So is one that the peers register but the
+# cn=admin data of this server does not: an enable that stopped half-way - its initialize of
+# cn=admin data failed, say - leaves BASE_DN replicated and this server registered at the
+# peer, and every enable after it exits 5, "already replicated", without registering it here.
+# Fails when a reset is due and could not be done in this round, which then enables nothing -
+# an enable registers nobody before it.
+reset_if_unregistered() { # <the state to publish meanwhile>
+ local rc=0
+ replicates_base_dn localhost || return 0
+ if ! registered_with_peers; then
+ reset_replication "$1" "the peers no longer register this server"
+ return
+ fi
+ registered_here || rc=$?
+ if [ "$rc" -eq 1 ]; then
+ reset_replication "$1" "the peers register this server but its own cn=admin data does not"
+ fi
+}
+
+if [ ! -f "$INITIALIZE_PENDING" ]; then
+ # This volume holds the data of the topology, or is the one that seeds it: it publishes
+ # that - unless a reset took its replication down and it has not rejoined yet, also on an
+ # earlier start
+ if [ -f "$REJOIN_PENDING" ]; then
+ publish_state rejoining
+ else
+ publish_state ready
+ fi
+ for i in $(seq 1 "$REPLICATION_RETRY_COUNT"); do
+ if ! reset_if_unregistered rejoining; then
+ : # the next round tries the reset again
+ elif is_member; then
+ echo "join: already a member of the replication topology"
+ joined
+ exit
+ else
+ # only through a peer that holds the topology, as on the pending road: a rejoining peer
+ # has no replication of its own, a pending one or one that still bootstraps none yet, and
+ # an enable with any of them registers the two in a registry of their own - which counts
+ # as membership, a peer of the pair lists this server - beside the survivors, whom
+ # nothing ever adds to it
+ for peer in "${PEERS[@]}"; do
+ is_self "$peer" && continue
+ state=$(peer_state "$peer")
+ if ! trusted "$state"; then
+ echo "join: $peer is $state, not a peer to join through"
+ continue
+ fi
+ echo "join: enabling replication with $peer"
+ enable_through "$peer"
+ rc=$?
+ if member_of_topology; then
+ echo "join: joined the replication topology through $peer"
+ cross_check "$peer" "; replication will not flow until this server or that one is initialized by hand"
+ joined
+ exit
+ fi
+ [ "$rc" -eq 75 ] || echo "join: dsreplication enable with $peer exited with $rc and membership is not there"
+ done
+ fi
+ # a list without a single other server leaves nothing to retry against, and peers that
+ # all wait for data join through this one rather than the other way round
+ if [ "$FIRST" = yes ] && all_others_pending; then
+ seed "no other peer holds the data of the topology" || exit 1
+ exit 0
+ fi
+ if [ "$i" -lt "$REPLICATION_RETRY_COUNT" ]; then
+ pause "join: no peer joined this server yet ($i of $REPLICATION_RETRY_COUNT)"
+ fi
+ done
+ # the same rule as at the end of the pending road below
+ if [ "$FIRST" = yes ] && ! any_other_may_hold_data; then
+ seed "no topology found after $REPLICATION_RETRY_COUNT attempts" || exit 1
+ exit 0
+ fi
+ give_up
+fi
+
+# This volume was bootstrapped and never received the data of the topology: it joins and
+# initializes only through a peer that holds that data, and only the first peer may decide
+# that nobody does. Its membership is asked of the peers as on the road above: a volume that
+# enabled once but never completed its initialize may have been removed by the survivors
+# while it was away, and its own cn=admin data still registers it until their delete reaches
+# it - after which every enable registers nobody, as reset_replication describes.
+publish_state pending
+member=no
+for i in $(seq 1 "$REPLICATION_RETRY_COUNT"); do
+ member=no
+ reset_due=no
+ if ! reset_if_unregistered pending; then
+ reset_due=yes
+ elif is_member; then
+ member=yes
+ fi
+ if [ "$member" = no ] && [ "$reset_due" = no ]; then
+ for peer in "${PEERS[@]}"; do
+ is_self "$peer" && continue
+ state=$(peer_state "$peer")
+ if ! trusted "$state"; then
+ echo "join: $peer is $state, not a peer to join through"
+ continue
+ fi
+ echo "join: enabling replication with $peer"
+ enable_through "$peer"
+ rc=$?
+ if member_of_topology; then
+ member=yes
+ echo "join: joined the replication topology through $peer"
+ break
+ fi
+ [ "$rc" -eq 75 ] || echo "join: dsreplication enable with $peer exited with $rc and membership is not there"
+ done
+ fi
+ if [ "$member" = yes ]; then
+ source=$(trusted_source)
+ if [ -n "$source" ]; then
+ echo "join: initializing from $source"
+ initialize_from "$source"
+ rc=$?
+ if [ "$rc" -eq 0 ]; then
+ # published before the marker goes, so that the marker is never gone while the peers
+ # still see this server pending; the next start initializes it again
+ publish_state ready || exit 1
+ rm -f "$INITIALIZE_PENDING"
+ cross_check "$source" " after initialize"
+ joined
+ exit
+ fi
+ echo "join: initialize from $source exited with $rc"
+ elif [ "$FIRST" = yes ] && all_others_pending; then
+ seed "every other peer waits for the data of the topology as well" || exit 1
+ joined
+ exit 0
+ else
+ echo "join: no peer holds the data of the topology yet"
+ fi
+ elif [ "$FIRST" = yes ] && all_others_pending; then
+ seed "every other peer waits for the data of the topology as well" || exit 1
+ exit 0
+ fi
+ if [ "$i" -lt "$REPLICATION_RETRY_COUNT" ]; then
+ pause "join: not joined and initialized yet ($i of $REPLICATION_RETRY_COUNT)"
+ fi
+done
+
+# Only the first peer may decide that there is no topology to join and that its own data
+# seeds one, and only while no peer that could hold the topology's data answers; anyone
+# else staying unhealthy is what surfaces a lost topology instead of forking it
+if [ "$FIRST" = yes ] && [ "$member" = no ] && ! any_other_may_hold_data; then
+ seed "no topology found after $REPLICATION_RETRY_COUNT attempts" || exit 1
+ exit 0
+fi
+
+give_up
diff --git a/opendj-packages/opendj-docker/bootstrap/replicate.sh b/opendj-packages/opendj-docker/bootstrap/replicate.sh
index 470e721..9f15ed7 100755
--- a/opendj-packages/opendj-docker/bootstrap/replicate.sh
+++ b/opendj-packages/opendj-docker/bootstrap/replicate.sh
@@ -15,12 +15,18 @@
# Replicate to the master server hostname defined in $1
# If that server is ourself this is a no-op
+#
+# This is the one-shot bootstrap path of the srs, sdsr and rg replication types, kept as it
+# is for compatibility; OPENDJ_REPLICATION_TYPE=simple is served by bootstrap/join.sh, which
+# runs in the background next to the server on every start (#1086)
# This is a bit kludgy.
# The hostname has to be a fully resolvable DNS name in the cluster
# If the service is called
MYHOSTNAME=${MYHOSTNAME:-$(hostname -f)}
+ADMIN_PORT=${ADMIN_PORT:-4444}
+REPLICATION_PORT=${REPLICATION_PORT:-8989}
export PATH=/opt/opendj/bin:$PATH
echo "Setting up replication from $MYHOSTNAME to $MASTER_SERVER"
@@ -38,12 +44,13 @@
# The tools read the root password from a file, so that it shows neither in the container log
# nor on the command line of a process while the tool runs. mktemp creates the file readable by
# its owner only. It goes to the tmpfs of /dev/shm where there is one: the EXIT trap does not
-# run if the script is killed, and a file left in /tmp would stay in the writable layer of the
-# container, where no later start removes it, since replicate.sh runs only on the first one.
-# On Kubernetes /dev/shm belongs to the pod: it is shared by all its containers and outlives a
-# restart of this one, so run.sh removes a file left there by name on every start; the name
-# carries ADMIN_PORT, so that it removes only the file of this container.
-PASSWORD_FILE=$(mktemp -p /dev/shm "opendj-replicate.$ADMIN_PORT.XXXXXX" 2>/dev/null || mktemp) || exit 1
+# run if the script is killed, and a file left in /tmp stays in the writable layer of the
+# container until run.sh removes it, by its name, on the next start. On Kubernetes /dev/shm belongs to the pod: it is shared by all
+# its containers and outlives a restart of this one, so run.sh removes a file left there by
+# name on every start; the name carries ADMIN_PORT, so that it removes only the file of this
+# container.
+PASSWORD_FILE=$(mktemp -p /dev/shm "opendj-replicate.$ADMIN_PORT.XXXXXX" 2>/dev/null \
+ || mktemp "/tmp/opendj-replicate.$ADMIN_PORT.XXXXXX") || exit 1
trap 'rm -f "$PASSWORD_FILE"' EXIT
printf '%s\n' "$ROOT_PASSWORD" >"$PASSWORD_FILE" || exit 1
@@ -77,43 +84,22 @@
sleep 5
-if [ "$OPENDJ_REPLICATION_TYPE" == "simple" ]; then
- echo "Enabling Standard Replication..."
- retry 8 /opt/opendj/bin/dsreplication \
- enable \
- --host1 $MASTER_SERVER \
- --port1 4444 \
- --bindDN1 "$ROOT_USER_DN" \
- --bindPasswordFile1 "$PASSWORD_FILE" --replicationPort1 8989 \
- --host2 $MYHOSTNAME --port2 4444 --bindDN2 "$ROOT_USER_DN" \
- --bindPasswordFile2 "$PASSWORD_FILE" --replicationPort2 8989 \
- --adminUID admin --adminPasswordFile "$PASSWORD_FILE" \
- --baseDN "$BASE_DN" -X -n || exit
-
- echo "initializing replication"
-
- # replicating data in MASTER_SERVER to MYHOSTNAME:
- retry any /opt/opendj/bin/dsreplication initialize --baseDN "$BASE_DN" \
- --adminUID admin --adminPasswordFile "$PASSWORD_FILE" \
- --hostSource $MASTER_SERVER --portSource 4444 \
- --hostDestination $MYHOSTNAME --portDestination 4444 -X -n
-
-elif [ "$OPENDJ_REPLICATION_TYPE" == "srs" ]; then
+if [ "$OPENDJ_REPLICATION_TYPE" == "srs" ]; then
echo "Enabling Standalone Replication Servers..."
retry 8 dsreplication enable \
--adminUID admin \
--adminPasswordFile "$PASSWORD_FILE" \
--baseDN "$BASE_DN" \
--host1 $MYHOSTNAME \
- --port1 4444 \
+ --port1 "$ADMIN_PORT" \
--bindDN1 "$ROOT_USER_DN" \
--bindPasswordFile1 "$PASSWORD_FILE" \
--noReplicationServer1 \
--host2 $MASTER_SERVER \
- --port2 4444 \
+ --port2 "$ADMIN_PORT" \
--bindDN2 "$ROOT_USER_DN" \
--bindPasswordFile2 "$PASSWORD_FILE" \
- --replicationPort2 8989 \
+ --replicationPort2 "$REPLICATION_PORT" \
--onlyReplicationServer2 \
--trustAll \
--no-prompt || exit
@@ -126,7 +112,7 @@
--adminPasswordFile "$PASSWORD_FILE" \
--baseDN "$BASE_DN" \
--hostname $MYHOSTNAME \
- --port 4444 \
+ --port "$ADMIN_PORT" \
--trustAll \
--no-prompt
@@ -138,11 +124,11 @@
--adminPasswordFile "$PASSWORD_FILE" \
--baseDN "$BASE_DN" \
--host1 $MASTER_SERVER \
- --port1 4444 \
+ --port1 "$ADMIN_PORT" \
--bindDN1 "$ROOT_USER_DN" \
--bindPasswordFile1 "$PASSWORD_FILE" \
--host2 $MYHOSTNAME \
- --port2 4444 \
+ --port2 "$ADMIN_PORT" \
--bindDN2 "$ROOT_USER_DN" \
--bindPasswordFile2 "$PASSWORD_FILE" \
--noReplicationServer2 \
@@ -157,9 +143,9 @@
--adminPasswordFile "$PASSWORD_FILE" \
--baseDN "$BASE_DN" \
--hostSource $MASTER_SERVER \
- --portSource 4444 \
+ --portSource "$ADMIN_PORT" \
--hostDestination $MYHOSTNAME \
- --portDestination 4444 \
+ --portDestination "$ADMIN_PORT" \
--trustAll \
--no-prompt
@@ -168,7 +154,7 @@
dsconfig \
set-replication-domain-prop \
- --port 4444 \
+ --port "$ADMIN_PORT" \
--hostname $MYHOSTNAME \
--bindDN "$ROOT_USER_DN" \
--bindPasswordFile "$PASSWORD_FILE" \
@@ -180,7 +166,7 @@
retry any dsconfig \
set-replication-server-prop \
- --port 4444 \
+ --port "$ADMIN_PORT" \
--hostname $MASTER_SERVER \
--bindDN "$ROOT_USER_DN" \
--bindPasswordFile "$PASSWORD_FILE" \
diff --git a/opendj-packages/opendj-docker/run.sh b/opendj-packages/opendj-docker/run.sh
index d7f1943..0ba6d02 100755
--- a/opendj-packages/opendj-docker/run.sh
+++ b/opendj-packages/opendj-docker/run.sh
@@ -41,9 +41,25 @@
# ADMIN_PORT in their name, which the containers of a pod cannot share as they share one
# network namespace. Containers in distinct network namespaces that share /dev/shm through
# --ipc=host, and listen on the same ADMIN_PORT, are not told apart. /tmp belongs to this
-# container alone, so a password file left there is removed whatever its name.
-rm -f /dev/shm/opendj-replicate."$ADMIN_PORT".*
-rm -f /dev/shm/opendj-setup-password."$ADMIN_PORT".* /tmp/opendj-setup-password.*
+# container alone, so a password file left there is removed whatever its ADMIN_PORT.
+rm -f /dev/shm/opendj-replicate."$ADMIN_PORT".* /dev/shm/opendj-join."$ADMIN_PORT".*
+rm -f /dev/shm/opendj-setup-password."$ADMIN_PORT".*
+rm -f /tmp/opendj-setup-password.* /tmp/opendj-replicate.* /tmp/opendj-join.*
+
+# What the volume went through is recorded next to the instance: a volume this script
+# bootstraps carries the marker from before its bootstrap until the join has initialized it
+# from the topology (see bootstrap/join.sh), however many restarts that takes
+export INITIALIZE_PENDING=${INITIALIZE_PENDING:-/opt/opendj/data/.replication-initialize-pending}
+# and a volume whose replication the join took down to enable it anew carries this one until
+# the enable registered it again: it holds the data of the topology, but takes writes that
+# replicate nowhere meanwhile
+export REJOIN_PENDING=${REJOIN_PENDING:-/opt/opendj/data/.replication-rejoin-pending}
+
+# The background join (bootstrap/join.sh) serves OPENDJ_REPLICATION_TYPE=simple on every
+# start; srs, sdsr and rg keep the one-shot replicate.sh of the first bootstrap
+join_requested() {
+ [ "$OPENDJ_REPLICATION_TYPE" = "simple" ] && [ -n "${REPLICATION_PEERS:-$MASTER_SERVER}" ]
+}
# Keystores and truststores mounted as a volume (a Kubernetes Secret, say) are copied into
# the instance on every start, not only on the one that bootstraps it: the instance lives on
@@ -122,7 +138,34 @@
# nothing is bootstrapped here, the instance is already there - but a half-migrated one
# is not ready to serve either, so the marker follows the upgrade
if sh ./upgrade -n; then
- touch "$BOOTSTRAP_COMPLETE"
+ # A server whose volume holds the data of the topology is ready as soon as it serves:
+ # gating it on its peers would deadlock a whole-cluster restart under OrderedReady,
+ # where -0 would wait for peers the StatefulSet starts only once -0 is ready - and so
+ # would gating it on the join, which binds with the bootstrap's ROOT_PASSWORD and can
+ # no longer do anything once that password was changed. A volume that never received
+ # that data - its bootstrap failed or was killed, its join or its initialize never
+ # completed - still carries the marker and waits for the join, so none of these turns
+ # into a healthy server with bootstrap data only, the way they did when the marker
+ # followed the upgrade alone. A seed that no peer joined yet holds the data without a
+ # replication domain, which is why the domain is not what decides. Nor does a volume
+ # whose replication the join took down turn healthy before the enable that follows
+ # registered it again: the join could bind with ROOT_PASSWORD when it took it down, and
+ # the peer it enables through answered then.
+ if ! join_requested || { [ ! -f "$INITIALIZE_PENDING" ] && [ ! -f "$REJOIN_PENDING" ]; }; then
+ # the hold on the writes is configuration, and only the join lets them in again
+ if ! join_requested && [ -s "$REJOIN_PENDING" ]; then
+ echo "The backend $(cat "$REJOIN_PENDING") may still refuse the writes of clients from a rejoin that did not complete, and no join runs to let them in again: once its replication is settled, set its writability-mode back to enabled with dsconfig set-backend-prop"
+ fi
+ touch "$BOOTSTRAP_COMPLETE"
+ elif [ -f "$INITIALIZE_PENDING" ]; then
+ echo "This instance never joined its replication topology or never received its data, the join decides whether it is healthy"
+ else
+ echo "This instance was taken out of its replication topology to join it anew, the join decides whether it is healthy"
+ fi
+ # the join also repairs membership that changed while no container ran, on every start
+ if join_requested; then
+ /opt/opendj/bootstrap/join.sh &
+ fi
else
echo "Upgrade failed, this container will not report itself healthy"
fi
@@ -140,13 +183,21 @@
BOOTSTRAP=${BOOTSTRAP:-/opt/opendj/bootstrap/setup.sh}
echo "Running $BOOTSTRAP"
BOOTSTRAPPED=true
+# marked before the bootstrap, so that one that fails or is killed half-way never leaves a
+# volume that counts as holding the data of the topology; a bootstrapped volume has entries
+# whether or not it was asked for any, so it is initialized from the topology (see
+# bootstrap/join.sh)
+if join_requested; then
+ touch "$INITIALIZE_PENDING"
+fi
if ! sh "${BOOTSTRAP}"; then
BOOTSTRAPPED=false
echo "$BOOTSTRAP failed, this container will not report itself healthy"
fi
-# Check if OPENDJ_REPLICATION_TYPE var is set. If it is - replicate to that server
-if [ -n "${MASTER_SERVER}" ] && [ -n "${OPENDJ_REPLICATION_TYPE}" ]; then
+# Check if OPENDJ_REPLICATION_TYPE var is set. If it is - replicate to that server; the
+# background join runs next to the server, below
+if ! join_requested && [ -n "${MASTER_SERVER}" ] && [ -n "${OPENDJ_REPLICATION_TYPE}" ]; then
if ! /opt/opendj/bootstrap/replicate.sh; then
BOOTSTRAPPED=false
echo "Replication setup failed, this container will not report itself healthy"
@@ -170,10 +221,17 @@
fi
# Everything the instance was asked to be set up with - its backend, its base entry, its
-# replication - is in place from here on, so the health check may start probing the server
+# replication - is in place from here on, so the health check may start probing the server.
+# With the background join, replication is the one thing still outstanding: the join writes
+# the health marker once this server is a member of its topology, and not before.
if [ "$BOOTSTRAPPED" = true ]; then
- touch "$BOOTSTRAP_COMPLETE"
- echo "The instance is bootstrapped, the health check may probe it"
+ if join_requested; then
+ /opt/opendj/bootstrap/join.sh &
+ echo "The instance is bootstrapped, joining the replication topology decides whether it is healthy"
+ else
+ touch "$BOOTSTRAP_COMPLETE"
+ echo "The instance is bootstrapped, the health check may probe it"
+ fi
fi
start_server
--
Gitblit v1.10.0