#!/usr/bin/bash
# @file
# @brief CII OLDB KVDB cluster creation tool
#
# @copyright
#   SPDX-FileCopyrightText: 2022,2024,2026 European Southern Observatory (ESO) @n
#   SPDX-License-Identifier: LGPL-3.0-only

# Sets up a redis/valkey cluster.
# Original: cii/srv/incubation/oldb-cluster/bash/src/oldb-cluster-ctl


# Usage
# =======================================================

function help {
    this=`basename $0`
    cat <<EOF
CII OLDB Main KVDB Cluster Maker Tool

Usage: [Options] [Command]+
Options:
  --size <n>     : number of masters/primaries
  --port <n>     : port of first node
  --host <ip>    : nic used by kvdb-server
  --exec <path>  : path to kvdb-executable
  --ccli <name>  : name of kvdb-client executable
  --tsec <secs>  : timeout for start/provision
  --trgd <path>  : where to create conf files
  --cfgd <path>  : where to create cluster files
  --dird <path>  : where to create work files
  --pidd <path>  : where to create pid files
  --logd <path>  : where to create log files
Commands:
  generate       : pre-configure cluster
  start          : start cluster nodes
  provision      : post-configure cluster
  ready          : check cluster ready
  stop           : stop cluster nodes
  list           : display files/folders
Example:
  spec="--size 3 --tsec 10"
  $this \$spec generate start provision
  $this \$spec ready && echo "ok" || echo "nope"
EOF
}

# Set-up and starting / stopping of the Cluster
# ===========================================================


# Put a config for each cluster-node
function configure_nodes {

    run install -d  -o ${props[user]} -g ${props[ugrp]}  ${props[trgd]}

    # These dirs must exist when we try to start a kvdb process.
    # I used to create them late (in the start function), but for
    # when we start through systemd, so we do this already here.
    # And they should be writable by the database user (or root)
    run install -d  -o ${props[user]} -g ${props[ugrp]}  ${props[cfgd]}
    run install -d  -o ${props[user]} -g ${props[ugrp]}  ${props[pidd]}
    run install -d  -o ${props[user]} -g ${props[ugrp]}  ${props[logd]}

    for i in $(seq 1 ${props[size]}) ; do
        let port=${props[port]}+i-1
        run install -d  -o ${props[user]} -g ${props[ugrp]}  ${props[dird]}/kvdb_$port
        create_node_config $port
    done
}

function create_node_config {
        $port=$1

        # shared config
        # impl-note: writing it here means we'll write this file repeatedly, but
        #            that won't hurt, and it allows me re-use of "main_conf_path"

        main_conf_path=${props[trgd]}/kvdb.conf

        run dd of=$main_conf_path <<EOF
            # cii oldb cluster shared config

            bind 127.0.0.1
            protected-mode yes

            save ""
            appendonly no

            loglevel notice

            # clustering

            cluster-enabled yes
            cluster-node-timeout 5000
EOF

        # per node

        node_conf_path=${props[trgd]}/node_$port.conf

        run dd of=$node_conf_path <<EOF

            include $main_conf_path

            port $port

            dir     ${props[dird]}/kvdb_$port
            pidfile ${props[pidd]}/kvdb_$port.pid
            logfile ${props[logd]}/kvdb_$port.log
            cluster-config-file ${props[cfgd]}/cluster-config-file_$port.conf

EOF

        # Dedent. Why not use "<<-EOF"? It needs tabs, which some editors will eat.
        sed "s/^[ \t]*//" -i $main_conf_path
        sed "s/^[ \t]*//" -i $node_conf_path

        # be owned by valkey/redis user
        chown ${props[user]}:${props[ugrp]} $main_conf_path
        chown ${props[user]}:${props[ugrp]} $node_conf_path
}



# Starts all kvdb-nodes
# Ret: 0 = all started
function start_nodes {
    rc=0
    for i in $(seq 1 ${props[size]}) ; do
        start_node $i
        rc=$(($?+rc))
    done
    return $rc
}


# Starts one kvdb-node on a dedicated cpu core
# Args:  <node-number starting with 1>
# Ret: 0 = started, 1 = otherwise
function start_node {
    node=$1
    let core=node
    let port=${props[port]}+node-1

    # 2024-01: there is no advantage in doing this (but disadvantages).
    # run taskset -c $core ${props[exec]}

    run ${props[exec]} \
        ${props[trgd]}/node_$port.conf  \
        --loglevel verbose \
        --supervised no \&

    for i in $(seq 1 $((${props[tsec]}*2)) ); do
        sleep 0.5
        ${props[ccli]} -p $port ping &>/dev/null && return 0
    done
    echo "node $node hasn't started (yet?), waited ${props[tsec]} seconds for it"
    return 1
}


# Stops all kvdb-nodes
function stop_nodes {
    for i in $(seq 1 ${props[size]}) ; do
        stop_node $i
    done
}

# Stops one kvdb-node
# Args: <node-number starting with 1>
function stop_node {
    node=$1
    let port=${props[port]}+node-1
    run ${props[ccli]} -h ${props[host]} -p $port  SHUTDOWN
}



# Un-introduce all cluster nodes from each other
function forget_nodes {
    for ((n=1; n<=${props[size]}; n++)) ; do
        forget_node $n
    done
}

# Removes one kvdb-node
# Args: <node-number starting with 1>
function forget_node {
    node=$1
    let port=${props[port]}+node-1
    node_id=`${props[ccli]} -h ${props[host]} -p $port  CLUSTER MYID`

    # Notes:
    # To completely remove a node, the command must be sent to all remaining nodes
    # A replica will refuse to forget its master. A node will refuse to forget itself.

    for ((i=1; i<=${props[size]}; i++)) ; do
        if [ $i == $node ] ; then continue ; fi
        let port=${props[port]}+i-1
        run ${props[ccli]} -h ${props[host]} -p $port  CLUSTER FORGET $node_id
    done
}


# Assigns slots to kvdb master nodes
function cluster_addslots {

    # hashslots will be distributed evenly across the nodes,
    # any modulo-remainder goes to the last node.

    let x=16384/${props[size]}
    for ((i=0; i<${props[size]}; i++)) ; do
        let j=i+1
        let from=i*x
        let to=j*x-1
        if [ $j == ${props[size]} ] ; then to=16383 ; fi

        let port=${props[port]}+i

        hash_def="CLUSTER ADDSLOTS `seq -s " " $from $to`"

        run ${props[ccli]} -h ${props[host]} \
            -p $port $hash_def
    done

}


# Introduce all cluster nodes to each other
# Ret: 0 = ok, 1 otherwise
function cluster_meet {

    # Pre-check they are all in cluster mode.
    # If the 1st is not, the meet command will fail
    # If the rest are not, the meet command won't have an effect
    local bad=
    for ((p=0; p<${props[size]}; p++)) ; do
        let port=${props[port]}+p
        ${props[ccli]} -p $port info | grep "cluster_enabled:1" &>/dev/null || bad="$bad$port "
    done
    [ -z "$bad" ] || { echo "cluster support not enabled: $bad" ; return 1 ; }


    # Notes:
    # - It's enough to talk to one node, it will share the info with all nodes it knows
    # - "all" nodes includes "self" as well as all replicas (aka. slaves)
    #
    # Update msc 2024-04:
    # This has become less reliable in redis 7 (DevEnv 5, on a VM with 2GB RAM).
    # Symptom: only the first node reports it is in a cluster of n nodes, while the others
    #   report they are in a cluster of 1. Seems redis gossip back-talk does not work!?!

    for ((p=0; p<${props[size]}; p++)) ; do
         let port=${props[port]}+p

         run ${props[ccli]} -h ${props[host]} -p ${props[port]} \
            CLUSTER MEET ${props[host]} $port

         # msc 2024-04: also do the reverse meet, this improves it by provoking more gossip.
         # it is nonetheless necessary to run a "cluster_ready" (newly added) to be sure.
         run ${props[ccli]} -h ${props[host]} -p $port \
            CLUSTER MEET ${props[host]} ${props[port]}
    done
}


# In particular on slow hosts, it takes some seconds until the internal
# communication has finished and the cluster is responsive.
# 2024-04: and may also just not work, so nodes do not meet.
function cluster_ready {

    # Post-check

    local bad=
    local bad_run=
    local bad_met=
    local bad_sta=
    for i in $(seq 1 $((${props[tsec]}*2)) ); do
        bad=
        bad_run=
        bad_met=
        bad_sta=

        sleep 0.5

        for ((p=0; p<${props[size]}; p++)) ; do
            let port=${props[port]}+p

            ${props[ccli]} -p $port ping &>/dev/null \
              || { bad=yes ; bad_run="$bad_run$port " ; continue ; }

            ${props[ccli]} -p $port cluster info \
              | grep "cluster_size:${props[size]}" &>/dev/null \
              || { bad=yes ; bad_met="$bad_met$port " ; continue ; }

            ${props[ccli]} -p $port cluster info \
              | grep "cluster_state:ok" &>/dev/null \
              || { bad=yes ; bad_sta="$bad_sta$port " ; continue ; }

        done
        [ -z "$bad" ] && break # nothing bad found, we're good

   done
   [ -z "$bad_run" ] || echo "cluster nodes are not running: $bad_run"
   [ -z "$bad_met" ] || echo "cluster nodes have not met (slow host?): $bad_met"
   [ -z "$bad_sta" ] || echo "cluster state not reported as ok: $bad_sta"

   [ -z "$bad"  ] || { echo "cluster is not ready" ; return 1 ; }

   echo "cluster is ready"
   return 0
}



# User commands
# ======================================================

function generate {

    ECHO "configure_nodes"
    configure_nodes
}

function start {

    ECHO "start_nodes"
    start_nodes || { rc=$? ; echo "failed with rc $rc" ; return $rc ; }
}

function provision {

    ECHO "cluster_addslots"
    cluster_addslots

    ECHO "cluster_meet"
    cluster_meet || { rc=$? ; echo "failed with rc $rc" ; return $rc ; }

    # 2024-04: redis 7: cluster meet is now more unrealiable
    for ((i=0;i<3;i++)) ; do
        ECHO "cluster_ready"
        cluster_ready && break || cluster_meet
    done
}

function ready {

    ECHO "cluster_ready"
    cluster_ready
}

function stop {

    ECHO "stop_nodes"
    stop_nodes
}

function list {
    run ls -l ${props[trgd]}
    run ls -l ${props[cfgd]}
    run ls -l ${props[dird]}
    run ls -l ${props[pidd]}
    run ls -l ${props[logd]}
}

# Internal Utils
# ===========================================================

cyan=`tput setaf 6` # 1=red, 2=green, 4=blue
undr=`tput smul`
bold=`tput bold`
norm=`tput sgr0`

function ECHO {
    echo "${undr}${@}${norm}"
}

# Prefixing commands with this "run" function comes with a few restrictions:
# foremost, redirects (>) and jobcontrol operators (&) must be escaped (\).
function run {

    str=${@} ; len=${#str}
    if [ $len -gt 320 ] ; then str=${str:0:300}"....."${str:len-15:15} ; fi
    echo "${cyan}${str}${norm}"

    eval $@
    return $?
}



# Defaults
# ===========================================================
declare -A props
props[size]=2
props[port]=6379
props[host]=127.0.0.1
props[tsec]=5  # configurable timeout for slow hosts
props[trgd]=$PWD/cluster       # /etc/cii/oldb/cluster
props[cfgd]=$PWD/cluster/conf  # /etc/cii/oldb/cluster
props[dird]=$PWD/cluster/inst  # /var/lib/cii/oldb
props[pidd]=$PWD/cluster/pids  # /var/run/cii/oldb
props[logd]=$PWD/cluster/logs  # /var/log/cii/oldb
props[exec]=`which valkey-server` 2>/dev/null || props[exec]=`which redis-server`
props[ccli]=`which valkey-cli`    2>/dev/null || props[ccli]=`which redis-cli`
props[user]=`basename "${props[exec]}" | cut -d- -f1`  # valkey
props[ugrp]=root

# Main
# ===========================================================

if [ "$1" == "-h" -o "$1" == "--help" ]; then help; exit 0; fi
while [[ $1 == --* ]]; do k="${1#--}"; declare "props[$k]=$2"; shift 2; done
if [ "$1" == "" ]; then help; exit 2; fi
while [ "$1" ]; do
  "$1"
  rc=$?
  shift
done

exit $rc
